Compare commits
163
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
67702fa250 | ||
|
|
8a38540312 | ||
|
|
54635eecf5 | ||
|
|
184d90c2c6 | ||
|
|
a32eed877d | ||
|
|
347a041d85 | ||
|
|
53e42837e0 | ||
|
|
f52c591f5e | ||
|
|
77d2e22ed1 | ||
|
|
a64a01a6a9 | ||
|
|
35e8921de8 | ||
|
|
3fb9d5936a | ||
|
|
452080e997 | ||
|
|
5b63a841ba | ||
|
|
4b9507bd59 | ||
|
|
7dec1b0767 | ||
|
|
45b8fd8614 | ||
|
|
dd22bfbc48 | ||
|
|
6e60daed6a | ||
|
|
446f655c69 | ||
|
|
a785efa4a3 | ||
|
|
767a2a9a7c | ||
|
|
dd044fb115 | ||
|
|
3ec14509a0 | ||
|
|
1f2c60b3ec | ||
|
|
d71a5f25d3 | ||
|
|
2a0f17ae86 | ||
|
|
9ef61a0844 | ||
|
|
688b9101e8 | ||
|
|
1d7ba814dc | ||
|
|
8aba86b4ce | ||
|
|
a0c83f4b3f | ||
|
|
1ea48ed5d6 | ||
|
|
dfc7d6a609 | ||
|
|
77a3ccd33d | ||
|
|
07da27dc39 | ||
|
|
d89657d0a4 | ||
|
|
849b4b62d4 | ||
|
|
8589bf713b | ||
|
|
738f6aa697 | ||
|
|
01e56e3f36 | ||
|
|
d5d15cefcb | ||
|
|
fd96a35eb9 | ||
|
|
e3fe3f3f45 | ||
|
|
60da667b42 | ||
|
|
36145e623b | ||
|
|
565415327d | ||
|
|
bf6bdb389d | ||
|
|
0628944b15 | ||
|
|
e6d0bb7ef0 | ||
|
|
5ca804d827 | ||
|
|
a299363b59 | ||
|
|
d8dd62c639 | ||
|
|
d106b6e8dc | ||
|
|
af2c766f42 | ||
|
|
d26262bf56 | ||
|
|
10b89c06ff | ||
|
|
a2a05d0b3d | ||
|
|
75d62660b0 | ||
|
|
a53c2b0193 | ||
|
|
bf481c080c | ||
|
|
3a78dcab3f | ||
|
|
470f93a83d | ||
|
|
7798706b41 | ||
|
|
ad40de03c2 | ||
|
|
d8778b4b70 | ||
|
|
79d859a111 | ||
|
|
37fb09f75e | ||
|
|
34ebeb968d | ||
|
|
3cc1d38dd0 | ||
|
|
36e804b848 | ||
|
|
be83a42d42 | ||
|
|
650a1b1595 | ||
|
|
d8c55c6f2f | ||
|
|
2ebfb0c3b0 | ||
|
|
888eaa74e1 | ||
|
|
ed76cbbc79 | ||
|
|
140229b88d | ||
|
|
cb2379fd06 | ||
|
|
1665b239b0 | ||
|
|
70ed0337f8 | ||
|
|
116b8f6c41 | ||
|
|
77901bbba6 | ||
|
|
78582d24d2 | ||
|
|
1cdffe21b1 | ||
|
|
4df90bc212 | ||
|
|
713e77354b | ||
|
|
abb7b1b634 | ||
|
|
f5f0e15769 | ||
|
|
8652b4a724 | ||
|
|
e8233127c7 | ||
|
|
88e92254e9 | ||
|
|
5725d35e5b | ||
|
|
8c95525793 | ||
|
|
5bba5d3363 | ||
|
|
80b72db676 | ||
|
|
aa0c97353a | ||
|
|
dd93204b44 | ||
|
|
c7b17aaa0e | ||
|
|
d7a154a596 | ||
|
|
1bf91115dd | ||
|
|
65244e3103 | ||
|
|
75ccfff171 | ||
|
|
2a583d55a8 | ||
|
|
d218d93f79 | ||
|
|
a5fe63c1dd | ||
|
|
6b3ae0c997 | ||
|
|
59104dd988 | ||
|
|
e499f500c3 | ||
|
|
2ab0d129a2 | ||
|
|
d702d2e9ae | ||
|
|
3ec2d1828a | ||
|
|
21657943a8 | ||
|
|
e612d948d2 | ||
|
|
4ef21fa083 | ||
|
|
125a3b4993 | ||
|
|
e7c7e7b94c | ||
|
|
a581712b09 | ||
|
|
8eb4210251 | ||
|
|
56110b0019 | ||
|
|
afbf10f7fc | ||
|
|
b61b7775b9 | ||
|
|
be81394be3 | ||
|
|
47610e8ee2 | ||
|
|
193fd71a50 | ||
|
|
1c2b3ae64d | ||
|
|
d19a0ae38d | ||
|
|
3d1de37d0e | ||
|
|
ceacc6b514 | ||
|
|
8754d4e46a | ||
|
|
15b70856c9 | ||
|
|
83881641ca | ||
|
|
b0f894f50c | ||
|
|
750a73f050 | ||
|
|
eabb81684a | ||
|
|
0a8c81b17a | ||
|
|
9316f9f1c3 | ||
|
|
7384a730be | ||
|
|
b9d9e1e523 | ||
|
|
37fb3cb0cf | ||
|
|
5d57e7b01c | ||
|
|
8a7235c725 | ||
|
|
f4590fcc19 | ||
|
|
bb7597ea0b | ||
|
|
f57a73e8a1 | ||
|
|
724de7bbd0 | ||
|
|
75bc429aa1 | ||
|
|
91aff2bc4a | ||
|
|
546dd44a2a | ||
|
|
7501bd1703 | ||
|
|
1b47de5058 | ||
|
|
26ac97df31 | ||
|
|
37f72a4df8 | ||
|
|
f02259cae0 | ||
|
|
5940864958 | ||
|
|
91f2cfa17b | ||
|
|
debe815a5c | ||
|
|
dba3939a0f | ||
|
|
43afe6bf2e | ||
|
|
ed7f542006 | ||
|
|
36c29d2d6d | ||
|
|
941ab091db | ||
|
|
cf7c6df41c |
@@ -0,0 +1,16 @@
|
|||||||
|
# EditorConfig: https://editorconfig.org/
|
||||||
|
# Follows the Zig style guide: https://ziglang.org/documentation/0.16.0/#Style-Guide
|
||||||
|
|
||||||
|
root = true
|
||||||
|
|
||||||
|
[*]
|
||||||
|
charset = utf-8
|
||||||
|
end_of_line = lf
|
||||||
|
indent_style = space
|
||||||
|
indent_size = 4
|
||||||
|
trim_trailing_whitespace = true
|
||||||
|
insert_final_newline = true
|
||||||
|
|
||||||
|
[*.zig]
|
||||||
|
# "Line length: aim for 100; use common sense."
|
||||||
|
max_line_length = 100
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
*.zig text eol=lf
|
||||||
@@ -4,3 +4,6 @@ zig-out/
|
|||||||
|
|
||||||
# JetBrains IDE
|
# JetBrains IDE
|
||||||
.idea/
|
.idea/
|
||||||
|
|
||||||
|
.claude/
|
||||||
|
.github/
|
||||||
@@ -2,12 +2,31 @@
|
|||||||
Codename: Shodan
|
Codename: Shodan
|
||||||
Version: 1
|
Version: 1
|
||||||
|
|
||||||
A small operating system, written from scratch in Zig — a bootloader (`src/boot/`)
|
A small resilient operating system, written from scratch in Zig.
|
||||||
and a microkernel (`src/kernel/`), sharing a neutral handoff contract (`src/root.zig`).
|
|
||||||
It boots x86-64 via UEFI, and so far has a framebuffer console, a physical frame
|
## Zen of DanOS:
|
||||||
allocator, its own paging with W^X permissions, interrupt/exception handling, a
|
|
||||||
LAPIC timer, a kernel heap, a fixed-priority preemptive scheduler, and in-kernel IPC
|
- Resilient Micro-Kernel Architecture.
|
||||||
channels. See [`docs/`](docs/README.md) for how each piece works.
|
- Every process run in an isolated user space not kernel space.
|
||||||
|
- Processes cannot take down the entire OS with it when they die or is killed
|
||||||
|
- Stable public runtime library, private OS ABI.
|
||||||
|
- Keeps a stable runtime for user space processes between OS versions (great for backwards compatibility)
|
||||||
|
- Allows the underlying OS to be changed without effecting applications
|
||||||
|
- Provides a boundary to enable compatibility between OS's e.g. POSIX, MUSL etc
|
||||||
|
- Drivers are just isolated processes in user space.
|
||||||
|
- Thin binaries that can be restarted like applications.
|
||||||
|
- Useful during driver development.
|
||||||
|
- Drivers can claim MMIO / ports
|
||||||
|
- Driver resources (e.g. IRQ/Port/MMIO) claims are automatically cleaned up if the driver dies or is killed
|
||||||
|
- Drivers can also hook into the process lifecyle to clean up or reset hardware
|
||||||
|
- No legacy to deal with
|
||||||
|
- Zig code uses a clean coding style (Zen of Zig)
|
||||||
|
- Favor reading code over writing code.
|
||||||
|
- No magic numbers.
|
||||||
|
- No shortend names unless its for ABI compatibility or acronyms
|
||||||
|
- Inter-Process Communication (IPC)
|
||||||
|
- Publish and subscribe to Asynchronous Messages
|
||||||
|
- Talk to services and processes synchronously
|
||||||
|
|
||||||
## Prerequisites
|
## Prerequisites
|
||||||
|
|
||||||
@@ -30,8 +49,10 @@ channels. See [`docs/`](docs/README.md) for how each piece works.
|
|||||||
zig build
|
zig build
|
||||||
```
|
```
|
||||||
|
|
||||||
Produces the UEFI bootloader (`zig-out/bin/BOOTX64.efi`) and the kernel ELF
|
Produces a FHS-shaped `zig-out/` that *is* the danos filesystem and the boot volume:
|
||||||
(`zig-out/bin/kernel`).
|
the UEFI bootloader at `zig-out/EFI/BOOT/BOOTX64.efi`, the kernel at
|
||||||
|
`zig-out/system/kernel`, init at `zig-out/system/services/init`, drivers under
|
||||||
|
`zig-out/system/drivers/`, and the initial-ramdisk at `zig-out/boot/`.
|
||||||
|
|
||||||
## Run
|
## Run
|
||||||
|
|
||||||
@@ -58,9 +79,13 @@ straight into CI.
|
|||||||
|
|
||||||
## Documentation
|
## Documentation
|
||||||
|
|
||||||
Design notes explaining the *why* behind the code live in
|
Design notes explaining *why* behind the code live in
|
||||||
[`docs/`](docs/README.md) — start with [`docs/README.md`](docs/README.md).
|
[`docs/`](docs/README.md) — start with [`docs/README.md`](docs/README.md).
|
||||||
|
|
||||||
|
For the hardware needed to run DanOS — minimum specs plus a plain-language guide
|
||||||
|
matching Intel/AMD CPU generations by name — see
|
||||||
|
[`docs/system-requirements.md`](docs/system-requirements.md).
|
||||||
|
|
||||||
## Logo
|
## Logo
|
||||||
|
|
||||||
San Serif Text "Dan OS" with a black karate belt around it.
|
San Serif Text "Dan OS" with a black karate belt around it.
|
||||||
|
|||||||
+268
-55
@@ -1,15 +1,24 @@
|
|||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const uefi = std.os.uefi;
|
const uefi = std.os.uefi;
|
||||||
const elf = std.elf;
|
const elf = std.elf;
|
||||||
const danos = @import("danos");
|
const boot_handoff = @import("boot-handoff");
|
||||||
const BootInfo = danos.BootInfo;
|
const BootInformation = boot_handoff.BootInformation;
|
||||||
const GraphicsOutput = uefi.protocol.GraphicsOutput;
|
const GraphicsOutput = uefi.protocol.GraphicsOutput;
|
||||||
const EdidActive = uefi.protocol.edid.Active;
|
const EdidActive = uefi.protocol.edid.Active;
|
||||||
const MemoryMapSlice = uefi.tables.MemoryMapSlice;
|
const MemoryMapSlice = uefi.tables.MemoryMapSlice;
|
||||||
|
|
||||||
/// Name of the kernel ELF on the boot volume (installed to the ESP root by
|
// The boot volume is the FHS-shaped zig-out (see build.zig / docs/README.md), so the
|
||||||
/// build.zig). UEFI wants a UTF-16, null-terminated path.
|
// loader reads each artifact from its addressed FHS path. UEFI paths use backslashes;
|
||||||
const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("kernel");
|
// the FAT driver walks the components itself, so no per-directory dance is needed.
|
||||||
|
|
||||||
|
/// The kernel image: /system/kernel.
|
||||||
|
const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("system\\kernel");
|
||||||
|
|
||||||
|
/// The init program: /system/services/init.
|
||||||
|
const init_file_name = std.unicode.utf8ToUtf16LeStringLiteral("system\\services\\init");
|
||||||
|
|
||||||
|
/// The initial-ramdisk (the VFS server + drivers), in /boot.
|
||||||
|
const initial_ramdisk_file_name = std.unicode.utf8ToUtf16LeStringLiteral("boot\\initial-ramdisk.img");
|
||||||
|
|
||||||
/// Physical page size, and the sentinel UEFI uses to seek to end-of-file.
|
/// Physical page size, and the sentinel UEFI uses to seek to end-of-file.
|
||||||
const page_size = 4096;
|
const page_size = 4096;
|
||||||
@@ -20,7 +29,7 @@ pub fn main() uefi.Status {
|
|||||||
// report the reason (boot services are still up) and park the machine so the
|
// report the reason (boot services are still up) and park the machine so the
|
||||||
// message stays on screen.
|
// message stays on screen.
|
||||||
boot() catch |err| {
|
boot() catch |err| {
|
||||||
log("\r\ndanos: boot failed: ");
|
log("\r\nEFI: boot failed: ");
|
||||||
logBytes(@errorName(err));
|
logBytes(@errorName(err));
|
||||||
log("\r\n");
|
log("\r\n");
|
||||||
while (true) asm volatile ("hlt");
|
while (true) asm volatile ("hlt");
|
||||||
@@ -33,10 +42,10 @@ fn boot() !noreturn {
|
|||||||
|
|
||||||
// Everything the kernel needs must be gathered *before* we exit boot
|
// Everything the kernel needs must be gathered *before* we exit boot
|
||||||
// services, since afterwards none of these calls are usable.
|
// services, since afterwards none of these calls are usable.
|
||||||
var boot_info: BootInfo = .{
|
var boot_information: BootInformation = .{
|
||||||
// A missing GOP (a headless machine) is not fatal — hand the kernel a
|
// A missing GOP (a headless machine) is not fatal — hand the kernel a
|
||||||
// "no framebuffer" descriptor (base 0) and let it log to serial instead.
|
// "no framebuffer" descriptor (base 0) and let it log to serial instead.
|
||||||
.framebuffer = queryFramebuffer(bs) catch danos.Framebuffer{
|
.framebuffer = queryFramebuffer(bs) catch boot_handoff.Framebuffer{
|
||||||
.base = 0,
|
.base = 0,
|
||||||
.width = 0,
|
.width = 0,
|
||||||
.height = 0,
|
.height = 0,
|
||||||
@@ -52,16 +61,38 @@ fn boot() !noreturn {
|
|||||||
.acpi_rsdp = if (acpiRootSystemDescriptorPointer()) |p| @intFromPtr(p) else 0,
|
.acpi_rsdp = if (acpiRootSystemDescriptorPointer()) |p| @intFromPtr(p) else 0,
|
||||||
};
|
};
|
||||||
|
|
||||||
const entry = try loadKernel(bs, &boot_info);
|
const entry = try loadKernel(bs, &boot_information);
|
||||||
|
|
||||||
log("danos: kernel loaded, exiting boot services\r\n");
|
// Best effort: a volume without /system/services/init still boots (kernel-only).
|
||||||
boot_info.memory_map = try exitBootServices(bs);
|
loadInit(bs, &boot_information) catch |err| {
|
||||||
|
log("EFI: no /system/services/init (");
|
||||||
|
logBytes(@errorName(err));
|
||||||
|
log(") - booting without user space\r\n");
|
||||||
|
};
|
||||||
|
|
||||||
// Hand control to the kernel. `danos.kernel_abi` is SysV, so the pointer is
|
// Best effort: the initial_ramdisk (VFS server + drivers) is optional too.
|
||||||
// passed in RDI as the kernel expects — not RCX, which this UEFI binary's
|
loadInitialRamdisk(bs, &boot_information) catch |err| {
|
||||||
// default `.c` convention (Microsoft x64) would use.
|
log("EFI: no initial_ramdisk (");
|
||||||
const kernel: *const fn (*const BootInfo) callconv(danos.kernel_abi) noreturn = @ptrFromInt(entry);
|
logBytes(@errorName(err));
|
||||||
kernel(&boot_info);
|
log(")\r\n");
|
||||||
|
};
|
||||||
|
|
||||||
|
// Build the page tables the kernel starts life on: identity + a physmap of
|
||||||
|
// low RAM, plus the higher-half kernel image once it links high. Allocated
|
||||||
|
// now, while boot services (and the memory map) are still stable — nothing
|
||||||
|
// is allocatable after ExitBootServices, and any allocation between fetching
|
||||||
|
// the map and exiting would invalidate the map key.
|
||||||
|
const cr3 = try buildBootstrapTables(bs, &boot_information);
|
||||||
|
|
||||||
|
log("EFI: kernel loaded, exiting boot services\r\n");
|
||||||
|
boot_information.memory_map = try exitBootServices(bs);
|
||||||
|
|
||||||
|
// Switch onto our tables and jump to the kernel in one uninterruptible step.
|
||||||
|
// We load RDI explicitly (SystemV first arg) rather than trusting this UEFI
|
||||||
|
// binary's Microsoft-x64 default, and jump straight to the (possibly
|
||||||
|
// higher-half) entry — the bootstrap tables map both the low loader code
|
||||||
|
// executing this and the kernel's link address.
|
||||||
|
handoff(cr3, entry, &boot_information);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A display resolution in pixels.
|
/// A display resolution in pixels.
|
||||||
@@ -69,7 +100,7 @@ const Resolution = struct { width: u32, height: u32 };
|
|||||||
|
|
||||||
/// Switch the GPU to the monitor's native resolution (when we can determine it)
|
/// Switch the GPU to the monitor's native resolution (when we can determine it)
|
||||||
/// and read the resulting graphics mode into our own framebuffer description.
|
/// and read the resulting graphics mode into our own framebuffer description.
|
||||||
fn queryFramebuffer(bs: *uefi.tables.BootServices) !danos.Framebuffer {
|
fn queryFramebuffer(bs: *uefi.tables.BootServices) !boot_handoff.Framebuffer {
|
||||||
// Enumerate the handles carrying the Graphics Output Protocol. We go through
|
// Enumerate the handles carrying the Graphics Output Protocol. We go through
|
||||||
// handles (rather than locateProtocol) so we can also ask them for their EDID,
|
// handles (rather than locateProtocol) so we can also ask them for their EDID,
|
||||||
// which is what tells us the panel's native resolution.
|
// which is what tells us the panel's native resolution.
|
||||||
@@ -101,7 +132,7 @@ fn queryFramebuffer(bs: *uefi.tables.BootServices) !danos.Framebuffer {
|
|||||||
|
|
||||||
/// Map a GOP pixel format to ours. bit_mask / blt_only have no linear 32bpp
|
/// Map a GOP pixel format to ours. bit_mask / blt_only have no linear 32bpp
|
||||||
/// layout we can paint into, so they're rejected.
|
/// layout we can paint into, so they're rejected.
|
||||||
fn pixelFormat(fmt: GraphicsOutput.PixelFormat) !danos.PixelFormat {
|
fn pixelFormat(fmt: GraphicsOutput.PixelFormat) !boot_handoff.PixelFormat {
|
||||||
return switch (fmt) {
|
return switch (fmt) {
|
||||||
.red_green_blue_reserved_8_bit_per_color => .rgbx,
|
.red_green_blue_reserved_8_bit_per_color => .rgbx,
|
||||||
.blue_green_red_reserved_8_bit_per_color => .bgrx,
|
.blue_green_red_reserved_8_bit_per_color => .bgrx,
|
||||||
@@ -166,7 +197,7 @@ fn edidNative(edid: []const u8) ?Resolution {
|
|||||||
|
|
||||||
/// Open the kernel on the volume we booted from, read it into a pool buffer,
|
/// Open the kernel on the volume we booted from, read it into a pool buffer,
|
||||||
/// load its segments, and return the physical entry-point address.
|
/// load its segments, and return the physical entry-point address.
|
||||||
fn loadKernel(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !usize {
|
fn loadKernel(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !usize {
|
||||||
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
|
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
|
||||||
return error.NoLoadedImage;
|
return error.NoLoadedImage;
|
||||||
const device = loaded.device_handle orelse return error.NoBootDevice;
|
const device = loaded.device_handle orelse return error.NoBootDevice;
|
||||||
@@ -195,13 +226,190 @@ fn loadKernel(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !usize {
|
|||||||
read_total += n;
|
read_total += n;
|
||||||
}
|
}
|
||||||
|
|
||||||
return loadElf(bs, image, boot_info);
|
return loadElf(bs, image, boot_information);
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- bootstrap page tables -------------------------------------------------
|
||||||
|
// The kernel is (or will be) linked in the higher half but loaded low; the
|
||||||
|
// firmware's identity map doesn't cover the higher half, so the loader builds
|
||||||
|
// the first set of real page tables and switches CR3 before jumping in. They
|
||||||
|
// carry: an identity map of low RAM (so the loader's own code/stack executing
|
||||||
|
// the switch stays valid, and the low-linked kernel keeps working during the
|
||||||
|
// staged move), a physmap at boot_handoff.physmap_base (the kernel's permanent way to
|
||||||
|
// reach physical memory), and 4 KiB mappings of any higher-half kernel segment.
|
||||||
|
// The kernel later builds its own precise tables (paging.init) and abandons
|
||||||
|
// these; they leak as reserved LoaderData (~a handful of frames).
|
||||||
|
|
||||||
|
const pte_present: u64 = 1 << 0;
|
||||||
|
const pte_write: u64 = 1 << 1;
|
||||||
|
const pte_ps: u64 = 1 << 7; // page-size: a 2 MiB leaf at the PD level
|
||||||
|
const pte_address: u64 = 0x000F_FFFF_FFFF_F000;
|
||||||
|
const gib: u64 = 1 << 30;
|
||||||
|
|
||||||
|
/// A bump allocator over a pre-reserved block of zeroed frames, for page tables.
|
||||||
|
const TablePool = struct {
|
||||||
|
base: usize,
|
||||||
|
next: usize,
|
||||||
|
cap: usize,
|
||||||
|
|
||||||
|
fn alloc(self: *TablePool) !u64 {
|
||||||
|
if (self.next >= self.cap) return error.OutOfBootstrapFrames;
|
||||||
|
const frame = self.base + self.next * page_size;
|
||||||
|
self.next += 1;
|
||||||
|
@memset(@as(*[512]u64, @ptrFromInt(frame)), 0);
|
||||||
|
return frame;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn table(physical: u64) *[512]u64 {
|
||||||
|
return @ptrFromInt(physical);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Return the next-level table an entry points at, creating it if absent.
|
||||||
|
fn descend(self: *TablePool, entry: *u64) !u64 {
|
||||||
|
if (entry.* & pte_present != 0) return entry.* & pte_address;
|
||||||
|
const frame = try self.alloc();
|
||||||
|
entry.* = frame | pte_present | pte_write;
|
||||||
|
return frame;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn map2M(self: *TablePool, pml4: u64, virtual: u64, physical: u64) !void {
|
||||||
|
const pml4e = &table(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
|
const pdpt = try self.descend(pml4e);
|
||||||
|
const pdpte = &table(pdpt)[(virtual >> 30) & 0x1FF];
|
||||||
|
const pd = try self.descend(pdpte);
|
||||||
|
table(pd)[(virtual >> 21) & 0x1FF] = (physical & ~@as(u64, 0x1F_FFFF)) | pte_present | pte_write | pte_ps;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn map4K(self: *TablePool, pml4: u64, virtual: u64, physical: u64) !void {
|
||||||
|
const pml4e = &table(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
|
const pdpt = try self.descend(pml4e);
|
||||||
|
const pdpte = &table(pdpt)[(virtual >> 30) & 0x1FF];
|
||||||
|
const pd = try self.descend(pdpte);
|
||||||
|
const pde = &table(pd)[(virtual >> 21) & 0x1FF];
|
||||||
|
const pt = try self.descend(pde);
|
||||||
|
table(pt)[(virtual >> 12) & 0x1FF] = (physical & pte_address) | pte_present | pte_write;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Build the bootstrap tables and return the physical PML4 address (for CR3).
|
||||||
|
/// No NX bits are set anywhere, so EFER.NXE (still off here) is irrelevant.
|
||||||
|
fn buildBootstrapTables(bs: *uefi.tables.BootServices, boot_information: *const BootInformation) !u64 {
|
||||||
|
// 64 frames (256 KiB) — comfortably covers a PML4, two PDPTs, eight PDs for
|
||||||
|
// the 4 GiB identity+physmap ranges, plus the kernel image's PTs.
|
||||||
|
const pool_pages = 64;
|
||||||
|
const block = try bs.allocatePages(.any, .loader_data, pool_pages);
|
||||||
|
var pool = TablePool{ .base = @intFromPtr(block.ptr), .next = 0, .cap = pool_pages };
|
||||||
|
|
||||||
|
const pml4 = try pool.alloc();
|
||||||
|
|
||||||
|
// Identity + physmap for low RAM. 4 GiB covers all of QEMU's RAM and MMIO
|
||||||
|
// (LAPIC/IOAPIC/HPET/ECAM/framebuffer under q35); a machine with RAM or a
|
||||||
|
// framebuffer above 4 GiB would extend this — see the fb window below.
|
||||||
|
var address: u64 = 0;
|
||||||
|
while (address < 4 * gib) : (address += 2 << 20) {
|
||||||
|
try pool.map2M(pml4, address, address); // identity
|
||||||
|
try pool.map2M(pml4, boot_handoff.physicalToVirtual(address), address); // physmap
|
||||||
|
}
|
||||||
|
|
||||||
|
// A framebuffer above the 4 GiB window needs its own identity + physmap
|
||||||
|
// pages (the kernel touches fb.base before it builds its own tables).
|
||||||
|
const fb = boot_information.framebuffer;
|
||||||
|
if (fb.present() and fb.base + @as(u64, fb.pitch) * fb.height > 4 * gib) {
|
||||||
|
var p: u64 = fb.base & ~@as(u64, 0x1F_FFFF);
|
||||||
|
const fb_end = fb.base + @as(u64, fb.pitch) * fb.height;
|
||||||
|
while (p < fb_end) : (p += 2 << 20) {
|
||||||
|
try pool.map2M(pml4, p, p);
|
||||||
|
try pool.map2M(pml4, boot_handoff.physicalToVirtual(p), p);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Higher-half kernel segments (virtual != physical). While the kernel still links
|
||||||
|
// low its segments sit in the identity range and need no separate mapping
|
||||||
|
// (and 4 KiB-mapping them would collide with the 2 MiB identity leaves), so
|
||||||
|
// only map segments that actually live in the higher half.
|
||||||
|
for (boot_information.kernel_segments[0..boot_information.kernel_segment_count]) |seg| {
|
||||||
|
if (seg.virtual < boot_handoff.kernel_virt_base) continue;
|
||||||
|
var off: u64 = 0;
|
||||||
|
while (off < seg.pages * page_size) : (off += page_size) {
|
||||||
|
try pool.map4K(pml4, seg.virtual + off, seg.physical + off);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return pml4;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Switch onto `cr3` and jump to the kernel `entry` with `boot_information` in RDI,
|
||||||
|
/// interrupts off, in one block so nothing runs between the CR3 load and the
|
||||||
|
/// jump. The identity mapping keeps this low loader code valid across the CR3
|
||||||
|
/// load; the jump target is mapped (identity while low, higher-half once high).
|
||||||
|
fn handoff(cr3: u64, entry: usize, boot_information: *const BootInformation) noreturn {
|
||||||
|
asm volatile (
|
||||||
|
\\cli
|
||||||
|
\\movq %[cr3], %%cr3
|
||||||
|
\\movq %[bi], %%rdi
|
||||||
|
\\callq *%[entry]
|
||||||
|
:
|
||||||
|
: [cr3] "r" (cr3),
|
||||||
|
[bi] "r" (boot_information),
|
||||||
|
[entry] "r" (entry),
|
||||||
|
: .{ .memory = true });
|
||||||
|
unreachable;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read a whole file off the boot volume into a pool buffer that outlives the
|
||||||
|
/// loader. The buffer is deliberately NOT freed: it's LoaderData, which the
|
||||||
|
/// memory-map conversion classifies as reserved, so the kernel identity-maps it
|
||||||
|
/// and reads from there. Returns the buffer (pointer + length).
|
||||||
|
fn loadFile(bs: *uefi.tables.BootServices, name: [*:0]const u16) ![]u8 {
|
||||||
|
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
|
||||||
|
return error.NoLoadedImage;
|
||||||
|
const device = loaded.device_handle orelse return error.NoBootDevice;
|
||||||
|
const fs = (try bs.handleProtocol(uefi.protocol.SimpleFileSystem, device)) orelse
|
||||||
|
return error.NoFileSystem;
|
||||||
|
|
||||||
|
const root = try fs.openVolume();
|
||||||
|
defer _ = root.close() catch {};
|
||||||
|
|
||||||
|
const file = try root.open(name, .read, .{});
|
||||||
|
defer _ = file.close() catch {};
|
||||||
|
|
||||||
|
try file.setPosition(seek_end);
|
||||||
|
const size: usize = @intCast(try file.getPosition());
|
||||||
|
try file.setPosition(0);
|
||||||
|
if (size == 0) return error.EmptyFile;
|
||||||
|
|
||||||
|
const image = try bs.allocatePool(.loader_data, size); // survives the handoff
|
||||||
|
|
||||||
|
var read_total: usize = 0;
|
||||||
|
while (read_total < size) {
|
||||||
|
const n = try file.read(image[read_total..]);
|
||||||
|
if (n == 0) return error.UnexpectedEof;
|
||||||
|
read_total += n;
|
||||||
|
}
|
||||||
|
return image[0..size];
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Ferry the init program (/system/services/init) to the kernel. The kernel does the ELF
|
||||||
|
/// loading itself (into ring-3 mappings) — the loader just carries the bytes.
|
||||||
|
fn loadInit(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||||
|
const image = try loadFile(bs, init_file_name);
|
||||||
|
boot_information.init_base = @intFromPtr(image.ptr);
|
||||||
|
boot_information.init_len = image.len;
|
||||||
|
log("EFI: /system/services/init loaded\r\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Ferry the initial_ramdisk (the VFS server + drivers) to the kernel, same as init.
|
||||||
|
fn loadInitialRamdisk(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||||
|
const image = try loadFile(bs, initial_ramdisk_file_name);
|
||||||
|
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
||||||
|
boot_information.initial_ramdisk_len = image.len;
|
||||||
|
log("EFI: initial_ramdisk loaded\r\n");
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
||||||
/// record each segment's layout so the kernel can re-map itself with the right
|
/// record each segment's layout so the kernel can re-map itself with the right
|
||||||
/// permissions.
|
/// permissions.
|
||||||
fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !usize {
|
fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_information: *BootInformation) !usize {
|
||||||
if (image.len < @sizeOf(elf.Elf64_Ehdr)) return error.NotElf;
|
if (image.len < @sizeOf(elf.Elf64_Ehdr)) return error.NotElf;
|
||||||
const ehdr: *const elf.Elf64_Ehdr = @ptrCast(@alignCast(image.ptr));
|
const ehdr: *const elf.Elf64_Ehdr = @ptrCast(@alignCast(image.ptr));
|
||||||
|
|
||||||
@@ -218,7 +426,9 @@ fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !us
|
|||||||
|
|
||||||
// Reserve the exact physical pages this segment is linked at. This
|
// Reserve the exact physical pages this segment is linked at. This
|
||||||
// requires the segment's p_paddr to be free in the firmware memory map;
|
// requires the segment's p_paddr to be free in the firmware memory map;
|
||||||
// if it collides, adjust `image_base` in build.zig.
|
// if it collides, adjust `image_base` in build.zig. (Once the kernel
|
||||||
|
// links high — M2 step 4 — p_paddr becomes a separate low load address
|
||||||
|
// via the linker's AT(), and this stays a valid physical allocation.)
|
||||||
const mem_sz: usize = @intCast(phdr.p_memsz);
|
const mem_sz: usize = @intCast(phdr.p_memsz);
|
||||||
const pages = (mem_sz + page_size - 1) / page_size;
|
const pages = (mem_sz + page_size - 1) / page_size;
|
||||||
const dest: [*]align(page_size) uefi.Page = @ptrFromInt(phdr.p_paddr);
|
const dest: [*]align(page_size) uefi.Page = @ptrFromInt(phdr.p_paddr);
|
||||||
@@ -231,15 +441,18 @@ fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !us
|
|||||||
@memcpy(bytes[0..file_sz], image[off..][0..file_sz]);
|
@memcpy(bytes[0..file_sz], image[off..][0..file_sz]);
|
||||||
@memset(bytes[file_sz..mem_sz], 0);
|
@memset(bytes[file_sz..mem_sz], 0);
|
||||||
|
|
||||||
// Record it (identity-loaded: virtual == physical) for the kernel's VMM.
|
// Record the virtual link address and the physical load address so the
|
||||||
const n = boot_info.kernel_segment_count;
|
// kernel can map itself with the right permissions post-switch. They're
|
||||||
if (n < boot_info.kernel_segments.len) {
|
// equal while the kernel links low; they diverge once it links high.
|
||||||
boot_info.kernel_segments[n] = .{
|
const n = boot_information.kernel_segment_count;
|
||||||
.virt = phdr.p_vaddr,
|
if (n < boot_information.kernel_segments.len) {
|
||||||
|
boot_information.kernel_segments[n] = .{
|
||||||
|
.virtual = phdr.p_vaddr,
|
||||||
|
.physical = phdr.p_paddr,
|
||||||
.pages = pages,
|
.pages = pages,
|
||||||
.flags = phdr.p_flags,
|
.flags = phdr.p_flags,
|
||||||
};
|
};
|
||||||
boot_info.kernel_segment_count = n + 1;
|
boot_information.kernel_segment_count = n + 1;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -250,27 +463,27 @@ fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !us
|
|||||||
/// neutral form. Allocating the buffers can itself change the map (invalidating
|
/// neutral form. Allocating the buffers can itself change the map (invalidating
|
||||||
/// the key), so retry until it takes. Both buffers are LoaderData, which survives
|
/// the key), so retry until it takes. Both buffers are LoaderData, which survives
|
||||||
/// the exit, so the returned map stays valid for the kernel.
|
/// the exit, so the returned map stays valid for the kernel.
|
||||||
fn exitBootServices(bs: *uefi.tables.BootServices) !danos.MemoryMap {
|
fn exitBootServices(bs: *uefi.tables.BootServices) !boot_handoff.MemoryMap {
|
||||||
var attempts: usize = 0;
|
var attempts: usize = 0;
|
||||||
while (attempts < 8) : (attempts += 1) {
|
while (attempts < 8) : (attempts += 1) {
|
||||||
const info = try bs.getMemoryMapInfo();
|
const info = try bs.getMemoryMapInfo();
|
||||||
// Spare descriptors to absorb the growth from the allocations below.
|
// Spare descriptors to absorb the growth from the allocations below.
|
||||||
const cap = info.len + 8;
|
const cap = info.len + 8;
|
||||||
const map_buf = try bs.allocatePool(.loader_data, cap * info.descriptor_size);
|
const map_buffer = try bs.allocatePool(.loader_data, cap * info.descriptor_size);
|
||||||
const regions_buf = try bs.allocatePool(.loader_data, cap * @sizeOf(danos.MemoryRegion));
|
const regions_buffer = try bs.allocatePool(.loader_data, cap * @sizeOf(boot_handoff.MemoryRegion));
|
||||||
const map = bs.getMemoryMap(map_buf) catch {
|
const map = bs.getMemoryMap(map_buffer) catch {
|
||||||
_ = bs.freePool(map_buf.ptr) catch {};
|
_ = bs.freePool(map_buffer.ptr) catch {};
|
||||||
_ = bs.freePool(regions_buf.ptr) catch {};
|
_ = bs.freePool(regions_buffer.ptr) catch {};
|
||||||
continue;
|
continue;
|
||||||
};
|
};
|
||||||
bs.exitBootServices(uefi.handle, map.info.key) catch {
|
bs.exitBootServices(uefi.handle, map.info.key) catch {
|
||||||
_ = bs.freePool(map_buf.ptr) catch {};
|
_ = bs.freePool(map_buffer.ptr) catch {};
|
||||||
_ = bs.freePool(regions_buf.ptr) catch {};
|
_ = bs.freePool(regions_buffer.ptr) catch {};
|
||||||
continue;
|
continue;
|
||||||
};
|
};
|
||||||
// Boot services are gone; do not touch `bs` again. Converting the map is
|
// Boot services are gone; do not touch `bs` again. Converting the map is
|
||||||
// pure computation on memory we already hold, so it's safe here.
|
// pure computation on memory we already hold, so it's safe here.
|
||||||
return convertMemoryMap(map, regions_buf);
|
return convertMemoryMap(map, regions_buffer);
|
||||||
}
|
}
|
||||||
return error.ExitBootServicesFailed;
|
return error.ExitBootServicesFailed;
|
||||||
}
|
}
|
||||||
@@ -279,8 +492,8 @@ fn exitBootServices(bs: *uefi.tables.BootServices) !danos.MemoryMap {
|
|||||||
/// into `out` (sized for at least `map.info.len` regions). Adjacent regions of
|
/// into `out` (sized for at least `map.info.len` regions). Adjacent regions of
|
||||||
/// the same kind are coalesced. This is the loader's job precisely so the kernel
|
/// the same kind are coalesced. This is the loader's job precisely so the kernel
|
||||||
/// never sees UEFI's vocabulary — the same seam the framebuffer already uses.
|
/// never sees UEFI's vocabulary — the same seam the framebuffer already uses.
|
||||||
fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
fn convertMemoryMap(map: MemoryMapSlice, out: []u8) boot_handoff.MemoryMap {
|
||||||
const regions: [*]danos.MemoryRegion = @ptrCast(@alignCast(out.ptr));
|
const regions: [*]boot_handoff.MemoryRegion = @ptrCast(@alignCast(out.ptr));
|
||||||
// We're about to call boot-services memory `usable`, but our own stack lives
|
// We're about to call boot-services memory `usable`, but our own stack lives
|
||||||
// in it and the kernel starts out running on it. Keep the region holding the
|
// in it and the kernel starts out running on it. Keep the region holding the
|
||||||
// current stack pointer reserved so it's never handed out.
|
// current stack pointer reserved so it's never handed out.
|
||||||
@@ -297,16 +510,16 @@ fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
|||||||
if (d.number_of_pages == 0) continue;
|
if (d.number_of_pages == 0) continue;
|
||||||
var kind = classify(d);
|
var kind = classify(d);
|
||||||
// The descriptor we're executing on stays reserved (see rsp above).
|
// The descriptor we're executing on stays reserved (see rsp above).
|
||||||
const region_end = d.physical_start + d.number_of_pages * danos.page_size;
|
const region_end = d.physical_start + d.number_of_pages * page_size;
|
||||||
if (kind == .usable and rsp >= d.physical_start and rsp < region_end) kind = .reserved;
|
if (kind == .usable and rsp >= d.physical_start and rsp < region_end) kind = .reserved;
|
||||||
|
|
||||||
// Coalesce with the previous region if it's the same kind and contiguous.
|
// Coalesce with the previous region if it's the same kind and contiguous.
|
||||||
if (count > 0) {
|
if (count > 0) {
|
||||||
const prev = ®ions[count - 1];
|
const previous = ®ions[count - 1];
|
||||||
if (prev.kind == kind and
|
if (previous.kind == kind and
|
||||||
prev.base + prev.pages * danos.page_size == d.physical_start)
|
previous.base + previous.pages * page_size == d.physical_start)
|
||||||
{
|
{
|
||||||
prev.pages += d.number_of_pages;
|
previous.pages += d.number_of_pages;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -322,7 +535,7 @@ fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
|||||||
|
|
||||||
/// Map a UEFI descriptor to danos's neutral kind. A region that isn't
|
/// Map a UEFI descriptor to danos's neutral kind. A region that isn't
|
||||||
/// writeback-cacheable (`wb`) isn't backed by real RAM — it's device registers or
|
/// writeback-cacheable (`wb`) isn't backed by real RAM — it's device registers or
|
||||||
/// a reserved address-space window (e.g. PCIe config space) — so it's `mmio`
|
/// a reserved address-space window (e.g. PCIe configuration space) — so it's `mmio`
|
||||||
/// regardless of type. UEFI overloads `reserved_memory_type` for both reserved RAM
|
/// regardless of type. UEFI overloads `reserved_memory_type` for both reserved RAM
|
||||||
/// and such holes, and the cache attribute is what actually tells them apart.
|
/// and such holes, and the cache attribute is what actually tells them apart.
|
||||||
///
|
///
|
||||||
@@ -331,7 +544,7 @@ fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
|||||||
/// ever the firmware's (the one live piece, our stack, is reserved by the caller).
|
/// ever the firmware's (the one live piece, our stack, is reserved by the caller).
|
||||||
/// Anything unrecognised is `reserved` — the safe default; our own LoaderData (the
|
/// Anything unrecognised is `reserved` — the safe default; our own LoaderData (the
|
||||||
/// kernel image and these buffers) lands there and stays reserved.
|
/// kernel image and these buffers) lands there and stays reserved.
|
||||||
fn classify(d: *const uefi.tables.MemoryDescriptor) danos.MemoryKind {
|
fn classify(d: *const uefi.tables.MemoryDescriptor) boot_handoff.MemoryKind {
|
||||||
if (!d.attribute.wb) return .mmio;
|
if (!d.attribute.wb) return .mmio;
|
||||||
return switch (d.type) {
|
return switch (d.type) {
|
||||||
.conventional_memory, .boot_services_code, .boot_services_data => .usable,
|
.conventional_memory, .boot_services_code, .boot_services_data => .usable,
|
||||||
@@ -343,33 +556,33 @@ fn classify(d: *const uefi.tables.MemoryDescriptor) danos.MemoryKind {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Write a compile-time string to the console (best effort).
|
/// Write a compile-time string to the console (best effort).
|
||||||
fn log(comptime msg: []const u8) void {
|
fn log(comptime message: []const u8) void {
|
||||||
const out = uefi.system_table.con_out orelse return;
|
const out = uefi.system_table.con_out orelse return;
|
||||||
_ = out.outputString(std.unicode.utf8ToUtf16LeStringLiteral(msg)) catch {};
|
_ = out.outputString(std.unicode.utf8ToUtf16LeStringLiteral(message)) catch {};
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Write a runtime ASCII byte string (e.g. an @errorName) by widening to UTF-16.
|
/// Write a runtime ASCII byte string (e.g. an @errorName) by widening to UTF-16.
|
||||||
fn logBytes(bytes: []const u8) void {
|
fn logBytes(bytes: []const u8) void {
|
||||||
const out = uefi.system_table.con_out orelse return;
|
const out = uefi.system_table.con_out orelse return;
|
||||||
var buf: [128]u16 = undefined;
|
var buffer: [128]u16 = undefined;
|
||||||
var i: usize = 0;
|
var i: usize = 0;
|
||||||
for (bytes) |b| {
|
for (bytes) |b| {
|
||||||
if (i + 1 >= buf.len) break;
|
if (i + 1 >= buffer.len) break;
|
||||||
buf[i] = b;
|
buffer[i] = b;
|
||||||
i += 1;
|
i += 1;
|
||||||
}
|
}
|
||||||
buf[i] = 0;
|
buffer[i] = 0;
|
||||||
_ = out.outputString(buf[0..i :0].ptr) catch {};
|
_ = out.outputString(buffer[0..i :0].ptr) catch {};
|
||||||
}
|
}
|
||||||
|
|
||||||
fn acpiRootSystemDescriptorPointer() ?*const anyopaque {
|
fn acpiRootSystemDescriptorPointer() ?*const anyopaque {
|
||||||
const table_entries = uefi.system_table.number_of_table_entries;
|
const table_entries = uefi.system_table.number_of_table_entries;
|
||||||
const config_tables = uefi.system_table.configuration_table;
|
const configuration_tables = uefi.system_table.configuration_table;
|
||||||
const acpi2 = uefi.tables.ConfigurationTable.acpi_20_table_guid;
|
const acpi2 = uefi.tables.ConfigurationTable.acpi_20_table_guid;
|
||||||
const acpi1 = uefi.tables.ConfigurationTable.acpi_10_table_guid;
|
const acpi1 = uefi.tables.ConfigurationTable.acpi_10_table_guid;
|
||||||
|
|
||||||
for (0..table_entries) |i| {
|
for (0..table_entries) |i| {
|
||||||
const entry = config_tables[i];
|
const entry = configuration_tables[i];
|
||||||
if (entry.vendor_guid.eql(acpi2) or entry.vendor_guid.eql(acpi1)) {
|
if (entry.vendor_guid.eql(acpi2) or entry.vendor_guid.eql(acpi1)) {
|
||||||
return entry.vendor_table;
|
return entry.vendor_table;
|
||||||
}
|
}
|
||||||
@@ -48,50 +48,246 @@ fn timestamp(b: *std.Build) []const u8 {
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Build one user-space binary the same way for every program (init, and later
|
||||||
|
/// the VFS server + drivers): freestanding, ReleaseSmall, `.large` code model
|
||||||
|
/// (the image base is above 4 GiB — smaller models emit 32-bit relocations that
|
||||||
|
/// can't reach), linked against the `runtime` runtime library with the shared user
|
||||||
|
/// link script. Pinned to LLVM + LLD so the script's PHDRS (segment permissions)
|
||||||
|
/// are authoritative — the kernel's W^X user-ELF loader requires exact perms.
|
||||||
|
fn addUserBinary(
|
||||||
|
b: *std.Build,
|
||||||
|
target: std.Build.ResolvedTarget,
|
||||||
|
runtime_module: *std.Build.Module,
|
||||||
|
mmio_module: *std.Build.Module,
|
||||||
|
xkeyboard_config_module: *std.Build.Module,
|
||||||
|
acpi_ids_module: *std.Build.Module,
|
||||||
|
name: []const u8,
|
||||||
|
root: []const u8,
|
||||||
|
) *std.Build.Step.Compile {
|
||||||
|
const exe = b.addExecutable(.{
|
||||||
|
.name = name,
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path(root),
|
||||||
|
.target = target,
|
||||||
|
.optimize = .ReleaseSmall,
|
||||||
|
.code_model = .large,
|
||||||
|
.single_threaded = true,
|
||||||
|
.sanitize_c = .off,
|
||||||
|
.stack_check = false,
|
||||||
|
.stack_protector = false,
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "runtime", .module = runtime_module },
|
||||||
|
// Typed volatile MMIO + memory barriers, for drivers. See library/mmio/.
|
||||||
|
.{ .name = "mmio", .module = mmio_module },
|
||||||
|
// Keyboard layouts (keycode + modifiers -> keysym/character), available
|
||||||
|
// to any program that wants it. See library/xkeyboard-config/.
|
||||||
|
.{ .name = "xkeyboard-config", .module = xkeyboard_config_module },
|
||||||
|
// ACPI/PnP hardware-ID registry, so drivers name devices
|
||||||
|
// (HardwareId.ps2_keyboard) instead of magic "_HID" strings.
|
||||||
|
.{ .name = "acpi-ids", .module = acpi_ids_module },
|
||||||
|
},
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
exe.setLinkerScript(b.path("library/runtime/user.ld"));
|
||||||
|
exe.entry = .{ .symbol_name = "_start" };
|
||||||
|
exe.image_base = 0x7000_0000_0000;
|
||||||
|
exe.use_llvm = true;
|
||||||
|
exe.use_lld = true;
|
||||||
|
return exe;
|
||||||
|
}
|
||||||
|
|
||||||
pub fn build(b: *std.Build) void {
|
pub fn build(b: *std.Build) void {
|
||||||
ensureZigVersion();
|
ensureZigVersion();
|
||||||
|
|
||||||
const target = b.standardTargetOptions(.{});
|
const target = b.standardTargetOptions(.{});
|
||||||
const optimize = b.standardOptimizeOption(.{});
|
const optimize = b.standardOptimizeOption(.{});
|
||||||
|
|
||||||
// Shared handoff definitions (BootInfo, Framebuffer, ...). No target is set,
|
// The three shared contracts, each with its own audience so every import
|
||||||
// so the module inherits the target of whichever binary imports it — the
|
// declares which one it speaks (no target is set, so each inherits the target of
|
||||||
// freestanding kernel or the UEFI bootloader.
|
// whichever binary imports it). See docs/coding-standards.md.
|
||||||
const mod = b.addModule("danos", .{
|
// boot-handoff : loader <-> kernel (BootInformation, framebuffer, VM layout)
|
||||||
.root_source_file = b.path("src/root.zig"),
|
// abi : kernel <-> runtime, core (SystemCall, mmap prot flags, page_size)
|
||||||
|
// device-abi : kernel <-> user, devices (DeviceDescriptor, DeviceClass, ...)
|
||||||
|
const boot_handoff_module = b.addModule("boot-handoff", .{
|
||||||
|
.root_source_file = b.path("system/boot-handoff.zig"),
|
||||||
|
});
|
||||||
|
const abi_module = b.addModule("abi", .{
|
||||||
|
.root_source_file = b.path("system/abi.zig"),
|
||||||
|
});
|
||||||
|
// The devices sub-project's public interface (the flat wire types), exposed as
|
||||||
|
// its own module like vfs-protocol — importable by user space, unlike the
|
||||||
|
// kernel-internal device model it also feeds (system/devices/device-model.zig).
|
||||||
|
const device_abi_module = b.addModule("device-abi", .{
|
||||||
|
.root_source_file = b.path("system/devices/device-abi.zig"),
|
||||||
|
});
|
||||||
|
// PCI class-code decoding (class/subclass/prog-IF -> names). Pure reference data,
|
||||||
|
// shared by kernel discovery (the device-tree dump) and any user-space PCI tool.
|
||||||
|
const pci_class_module = b.addModule("pci-class", .{
|
||||||
|
.root_source_file = b.path("system/devices/pci-class.zig"),
|
||||||
|
});
|
||||||
|
// ACPI/PnP hardware-ID (_HID) names — the flat analog of pci-class for acpi_device
|
||||||
|
// nodes. Also shared reference data.
|
||||||
|
// The AML interpreter, a build module so the ring-3 acpi service can run the
|
||||||
|
// same parser the kernel does (docs/discovery.md — the shared AML module).
|
||||||
|
// Pure Zig, no kernel imports — one source, two builds.
|
||||||
|
const aml_module = b.addModule("aml", .{
|
||||||
|
.root_source_file = b.path("system/devices/aml/aml.zig"),
|
||||||
|
});
|
||||||
|
|
||||||
|
const acpi_ids_module = b.addModule("acpi-ids", .{
|
||||||
|
.root_source_file = b.path("system/devices/acpi-ids.zig"),
|
||||||
|
});
|
||||||
|
|
||||||
|
// The USB device-framework wire ABI (chapter-9 set-up packets, standard +
|
||||||
|
// class requests, descriptors) and the USB class-code taxonomy — the flat
|
||||||
|
// reference the xHCI bus driver, the USB class drivers, and the device
|
||||||
|
// manager's identity matcher all share. Pure data, like pci-class/acpi-ids.
|
||||||
|
const usb_abi_module = b.addModule("usb-abi", .{
|
||||||
|
.root_source_file = b.path("system/devices/usb-abi.zig"),
|
||||||
|
});
|
||||||
|
const usb_ids_module = b.addModule("usb-ids", .{
|
||||||
|
.root_source_file = b.path("system/devices/usb-ids.zig"),
|
||||||
|
});
|
||||||
|
// The USB transfer protocol: what a USB class driver says to the xHCI bus
|
||||||
|
// driver to drive its device (open / control / interrupt / bulk). A protocol
|
||||||
|
// module like vfs-protocol, shared by the bus driver and every class driver.
|
||||||
|
const usb_transfer_protocol_module = b.addModule("usb-transfer-protocol", .{
|
||||||
|
.root_source_file = b.path("system/drivers/usb-xhci-bus/usb-transfer-protocol.zig"),
|
||||||
|
});
|
||||||
|
// The block-device protocol: read/write of fixed-size blocks, spoken between a
|
||||||
|
// filesystem and a block driver (usb-storage). A protocol module like the rest.
|
||||||
|
const block_protocol_module = b.addModule("block-protocol", .{
|
||||||
|
.root_source_file = b.path("system/services/block/protocol.zig"),
|
||||||
|
});
|
||||||
|
|
||||||
|
// Kernel tunables (maximum_cpus, stack sizes, tick rate). A dependency-free module of
|
||||||
|
// compile-time constants, imported wherever a knob is read; keeps the trade-offs
|
||||||
|
// in one place instead of scattered across the tree. See system/parameters.zig.
|
||||||
|
const parameters_module = b.addModule("parameters", .{
|
||||||
|
.root_source_file = b.path("system/parameters.zig"),
|
||||||
});
|
});
|
||||||
|
|
||||||
// Architecture-specific kernel code (CPU ops, entry, later GDT/IDT/paging).
|
// Architecture-specific kernel code (CPU ops, entry, later GDT/IDT/paging).
|
||||||
// The generic kernel imports this as "arch" and never names x86_64, so a new
|
// The generic kernel imports this as "architecture" and never names x86_64, so a new
|
||||||
// architecture is a matter of pointing this module at a different directory.
|
// architecture is a matter of pointing this module at a different directory.
|
||||||
const arch_mod = b.addModule("arch", .{
|
const architecture_module = b.addModule("architecture", .{
|
||||||
.root_source_file = b.path("src/kernel/arch/x86_64/cpu.zig"),
|
.root_source_file = b.path("system/kernel/architecture/x86_64/cpu.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
.{ .name = "danos", .module = mod }, // paging uses the shared BootInfo/memory-map types
|
.{ .name = "boot-handoff", .module = boot_handoff_module }, // paging uses BootInformation/memory-map + physicalToVirtual
|
||||||
|
.{ .name = "abi", .module = abi_module }, // paging works in page_size units
|
||||||
|
.{ .name = "parameters", .module = parameters_module }, // maximum_cpus, ist_stack_size, timer_hz
|
||||||
},
|
},
|
||||||
});
|
});
|
||||||
// CPU-exception stubs — real assembly, since they need cross-symbol
|
// CPU-exception stubs — real assembly, since they need cross-symbol
|
||||||
// jumps/calls that Zig inline asm can't express (see the file's header).
|
// jumps/calls that Zig inline asm can't express (see the file's header).
|
||||||
arch_mod.addAssemblyFile(b.path("src/kernel/arch/x86_64/isr.s"));
|
architecture_module.addAssemblyFile(b.path("system/kernel/architecture/x86_64/isr.s"));
|
||||||
|
// The AP bring-up trampoline: 16-/32-/64-bit mode-switch code that can't be
|
||||||
|
// inline asm (it runs relocated to a low page, not at its link address).
|
||||||
|
architecture_module.addAssemblyFile(b.path("system/kernel/architecture/x86_64/trampoline.s"));
|
||||||
|
|
||||||
// Firmware-agnostic device discovery. The generic kernel imports this as
|
// Firmware-agnostic device discovery. The generic kernel imports this as
|
||||||
// "platform" and asks it to enumerate hardware into a backend-neutral device
|
// "platform" and asks it to enumerate hardware into a backend-neutral device
|
||||||
// tree, never naming ACPI (or, later, device-tree) — the same discipline the
|
// tree, never naming ACPI (or, later, device-tree) — the same discipline the
|
||||||
// arch module applies to CPU code. The backend is selected at runtime from
|
// architecture module applies to CPU code. The backend is selected at runtime from
|
||||||
// the boot handoff (see src/device/platform.zig).
|
// the boot handoff (see system/devices/platform.zig).
|
||||||
const platform_mod = b.addModule("platform", .{
|
const platform_module = b.addModule("platform", .{
|
||||||
.root_source_file = b.path("src/device/platform.zig"),
|
.root_source_file = b.path("system/devices/platform.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
.{ .name = "danos", .module = mod }, // BootInfo (carries the ACPI RSDP)
|
.{ .name = "boot-handoff", .module = boot_handoff_module }, // BootInformation (carries the ACPI RSDP), physicalToVirtual
|
||||||
|
.{ .name = "abi", .module = abi_module }, // acpi.zig works in page_size units
|
||||||
|
.{ .name = "device-abi", .module = device_abi_module }, // device-model's DeviceClass/ResourceKind live here
|
||||||
|
.{ .name = "pci-class", .module = pci_class_module }, // decode PCI class codes in the device dump
|
||||||
|
.{ .name = "acpi-ids", .module = acpi_ids_module }, // decode ACPI _HID names in the device dump
|
||||||
|
.{ .name = "parameters", .module = parameters_module }, // maximum_cpus (the discovery pool)
|
||||||
},
|
},
|
||||||
});
|
});
|
||||||
|
|
||||||
// Compile-time config the kernel reads as `@import("build_options")`. The
|
// The VFS wire protocol: the vfs sub-project's public interface, exposed as its
|
||||||
|
// own module. Both the vfs server and the runtime's file layer (unistd/stdio)
|
||||||
|
// depend on this contract by name — neither reaches into the other's files. This
|
||||||
|
// is the first "protocol module" (see docs/driver-model.md); usb/block will
|
||||||
|
// expose theirs the same way.
|
||||||
|
const vfs_protocol_module = b.addModule("vfs-protocol", .{
|
||||||
|
.root_source_file = b.path("system/services/vfs/protocol.zig"),
|
||||||
|
});
|
||||||
|
|
||||||
|
// The input wire protocol: the input service's public interface, exposed as its own
|
||||||
|
// module the same way vfs-protocol is. Shared by the input service, the runtime's
|
||||||
|
// `input` helper (subscribe/publish), and every source and subscriber.
|
||||||
|
const input_protocol_module = b.addModule("input-protocol", .{
|
||||||
|
.root_source_file = b.path("system/services/input/protocol.zig"),
|
||||||
|
});
|
||||||
|
|
||||||
|
// The danos-native user-space runtime: system_call wrappers, the C-convention
|
||||||
|
// heap, IPC helpers, the process start shim, device access. This is the stable
|
||||||
|
// application ABI; POSIX compatibility is a separate library on top (see below).
|
||||||
|
// Compiled into every user binary (see addUserBinary), so it inherits each exe's
|
||||||
|
// `.large` code model — do NOT set a target/code_model here. It imports `abi`
|
||||||
|
// for the shared SystemCall numbers / mmap flags, `device-abi` for the device
|
||||||
|
// types its `device` helper wraps, and re-exports `vfs-protocol` for the VFS
|
||||||
|
// server. It never touches `boot-handoff` — user space has no business with the
|
||||||
|
// loader↔kernel handoff.
|
||||||
|
const runtime_module = b.addModule("runtime", .{
|
||||||
|
.root_source_file = b.path("library/runtime/runtime.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "abi", .module = abi_module },
|
||||||
|
.{ .name = "device-abi", .module = device_abi_module },
|
||||||
|
.{ .name = "vfs-protocol", .module = vfs_protocol_module },
|
||||||
|
.{ .name = "input-protocol", .module = input_protocol_module },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
// The device-manager protocol: hello + (M18.2) tree reports, exposed as its
|
||||||
|
// own module like the other protocol modules. Imported through the runtime.
|
||||||
|
const device_manager_protocol_module = b.addModule("device-manager-protocol", .{
|
||||||
|
.root_source_file = b.path("system/services/device-manager/device-manager-protocol.zig"),
|
||||||
|
});
|
||||||
|
runtime_module.addImport("device-manager-protocol", device_manager_protocol_module);
|
||||||
|
// The USB transfer protocol, so runtime.usb (the class-driver client) can speak
|
||||||
|
// it, the way runtime.input speaks the input protocol.
|
||||||
|
runtime_module.addImport("usb-transfer-protocol", usb_transfer_protocol_module);
|
||||||
|
// The block protocol, so runtime.block (the block-device client) can speak it.
|
||||||
|
runtime_module.addImport("block-protocol", block_protocol_module);
|
||||||
|
|
||||||
|
// The power protocol: system power's domain-named surface (docs/power.md).
|
||||||
|
const power_protocol_module = b.addModule("power-protocol", .{
|
||||||
|
.root_source_file = b.path("system/services/power/protocol.zig"),
|
||||||
|
});
|
||||||
|
runtime_module.addImport("power-protocol", power_protocol_module);
|
||||||
|
|
||||||
|
// Typed volatile MMIO register access + memory-ordering barriers, for drivers on
|
||||||
|
// top of an mmio_map grant. Depends only on `builtin` (arch-conditional barriers);
|
||||||
|
// no target set, so it inherits each driver's. See library/mmio/mmio.zig.
|
||||||
|
const mmio_module = b.addModule("mmio", .{
|
||||||
|
.root_source_file = b.path("library/mmio/mmio.zig"),
|
||||||
|
});
|
||||||
|
|
||||||
|
// Keyboard layouts compiled from the X11 xkeyboard-config database into native Zig
|
||||||
|
// (keycode + modifiers -> keysym/character). The `layouts` tables are generated by
|
||||||
|
// tools/make-xkeyboard-config.py; `xkeyboard-config` is the hand-written API over them.
|
||||||
|
// No target set, so each inherits its importer's. See library/xkeyboard-config/.
|
||||||
|
const xkb_layouts_module = b.addModule("layouts", .{
|
||||||
|
.root_source_file = b.path("library/xkeyboard-config/generated/layouts.zig"),
|
||||||
|
});
|
||||||
|
const xkeyboard_config_module = b.addModule("xkeyboard-config", .{
|
||||||
|
.root_source_file = b.path("library/xkeyboard-config/xkeyboard-config.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "layouts", .module = xkb_layouts_module },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
// The initial_ramdisk container format, shared by the kernel (unpacks it) and the
|
||||||
|
// build-time packer tools/make-initial-ramdisk.py (produces it). No dependencies.
|
||||||
|
const initial_ramdisk_module = b.addModule("initial-ramdisk", .{
|
||||||
|
.root_source_file = b.path("system/initial-ramdisk.zig"),
|
||||||
|
});
|
||||||
|
|
||||||
|
// Compile-time configuration the kernel reads as `@import("build_options")`. The
|
||||||
// QEMU test harness sets -Dtest-case=<name> to run one self-test at boot.
|
// QEMU test harness sets -Dtest-case=<name> to run one self-test at boot.
|
||||||
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see src/kernel/tests.zig)");
|
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see system/kernel/tests.zig)");
|
||||||
const build_options = b.addOptions();
|
const build_options = b.addOptions();
|
||||||
build_options.addOption(?[]const u8, "test_case", test_case);
|
build_options.addOption(?[]const u8, "test_case", test_case);
|
||||||
const build_options_mod = build_options.createModule();
|
const build_options_module = build_options.createModule();
|
||||||
|
|
||||||
// --- Kernel: freestanding x86_64 ELF, jumped to by the bootloader ---
|
// --- Kernel: freestanding x86_64 ELF, jumped to by the bootloader ---
|
||||||
// SSE2 is part of the x86_64 baseline and UEFI leaves it enabled at handoff,
|
// SSE2 is part of the x86_64 baseline and UEFI leaves it enabled at handoff,
|
||||||
@@ -106,62 +302,263 @@ pub fn build(b: *std.Build) void {
|
|||||||
const exe = b.addExecutable(.{
|
const exe = b.addExecutable(.{
|
||||||
.name = "kernel",
|
.name = "kernel",
|
||||||
.root_module = b.createModule(.{
|
.root_module = b.createModule(.{
|
||||||
.root_source_file = b.path("src/kernel/main.zig"),
|
.root_source_file = b.path("system/kernel/kernel.zig"),
|
||||||
.target = kernel_target,
|
.target = kernel_target,
|
||||||
.optimize = optimize,
|
.optimize = optimize,
|
||||||
.code_model = .small, // kernel is linked in the low 2 GiB (see image_base)
|
.code_model = .kernel, // kernel runs in the top 2 GiB (higher half)
|
||||||
.red_zone = false, // interrupts would corrupt the SysV red zone
|
.red_zone = false, // interrupts would corrupt the SystemV red zone
|
||||||
.single_threaded = true, // no scheduler yet; avoids pulling in TLS/atomics
|
.single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores
|
||||||
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
||||||
.stack_check = false, // stack-probe calls have no runtime to land in
|
.stack_check = false, // stack-probe calls have no runtime to land in
|
||||||
.stack_protector = false,
|
.stack_protector = false,
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
.{ .name = "danos", .module = mod },
|
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
||||||
.{ .name = "arch", .module = arch_mod },
|
.{ .name = "abi", .module = abi_module },
|
||||||
.{ .name = "platform", .module = platform_mod },
|
.{ .name = "device-abi", .module = device_abi_module },
|
||||||
.{ .name = "build_options", .module = build_options_mod },
|
.{ .name = "architecture", .module = architecture_module },
|
||||||
|
.{ .name = "platform", .module = platform_module },
|
||||||
|
.{ .name = "parameters", .module = parameters_module },
|
||||||
|
.{ .name = "build_options", .module = build_options_module },
|
||||||
|
.{ .name = "initial-ramdisk", .module = initial_ramdisk_module },
|
||||||
},
|
},
|
||||||
}),
|
}),
|
||||||
});
|
});
|
||||||
exe.setLinkerScript(b.path("src/kernel/arch/x86_64/linker.ld"));
|
exe.setLinkerScript(b.path("system/kernel/architecture/x86_64/linker.ld"));
|
||||||
exe.entry = .{ .symbol_name = "_start" };
|
exe.entry = .{ .symbol_name = "_start" };
|
||||||
// Physical address the bootloader loads the kernel to (identity-mapped under
|
// The self-hosted linker ignores parts of the linker script (PHDRS,
|
||||||
// UEFI). Overrides Zig's default image base so the linker script's layout is
|
// /DISCARD/, AT(), section order); the higher-half layout depends on the
|
||||||
// honoured; adjust here if it collides with firmware-reserved memory.
|
// script being authoritative, so pin the kernel to LLVM + LLD.
|
||||||
exe.image_base = 0x100000; // 1 MiB
|
exe.use_llvm = true;
|
||||||
|
exe.use_lld = true;
|
||||||
|
// Higher-half virtual base (matches KERNEL_VIRT_BASE in linker.ld); the
|
||||||
|
// linker's AT() clauses give each segment a low physical load address
|
||||||
|
// (.text at 1 MiB), which the loader allocates and copies into.
|
||||||
|
exe.image_base = 0xFFFFFFFF80100000;
|
||||||
|
|
||||||
b.installArtifact(exe);
|
// Everything installs into a FHS-shaped zig-out: it IS the danos filesystem *and*
|
||||||
|
// the boot volume. Each binary lands at its addressed, leaf-collapsed path — the
|
||||||
|
// kernel at zig-out/system/kernel (from system/kernel/kernel.zig), init at
|
||||||
|
// zig-out/system/services/init, and so on (see docs/README.md). The bootloader
|
||||||
|
// then loads these FHS paths off the volume.
|
||||||
|
const kernel_install = b.addInstallArtifact(exe, .{ .dest_dir = .{ .override = .{ .custom = "system" } } });
|
||||||
|
b.getInstallStep().dependOn(&kernel_install.step);
|
||||||
|
|
||||||
// Boot methods live in src/boot/, one per way of getting the kernel running.
|
// --- init: the first user-space program (a system service) ---
|
||||||
|
// Built by the shared user-binary recipe (see addUserBinary): freestanding,
|
||||||
|
// linked into the kernel's user region against the `runtime` runtime library, and
|
||||||
|
// started in ring 3 by the kernel's user-ELF loader.
|
||||||
|
const init_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "init", "system/services/init/init.zig");
|
||||||
|
const init_install = b.addInstallArtifact(init_exe, .{ .dest_dir = .{ .override = .{ .custom = "system/services" } } });
|
||||||
|
b.getInstallStep().dependOn(&init_install.step);
|
||||||
|
|
||||||
|
// --- initial_ramdisk: a bundle of extra user binaries (VFS server + drivers) ---
|
||||||
|
// Each is built by the same user-binary recipe, then packed into one image by
|
||||||
|
// the host-side make-initial-ramdisk tool. The bootloader ferries the image to the kernel,
|
||||||
|
// which unpacks it and spawns each program (system/initial-ramdisk.zig).
|
||||||
|
const vfs_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs", "system/services/vfs/vfs.zig");
|
||||||
|
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs-test", "system/services/vfs/vfs-test.zig");
|
||||||
|
const ps2_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-bus", "system/drivers/ps2-bus/ps2-bus.zig");
|
||||||
|
const ps2_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig");
|
||||||
|
const ps2_mouse_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig");
|
||||||
|
const usb_xhci_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-xhci-bus", "system/drivers/usb-xhci-bus/usb-xhci-bus.zig");
|
||||||
|
// The xHCI bus driver builds chapter-9 requests and decodes descriptors from
|
||||||
|
// usb-abi, and reports each interface's (class,subclass,protocol) identity via
|
||||||
|
// usb-ids.packTriple.
|
||||||
|
usb_xhci_bus_exe.root_module.addImport("usb-abi", usb_abi_module);
|
||||||
|
usb_xhci_bus_exe.root_module.addImport("usb-ids", usb_ids_module);
|
||||||
|
usb_xhci_bus_exe.root_module.addImport("usb-transfer-protocol", usb_transfer_protocol_module);
|
||||||
|
// The USB HID class drivers: keyboard and mouse. They own no hardware — each
|
||||||
|
// opens its device through runtime.usb (the transfer protocol) and publishes to
|
||||||
|
// the input service. They build chapter-9 class requests from usb-abi.
|
||||||
|
const usb_hid_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-hid-keyboard", "system/drivers/usb-hid/keyboard.zig");
|
||||||
|
usb_hid_keyboard_exe.root_module.addImport("usb-abi", usb_abi_module);
|
||||||
|
const usb_hid_mouse_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-hid-mouse", "system/drivers/usb-hid/mouse.zig");
|
||||||
|
usb_hid_mouse_exe.root_module.addImport("usb-abi", usb_abi_module);
|
||||||
|
// The USB mass-storage class driver: opens its device via runtime.usb, drives it
|
||||||
|
// with Bulk-Only Transport + SCSI, and serves the block protocol under `.block`.
|
||||||
|
const usb_storage_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-storage", "system/drivers/usb-storage/usb-storage.zig");
|
||||||
|
usb_storage_exe.root_module.addImport("block-protocol", block_protocol_module);
|
||||||
|
// The FAT filesystem server: mounts the block device and serves it into the VFS
|
||||||
|
// at /mnt/usb. Its engine (engine.zig / on-disk.zig) is imported relatively.
|
||||||
|
const fat_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat", "system/services/fat/fat.zig");
|
||||||
|
const fat_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat-test", "system/services/fat/fat-test.zig");
|
||||||
|
const pci_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "pci-bus", "system/drivers/pci-bus/pci-bus.zig");
|
||||||
|
// The PCI bus driver decodes each function's class triple to human names in its
|
||||||
|
// boot log (class/subclass/prog-IF), so pull in the shared pci-class reference.
|
||||||
|
pci_bus_exe.root_module.addImport("pci-class", pci_class_module);
|
||||||
|
// A test fixture, not a real driver: hellos to the device manager, then faults —
|
||||||
|
// what the driver-restart scenario drives the crash-loop cap with.
|
||||||
|
const crash_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "crash-test", "system/services/crash-test/crash-test.zig");
|
||||||
|
const device_list_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-list", "system/services/device-list/device-list.zig");
|
||||||
|
// The discovery service: one swappable process per firmware
|
||||||
|
// (docs/discovery.md), bundled under the neutral ramdisk name
|
||||||
|
// "discovery" so the device manager never learns which firmware it is on.
|
||||||
|
// x86 boots describe hardware with ACPI; the Raspberry Pis hand over a
|
||||||
|
// flattened device tree — the aarch64 target flips the default when it
|
||||||
|
// lands (docs/arm.md). Both are placeholders until M20.1 (acpi) and the
|
||||||
|
// ARM bring-up (fdt).
|
||||||
|
const Discovery = enum { acpi, fdt };
|
||||||
|
const discovery = b.option(Discovery, "discovery", "Which discovery service fills the ramdisk's 'discovery' slot (default: acpi)") orelse Discovery.acpi;
|
||||||
|
const discovery_source: []const u8 = switch (discovery) {
|
||||||
|
.acpi => "system/services/acpi/acpi.zig",
|
||||||
|
.fdt => "system/services/fdt/fdt.zig",
|
||||||
|
};
|
||||||
|
const discovery_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "discovery", discovery_source);
|
||||||
|
if (discovery == .acpi) discovery_exe.root_module.addImport("aml", aml_module);
|
||||||
|
const device_manager_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-manager", "system/services/device-manager/device-manager.zig");
|
||||||
|
// Names the xHCI PCI class triple from the shared taxonomy instead of a bare 0x0C0330.
|
||||||
|
device_manager_exe.root_module.addImport("pci-class", pci_class_module);
|
||||||
|
// The manager matches reported USB interfaces by their (class,subclass,protocol)
|
||||||
|
// triple (usbDriverForIdentity), built from the named usb-ids codes.
|
||||||
|
device_manager_exe.root_module.addImport("usb-ids", usb_ids_module);
|
||||||
|
// The input service and its exercisers: the fan-out server, a hardware-free synthetic
|
||||||
|
// source, and a subscriber that doubles as the `input` test's oracle. See docs/input.md.
|
||||||
|
const input_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input", "system/services/input/input.zig");
|
||||||
|
const input_source_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-source", "system/services/input-source/input-source.zig");
|
||||||
|
const input_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-test", "system/services/input-test/input-test.zig");
|
||||||
|
const args_echo_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "args-echo", "system/services/args-echo/args-echo.zig");
|
||||||
|
const process_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "process-test", "system/services/process-test/process-test.zig");
|
||||||
|
const log_flush_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "log-flush", "system/services/log-flush/log-flush.zig");
|
||||||
|
|
||||||
|
// Pack the user binaries into the initial_ramdisk image with the host-side Python tool
|
||||||
|
// (the container format is trivial, and Python sidesteps std API churn). Args:
|
||||||
|
// make-initial-ramdisk.py <out> [<name> <file>]... — one name/file pair per binary.
|
||||||
|
const mk_run = b.addSystemCommand(&.{"python3"});
|
||||||
|
mk_run.addFileArg(b.path("tools/make-initial-ramdisk.py"));
|
||||||
|
const initial_ramdisk_img = mk_run.addOutputFileArg("initial-ramdisk.img");
|
||||||
|
mk_run.addArg("vfs");
|
||||||
|
mk_run.addFileArg(vfs_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("vfs-test");
|
||||||
|
mk_run.addFileArg(vfstest_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("ps2-bus");
|
||||||
|
mk_run.addFileArg(ps2_bus_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("ps2-keyboard");
|
||||||
|
mk_run.addFileArg(ps2_keyboard_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("ps2-mouse");
|
||||||
|
mk_run.addFileArg(ps2_mouse_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("usb-xhci-bus");
|
||||||
|
mk_run.addFileArg(usb_xhci_bus_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("usb-hid-keyboard");
|
||||||
|
mk_run.addFileArg(usb_hid_keyboard_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("usb-hid-mouse");
|
||||||
|
mk_run.addFileArg(usb_hid_mouse_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("usb-storage");
|
||||||
|
mk_run.addFileArg(usb_storage_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("fat");
|
||||||
|
mk_run.addFileArg(fat_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("fat-test");
|
||||||
|
mk_run.addFileArg(fat_test_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("pci-bus");
|
||||||
|
mk_run.addFileArg(pci_bus_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("crash-test");
|
||||||
|
mk_run.addFileArg(crash_test_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("device-list");
|
||||||
|
mk_run.addFileArg(device_list_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("discovery");
|
||||||
|
mk_run.addFileArg(discovery_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("device-manager");
|
||||||
|
mk_run.addFileArg(device_manager_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("input");
|
||||||
|
mk_run.addFileArg(input_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("input-source");
|
||||||
|
mk_run.addFileArg(input_source_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("input-test");
|
||||||
|
mk_run.addFileArg(input_test_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("args-echo");
|
||||||
|
mk_run.addFileArg(args_echo_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("process-test");
|
||||||
|
mk_run.addFileArg(process_test_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("log-flush");
|
||||||
|
mk_run.addFileArg(log_flush_exe.getEmittedBin());
|
||||||
|
|
||||||
|
// Also install the packed binaries to their FHS homes, so zig-out is a true image
|
||||||
|
// of the filesystem — even though at boot they arrive inside the initial-ramdisk.
|
||||||
|
for ([_]struct { *std.Build.Step.Compile, []const u8 }{
|
||||||
|
.{ vfs_exe, "system/services" },
|
||||||
|
.{ device_manager_exe, "system/services" },
|
||||||
|
.{ input_exe, "system/services" },
|
||||||
|
.{ ps2_bus_exe, "system/drivers" },
|
||||||
|
.{ ps2_keyboard_exe, "system/drivers" },
|
||||||
|
.{ ps2_mouse_exe, "system/drivers" },
|
||||||
|
.{ usb_xhci_bus_exe, "system/drivers" },
|
||||||
|
.{ usb_hid_keyboard_exe, "system/drivers" },
|
||||||
|
.{ usb_hid_mouse_exe, "system/drivers" },
|
||||||
|
.{ usb_storage_exe, "system/drivers" },
|
||||||
|
.{ fat_exe, "system/services" },
|
||||||
|
.{ log_flush_exe, "system/services" },
|
||||||
|
}) |entry| {
|
||||||
|
const step = b.addInstallArtifact(entry[0], .{ .dest_dir = .{ .override = .{ .custom = entry[1] } } });
|
||||||
|
b.getInstallStep().dependOn(&step.step);
|
||||||
|
}
|
||||||
|
|
||||||
|
// The initial-ramdisk itself installs to /boot (with the loaders).
|
||||||
|
const initial_ramdisk_install = b.addInstallFile(initial_ramdisk_img, "boot/initial-ramdisk.img");
|
||||||
|
b.getInstallStep().dependOn(&initial_ramdisk_install.step);
|
||||||
|
|
||||||
|
// Boot methods live in boot/, one per way of getting the kernel running.
|
||||||
// Each is its own binary/entry (a loader is built for its own target); today
|
// Each is its own binary/entry (a loader is built for its own target); today
|
||||||
// that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis.
|
// that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis.
|
||||||
const efiexe = b.addExecutable(.{
|
const efiexe = b.addExecutable(.{
|
||||||
.name = "BOOTX64",
|
.name = "BOOTX64",
|
||||||
.root_module = b.createModule(.{
|
.root_module = b.createModule(.{
|
||||||
.root_source_file = b.path("src/boot/efi.zig"),
|
.root_source_file = b.path("boot/efi.zig"),
|
||||||
.target = b.resolveTargetQuery(.{
|
.target = b.resolveTargetQuery(.{
|
||||||
.cpu_arch = .x86_64,
|
.cpu_arch = .x86_64,
|
||||||
.os_tag = .uefi,
|
.os_tag = .uefi,
|
||||||
}),
|
}),
|
||||||
.optimize = optimize,
|
.optimize = optimize,
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
.{ .name = "danos", .module = mod },
|
// The bootloader speaks only the handoff contract — never the user ABI.
|
||||||
|
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
||||||
},
|
},
|
||||||
}),
|
}),
|
||||||
});
|
});
|
||||||
|
|
||||||
b.installArtifact(efiexe);
|
// UEFI firmware requires the removable-media loader at exactly \EFI\BOOT\BOOTX64.efi,
|
||||||
|
// so that path is fixed by the firmware (it is /boot's EFI stub, conceptually).
|
||||||
|
const efi_install = b.addInstallArtifact(efiexe, .{ .dest_dir = .{ .override = .{ .custom = "EFI/BOOT" } } });
|
||||||
|
b.getInstallStep().dependOn(&efi_install.step);
|
||||||
|
|
||||||
|
// --- danos-usb.img: the bootable FAT32 USB image ---
|
||||||
|
// Format a real FAT32 image (the in-repo Python builder, no external tools)
|
||||||
|
// holding exactly what the firmware and bootloader need off the ESP: the EFI
|
||||||
|
// stub, the kernel, init, and the initial-ramdisk. QEMU presents this image as
|
||||||
|
// a USB mass-storage device the guest boots from (see run-x86-64 and the test
|
||||||
|
// harness), and the danos fat driver mounts the same image at /mnt/usb.
|
||||||
|
const mk_fat = b.addSystemCommand(&.{"python3"});
|
||||||
|
mk_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||||
|
const fat_image = mk_fat.addOutputFileArg("danos-usb.img");
|
||||||
|
mk_fat.addArg("64"); // MiB
|
||||||
|
mk_fat.addArg("EFI/BOOT/BOOTX64.efi");
|
||||||
|
mk_fat.addFileArg(efiexe.getEmittedBin());
|
||||||
|
mk_fat.addArg("system/kernel");
|
||||||
|
mk_fat.addFileArg(exe.getEmittedBin());
|
||||||
|
mk_fat.addArg("system/services/init");
|
||||||
|
mk_fat.addFileArg(init_exe.getEmittedBin());
|
||||||
|
mk_fat.addArg("boot/initial-ramdisk.img");
|
||||||
|
mk_fat.addFileArg(initial_ramdisk_img);
|
||||||
|
const fat_image_install = b.addInstallFile(fat_image, "danos-usb.img");
|
||||||
|
b.getInstallStep().dependOn(&fat_image_install.step);
|
||||||
|
|
||||||
|
// `zig build check-fat-image` — validate the produced image is a real FAT32
|
||||||
|
// with the EFI stub present (the builder's own --verify, no external tools).
|
||||||
|
const check_fat = b.addSystemCommand(&.{"python3"});
|
||||||
|
check_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||||
|
check_fat.addArg("--verify");
|
||||||
|
check_fat.addFileArg(fat_image);
|
||||||
|
const check_fat_step = b.step("check-fat-image", "Verify the FAT32 USB image is valid and bootable");
|
||||||
|
check_fat_step.dependOn(&check_fat.step);
|
||||||
|
|
||||||
// --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF ---
|
// --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF ---
|
||||||
// Firmware lives in different places per OS/distro, so probe the known
|
// Firmware lives in different places per OS/distro, so probe the known
|
||||||
// layouts (Arch, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
|
// layouts (Architecture, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
|
||||||
// that exists. Override with -Dovmf-code / -Dovmf-vars if yours is elsewhere.
|
// that exists. Override with -Dovmf-code / -Dovmf-vars if yours is elsewhere.
|
||||||
const ovmf_code = b.option(
|
const ovmf_code = b.option(
|
||||||
[]const u8,
|
[]const u8,
|
||||||
"ovmf-code",
|
"ovmf-code",
|
||||||
"Path to the OVMF_CODE firmware image",
|
"Path to the OVMF_CODE firmware image",
|
||||||
) orelse firstExisting(b.graph.io, &.{
|
) orelse firstExisting(b.graph.io, &.{
|
||||||
"/usr/share/edk2/x64/OVMF_CODE.4m.fd", // Arch
|
"/usr/share/edk2/x64/OVMF_CODE.4m.fd", // Architecture
|
||||||
"/usr/share/OVMF/OVMF_CODE_4M.fd", // Debian/Ubuntu
|
"/usr/share/OVMF/OVMF_CODE_4M.fd", // Debian/Ubuntu
|
||||||
"/usr/share/OVMF/OVMF_CODE.fd", // older Debian/Ubuntu
|
"/usr/share/OVMF/OVMF_CODE.fd", // older Debian/Ubuntu
|
||||||
"/usr/share/edk2-ovmf/x64/OVMF_CODE.fd", // Fedora
|
"/usr/share/edk2-ovmf/x64/OVMF_CODE.fd", // Fedora
|
||||||
@@ -173,7 +570,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
"ovmf-vars",
|
"ovmf-vars",
|
||||||
"Path to the OVMF_VARS firmware image (a writable copy is made)",
|
"Path to the OVMF_VARS firmware image (a writable copy is made)",
|
||||||
) orelse firstExisting(b.graph.io, &.{
|
) orelse firstExisting(b.graph.io, &.{
|
||||||
"/usr/share/edk2/x64/OVMF_VARS.4m.fd", // Arch
|
"/usr/share/edk2/x64/OVMF_VARS.4m.fd", // Architecture
|
||||||
"/usr/share/OVMF/OVMF_VARS_4M.fd", // Debian/Ubuntu
|
"/usr/share/OVMF/OVMF_VARS_4M.fd", // Debian/Ubuntu
|
||||||
"/usr/share/OVMF/OVMF_VARS.fd", // older Debian/Ubuntu
|
"/usr/share/OVMF/OVMF_VARS.fd", // older Debian/Ubuntu
|
||||||
"/usr/share/edk2-ovmf/x64/OVMF_VARS.fd", // Fedora
|
"/usr/share/edk2-ovmf/x64/OVMF_VARS.fd", // Fedora
|
||||||
@@ -181,15 +578,8 @@ pub fn build(b: *std.Build) void {
|
|||||||
"/usr/local/share/qemu/edk2-i386-vars.fd", // macOS Homebrew (Intel)
|
"/usr/local/share/qemu/edk2-i386-vars.fd", // macOS Homebrew (Intel)
|
||||||
});
|
});
|
||||||
|
|
||||||
// Assemble an EFI System Partition layout: esp/EFI/BOOT/BOOTX64.efi
|
// The FHS zig-out (installed above) *is* the boot volume — no separate ESP to
|
||||||
const efi_install = b.addInstallArtifact(efiexe, .{
|
// assemble. QEMU presents it to the guest as a FAT drive below.
|
||||||
.dest_dir = .{ .override = .{ .custom = "esp/EFI/BOOT" } },
|
|
||||||
});
|
|
||||||
// The bootloader loads the kernel by name from the volume root, so drop the
|
|
||||||
// kernel ELF at esp/kernel.
|
|
||||||
const kernel_install = b.addInstallArtifact(exe, .{
|
|
||||||
.dest_dir = .{ .override = .{ .custom = "esp" } },
|
|
||||||
});
|
|
||||||
|
|
||||||
// The firmware needs to write NVRAM, so give it a writable copy of the vars.
|
// The firmware needs to write NVRAM, so give it a writable copy of the vars.
|
||||||
const vars_copy = b.addSystemCommand(&.{ "cp", "-f", ovmf_vars });
|
const vars_copy = b.addSystemCommand(&.{ "cp", "-f", ovmf_vars });
|
||||||
@@ -197,6 +587,19 @@ pub fn build(b: *std.Build) void {
|
|||||||
|
|
||||||
const run_efi = b.addSystemCommand(&.{
|
const run_efi = b.addSystemCommand(&.{
|
||||||
"qemu-system-x86_64",
|
"qemu-system-x86_64",
|
||||||
|
"-device",
|
||||||
|
"qemu-xhci,id=xhci",
|
||||||
|
"-device",
|
||||||
|
"usb-mouse,bus=xhci.0",
|
||||||
|
"-device",
|
||||||
|
"usb-kbd,bus=xhci.0",
|
||||||
|
// "-usb",
|
||||||
|
// "-device",
|
||||||
|
// "usb-ehci,id=ehci",
|
||||||
|
// "-device",
|
||||||
|
// "usb-tablet,bus=usb-bus.0",
|
||||||
|
// "-device",
|
||||||
|
// "usb-mouse,bus=ehci.0",
|
||||||
"-machine",
|
"-machine",
|
||||||
"q35",
|
"q35",
|
||||||
"-m",
|
"-m",
|
||||||
@@ -206,10 +609,13 @@ pub fn build(b: *std.Build) void {
|
|||||||
});
|
});
|
||||||
run_efi.addArg("-drive");
|
run_efi.addArg("-drive");
|
||||||
run_efi.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
run_efi.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
||||||
// Present the ESP directory to the guest as a FAT drive.
|
// Boot off the FAT32 USB image: a mass-storage device on the same xHCI bus as
|
||||||
|
// the keyboard and mouse. OVMF finds \EFI\BOOT\BOOTX64.efi on it and boots.
|
||||||
|
run_efi.addArg("-drive");
|
||||||
|
run_efi.addPrefixedFileArg("if=none,id=bootusb,format=raw,file=", fat_image);
|
||||||
run_efi.addArgs(&.{
|
run_efi.addArgs(&.{
|
||||||
"-drive",
|
"-device",
|
||||||
b.fmt("format=raw,file=fat:rw:{s}/esp", .{b.install_path}),
|
"usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
||||||
"-net",
|
"-net",
|
||||||
"none",
|
"none",
|
||||||
// Emulated display advertising 1280x720 as its native (EDID preferred)
|
// Emulated display advertising 1280x720 as its native (EDID preferred)
|
||||||
@@ -220,14 +626,19 @@ pub fn build(b: *std.Build) void {
|
|||||||
"-device",
|
"-device",
|
||||||
"VGA,edid=on,xres=1280,yres=720",
|
"VGA,edid=on,xres=1280,yres=720",
|
||||||
});
|
});
|
||||||
// Always capture the guest's serial0 (the kernel's machine-readable log) to a
|
// Capture the guest's serial0 (danos's machine-readable log) to the qemu-test
|
||||||
// timestamped file under zig-out, so each run leaves its own log behind.
|
// scratch area — a dev/host artifact, kept out of the FHS boot volume we mount.
|
||||||
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ b.install_path, timestamp(b) });
|
// (/var/log/system is reserved for the kernel's own logging system later.) One
|
||||||
|
// timestamped file per run.
|
||||||
|
const log_dir = b.fmt("{s}/qemu-test", .{b.install_path});
|
||||||
|
const make_log_dir = b.addSystemCommand(&.{ "mkdir", "-p", log_dir });
|
||||||
|
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
||||||
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
|
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
|
||||||
run_efi.step.dependOn(&efi_install.step);
|
// The whole FHS zig-out must be installed (and the scratch dir created) before we mount it.
|
||||||
run_efi.step.dependOn(&kernel_install.step);
|
run_efi.step.dependOn(b.getInstallStep());
|
||||||
|
run_efi.step.dependOn(&make_log_dir.step);
|
||||||
|
|
||||||
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/run-x86-64-serial0-<timestamp>.log");
|
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/qemu-test/run-x86-64-serial0-<timestamp>.log");
|
||||||
run_efi_step.dependOn(&run_efi.step);
|
run_efi_step.dependOn(&run_efi.step);
|
||||||
|
|
||||||
// const run_cmd = b.addRunArtifact(exe);
|
// const run_cmd = b.addRunArtifact(exe);
|
||||||
@@ -240,18 +651,73 @@ pub fn build(b: *std.Build) void {
|
|||||||
// }
|
// }
|
||||||
|
|
||||||
// Tests run on the host. The kernel and bootloader target freestanding/UEFI
|
// Tests run on the host. The kernel and bootloader target freestanding/UEFI
|
||||||
// and can't be executed natively, so only the shared module is unit-tested
|
// and can't be executed natively, so only the shared contracts are unit-tested
|
||||||
// here (compiled for the host rather than inheriting a freestanding target).
|
// here (compiled for the host rather than inheriting a freestanding target) —
|
||||||
const mod_tests = b.addTest(.{
|
// which also compile-checks that the three-way split stays self-consistent.
|
||||||
|
const test_step = b.step("test", "Run tests");
|
||||||
|
for ([_][]const u8{
|
||||||
|
"system/boot-handoff.zig",
|
||||||
|
"system/abi.zig",
|
||||||
|
"system/devices/device-abi.zig",
|
||||||
|
"system/devices/pci-class.zig", // class/subclass/prog-IF name decoding
|
||||||
|
"system/devices/acpi-ids.zig", // _HID name decoding
|
||||||
|
"system/devices/aml/aml.zig", // AML parse + interpret, incl. Notify dispatch (M21)
|
||||||
|
"system/devices/usb-abi.zig", // wire sizes + bit packings + set-up packet encodings
|
||||||
|
"system/devices/usb-ids.zig", // class/subclass/protocol code assignments
|
||||||
|
"library/mmio/mmio.zig", // barriers assemble + registers round-trip
|
||||||
|
"system/drivers/ps2-bus/scancode.zig", // set-2 decode + keyboard state machine
|
||||||
|
"system/drivers/ps2-bus/mouse-packet.zig", // 3-byte mouse packet assembly
|
||||||
|
"system/drivers/usb-hid/hid-report.zig", // HID boot-report keyboard/mouse decode
|
||||||
|
"system/drivers/usb-storage/bulk-only-transport.zig", // CBW/CSW wrapper sizes
|
||||||
|
"system/drivers/usb-storage/scsi.zig", // SCSI CDB encodings (big-endian)
|
||||||
|
"system/services/vfs/path.zig", // mount-prefix path matching
|
||||||
|
"system/services/vfs/protocol.zig", // NodeKind / DirectoryEntry sizes + op values
|
||||||
|
"system/services/fat/on-disk.zig", // FAT on-disk struct sizes + type detection
|
||||||
|
"system/services/fat/engine.zig", // FAT read/write over a RAM-backed image
|
||||||
|
}) |root| {
|
||||||
|
const mod_tests = b.addTest(.{
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path(root),
|
||||||
|
.target = target,
|
||||||
|
.optimize = optimize,
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
test_step.dependOn(&b.addRunArtifact(mod_tests).step);
|
||||||
|
}
|
||||||
|
|
||||||
|
// The xkeyboard-config keymap tests need its generated `layouts` import wired, so they
|
||||||
|
// don't fit the plain loop above. Its keycode->character assertions are the end-to-end
|
||||||
|
// proof that the xkb-data -> generator -> Zig-lookup pipeline is correct.
|
||||||
|
const xkb_tests = b.addTest(.{
|
||||||
.root_module = b.createModule(.{
|
.root_module = b.createModule(.{
|
||||||
.root_source_file = b.path("src/root.zig"),
|
.root_source_file = b.path("library/xkeyboard-config/xkeyboard-config.zig"),
|
||||||
.target = target,
|
.target = target,
|
||||||
.optimize = optimize,
|
.optimize = optimize,
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "layouts", .module = xkb_layouts_module },
|
||||||
|
},
|
||||||
}),
|
}),
|
||||||
});
|
});
|
||||||
|
test_step.dependOn(&b.addRunArtifact(xkb_tests).step);
|
||||||
|
|
||||||
const run_mod_tests = b.addRunArtifact(mod_tests);
|
// runtime.time's Instant/Duration arithmetic. time.zig pulls in system.zig (the
|
||||||
|
// syscall wrappers), which needs the `abi` module, so it doesn't fit the plain
|
||||||
|
// loop above.
|
||||||
|
const time_tests = b.addTest(.{
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path("library/runtime/time.zig"),
|
||||||
|
.target = target,
|
||||||
|
.optimize = optimize,
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "abi", .module = abi_module },
|
||||||
|
},
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
test_step.dependOn(&b.addRunArtifact(time_tests).step);
|
||||||
|
|
||||||
const test_step = b.step("test", "Run tests");
|
// Convenience: `zig build gen-xkeyboard-config` regenerates the layout tables from the
|
||||||
test_step.dependOn(&run_mod_tests.step);
|
// vendored data (offline). `fetch` (the network step) stays a manual script run.
|
||||||
|
const gen_xkb = b.addSystemCommand(&.{ "python3", "tools/make-xkeyboard-config.py", "generate" });
|
||||||
|
const gen_xkb_step = b.step("gen-xkeyboard-config", "Regenerate library/xkeyboard-config/generated from the vendored data");
|
||||||
|
gen_xkb_step.dependOn(&gen_xkb.step);
|
||||||
}
|
}
|
||||||
|
|||||||
+145
-14
@@ -35,9 +35,41 @@ rather than restate it. Roughly in the order things happen at runtime:
|
|||||||
multitasking: kernel threads, the context switch, O(1) priority selection, and
|
multitasking: kernel threads, the context switch, O(1) priority selection, and
|
||||||
blocking (sleep, wait queues) — the leap to a running system.
|
blocking (sleep, wait queues) — the leap to a running system.
|
||||||
11. **[ipc.md](ipc.md) — inter-process communication.** Bounded blocking
|
11. **[ipc.md](ipc.md) — inter-process communication.** Bounded blocking
|
||||||
message-passing channels — the backbone the microkernel's isolated servers will
|
message-passing channels, then synchronous call/reply between *processes* over
|
||||||
talk over.
|
endpoints — the backbone the microkernel's isolated servers talk over.
|
||||||
12. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
12. **[syscall.md](syscall.md) — system calls.** How ring 3 asks the kernel for
|
||||||
|
something: the `syscall`/`sysret` fast path, the trap frame, and why the table is
|
||||||
|
deliberately tiny.
|
||||||
|
13. **[drivers.md](drivers.md) — writing a driver.** The payoff: a driver is an
|
||||||
|
ordinary ring-3 process that claims a device, maps its registers, and **sleeps
|
||||||
|
until its hardware interrupts it**. The claim is the capability; `irq_ack` is the
|
||||||
|
unmask.
|
||||||
|
14. **[driver-model.md](driver-model.md) — buses, classes and host controllers.** How
|
||||||
|
real driver stacks factor into three shapes and how families share code. The
|
||||||
|
three primitives it proposed are long since built (M13 capability passing,
|
||||||
|
M14 DMA + barriers, M15 MSI), and the driver *contract* on top of them —
|
||||||
|
hello, supervision, restart — is built too (device-manager.md, M18).
|
||||||
|
15. **[process-management.md](process-management.md) — process management.** The
|
||||||
|
microkernel's `ps`/`kill`/SIGCHLD: enumerate as a table snapshot, the
|
||||||
|
supervision link as the kill authority, and child-exit notifications over the
|
||||||
|
same endpoints IRQs arrive on.
|
||||||
|
16. **[process-lifecycle.md](process-lifecycle.md) — the process lifecycle.** Built
|
||||||
|
(M17): signals over IPC as the one lifecycle vocabulary every process speaks — the
|
||||||
|
POSIX.1-1990 words with message delivery instead of stack hijack, the stable
|
||||||
|
`runtime.process` interface, exit reasons, published exit events any stateful
|
||||||
|
service can subscribe to (the VFS releasing dead clients' handles), and the two
|
||||||
|
iron rules (cleanup is the kernel's job; kill is not a signal).
|
||||||
|
17. **[device-manager.md](device-manager.md) — the device manager.** Built (M18,
|
||||||
|
through the app surface): the
|
||||||
|
tree, the matcher, and the supervisor. Tree structure lives in the manager,
|
||||||
|
authority stays in the kernel; bus drivers report what they see; drivers are
|
||||||
|
restarted through the lifecycle vocabulary — the plan that turns
|
||||||
|
[resilience.md](resilience.md)'s restart goal into increments.
|
||||||
|
18. **[input.md](input.md) — the input module.** Broadcasting input events (keyboard,
|
||||||
|
mouse, joystick): why a synchronous rendezvous can't fan out to many listeners, the
|
||||||
|
asynchronous `ipc_send` primitive built to fix it, and the per-device subscribe/publish
|
||||||
|
service layered on top.
|
||||||
|
19. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
||||||
how `while (true) hlt` parks the CPU safely once there's nothing left to do.
|
how `while (true) hlt` parks the CPU safely once there's nothing left to do.
|
||||||
|
|
||||||
Start with the north star:
|
Start with the north star:
|
||||||
@@ -50,9 +82,19 @@ Start with the north star:
|
|||||||
- **[resilience.md](resilience.md) — resilience.** A design note (not built yet) on
|
- **[resilience.md](resilience.md) — resilience.** A design note (not built yet) on
|
||||||
fault isolation + live restart — the reincarnation-server + capability model that
|
fault isolation + live restart — the reincarnation-server + capability model that
|
||||||
makes "if I break it, I can restart it" real. danos's core motivation.
|
makes "if I break it, I can restart it" real. danos's core motivation.
|
||||||
|
- **[zig-self-hosting.md](zig-self-hosting.md) — running Zig on danos.** A design note
|
||||||
|
(not built yet) on making danos a real Zig target (`-target x86_64-danos`) and
|
||||||
|
eventually running the compiler on it. The key realisation: Zig 0.16 reduces an OS
|
||||||
|
port to **one seam** (`std.os.danos`), so we build `runtime.os` (→ that seam) plus a
|
||||||
|
thin `runtime.fs`, retire the `posix` shim, and follow a phased path to
|
||||||
|
`zig build-exe hello.zig` running on danos — **not** Linux-ABI emulation.
|
||||||
|
|
||||||
Cutting across all of these:
|
Cutting across all of these:
|
||||||
|
|
||||||
|
- **[system-requirements.md](system-requirements.md) — system requirements.** The
|
||||||
|
hardware needed to run danos: minimum specs (UEFI x86-64, ACPI, PCIe ECAM,
|
||||||
|
xHCI, ~128 MiB RAM) grounded in what the boot path actually assumes, plus a
|
||||||
|
plain-language guide matching Intel/AMD CPU generations by name.
|
||||||
- **[arch.md](arch.md) — the architecture split.** How CPU-specific code is kept
|
- **[arch.md](arch.md) — the architecture split.** How CPU-specific code is kept
|
||||||
behind a build-time `arch` module so the generic kernel never names x86_64,
|
behind a build-time `arch` module so the generic kernel never names x86_64,
|
||||||
leaving room for other systems (e.g. an AArch64 Raspberry Pi) later.
|
leaving room for other systems (e.g. an AArch64 Raspberry Pi) later.
|
||||||
@@ -64,10 +106,23 @@ Cutting across all of these:
|
|||||||
when to build it, and how to keep it architecture-agnostic.
|
when to build it, and how to keep it architecture-agnostic.
|
||||||
- **[acpi.md](acpi.md) — finding the ACPI tables.** The concrete x86 locator chain:
|
- **[acpi.md](acpi.md) — finding the ACPI tables.** The concrete x86 locator chain:
|
||||||
how the loader captures the **RSDP**, hands its physical address across in `BootInfo`,
|
how the loader captures the **RSDP**, hands its physical address across in `BootInfo`,
|
||||||
and how the platform derives the **RSDT/XSDT** from it and walks the SDTs.
|
and how the platform derives the **RSDT/XSDT** from it and walks the SDTs — plus the
|
||||||
|
live event side (the SCI, the power button, GPE/Notify) the ring-3 acpi service runs.
|
||||||
|
- **[power.md](power.md) — the power service.** System power as a domain-named
|
||||||
|
service: button/lid/battery events published to subscribers, and init's orderly
|
||||||
|
shutdown composing the [lifecycle](process-lifecycle.md) stop sequence with an ACPI
|
||||||
|
S5 write. Firmware-neutral — a PSCI backend drops in on ARM.
|
||||||
|
- **[timers.md](timers.md) — timers and time.** The ring-3 surface for reading the
|
||||||
|
clock and waiting: why `now()` is a syscall rather than a service, and the one-shot
|
||||||
|
timer notification (`timer_bind`) that gives supervisors a timed wait — built on the
|
||||||
|
LAPIC heartbeat and calibrated TSC of [device-interrupts.md](device-interrupts.md).
|
||||||
- **[smp.md](smp.md) — multiple cores.** A design/research note on how microkernels
|
- **[smp.md](smp.md) — multiple cores.** A design/research note on how microkernels
|
||||||
(L4, seL4) handle SMP — big kernel lock vs per-CPU vs multikernel — and how the
|
(L4, seL4) handle SMP — big kernel lock vs per-CPU vs multikernel — and how the
|
||||||
right choice depends on whether danos is chasing real-time or resilience.
|
right choice depends on whether danos is chasing real-time or resilience.
|
||||||
|
- **[coding-standards.md](coding-standards.md) — coding standards.** The naming rule the
|
||||||
|
tree follows: non-acronyms are spelled out in full (`message`, not `msg`), files are
|
||||||
|
`kebab-case`, code follows Zig's case conventions, and the handful of exceptions
|
||||||
|
(POSIX/C ABI names, `init`/`len`/`ptr`, acronyms).
|
||||||
- **[sysv.md](sysv.md) — the calling convention.** What "the kernel is SysV" means,
|
- **[sysv.md](sysv.md) — the calling convention.** What "the kernel is SysV" means,
|
||||||
and why the loader→kernel boundary has to pin it (the RDI-vs-RCX handoff).
|
and why the loader→kernel boundary has to pin it (the RDI-vs-RCX handoff).
|
||||||
- **[testing.md](testing.md) — testing.** How the kernel is tested by booting it in
|
- **[testing.md](testing.md) — testing.** How the kernel is tested by booting it in
|
||||||
@@ -94,19 +149,95 @@ passing messages over **[IPC](ipc.md)** channels — runs, its CPU-specific bits
|
|||||||
behind the [arch](arch.md) boundary, and when idle, or on a panic, it **halts**
|
behind the [arch](arch.md) boundary, and when idle, or on a panic, it **halts**
|
||||||
([halting.md](halting.md)).
|
([halting.md](halting.md)).
|
||||||
|
|
||||||
|
Above that line the microkernel proper begins: **discovery** ([discovery.md](discovery.md),
|
||||||
|
[acpi.md](acpi.md)) learns what hardware exists, ring-3 processes ask the kernel for
|
||||||
|
things through the small **[syscall](syscall.md)** table, isolated servers reach each
|
||||||
|
other over IPC **endpoints** ([ipc.md](ipc.md)), and a **[driver](drivers.md)** claims
|
||||||
|
a device, maps its registers, and sleeps until the hardware interrupts it — which is
|
||||||
|
the whole reason for the arrangement ([vision.md](vision.md)).
|
||||||
|
|
||||||
|
## Repository layout
|
||||||
|
|
||||||
|
danos is a **monorepo of sub-projects**. Each service or driver is a directory that is
|
||||||
|
its own Zig module — it can hold as many files as it needs, and other sub-projects
|
||||||
|
reach it *by module name*, never by a path into its files. The source tree deliberately
|
||||||
|
**mirrors the runtime FHS** ([danos-file-system-hierarchy-FSH.md](danos-file-system-hierarchy-FSH.md)):
|
||||||
|
what you see under `system/` in the source is what a running danos represents under
|
||||||
|
`/system`.
|
||||||
|
|
||||||
|
**A sub-project is addressed by its directory; its entry point repeats the directory's
|
||||||
|
name.** `system/services/init/` contains `init.zig` (its root), and produces a binary
|
||||||
|
addressed as **`system/services/init`** — the repeated leaf resolves away:
|
||||||
|
|
||||||
|
| Source (root file) | Addressed as (module / binary / FHS path) |
|
||||||
|
|----------------------------------------|--------------------------------------------|
|
||||||
|
| `system/services/init/init.zig` | `system/services/init` → `/system/services/init` |
|
||||||
|
| `system/drivers/ps2-bus/ps2-bus.zig` | `system/drivers/ps2-bus` → `/system/drivers/ps2-bus` |
|
||||||
|
| `library/runtime/runtime.zig` | `library/runtime` (the `runtime` module) |
|
||||||
|
|
||||||
|
In **source**, a sub-project is a directory so it can hold many files — the entry is
|
||||||
|
`init/init.zig`, beside it `vfs/vfs-test.zig`, `vfs/protocol.zig`, and so on. When
|
||||||
|
**addressed or installed**, that collapses to the single canonical path: the `init`
|
||||||
|
binary installs to `/system/services/init` (a file at that path), not
|
||||||
|
`/system/services/init/init`. The repeated leaf exists only in source; the directory is
|
||||||
|
the identity, the entry file is its implementation. (Same idea as a Go package being its
|
||||||
|
directory, or a macOS `.app` bundle addressed by the bundle, not the executable within.)
|
||||||
|
A sub-project's extra files are reached through the module, never as separate paths.
|
||||||
|
|
||||||
|
```
|
||||||
|
system/ → /system danos's own internals (the self-representation)
|
||||||
|
boot-handoff.zig the loader↔kernel contract (the `boot-handoff` module)
|
||||||
|
abi.zig the private kernel↔runtime syscall ABI (the `abi` module)
|
||||||
|
parameters.zig initial-ramdisk.zig shared contracts
|
||||||
|
kernel/ IPC, memory, scheduling, the private syscall dispatch
|
||||||
|
architecture/x86_64/ the `architecture` module (never named by generic code)
|
||||||
|
devices/ the device model /system/devices reflects (+ aml/)
|
||||||
|
device-abi.zig the device wire types (the `device-abi` module)
|
||||||
|
drivers/ hpet/ bus/ one sub-project per driver → /system/drivers
|
||||||
|
services/ init/ vfs/ device-manager/ system servers → /system/services (vfs/ holds
|
||||||
|
vfs.zig, vfs-test.zig, protocol.zig)
|
||||||
|
library/ → /lib libraries, one sub-directory each
|
||||||
|
runtime/ the danos-native runtime + file API (fs) — the stable application ABI
|
||||||
|
boot/ → /boot the loaders
|
||||||
|
tools/ test/ host-side build + QEMU test harness
|
||||||
|
```
|
||||||
|
|
||||||
|
A sub-project exposes its **public interface as a module**: `system/services/vfs/` owns
|
||||||
|
the VFS wire protocol (`protocol.zig`, the `vfs-protocol` module), which the runtime's
|
||||||
|
file API (`runtime.fs`) imports by name. `usb`/`block` drivers expose their protocols the
|
||||||
|
same way.
|
||||||
|
|
||||||
|
There is **no POSIX/C compatibility layer today**: danos programs do file I/O through the
|
||||||
|
danos-native `runtime.fs` (open/read/write/list over the VFS). A hand-rolled POSIX shim
|
||||||
|
(`library/posix/`) was retired as premature — the real POSIX/C surface will come later
|
||||||
|
from the `std.os.danos` seam (and, eventually, musl) when danos becomes a Zig target (see
|
||||||
|
[zig-self-hosting.md](zig-self-hosting.md)). When it does, the foreign-ABI naming
|
||||||
|
exception in [coding-standards.md](coding-standards.md) applies to that seam.
|
||||||
|
|
||||||
## Source map
|
## Source map
|
||||||
|
|
||||||
| Area | Code |
|
| Area | Code |
|
||||||
|------|------|
|
|------|------|
|
||||||
| Boot methods (one per way of booting the kernel) | `src/boot/` — `efi.zig` (UEFI) → `BOOTX64.efi` |
|
| Boot methods (one per way of booting the kernel) | `boot/` — `efi.zig` (UEFI) → `BOOTX64.efi` |
|
||||||
| Kernel entry, panic, bring-up | `src/kernel/main.zig` |
|
| Kernel entry, panic, bring-up | `system/kernel/kernel.zig` |
|
||||||
| Shared loader↔kernel contract (`BootInfo`, `Framebuffer`, `MemoryMap`, ABI) | `src/root.zig` |
|
| Loader↔kernel handoff (`BootInfo`, `Framebuffer`, `MemoryMap`, VM layout) | `system/boot-handoff.zig` |
|
||||||
| Physical frame allocator | `src/kernel/pmm.zig` |
|
| Private kernel↔runtime syscall ABI (`SystemCall`, mmap prot flags, `page_size`) — the runtime speaks it, not apps | `system/abi.zig` |
|
||||||
| Kernel heap (`std.mem.Allocator`) | `src/kernel/heap.zig` |
|
| Device wire types (`DeviceDescriptor`, `DeviceClass`, …) | `system/devices/device-abi.zig` |
|
||||||
| Scheduler (fixed-priority preemptive; blocking, wait queues) | `src/kernel/sched.zig` |
|
| Physical frame allocator | `system/kernel/pmm.zig` |
|
||||||
| IPC channels (message passing) | `src/kernel/ipc.zig` |
|
| Kernel heap (`std.mem.Allocator`) | `system/kernel/heap.zig` |
|
||||||
| Framebuffer text console (mirrors to serial) | `src/kernel/console.zig` |
|
| Scheduler (fixed-priority preemptive; blocking, wait queues) | `system/kernel/scheduler.zig` |
|
||||||
| In-kernel test cases | `src/kernel/tests.zig` |
|
| Big kernel lock + interrupt-safe critical sections | `system/kernel/sync.zig` |
|
||||||
| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/timer, serial, linker script) | `src/kernel/arch/x86_64/` |
|
| IPC channels between kernel threads (message passing) | `system/kernel/ipc.zig` |
|
||||||
|
| IPC endpoints: cross-address-space call/reply, handles, notifications | `system/kernel/ipc-synchronous.zig` |
|
||||||
|
| User processes: ELF loading, address spaces, the syscall table | `system/kernel/process.zig` |
|
||||||
|
| Device tree + claim capability + `device_register` containment | `system/kernel/devices-broker.zig` |
|
||||||
|
| IRQ-as-IPC: routing a device interrupt to a driver's endpoint | `system/kernel/irq.zig` |
|
||||||
|
| Hardware discovery (ACPI/device tree) behind one neutral device model | `system/devices/` |
|
||||||
|
| Framebuffer text console (mirrors to serial) | `system/kernel/console.zig` |
|
||||||
|
| In-kernel test cases | `system/kernel/tests.zig` |
|
||||||
|
| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/IO-APIC/timer, serial, linker script) | `system/kernel/architecture/x86_64/` |
|
||||||
|
| danos-native runtime (`runtime`): syscall wrappers, heap, IPC, device access, the file API (`fs`) — the stable application ABI | `library/runtime/` |
|
||||||
|
| System services (init, the VFS server + `protocol`, the device-manager) | `system/services/` |
|
||||||
|
| Device drivers, one sub-project each (`pci-bus`, `ps2-bus`, `usb-xhci-bus` bus drivers) | `system/drivers/` |
|
||||||
| Build + `run-x86-64` (QEMU/OVMF) | `build.zig` |
|
| Build + `run-x86-64` (QEMU/OVMF) | `build.zig` |
|
||||||
| QEMU integration test harness | `test/qemu_test.py` |
|
| QEMU integration test harness | `test/qemu_test.py` |
|
||||||
|
|||||||
+61
-7
@@ -16,13 +16,13 @@ the RSDT's address is a field *inside* the RSDP. The platform follows that point
|
|||||||
UEFI configuration table
|
UEFI configuration table
|
||||||
│ the loader reads the RSDP's physical address
|
│ the loader reads the RSDP's physical address
|
||||||
▼
|
▼
|
||||||
BootInfo.acpi_rsdp (u64, in the shared `danos` module) src/root.zig
|
BootInfo.acpi_rsdp (u64, in the loader↔kernel handoff) system/boot-handoff.zig
|
||||||
│ the kernel forwards the whole BootInfo
|
│ the kernel forwards the whole BootInfo
|
||||||
▼
|
▼
|
||||||
platform.discover(boot_info, …) src/device/platform.zig
|
platform.discover(boot_info, …) system/devices/platform.zig
|
||||||
│ reads boot_info.acpi_rsdp, hands it to the ACPI backend
|
│ reads boot_info.acpi_rsdp, hands it to the ACPI backend
|
||||||
▼
|
▼
|
||||||
acpi.discover(rsdp_phys, …) src/device/acpi.zig
|
acpi.discover(rsdp_phys, …) system/devices/acpi.zig
|
||||||
│ dereferences the RSDP, reads the pointer it contains
|
│ dereferences the RSDP, reads the pointer it contains
|
||||||
▼
|
▼
|
||||||
RSDP ──(a field in the struct)──► RSDT / XSDT ──► SDTs (MADT, MCFG, FADT, HPET, DSDT…)
|
RSDP ──(a field in the struct)──► RSDT / XSDT ──► SDTs (MADT, MCFG, FADT, HPET, DSDT…)
|
||||||
@@ -35,7 +35,7 @@ successor the **XSDT** (ACPI 2.0+) — which in turn lists every other SDT.
|
|||||||
## Step 1 — the loader finds the RSDP
|
## Step 1 — the loader finds the RSDP
|
||||||
|
|
||||||
Only the firmware knows where ACPI lives, so the RSDP must be grabbed while UEFI is
|
Only the firmware knows where ACPI lives, so the RSDP must be grabbed while UEFI is
|
||||||
still up. `acpiRootSystemDescriptorPointer()` in `src/boot/efi.zig` walks the UEFI
|
still up. `acpiRootSystemDescriptorPointer()` in `boot/efi.zig` walks the UEFI
|
||||||
**configuration table** for the ACPI GUID and returns the vendor pointer — the same
|
**configuration table** for the ACPI GUID and returns the vendor pointer — the same
|
||||||
"grab it before `ExitBootServices`" pattern as the [framebuffer](framebuffer.md) and
|
"grab it before `ExitBootServices`" pattern as the [framebuffer](framebuffer.md) and
|
||||||
the [memory map](memory-map.md).
|
the [memory map](memory-map.md).
|
||||||
@@ -44,11 +44,11 @@ the [memory map](memory-map.md).
|
|||||||
|
|
||||||
The loader can't just call the device module: the bootloader binary and the kernel
|
The loader can't just call the device module: the bootloader binary and the kernel
|
||||||
binary are compiled separately, and **the loader isn't linked against the `platform`
|
binary are compiled separately, and **the loader isn't linked against the `platform`
|
||||||
module at all** (it imports only the shared `danos` module). So instead of a call, it
|
module at all** (it imports only the `boot-handoff` contract). So instead of a call, it
|
||||||
deposits a value in the handoff struct:
|
deposits a value in the handoff struct:
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
// src/boot/efi.zig — while boot services are still up
|
// boot/efi.zig — while boot services are still up
|
||||||
.acpi_rsdp = if (acpiRootSystemDescriptorPointer()) |p| @intFromPtr(p) else 0,
|
.acpi_rsdp = if (acpiRootSystemDescriptorPointer()) |p| @intFromPtr(p) else 0,
|
||||||
```
|
```
|
||||||
|
|
||||||
@@ -107,12 +107,66 @@ firmware-agnostic [device model](discovery.md) gets populated; this note stops a
|
|||||||
part that answers "where are the tables?" — everything past the RSDP is just following
|
part that answers "where are the tables?" — everything past the RSDP is just following
|
||||||
more pointers the tables themselves provide.
|
more pointers the tables themselves provide.
|
||||||
|
|
||||||
|
## ACPI events: the SCI, the power button, and GPEs (M21)
|
||||||
|
|
||||||
|
The tables above are static description; ACPI is also a *live* channel. Hardware
|
||||||
|
raises the **SCI** (System Control Interrupt) — one shared, level-triggered line
|
||||||
|
whose vector the FADT names — and the OS reads status registers to learn what
|
||||||
|
happened: a fixed event like the power button, or a **General-Purpose Event**
|
||||||
|
(GPE) whose handler is an AML method. Since [discovery](discovery.md) moved AML
|
||||||
|
to ring 3, the event side lives there too, in the same **acpi service** — the
|
||||||
|
device discoverer and the event source are one process, because both need the
|
||||||
|
namespace and the port grant.
|
||||||
|
|
||||||
|
**The kernel hands the service what it needs and no more.** Reading PM1 event
|
||||||
|
blocks and GPE blocks requires the FADT, which the kernel already parses for its
|
||||||
|
own `\_S5` poweroff. Rather than re-parse, the kernel appends the **FADT as one
|
||||||
|
more memory resource** on the `acpi-tables` node; the service tells it apart
|
||||||
|
from the AML blob resources by signature — the FADT keeps its intact `"FACP"`
|
||||||
|
header, while the blob resources are header-stripped bytecode that starts with
|
||||||
|
no signature. The kernel's own FADT parse is untouched; the service reads the
|
||||||
|
PM1 *event* blocks (which the kernel never parsed — it only needs PM1 *control*
|
||||||
|
for `\_S5`) and the GPE0/GPE1 blocks straight from its copy. The **SCI itself**
|
||||||
|
arrives as the node's one `len == 1` irq resource (distinct from the broad
|
||||||
|
`[0, 256)` window that covers children's legacy lines), which is how the service
|
||||||
|
finds the line to `irq_bind`.
|
||||||
|
|
||||||
|
With those in hand the service enables ACPI mode (only if `SCI_EN` is clear —
|
||||||
|
some firmwares boot with it already set), sets `PWRBTN_EN`, and on each SCI:
|
||||||
|
|
||||||
|
- **The power button** is a *fixed* event: a set `PWRBTN_STS` bit in PM1 status.
|
||||||
|
The handler clears it (write-1-to-clear), logs the press, and publishes a
|
||||||
|
[`power`](power.md) `power_button` event to subscribers.
|
||||||
|
- **GPEs** are the general path: for each set-and-enabled GPE bit `n`, the
|
||||||
|
service evaluates its `\_GPE._L%02X` (level) or `_E%02X` (edge) handler
|
||||||
|
method, drains the **Notify** queue that method produced, maps each notified
|
||||||
|
device to an event (battery, AC, lid, or a generic `notify` with its code),
|
||||||
|
and clears the status bit. A missing handler method is clear-and-log, not an
|
||||||
|
error. Making GPEs work required teaching the interpreter one opcode it never
|
||||||
|
handled — `Notify` (`0x86`) — which it now folds into a bounded queue drained
|
||||||
|
per evaluation; everything else a handler needs (field access, control flow,
|
||||||
|
method calls) was already proven by the ring-3 `_STA`/`_CRS` work.
|
||||||
|
|
||||||
|
**How this is tested.** QEMU cannot raise GPEs deterministically on this config,
|
||||||
|
so GPE/Notify correctness is proven by **host unit tests** — hand-encoded AML
|
||||||
|
with a `Notify` inside a method body, run under `zig build test`. The QEMU
|
||||||
|
`power-button` scenario proves the fixed-event path end to end: a QMP
|
||||||
|
`system_powerdown` injects a real ACPI power-button press, and the service's SCI
|
||||||
|
handler must log it. Battery/AC/lid and the embedded controller's `_Qxx` queries
|
||||||
|
are interface-complete but validated on real hardware later.
|
||||||
|
|
||||||
|
The service surface these events are *published on* — subscription, the event
|
||||||
|
vocabulary, and orderly shutdown — is the power service, [power.md](power.md).
|
||||||
|
|
||||||
## Related
|
## Related
|
||||||
|
|
||||||
- [efi.md](efi.md) — the loader that captures the RSDP before `ExitBootServices`.
|
- [efi.md](efi.md) — the loader that captures the RSDP before `ExitBootServices`.
|
||||||
- [memory-map.md](memory-map.md) — the same loader-captures / kernel-consumes seam, and
|
- [memory-map.md](memory-map.md) — the same loader-captures / kernel-consumes seam, and
|
||||||
the ACPI-reclaim memory the RSDP lives in.
|
the ACPI-reclaim memory the RSDP lives in.
|
||||||
- [discovery.md](discovery.md) — the broader (still-evolving) plan for turning these
|
- [discovery.md](discovery.md) — the broader (still-evolving) plan for turning these
|
||||||
tables into one neutral device model shared with the ARM device-tree path.
|
tables into one neutral device model shared with the ARM device-tree path, and how
|
||||||
|
ACPI enumeration and events moved to the ring-3 acpi service.
|
||||||
|
- [power.md](power.md) — the domain-named power service the ACPI event side publishes
|
||||||
|
to (button, lid, battery) and its orderly-shutdown path into S5.
|
||||||
- [arch.md](arch.md) — why the kernel reaches the device code through a `platform`
|
- [arch.md](arch.md) — why the kernel reaches the device code through a `platform`
|
||||||
module and never names ACPI directly.
|
module and never names ACPI directly.
|
||||||
|
|||||||
+13
-13
@@ -13,7 +13,7 @@ runtime dispatch. `build.zig` exposes one architecture's code as a module called
|
|||||||
|
|
||||||
```zig
|
```zig
|
||||||
const arch_mod = b.addModule("arch", .{
|
const arch_mod = b.addModule("arch", .{
|
||||||
.root_source_file = b.path("src/kernel/arch/x86_64/cpu.zig"),
|
.root_source_file = b.path("system/kernel/architecture/x86_64/cpu.zig"),
|
||||||
});
|
});
|
||||||
```
|
```
|
||||||
|
|
||||||
@@ -26,7 +26,7 @@ arch.halt(); // never says "x86_64"
|
|||||||
```
|
```
|
||||||
|
|
||||||
Adding a second architecture is then a build-time choice: create
|
Adding a second architecture is then a build-time choice: create
|
||||||
`src/kernel/arch/aarch64/`, and point the `arch` module at it when the target CPU is
|
`system/kernel/arch/aarch64/`, and point the `arch` module at it when the target CPU is
|
||||||
AArch64. `main.zig` and `console.zig` don't change. **That compiler-checked module
|
AArch64. `main.zig` and `console.zig` don't change. **That compiler-checked module
|
||||||
boundary _is_ the architecture interface** — when a new arch is missing a function
|
boundary _is_ the architecture interface** — when a new arch is missing a function
|
||||||
the generic kernel calls, the build fails and names exactly what's missing.
|
the generic kernel calls, the build fails and names exactly what's missing.
|
||||||
@@ -36,7 +36,7 @@ the generic kernel calls, the build fails and names exactly what's missing.
|
|||||||
The split follows a simple test: does it name a CPU instruction, a hardware
|
The split follows a simple test: does it name a CPU instruction, a hardware
|
||||||
register, or a memory-management structure? If so, it's arch-specific.
|
register, or a memory-management structure? If so, it's arch-specific.
|
||||||
|
|
||||||
| Arch-specific — `src/kernel/arch/x86_64/` | Generic — kernel core |
|
| Arch-specific — `system/kernel/architecture/x86_64/` | Generic — kernel core |
|
||||||
|---|---|
|
|---|---|
|
||||||
| `cpu.zig`: `halt()` (`hlt`), later GDT/IDT/paging | `console.zig` — pure pixel math, works anywhere |
|
| `cpu.zig`: `halt()` (`hlt`), later GDT/IDT/paging | `console.zig` — pure pixel math, works anywhere |
|
||||||
| `linker.ld` — link layout, load address | `main.zig` — `kmain` orchestration, panic handler |
|
| `linker.ld` — link layout, load address | `main.zig` — `kmain` orchestration, panic handler |
|
||||||
@@ -51,9 +51,9 @@ should end up on the generic side; the arch module stays small.
|
|||||||
There are really two independent questions, and it's worth not conflating them:
|
There are really two independent questions, and it's worth not conflating them:
|
||||||
|
|
||||||
- **CPU architecture** (x86_64 vs AArch64): instructions, MMU, interrupts →
|
- **CPU architecture** (x86_64 vs AArch64): instructions, MMU, interrupts →
|
||||||
`src/kernel/arch/<cpu>/`.
|
`system/kernel/arch/<cpu>/`.
|
||||||
- **Boot protocol** (UEFI vs Raspberry Pi firmware + device tree): handled
|
- **Boot protocol** (UEFI vs Raspberry Pi firmware + device tree): handled
|
||||||
*separately*, because loaders are their own binaries. `src/boot/efi.zig` builds
|
*separately*, because loaders are their own binaries. `boot/efi.zig` builds
|
||||||
`BOOTX64.efi`, a distinct executable from the kernel ELF. On a Pi there is no
|
`BOOTX64.efi`, a distinct executable from the kernel ELF. On a Pi there is no
|
||||||
separate loader at all — the firmware jumps straight into the kernel with a
|
separate loader at all — the firmware jumps straight into the kernel with a
|
||||||
device-tree pointer, so that entry work would live in the AArch64 arch code.
|
device-tree pointer, so that entry work would live in the AArch64 arch code.
|
||||||
@@ -61,29 +61,29 @@ There are really two independent questions, and it's worth not conflating them:
|
|||||||
|
|
||||||
## Current x86_64 contents
|
## Current x86_64 contents
|
||||||
|
|
||||||
- **`src/kernel/arch/x86_64/cpu.zig`** — the `arch` module root. Exposes `halt()` (see
|
- **`system/kernel/architecture/x86_64/cpu.zig`** — the `arch` module root. Exposes `halt()` (see
|
||||||
[halting.md](halting.md)), `init()` (bring up the descriptor tables),
|
[halting.md](halting.md)), `init()` (bring up the descriptor tables),
|
||||||
`enablePaging()`, `setFaultHandler`, `readCr2`/`readCr3`, and the `CpuState`
|
`enablePaging()`, `setFaultHandler`, `readCr2`/`readCr3`, and the `CpuState`
|
||||||
trap frame.
|
trap frame.
|
||||||
- **`src/kernel/arch/x86_64/gdt.zig`** / **`idt.zig`** / **`tss.zig`** — the GDT, IDT and
|
- **`system/kernel/architecture/x86_64/gdt.zig`** / **`idt.zig`** / **`tss.zig`** — the GDT, IDT and
|
||||||
TSS plus CPU-exception handling (see [interrupts.md](interrupts.md)).
|
TSS plus CPU-exception handling (see [interrupts.md](interrupts.md)).
|
||||||
- **`src/kernel/arch/x86_64/paging.zig`** — the kernel's page tables (see
|
- **`system/kernel/architecture/x86_64/paging.zig`** — the kernel's page tables (see
|
||||||
[paging.md](paging.md)).
|
[paging.md](paging.md)).
|
||||||
- **`src/kernel/arch/x86_64/apic.zig`** — the Local APIC and its timer, the source of
|
- **`system/kernel/architecture/x86_64/apic.zig`** — the Local APIC and its timer, the source of
|
||||||
device interrupts (see [device-interrupts.md](device-interrupts.md)).
|
device interrupts (see [device-interrupts.md](device-interrupts.md)).
|
||||||
- **`src/kernel/arch/x86_64/serial.zig`** / **`io.zig`** — the COM1 UART (the kernel's
|
- **`system/kernel/architecture/x86_64/serial.zig`** / **`io.zig`** — the COM1 UART (the kernel's
|
||||||
machine-readable log channel, see [testing.md](testing.md)) and the shared
|
machine-readable log channel, see [testing.md](testing.md)) and the shared
|
||||||
port-I/O + MSR primitives.
|
port-I/O + MSR primitives.
|
||||||
- **`src/kernel/arch/x86_64/isr.s`** — the exception stubs, the `lgdt`/`lidt`/`ltr` load
|
- **`system/kernel/architecture/x86_64/isr.s`** — the exception stubs, the `lgdt`/`lidt`/`ltr` load
|
||||||
helpers, and the context switch (`switch_context` / `task_trampoline`, see
|
helpers, and the context switch (`switch_context` / `task_trampoline`, see
|
||||||
[scheduling.md](scheduling.md)) — real assembly, since Zig inline asm can't
|
[scheduling.md](scheduling.md)) — real assembly, since Zig inline asm can't
|
||||||
express them.
|
express them.
|
||||||
- **`src/kernel/arch/x86_64/linker.ld`** — the kernel link layout (fixed low load
|
- **`system/kernel/architecture/x86_64/linker.ld`** — the kernel link layout (fixed low load
|
||||||
address, one PT_LOAD per permission set).
|
address, one PT_LOAD per permission set).
|
||||||
|
|
||||||
The kernel entry point `_start` currently still lives in the generic `main.zig` as
|
The kernel entry point `_start` currently still lives in the generic `main.zig` as
|
||||||
a thin trampoline into `kmain`. It's arch-adjacent (its calling convention is
|
a thin trampoline into `kmain`. It's arch-adjacent (its calling convention is
|
||||||
x86_64 [SysV](sysv.md), via the shared `danos.kernel_abi`), but it's three lines
|
x86_64 [SysV](sysv.md), via the shared `system.kernel_abi`), but it's three lines
|
||||||
and mostly generic, so it stays put for now. When AArch64 arrives — where entry means setting
|
and mostly generic, so it stays put for now. When AArch64 arrives — where entry means setting
|
||||||
up a stack and reading a device-tree pointer from a register — the entry work will
|
up a stack and reading a device-tree pointer from a register — the entry work will
|
||||||
be substantial and per-arch, and *that* is when we extract an entry interface into
|
be substantial and per-arch, and *that* is when we extract an entry interface into
|
||||||
|
|||||||
+4
-4
@@ -18,7 +18,7 @@ matters for understanding why. This page maps the landscape so the
|
|||||||
new ISA.
|
new ISA.
|
||||||
|
|
||||||
They are as different from each other as either is from x86-64: separate registers,
|
They are as different from each other as either is from x86-64: separate registers,
|
||||||
page-table formats, and calling conventions. Each needs its own `src/kernel/arch/<name>/`.
|
page-table formats, and calling conventions. Each needs its own `system/kernel/arch/<name>/`.
|
||||||
|
|
||||||
## The Raspberry Pi models
|
## The Raspberry Pi models
|
||||||
|
|
||||||
@@ -57,16 +57,16 @@ the DTB/ACPI tells you what devices exist.
|
|||||||
|
|
||||||
## What danos needs, layer by layer
|
## What danos needs, layer by layer
|
||||||
|
|
||||||
- **One CPU arch module: `src/kernel/arch/aarch64/`** — covering the Zero 2 W and Pi 3-5,
|
- **One CPU arch module: `system/kernel/arch/aarch64/`** — covering the Zero 2 W and Pi 3-5,
|
||||||
providing the same `arch` interface as x86_64: `halt`, context switch,
|
providing the same `arch` interface as x86_64: `halt`, context switch,
|
||||||
interrupt/exception vectors, page tables, a UART, a timer. No `src/kernel/arch/arm/` is
|
interrupt/exception vectors, page tables, a UART, a timer. No `system/kernel/arch/arm/` is
|
||||||
planned (see the decision above), so there's a single ARM backend to write.
|
planned (see the decision above), so there's a single ARM backend to write.
|
||||||
- **A device-tree boot path.** Since stock Pis boot via DTB, danos needs an entry
|
- **A device-tree boot path.** Since stock Pis boot via DTB, danos needs an entry
|
||||||
that parses the DTB's `/memory` and `/reserved-memory` into the neutral
|
that parses the DTB's `/memory` and `/reserved-memory` into the neutral
|
||||||
[`MemoryMap`](memory-map.md) — the same neutral handoff `efi.zig` produces, just
|
[`MemoryMap`](memory-map.md) — the same neutral handoff `efi.zig` produces, just
|
||||||
from a different source. This is where keeping boot-protocol knowledge on the
|
from a different source. This is where keeping boot-protocol knowledge on the
|
||||||
loader side (as we did for the UEFI memory-map classification) pays off.
|
loader side (as we did for the UEFI memory-map classification) pays off.
|
||||||
- **The UEFI loader mostly carries over.** `src/boot/efi.zig` is largely
|
- **The UEFI loader mostly carries over.** `boot/efi.zig` is largely
|
||||||
boot-*protocol* code (`std.os.uefi` protocol calls), not x86 code. Its only truly
|
boot-*protocol* code (`std.os.uefi` protocol calls), not x86 code. Its only truly
|
||||||
x86-specific bits are the ELF machine check (`.X86_64`) and the SysV calling
|
x86-specific bits are the ELF machine check (`.X86_64`) and the SysV calling
|
||||||
convention for the kernel jump. So an `aarch64`-UEFI target (QEMU `virt` + AAVMF)
|
convention for the kernel jump. So an `aarch64`-UEFI target (QEMU `virt` + AAVMF)
|
||||||
|
|||||||
@@ -0,0 +1,197 @@
|
|||||||
|
# Coding standards
|
||||||
|
|
||||||
|
Conventions for danos source. The overriding one, from which most of the rest follows:
|
||||||
|
|
||||||
|
> **Names are spelled out in full. An identifier is not abbreviated unless the
|
||||||
|
> abbreviation is an acronym.**
|
||||||
|
|
||||||
|
`interruptDispatch`, not `intDisp`. `message_len`, not `message_len` (`msg` expands, `len`
|
||||||
|
is a Zig idiom — see the exceptions). `devices_broker`, not `devices_broker`. `scheduler`, not
|
||||||
|
`sched`. The cost of a longer name is paid once, at the keyboard; the cost of a
|
||||||
|
cryptic one is paid every time the code is read, by everyone who reads it. In a
|
||||||
|
microkernel whose whole argument is that a human can hold each piece in their head,
|
||||||
|
that trade is not close.
|
||||||
|
|
||||||
|
## The rule, precisely
|
||||||
|
|
||||||
|
**Acronyms and initialisms stay.** They *are* the full name — expanding them would make
|
||||||
|
the code worse, not better. `IPC`, `MMIO`, `DMA`, `IRQ`, `TSS`, `GDT`, `IDT`, `APIC`,
|
||||||
|
`GSI`, `HPET`, `ACPI`, `PCI`, `EOI`, `BAR`, `ECAM`, `MSI`, `CPU`, `ELF`, `ABI`, `UEFI`,
|
||||||
|
`MMU`, `TLB`, `ISR`, `ISA`, `GAS`, `HAL`, `PMM`, `VMM`, `VFS`, `HID`, `HCD`, `SMP`,
|
||||||
|
`AML`, `MADT`, `MCFG`, `FADT`, `RSDP`, `XSDT`, `RSDT`, `GOP`, `EDID`, `TSC`, `PIT`,
|
||||||
|
`RTC`, `LAPIC`, `SIPI`. In code they carry whatever case the surrounding convention
|
||||||
|
demands: `Hal` the type, `hal` the variable, `mapMmio` the function.
|
||||||
|
|
||||||
|
**Everything else is spelled out.** If it's a word with letters removed, restore them:
|
||||||
|
|
||||||
|
| Abbreviation | Full |
|
||||||
|
|---|---|
|
||||||
|
| `proto` | `protocol` |
|
||||||
|
| `msg` | `message` |
|
||||||
|
| `desc` | `descriptor` |
|
||||||
|
| `res` | `resource` |
|
||||||
|
| `recv` | `receive` |
|
||||||
|
| `buf` | `buffer` |
|
||||||
|
| `cur` | `current` |
|
||||||
|
| `src` / `dst` | `source` / `destination` |
|
||||||
|
| `idx` | `index` |
|
||||||
|
| `addr` | `address` |
|
||||||
|
| `reg` | `register` |
|
||||||
|
| `prev` | `previous` |
|
||||||
|
| `cfg` / `config` | `configuration` |
|
||||||
|
| `arch` | `architecture` |
|
||||||
|
| `sched` | `scheduler` |
|
||||||
|
| `dev` | `device` |
|
||||||
|
| `sys` / `syscall` | `system` / `system_call` |
|
||||||
|
| `info` | `information` |
|
||||||
|
| `dt` | `device_tree` |
|
||||||
|
| `ep` | `endpoint` |
|
||||||
|
| `rt` | `runtime` |
|
||||||
|
| `func` | `function` |
|
||||||
|
| `phys` / `virt` | `physical` / `virtual` |
|
||||||
|
| `wq` | `wait_queue` |
|
||||||
|
|
||||||
|
This list is illustrative, not exhaustive. The rule is the rule; when you meet a new
|
||||||
|
abbreviation, expand it.
|
||||||
|
|
||||||
|
## Exceptions
|
||||||
|
|
||||||
|
Three, and only three.
|
||||||
|
|
||||||
|
1. **Foreign ABI names are spelled exactly as the ABI spells them — but only inside
|
||||||
|
the layer that *is* that ABI.** A function that *is* the C or POSIX interface keeps
|
||||||
|
its name: `fopen`, `fwrite`, `fread`, `malloc`, `calloc`, `realloc`, `free`,
|
||||||
|
`memcpy`, `mmap`, `munmap`, `open`, `read`, `write`, `close`, `lseek`, `stat`,
|
||||||
|
`errno`, `O_CREAT`. We don't get to rename `fwrite` to `fileWrite` — it wouldn't be
|
||||||
|
`fwrite` any more.
|
||||||
|
|
||||||
|
**This exception is scoped to a file that *is* a foreign ABI, and nothing else.**
|
||||||
|
danos has no such file today: the old `library/posix/` compatibility shim was retired
|
||||||
|
once its callers moved to the danos-native `runtime.fs`, since a hand-rolled POSIX
|
||||||
|
layer is premature until danos actually needs it (see
|
||||||
|
[zig-self-hosting.md](zig-self-hosting.md)). The exception will apply again to the
|
||||||
|
`std.os.danos` seam when danos becomes a real Zig target — that module *is* the C-ABI
|
||||||
|
`system` interface, so it keeps `open`/`read`/`errno`/`O_CREAT`. **Everywhere else,
|
||||||
|
Zig/danos naming applies with no exception**: a concept POSIX also has gets a danos
|
||||||
|
name — the VFS wire protocol carries a `FileStatus`, not a `Stat`, and a `create`
|
||||||
|
flag, not `O_CREAT`; the boundary is where `stat`→`status` and `O_CREAT`→`create` get
|
||||||
|
mapped. (The `syscall` *wrappers* elsewhere are not an exception — they wrap the
|
||||||
|
private danos ABI, so they use danos names.)
|
||||||
|
|
||||||
|
2. **Zig idioms are spelled the way Zig spells them.** Three names are the language's,
|
||||||
|
not ours, and are left alone:
|
||||||
|
- **`init` / `deinit`** — the constructor convention (`std.ArrayList.init`), not a
|
||||||
|
shortening of "initialize".
|
||||||
|
- **`len` / `ptr`** — the slice field names (`slice.len`, `slice.ptr`). Our own
|
||||||
|
structs use bare `len`/`ptr` fields to mirror them, so a reader carries one
|
||||||
|
mental model. (Compounds still expand: a field is `message_len`, not
|
||||||
|
`message_length` — `len` is kept, `msg` is not.)
|
||||||
|
- The builtins (`@min`, `@max`, `@memcpy`) and `allocator.alloc` / `.create` are
|
||||||
|
Zig's spelling.
|
||||||
|
|
||||||
|
The rule governs the names *we* coin.
|
||||||
|
|
||||||
|
3. **Single-letter variables in a trivial local scope.** `for (items) |item, i|` may
|
||||||
|
keep `i`; a coordinate may be `x`, `y`. The moment the scope is big enough that the
|
||||||
|
letter's meaning isn't obvious on sight, give it a real name. When in doubt, name it.
|
||||||
|
|
||||||
|
That's all — no Unix-abbreviation exception. The source directories are full words
|
||||||
|
(`system`, `library`, not `src`/`lib`), and there is no daemon `d` suffix: a driver
|
||||||
|
lives in `system/drivers/` and a service in `system/services/`, so the *location*
|
||||||
|
already says what it is. Encoding the role in the name too (`busd`, `vfsd`) is
|
||||||
|
redundant — the program is just `ps2-bus`, `vfs`. Don't put in a name what its directory
|
||||||
|
already tells you.
|
||||||
|
|
||||||
|
## A note on collisions
|
||||||
|
|
||||||
|
Two identifiers can legitimately expand to the same word. When they do, keep both
|
||||||
|
meaningful by renaming one to its *specific* identity rather than the generic
|
||||||
|
expansion. Two cases resolved this way:
|
||||||
|
|
||||||
|
- The `config` module (compile-time tunables — `maximum_cpus`, `timer_hz`) would
|
||||||
|
collide with `cfg` (a `PlatformConfiguration` value) at `configuration`. The module
|
||||||
|
became **`parameters`**, which is what it holds.
|
||||||
|
- The kernel `device.zig` module would collide with `dev` (a device value) at
|
||||||
|
`device`. The module alias became **`device_model`**, which is what it is — the
|
||||||
|
device data model (`Device`, `DeviceTree`, `ResourceKind`).
|
||||||
|
- The `Namespace` module alias (`ns`/`nsp` across the AML files) collides with a
|
||||||
|
`Namespace` **instance**. Resolved by dropping the module alias entirely — the two
|
||||||
|
types it provided are imported directly (`const Node = @import("namespace.zig").Node;`)
|
||||||
|
— which frees `namespace` for the instance.
|
||||||
|
|
||||||
|
A related case is one abbreviation with two meanings. In the AML code, `op` means
|
||||||
|
**opcode** (`opcodes.zig`, the `*_opcode` constants) but `Op` in `BinaryOperation` /
|
||||||
|
`LogicOperation` means **operation** — distinguished by case. The per-opcode parser
|
||||||
|
handlers, formerly `opName`/`opField`, are `parseName`/`parseField`: they *parse* the
|
||||||
|
opcode's structure, which says what they do without overloading "op".
|
||||||
|
|
||||||
|
## Case and file names
|
||||||
|
|
||||||
|
Within those spelling rules, follow Zig's own conventions:
|
||||||
|
|
||||||
|
- **Types** — `PascalCase`: `DeviceDescriptor`, `Endpoint`, `WaitQueue`.
|
||||||
|
- **Functions** — `camelCase`: `mapUserDeviceInto`, `notifyFromIsr`.
|
||||||
|
- **Variables, fields, constants** — `snake_case`: `message_length`, `devices_broker`,
|
||||||
|
`notify_badge_bit`.
|
||||||
|
|
||||||
|
**File names are `kebab-case`.** A file named for a multi-word thing hyphenates it:
|
||||||
|
`device-tree.zig`, `ipc-synchronous.zig`, `vfs-protocol.zig`, `devices-broker.zig`. A
|
||||||
|
single word or acronym needs no hyphen: `scheduler.zig`, `paging.zig`, `apic.zig`,
|
||||||
|
`idt.zig`. (The module *alias* a file is imported under still follows the code
|
||||||
|
conventions above — `snake_case` — because it's an identifier, not a filename.)
|
||||||
|
|
||||||
|
**A sub-project's entry point repeats its directory's name** — `init/init.zig`,
|
||||||
|
`runtime/runtime.zig`, `ps2-bus/ps2-bus.zig` — and the sub-project is addressed by the
|
||||||
|
*directory* (`system/services/init`, `library/runtime`), with the repeated leaf
|
||||||
|
resolving away. See the repository-layout section of [README.md](README.md).
|
||||||
|
|
||||||
|
## Named values, not magic numbers
|
||||||
|
|
||||||
|
The naming rule has a twin: **a value with meaning gets a name, too.** The same
|
||||||
|
principle drives both — a reader should never have to leave the code to understand it.
|
||||||
|
An abbreviated *name* forces a reader to guess; a bare *number* forces them worse, out
|
||||||
|
to a spec or a header or a comment three files away, to learn what the value even *is*.
|
||||||
|
If `0x0C` is the PCI serial-bus class, the code says `BaseClass.serial_bus`, not `0x0C`;
|
||||||
|
if `0x04` is the ACPI IRQ resource descriptor, it says `SmallResourceType.irq`, not
|
||||||
|
`0x04`. The number is an implementation detail of the name — recorded once, where the
|
||||||
|
name is defined, and never spelled again at a use site.
|
||||||
|
|
||||||
|
**Prefer an `enum`** when the values form a set (device classes, AML opcodes, resource
|
||||||
|
descriptor types, states): the type then also says *which* set a value belongs to, and
|
||||||
|
the compiler rejects a value from the wrong one. A lone `pub const` with a descriptive
|
||||||
|
name suffices for a one-off (`const large_descriptor_bit = 0x80`). Reach for the enum
|
||||||
|
the moment code elsewhere compares against, packs, or produces the value — a packed PCI
|
||||||
|
class triple is written from named parts (`.serial_bus`, `.usb`, `.xhci`), never as
|
||||||
|
`0x0C_03_30` under a comment that decodes the bytes.
|
||||||
|
|
||||||
|
The exceptions are the numbers that carry no hidden meaning: `0` and `1` as plain zero
|
||||||
|
and one, an index step, a field width, a bit shift. `x + 1`, `buffer[0]`, and `<< 8`
|
||||||
|
need no christening — there is nothing to look up. The test is exactly the naming test:
|
||||||
|
*would a reader have to look this up to know what it means?* If yes, name it. This is
|
||||||
|
what `opcodes.zig`'s `*_opcode` constants, `acpi-ids`'s `HardwareId`, and `pci-class`'s
|
||||||
|
class enums already are — reference data defined once and named everywhere it is used.
|
||||||
|
|
||||||
|
## Why acronyms are the line
|
||||||
|
|
||||||
|
Because an acronym has no letters to restore. `MMIO` doesn't become "memory mapped
|
||||||
|
input output" in code — that expansion is what the acronym *is for*. But `msg` is just
|
||||||
|
`message` with three letters stolen, and stealing them buys nothing a reader wants. The
|
||||||
|
test for "is this an abbreviation I must expand" is simply: *is there a longer word this
|
||||||
|
is a clipped form of?* If yes, write the word. If it's an initialism standing in for a
|
||||||
|
phrase, leave it.
|
||||||
|
|
||||||
|
## Zen of Zig
|
||||||
|
|
||||||
|
* Communicate intent precisely.
|
||||||
|
* Edge cases matter.
|
||||||
|
* Favor reading code over writing code.
|
||||||
|
* Only one obvious way to do things.
|
||||||
|
* Runtime crashes are better than bugs.
|
||||||
|
* Compile errors are better than runtime crashes.
|
||||||
|
* Incremental improvements.
|
||||||
|
* Avoid local maximums.
|
||||||
|
* Reduce the amount one must remember.
|
||||||
|
* Focus on code rather than style.
|
||||||
|
* Resource allocation may fail; resource deallocation must succeed.
|
||||||
|
* Memory is a resource.
|
||||||
|
* Together we serve the users.
|
||||||
@@ -0,0 +1,121 @@
|
|||||||
|
# DanOS Filesystem Hierarchy Standard (DFHS)
|
||||||
|
|
||||||
|
Most modern Unix and Unix-like operating systems follow the FHS. DanOS has its own FHS structure which extends the unix FHS. This is provided by virtual file system driver (VFS).
|
||||||
|
|
||||||
|
## Directory structure
|
||||||
|
|
||||||
|
| Path | Description |
|
||||||
|
|------------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||||
|
| / | Primary hierarchy root and root directory of the entire file system hierarchy. |
|
||||||
|
| /bin | Essential command binaries that need to be available in single-user mode, including to bring up the system or repair it, for all users (e.g., cat, ls, cp). |
|
||||||
|
| /boot | Boot loader files (e.g., EFI, initial-ramdisk.img ). |
|
||||||
|
| /dev | POSIX Device files (e.g., /dev/null, /dev/disk0, /dev/tty, /dev/random). |
|
||||||
|
| /etc | Host-specific system-wide configuration files. |
|
||||||
|
| /home | Users' home directories, containing saved files, personal settings, etc. |
|
||||||
|
| /lib | Libraries essential for the binaries in /bin and /sbin. eg realtime, system, ipc etc. |
|
||||||
|
| /sbin | Essential system binaries (e.g init) |
|
||||||
|
| /srv | Site-specific data served by this system, such as data and scripts for web servers, data offered by FTP servers, and repositories for version control systems |
|
||||||
|
| /system | DanOS operating system files (similar idea to C:\Windows). A true representation of danos — its layout mirrors the source tree, so `/system` is what danos *is*. |
|
||||||
|
| /system/devices | danos virtual device tree e.g. similar to /sys on linux but with danos device tree conventions (the structures in the devices module) |
|
||||||
|
| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/pci-bus, /system/drivers/ps2-bus) |
|
||||||
|
| /system/services | system-service binaries — the VFS server, init, and other user-mode servers (e.g. /system/services/vfs, /system/services/init) |
|
||||||
|
| /system/kernel | the kernel image |
|
||||||
|
| /tmp | Directory for temporary files (see also /var/tmp). Often not preserved between system reboots and may be severely size-restricted. |
|
||||||
|
| /usr | Secondary hierarchy for read-only user data; contains the majority of (multi-)user utilities and applications. Should be shareable and read-only. |
|
||||||
|
| /var | Variable files: files whose content is expected to continually change during normal operation of the system, such as logs, spool files, and temporary e-mail files. |
|
||||||
|
|
||||||
|
## File types
|
||||||
|
|
||||||
|
POSIX specifies the long format of the ls command to represent the Unix file type as the first letter for an entry.
|
||||||
|
|
||||||
|
| type | symbol | Description |
|
||||||
|
|-------------------|--------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||||
|
| regular | - | An ordinary file holding an uninterpreted byte stream. Reads and writes are positional, and the file grows on demand (e.g., a binary in /bin, a config file in /etc). |
|
||||||
|
| directory | d | A container mapping names to other files. It may only be modified through directory operations, never written to directly. |
|
||||||
|
| symbolic link | l | A file whose contents are a path that is resolved in its place. The target need not exist, and may cross mount points. |
|
||||||
|
| FIFO special | p | A named pipe: an in-order byte stream between processes, where writers block until a reader opens the other end. |
|
||||||
|
| block special | b | A device node addressed in fixed-size blocks with the kernel free to buffer and reorder access (e.g., /dev/disk0). |
|
||||||
|
| character special | c | A device node addressed as an unbuffered byte stream, delivered to the driver in order (e.g., /dev/tty, /dev/null). |
|
||||||
|
| socket | s | A named endpoint for bidirectional message-passing between processes, bound to a path rather than an address. |
|
||||||
|
|
||||||
|
## /dev
|
||||||
|
|
||||||
|
`/dev` holds the names through which processes reach devices. It is deliberately not
|
||||||
|
the device tree: the tree — every node discovered by ACPI or PCI enumeration, with its
|
||||||
|
resources and its parent — lives under [/system/devices](#directory-structure) and is
|
||||||
|
addressed by device id. `/dev` is the much smaller set of devices that have a driver
|
||||||
|
willing to serve them, addressed by name.
|
||||||
|
|
||||||
|
A device node is not a file the VFS can read. The bytes live in a driver process
|
||||||
|
([drivers.md](drivers.md)), so opening a `/dev` name has to resolve to that driver's
|
||||||
|
IPC endpoint, and subsequent reads and writes are calls against it. This is what
|
||||||
|
`system/services/vfs/vfs.zig` reserves for M10 and what the `Stat.kind` field is for; **none of it is
|
||||||
|
implemented today.** The current VFS is a flat, in-memory ramfs of eight nodes, with no
|
||||||
|
directories at all and `kind` hardcoded to zero. The three sections below describe the
|
||||||
|
intended shape, and are honest about which parts the kernel can already support.
|
||||||
|
|
||||||
|
### Character devices
|
||||||
|
|
||||||
|
A character device is a byte stream with no addressable position: bytes are delivered
|
||||||
|
to the driver in the order written, and a read consumes what is there. Terminals,
|
||||||
|
serial lines, keyboards and mice are all of this shape. These are the natural first
|
||||||
|
device nodes in danos, because a character driver needs nothing the kernel doesn't
|
||||||
|
already provide — it claims its device, maps its registers with `mmio_map`, and blocks
|
||||||
|
on `replyWait` for either an interrupt or a client request. `system/drivers/ps2-bus/ps2-bus.zig`
|
||||||
|
is already that program, minus the file-node client half.
|
||||||
|
|
||||||
|
The obstacle was never the file type; it is which hardware a ring-3 driver can reach.
|
||||||
|
Direct `in`/`out` from user space is still a #GP (no TSS I/O bitmap, IOPL never raised),
|
||||||
|
but a driver no longer needs it: **`io_read`/`io_write`** grant port access the same way
|
||||||
|
`mmio_map` grants memory — gated by `device_claim` and the device's discovered `io_port`
|
||||||
|
resource. So the 16550 UART at `0x3F8` and the PS/2 controller at `0x60`/`0x64` (and thus
|
||||||
|
`/dev/ttyS0` and a keyboard node) are now writable as ordinary ring-3 drivers; the
|
||||||
|
low-rate legacy hardware that needs port I/O is fine with a syscall per access. A
|
||||||
|
memory-mapped device such as the framebuffer, needing no port I/O at all, remains the
|
||||||
|
easiest first entry.
|
||||||
|
|
||||||
|
### Block devices
|
||||||
|
|
||||||
|
A block device is addressed in fixed-size blocks and, unlike a character device, the
|
||||||
|
layer above is free to buffer, reorder, coalesce and retry requests against it. Disks
|
||||||
|
and other persistent storage are the whole population of this class.
|
||||||
|
|
||||||
|
A block driver is now **writable, but not yet memory-safe.** Every storage controller
|
||||||
|
worth naming is a bus master: it is programmed by handing it the physical address of a
|
||||||
|
descriptor ring and left to read and write memory on its own. That ring is exactly what
|
||||||
|
**`dma_alloc`** now provides — physically contiguous, pinned, uncacheable, with its
|
||||||
|
physical address disclosed — and **`/lib/mmio`**'s barriers order the descriptor writes
|
||||||
|
against the doorbell, and **`msi_bind`** delivers completions. So an AHCI or NVMe driver
|
||||||
|
can be written today (the M14/M15 work in [driver-model.md](driver-model.md); the earlier
|
||||||
|
"cannot host a block driver at all" is no longer true).
|
||||||
|
|
||||||
|
What is *not* yet true is that it is safe. A device programmed with an arbitrary physical
|
||||||
|
address writes to arbitrary physical memory, and page tables do not sit between a device
|
||||||
|
and RAM — an IOMMU does. The IOMMU is now *detected* (M16), but no translation domains
|
||||||
|
are programmed, so granting a DMA-capable device to a driver process is still equivalent
|
||||||
|
to granting ring 0. Until per-device domains confine a driver's DMA to the buffers it
|
||||||
|
`dma_alloc`'d, a block driver works but forfeits the isolation that motivates user-space
|
||||||
|
drivers — enforcement is the next step, and lands with that first driver. A ramdisk over
|
||||||
|
the initial ramdisk remains the one block-shaped thing that needs no driver process at all.
|
||||||
|
|
||||||
|
### Pseudo-devices
|
||||||
|
|
||||||
|
A pseudo-device has the interface of a device and no hardware behind it: `/dev/null`
|
||||||
|
discarding writes and reading as end-of-file, `/dev/zero` reading as an endless run of
|
||||||
|
zero bytes, `/dev/full` failing writes with `ENOSPC`, `/dev/random` and `/dev/urandom`
|
||||||
|
yielding unpredictable bytes.
|
||||||
|
|
||||||
|
These are the only `/dev` entries danos can implement immediately, and they are the
|
||||||
|
sensible place to start, because they are exactly the entries that need no driver
|
||||||
|
process, no `device_claim`, no MMIO grant and no interrupt. The VFS server answers them
|
||||||
|
out of its own address space — `null` and `zero` are a few lines each in
|
||||||
|
`system/services/vfs/vfs.zig`'s `read` and `write` handlers. Doing so forces the two pieces of
|
||||||
|
structure that every later device node depends on and that the flat ramfs currently
|
||||||
|
lacks: a directory, so that `/dev/null` is a path rather than a name; and a populated
|
||||||
|
`Stat.kind`, so that a caller can tell a character device from a regular file.
|
||||||
|
|
||||||
|
`/dev/random` is the one that is not free. It needs an entropy source, and the honest
|
||||||
|
options on this kernel are `RDRAND`/`RDSEED` where CPUID advertises them, and the HPET
|
||||||
|
counter's low bits as a poor fallback. Neither is a seeded CSPRNG, and a `/dev/random`
|
||||||
|
that is merely unpredictable-looking is worse than none — nothing should be keyed from
|
||||||
|
it until it is a real one.
|
||||||
+68
-12
@@ -18,7 +18,7 @@ Interrupt delivery on modern x86 goes through the **APIC**, not the legacy 8259
|
|||||||
PIC. There are two halves; we only need one so far:
|
PIC. There are two halves; we only need one so far:
|
||||||
|
|
||||||
- The **Local APIC** (per-CPU, memory-mapped at physical `0xFEE00000`) handles the
|
- The **Local APIC** (per-CPU, memory-mapped at physical `0xFEE00000`) handles the
|
||||||
CPU's own timer and receives interrupts routed to it. `src/kernel/arch/x86_64/apic.zig`.
|
CPU's own timer and receives interrupts routed to it. `system/kernel/architecture/x86_64/apic.zig`.
|
||||||
- The **IO-APIC** routes *external* device lines (keyboard, etc.) to LAPIC vectors.
|
- The **IO-APIC** routes *external* device lines (keyboard, etc.) to LAPIC vectors.
|
||||||
Not needed for the timer — it'll arrive with the keyboard.
|
Not needed for the timer — it'll arrive with the keyboard.
|
||||||
|
|
||||||
@@ -78,6 +78,40 @@ preemption and wakeups (1 ms granularity); the **TSC** is the resolution you rea
|
|||||||
time at. Making `sleep` itself sub-millisecond would take a tickless one-shot
|
time at. Making `sleep` itself sub-millisecond would take a tickless one-shot
|
||||||
timer — a later step.
|
timer — a later step.
|
||||||
|
|
||||||
|
### Is the TSC trustworthy? Invariant, and synchronized
|
||||||
|
|
||||||
|
A cycle counter is only a valid *clock* if two things hold, and danos checks both,
|
||||||
|
because they decide whether we read time with a cheap `rdtsc` or fall back to the HPET.
|
||||||
|
|
||||||
|
**Invariant.** An old TSC counted core clock cycles, so it sped up and slowed down with
|
||||||
|
frequency scaling — useless as wall time. Modern CPUs (all of danos's targets) provide an
|
||||||
|
**invariant TSC**: a constant rate across P/C-states that never stops. The guarantee is a
|
||||||
|
CPUID bit — leaf `0x80000007`, EDX bit 8 — on both Intel *and* AMD. danos reads it in
|
||||||
|
`calibrate`, and a TSC that doesn't advertise it is not used as the clocksource. AMD is
|
||||||
|
why this matters in practice: it doesn't populate the Intel leaf `0x15` that enumerates
|
||||||
|
the TSC *frequency*, so danos already measures AMD's rate against the HPET — but a
|
||||||
|
measured frequency without the invariance guarantee is not enough.
|
||||||
|
|
||||||
|
**Synchronized.** Each core has its own TSC. Even invariant ones can start at different
|
||||||
|
values (a second socket, some firmware), so a thread migrating from a core reading
|
||||||
|
`1_000_000` to one reading `999_000` would see time jump *backward*. danos runs a **warp
|
||||||
|
check** as each application processor comes online (`checkWarpSource`, adapted from
|
||||||
|
Linux's): the waking core and the BSP hammer a shared "highest seen" TSC under a lock,
|
||||||
|
and if either ever reads below it, the cores' TSCs are skewed. It's pairwise because APs
|
||||||
|
come up one at a time ([smp.md](smp.md)).
|
||||||
|
|
||||||
|
**The fallback.** When the TSC fails either test — non-invariant (a bare VM such as the
|
||||||
|
default qemu64), or warped between cores — danos moves the monotonic clock onto the
|
||||||
|
**HPET** main counter: one fixed-rate counter, so it can neither skew between cores nor
|
||||||
|
drift with frequency. It costs a memory-mapped read instead of a register read, but it
|
||||||
|
keeps time *accurate*, which is the whole point. The switch preserves the current value,
|
||||||
|
so the clock never jumps. The boot log names the outcome:
|
||||||
|
|
||||||
|
```
|
||||||
|
/system/kernel: clocksource tsc (TSC invariant: yes, synchronized: yes) # real Intel/AMD
|
||||||
|
/system/kernel: clocksource hpet (TSC invariant: no, synchronized: yes) # a bare VM (TCG)
|
||||||
|
```
|
||||||
|
|
||||||
## Two kinds of vector, one dispatch
|
## Two kinds of vector, one dispatch
|
||||||
|
|
||||||
The IDT now installs gates `0-47`: the 32 exceptions plus the device range. Every
|
The IDT now installs gates `0-47`: the 32 exceptions plus the device range. Every
|
||||||
@@ -89,7 +123,6 @@ if (state.vector < 32) {
|
|||||||
on_fault(state); // exception: report and halt (never returns)
|
on_fault(state); // exception: report and halt (never returns)
|
||||||
} else if (handlers[state.vector]) |handler| {
|
} else if (handlers[state.vector]) |handler| {
|
||||||
handler(); // device: run the registered handler
|
handler(); // device: run the registered handler
|
||||||
apic.eoi(); // ...acknowledge the LAPIC
|
|
||||||
}
|
}
|
||||||
// else: spurious/unhandled — deliberately no EOI
|
// else: spurious/unhandled — deliberately no EOI
|
||||||
```
|
```
|
||||||
@@ -100,10 +133,23 @@ Two things make device interrupts *return* where exceptions don't:
|
|||||||
flows back to `isr_common`, which restores every register it saved and executes
|
flows back to `isr_common`, which restores every register it saved and executes
|
||||||
`iretq` — resuming the interrupted instruction exactly. (This is why the stub
|
`iretq` — resuming the interrupted instruction exactly. (This is why the stub
|
||||||
saves *all* the general registers.)
|
saves *all* the general registers.)
|
||||||
2. **End-of-interrupt.** After handling, we write the LAPIC's EOI register. Miss
|
2. **End-of-interrupt.** Somewhere in there we write the LAPIC's EOI register. Miss
|
||||||
this and the LAPIC thinks we're still busy and never delivers the next
|
this and the LAPIC thinks we're still busy and never delivers the next
|
||||||
interrupt. It's the single most common "my timer fired once and stopped" bug.
|
interrupt. It's the single most common "my timer fired once and stopped" bug.
|
||||||
|
|
||||||
|
**Each handler issues its own EOI**, rather than the dispatcher doing it around the
|
||||||
|
call. That looks like a needless devolution while the timer is the only device, and
|
||||||
|
`apic.timerTick` indeed does nothing but `eoi()` before bumping its counter (early,
|
||||||
|
because the tick hook is the scheduler, which may switch tasks and not return
|
||||||
|
promptly — the LAPIC mustn't wait on it).
|
||||||
|
|
||||||
|
It stops looking needless with the second device. A *routed* interrupt — one arriving
|
||||||
|
through the I/O APIC from a real device line — must be **masked before it is
|
||||||
|
acknowledged**, because a level-triggered line is still asserted at EOI time and would
|
||||||
|
redeliver instantly, forever. Only the handler knows which discipline its source
|
||||||
|
needs, so only the handler can sequence it. See [drivers.md](drivers.md), where the
|
||||||
|
device is quieted by a driver in ring 3, long after the ISR has returned.
|
||||||
|
|
||||||
A device handler is a plain `fn () void` — a timer or keyboard handler doesn't need
|
A device handler is a plain `fn () void` — a timer or keyboard handler doesn't need
|
||||||
the interrupted registers. (Note: the stubs don't save the SSE/vector registers, so
|
the interrupted registers. (Note: the stubs don't save the SSE/vector registers, so
|
||||||
a handler must not use them; ours don't.)
|
a handler must not use them; ours don't.)
|
||||||
@@ -131,14 +177,24 @@ If the APIC weren't enabled, or `sti` were missing, or EOI were forgotten, the
|
|||||||
count would stay put and the test would fail. That it advances — while the CPU was
|
count would stay put and the test would fail. That it advances — while the CPU was
|
||||||
spinning in unrelated code — is the whole mechanism working end to end.
|
spinning in unrelated code — is the whole mechanism working end to end.
|
||||||
|
|
||||||
|
## Since (done elsewhere)
|
||||||
|
|
||||||
|
- **Preemption**: the timer handler is where the scheduler decides to switch — the
|
||||||
|
reason a *returning* interrupt matters. See [scheduling.md](scheduling.md).
|
||||||
|
- **`sleep()` / timeouts** built on the calibrated clock.
|
||||||
|
- **The I/O APIC, routed**: external device lines now reach a vector, and the
|
||||||
|
interrupt is delivered onward to a *user-space* driver as an IPC message. See
|
||||||
|
[drivers.md](drivers.md).
|
||||||
|
- **Uncacheable MMIO**: device grants are mapped `PCD|PWT` (strong-uncacheable) for
|
||||||
|
user drivers — see [paging.md](paging.md).
|
||||||
|
|
||||||
## What's next (not done here)
|
## What's next (not done here)
|
||||||
|
|
||||||
- **The keyboard**: bring up the IO-APIC, route its IRQ to a vector, and read
|
- **The keyboard**: the PS/2 controller is port-mapped (`0x60`/`0x64`), and port I/O is
|
||||||
scancodes from the PS/2 controller — the first *input* device.
|
now available to ring 3 via the claim-gated `io_read`/`io_write` syscalls
|
||||||
- **`sleep()` / timeouts** built on the calibrated clock (the monotonic
|
([drivers.md](drivers.md)) — so the first *input* device is unblocked; it just needs
|
||||||
`uptimeMs()` is in place).
|
writing (claim the controller, `irq_bind` GSI 1, read scancodes from `0x60`).
|
||||||
- **Uncacheable MMIO**: the LAPIC page is currently mapped writeback-cacheable like
|
- **MSI-X**: `msi_bind` gives one per-device edge-triggered vector (M15); MSI-X's
|
||||||
the rest of the identity map. QEMU tolerates it, but real hardware wants MMIO
|
multi-vector table (many queues per device, e.g. NVMe) is the remaining extension.
|
||||||
marked uncacheable (via the page's cache bits or an MTRR).
|
- **The LAPIC's own page** is still mapped writeback-cacheable like the rest of the
|
||||||
- **Preemption**: once there are tasks, the timer handler is where the scheduler
|
identity map. QEMU tolerates it; real hardware wants it uncacheable.
|
||||||
decides to switch — the reason a *returning* interrupt matters.
|
|
||||||
|
|||||||
@@ -0,0 +1,172 @@
|
|||||||
|
# The device manager
|
||||||
|
|
||||||
|
**Status: the protocol and supervision are built** (M18.1, 2026-07-13): `hello`
|
||||||
|
with its deadline, supervised spawn, restart with backoff, and the crash-loop
|
||||||
|
cap are in — usb-xhci-bus is the first conforming driver, and the
|
||||||
|
`driver-restart` scenario proves fault → backoff → re-claim → cap end to end.
|
||||||
|
Tree reports are built too (M18.2, 2026-07-13): the xHCI driver scans its
|
||||||
|
root-hub ports and reports each connected device (`child_added`); the manager
|
||||||
|
mirrors them and prunes a dead reporter's children, and the `usb-report`
|
||||||
|
scenario proves report → prune → respawn → re-report. The application surface is built (M18.3, 2026-07-13):
|
||||||
|
`enumerate` and `subscribe` over IPC, with `device-list` as the first client —
|
||||||
|
the manager is now the one answer to "what devices exist" for applications.
|
||||||
|
The primitives underneath are real ([process-management.md](process-management.md):
|
||||||
|
spawn/supervise/kill/exit-notification; [driver-model.md](driver-model.md): the device
|
||||||
|
table as a capability system; [drivers.md](drivers.md): claim/map/IRQ), and the first
|
||||||
|
per-device driver spawn works (the device manager matches the xHCI controller by PCI
|
||||||
|
class and spawns `usb-xhci-bus` with the device id as argv[1]). This document designs
|
||||||
|
the rest: the device manager as **the tree, the matcher, and the supervisor** — the
|
||||||
|
policy process that turns [resilience.md](resilience.md)'s restart goal into practice
|
||||||
|
for drivers.
|
||||||
|
|
||||||
|
How processes stop, reload, and report their deaths is deliberately **not** in this
|
||||||
|
document: that is the universal lifecycle every danos process speaks —
|
||||||
|
[process-lifecycle.md](process-lifecycle.md), signals over IPC and the stable
|
||||||
|
`runtime.process` interface. The device manager is that design's first serious
|
||||||
|
customer, not its owner. Its own protocol contains nothing lifecycle-shaped; a
|
||||||
|
driver is stopped, health-checked, and buried exactly like any other process.
|
||||||
|
|
||||||
|
## The tree: structure in the manager, authority in the kernel
|
||||||
|
|
||||||
|
The device tree is two things fused: *information* (what exists, how it nests) and
|
||||||
|
*authority* (a descriptor is a licence to map physical memory). They separate:
|
||||||
|
|
||||||
|
- The **kernel keeps the capability system** — device, I/O-port, and interrupt
|
||||||
|
claims, resource containment on `device_register`, the
|
||||||
|
`mmio_map`/`irq_bind`/`msi_bind` gates — and **cleans all of it up when a process
|
||||||
|
dies** (settled; it is increment 1 of
|
||||||
|
[process-lifecycle.md](process-lifecycle.md)). The three invariants in
|
||||||
|
[driver-model.md](driver-model.md) stay exactly where they are. A device manager
|
||||||
|
that could mint MMIO mappings by its own say-so would be a second kernel, and a
|
||||||
|
buggy one would un-earn everything the microkernel bought.
|
||||||
|
- The **device manager owns the tree as data** — identity, topology, naming, driver
|
||||||
|
matching, hotplug events, and being the one process everything else asks about
|
||||||
|
devices. Firmware discovery seeds it (today via the kernel's snapshot); **bus
|
||||||
|
drivers grow it** by reporting what they see; applications query and watch it.
|
||||||
|
`device_enumerate` fades to a manager-internal (then deleted) seam.
|
||||||
|
|
||||||
|
Long-term, discovery itself leaves the kernel — but not *into* the manager. PCI
|
||||||
|
enumeration is a **pci-bus driver**: the manager spawns it against the host bridge
|
||||||
|
(already a device with the ECAM window as a resource), it scans, it reports functions
|
||||||
|
like any bus reports children. ACPI becomes an **acpi service** that interprets the
|
||||||
|
tables and reports the namespace. The manager only orchestrates and merges. Moving
|
||||||
|
AML interpretation out of ring 0 is its own project on its own track; nothing here
|
||||||
|
depends on when it lands. (It landed: [discovery.md](discovery.md), M19–M20.)
|
||||||
|
|
||||||
|
`device_register` is **idempotent on exact match**: a re-registration with an
|
||||||
|
identical (parent, class, identity, resources) tuple returns the existing id
|
||||||
|
instead of appending a duplicate. The kernel table has no unregister, so without
|
||||||
|
this a restarted registering bus would re-report its children as fresh nodes on
|
||||||
|
every respawn. Idempotence is what makes restart-and-re-report sound for *every*
|
||||||
|
reporting bus — pci-bus, the acpi service, a future fdt service — not just one,
|
||||||
|
and it is why supervision (below) can prune a dead bus's subtree and trust the
|
||||||
|
restarted instance to rebuild exactly the same ids.
|
||||||
|
|
||||||
|
## The protocol
|
||||||
|
|
||||||
|
A `device-manager-protocol` module (the vfs-protocol pattern): extern-struct
|
||||||
|
messages, a version in the handshake, reserved fields everywhere. The manager is a
|
||||||
|
well-known endpoint (`ipc.register(.device_manager)`); the badge tells it who is
|
||||||
|
talking; the same endpoint receives its children's exit notifications — one loop,
|
||||||
|
one world.
|
||||||
|
|
||||||
|
| Direction | Message | Purpose |
|
||||||
|
|---|---|---|
|
||||||
|
| driver → manager | `hello { version, role, device_id }` | confirms the argv assignment, starts the deadline clock |
|
||||||
|
| bus → manager | `child_added { parent, identity, resources }` | one node the bus discovered |
|
||||||
|
| bus → manager | `child_removed { id }` | unplug, or the bus lost it |
|
||||||
|
| app → manager | `enumerate` | snapshot of the tree (read-only) |
|
||||||
|
| app → manager | `subscribe` | receive published add/remove events |
|
||||||
|
|
||||||
|
`hello` is the one deadline the manager enforces itself: spawned and silent past the
|
||||||
|
deadline means wrong binary, wrong protocol version, or wedged before main — apply
|
||||||
|
the stop sequence and the restart policy. Everything else lifecycle-shaped
|
||||||
|
(terminate, the common `ping` liveness call, exit reasons) arrives through
|
||||||
|
[process-lifecycle.md](process-lifecycle.md)'s vocabulary, not this protocol.
|
||||||
|
|
||||||
|
Assignment stays argv (`usb-xhci-bus <device id>`) for now — simple, and it works.
|
||||||
|
The step after `hello` exists is delegation: the manager claims (or is granted) the
|
||||||
|
devices and passes the claim to the driver over IPC (the M13 capability-transfer
|
||||||
|
mechanism), replacing first-come-first-served `device_claim` with policy. Identity in
|
||||||
|
`child_added` is per-bus: PCI children carry the class triple (`pci_class`, as the
|
||||||
|
xHCI match already uses); USB children carry the (class, subclass, protocol) triple
|
||||||
|
from usb-ids.zig — each bus's native language, decoded by the shared ids modules.
|
||||||
|
|
||||||
|
## Supervision and restart
|
||||||
|
|
||||||
|
Every driver is spawned with the manager's exit endpoint (`spawnSupervised` — built).
|
||||||
|
On a death notification:
|
||||||
|
|
||||||
|
1. **Read the reason** ([process-lifecycle.md](process-lifecycle.md) increment 2).
|
||||||
|
Clean exit → it meant to; don't restart. Fault or missed `hello` deadline →
|
||||||
|
restart with **backoff**, and a crash-loop cap (three fast deaths → mark failed,
|
||||||
|
stop respawning, log loudly; a later `reload` to the manager can retry).
|
||||||
|
2. **Prune the subtree** the dead bus driver reported. Its children describe
|
||||||
|
protocol state (xHCI slot ids, transfer rings) that died with the process;
|
||||||
|
keeping the nodes would be keeping a lie. Watchers receive `child_removed` — the
|
||||||
|
input service losing, then regaining, a keyboard is the *honest* description of
|
||||||
|
what happened. The restarted instance rediscovers and re-reports.
|
||||||
|
3. **The claim is already free** because the kernel released it at death — the
|
||||||
|
restarted instance claims the same controller and comes up.
|
||||||
|
|
||||||
|
Who supervises the supervisor: **init** (PID 1), which already supervises the
|
||||||
|
services it starts. If the manager dies, drivers keep running (they hold their
|
||||||
|
claims; the kernel doesn't care who their supervisor was — though their exit
|
||||||
|
notifications now dangle harmlessly). The restarted manager re-learns the world:
|
||||||
|
kernel snapshot, then a re-`hello` round — drivers answer a broadcast or are stopped
|
||||||
|
and respawned. Full state handoff is deliberately not attempted.
|
||||||
|
|
||||||
|
## Thin drivers, class protocols
|
||||||
|
|
||||||
|
The [driver-model.md](driver-model.md) three-shape split, restated as processes:
|
||||||
|
|
||||||
|
- A **bus driver** (usb-xhci-bus) owns its controller — claim, MMIO, IRQ/MSI, DMA
|
||||||
|
rings — and offers a *transfer* protocol ("submit a control transfer to device N",
|
||||||
|
built from the usb-abi request constructors) plus tree reports to the manager.
|
||||||
|
- A **class driver** (usb-hid, usb-storage) owns nothing: it is matched to a reported
|
||||||
|
child by its identity triple, speaks the bus's transfer protocol downward and its
|
||||||
|
service's protocol upward — HID reports to the input service, blocks to the block
|
||||||
|
service. It works unchanged over any controller.
|
||||||
|
- **Services** (input, display, block) aggregate class drivers and face applications.
|
||||||
|
|
||||||
|
Each arrow is a protocol module. The manager routes none of the data plane — it
|
||||||
|
introduces the parties (matching), supervises them (lifecycle), and gets out of the
|
||||||
|
way.
|
||||||
|
|
||||||
|
## Increments
|
||||||
|
|
||||||
|
Increments 1–4 are the lifecycle prerequisites and live in
|
||||||
|
[process-lifecycle.md](process-lifecycle.md) (claim cleanup on death, exit reasons,
|
||||||
|
published exit events, signals + `runtime.process`). On top of those:
|
||||||
|
|
||||||
|
5. **device-manager-protocol**: `hello`, supervised spawn with restart policy;
|
||||||
|
usb-xhci-bus becomes the first conforming driver.
|
||||||
|
6. **Tree reports**: `child_added`/`child_removed`; the manager mirrors; xHCI reports
|
||||||
|
the mouse and keyboard QEMU already hangs off it.
|
||||||
|
7. **App surface**: `enumerate`/`subscribe` over IPC; `device_enumerate` retreats
|
||||||
|
to a manager-internal seam.
|
||||||
|
8. **Discovery migration** — DONE (M19–M20, 2026-07-13): enumeration moved to
|
||||||
|
ring 3 as swappable per-firmware discoverers — the pci-bus driver (M19) then
|
||||||
|
the acpi service (M20), see [discovery.md](discovery.md); the kernel seeds
|
||||||
|
only the host bridge and the acpi-tables node. Matching moved with it:
|
||||||
|
`child_added` grew a `device_id` (the kernel-registered id, `no_device` for
|
||||||
|
unregistered leaves like USB ports) and a firmware `hid`, and the manager now
|
||||||
|
matches drivers from those **reports** rather than its boot-time snapshot. The
|
||||||
|
PCI arm flipped in M19.3, the ACPI arm (ps2-bus matched from `_HID`) in M20.3
|
||||||
|
— each in a single phase so no device is ever matched from both sources at
|
||||||
|
once. The acpi service reports only the non-PCI `_HID` devices, since pci-bus
|
||||||
|
already reports PCI functions (M20.2).
|
||||||
|
|
||||||
|
## Settled questions (2026-07-12)
|
||||||
|
|
||||||
|
- **Stateful buses**: pruning the subtree on bus-driver death is right for USB. A
|
||||||
|
future storage bus with in-flight writes wants drain-before-terminate — which is
|
||||||
|
exactly the `deadline_ms` parameter `stop()` already has; a per-driver deadline
|
||||||
|
is one value in the manager's policy table when such a bus arrives. No design
|
||||||
|
change.
|
||||||
|
- **Manager death**: drivers survive the manager; the restarted manager re-learns
|
||||||
|
the world (above). Checkpointing driver state with the manager is deferred until
|
||||||
|
something demonstrates the need.
|
||||||
|
- **Matching stays code until the third bus.** `driverFor`/`pciDriverFor` are
|
||||||
|
honest at two bus types; the third triggers the manifest (a driver declares what
|
||||||
|
it binds: a PCI class triple, a USB class triple, an ACPI `_HID`).
|
||||||
@@ -167,3 +167,81 @@ free; discovery on x86 is partly about *finding* what ARM just tells you.
|
|||||||
- [ipc.md](ipc.md) — the channels that interrupts-as-messages and the device manager
|
- [ipc.md](ipc.md) — the channels that interrupts-as-messages and the device manager
|
||||||
will ride on.
|
will ride on.
|
||||||
- [vision.md](vision.md) — why drivers belong in isolated user space at all.
|
- [vision.md](vision.md) — why drivers belong in isolated user space at all.
|
||||||
|
|
||||||
|
## Update (M19.3, 2026-07-13): PCI enumeration left the kernel
|
||||||
|
|
||||||
|
The kernel now seeds only the `pci_host_bridge` node (ECAM window, MMIO
|
||||||
|
apertures derived from the memory map's holes, bus range, and the 16-bit I/O
|
||||||
|
window). The per-function walk moved to the ring-3 `pci-bus` driver
|
||||||
|
([device-manager.md](device-manager.md)): it claims the bridge, repeats the
|
||||||
|
ECAM scan through its mmio grant, and `device_register`s what it finds, which
|
||||||
|
the device manager mirrors and matches. The ACPI namespace walk follows in M20;
|
||||||
|
the static tables (MADT, HPET, MCFG, FADT + `\\_S5`) stay kernel-side.
|
||||||
|
|
||||||
|
## Update (M20.3, 2026-07-13): ACPI enumeration left the kernel too
|
||||||
|
|
||||||
|
The kernel no longer folds the AML namespace's Device objects into the device
|
||||||
|
tree. It still parses the *static* tables (MADT for SMP, HPET for the tick, MCFG
|
||||||
|
for the host bridge, FADT) and still builds the AML namespace — but only to read
|
||||||
|
the `\\_S5` sleep type for poweroff. Device discovery is the ring-3 **acpi
|
||||||
|
service** ([device-manager.md](device-manager.md)): it claims the `acpi-tables`
|
||||||
|
node the kernel publishes (the AML blobs, a broad io_port grant, the SCI),
|
||||||
|
re-parses the same blobs with the shared AML module, evaluates `_STA`/`_CRS`,
|
||||||
|
and registers + reports each `_HID` device — the device manager matches drivers
|
||||||
|
(ps2-bus) from those reports. With M19's pci-bus driver, discovery now runs
|
||||||
|
entirely in user space; the kernel seeds only the host bridge and the
|
||||||
|
acpi-tables node.
|
||||||
|
|
||||||
|
## Discovery is a swappable process per firmware (M19–M20)
|
||||||
|
|
||||||
|
Moving PCI and ACPI enumeration out of ring 0 was not just a relocation — it
|
||||||
|
made discovery **firmware-neutral by construction**, which is the whole reason
|
||||||
|
to do it before the second architecture rather than after. Everything at and
|
||||||
|
above the [device-manager](device-manager.md) protocol — descriptors,
|
||||||
|
containment, reports, matching, supervision — is generic and may never become
|
||||||
|
x86-specific. Discovery is the single firmware-specific piece, and it is
|
||||||
|
isolated as **one swappable process per firmware**:
|
||||||
|
|
||||||
|
- **x86** boots describe hardware with ACPI, so the discoverer is the **acpi
|
||||||
|
service** ([acpi.md](acpi.md)): it claims the `acpi-tables` node and runs AML.
|
||||||
|
- **The Raspberry Pis** hand over a flattened device tree, so the discoverer is
|
||||||
|
an **fdt service**: it claims a `devicetree-blob` node and walks the tree —
|
||||||
|
pure data, no bytecode, so it needs neither a port grant nor an interpreter,
|
||||||
|
strictly simpler than ACPI. (A placeholder until the [aarch64](arm.md)
|
||||||
|
bring-up fills it in.)
|
||||||
|
|
||||||
|
The device manager spawns the discoverer under the **neutral ramdisk name
|
||||||
|
`discovery`** and never learns which firmware it is on; the build's
|
||||||
|
`-Ddiscovery=acpi|fdt` option fills that slot (x86 defaults to `acpi`, the
|
||||||
|
aarch64 target flips the default when it lands). The manager owns the device
|
||||||
|
tree as *data* and touches no hardware, ever — firmware bytecode runs only
|
||||||
|
inside the crashable, supervised discoverer, so an AML fault can never take
|
||||||
|
down the supervisor.
|
||||||
|
|
||||||
|
Two consequences of neutrality bind on later work:
|
||||||
|
|
||||||
|
- **Cross-firmware surfaces are named by domain, not firmware.** System power is
|
||||||
|
a [`power`](power.md) protocol, not an "ACPI events" protocol: on x86 the acpi
|
||||||
|
service registers it, on ARM a PSCI/mailbox service registers the same
|
||||||
|
`ServiceId.power`, and subscribers never learn the difference.
|
||||||
|
- **Identity must widen before the fdt service exists.** `DeviceDescriptor`'s
|
||||||
|
8-byte `hid` holds an EISA id but cannot hold an FDT `compatible` string
|
||||||
|
(`"brcm,bcm2835-aux-uart"`); the identity field grows before the ARM path can
|
||||||
|
report a real node.
|
||||||
|
|
||||||
|
Two supporting decisions keep the kernel's remaining slice honest:
|
||||||
|
|
||||||
|
- **The AML interpreter is a shared build module**, compiled into both the
|
||||||
|
kernel and the acpi service — one source, two builds, no fork. The kernel
|
||||||
|
links it for the `\_S5` poweroff evaluation, the service links it for
|
||||||
|
everything else, and the `acpi-parse` test asserts the two produce the same
|
||||||
|
device count across the ring-3 move.
|
||||||
|
- **Bridge apertures come from the firmware memory map, not AML.** Registered
|
||||||
|
PCI functions carry BAR resources, and `device_register` containment demands
|
||||||
|
the bridge own windows that cover them. Those apertures are derived
|
||||||
|
kernel-side from the boot memory map's MMIO holes (regions that are neither
|
||||||
|
RAM nor tables) — mechanical, AML-free, and available at boot regardless of
|
||||||
|
what later moved to user space. The acpi service's authority is likewise
|
||||||
|
exactly one node: the `acpi-tables` node, whose broad io_port grant is the
|
||||||
|
documented trust boundary for the one process allowed to run firmware
|
||||||
|
bytecode.
|
||||||
|
|||||||
@@ -0,0 +1,368 @@
|
|||||||
|
# The driver model: buses, classes, and host controllers
|
||||||
|
|
||||||
|
[drivers.md](drivers.md) shows how to write *a* driver — claim a device, map its
|
||||||
|
registers, sleep on its interrupt. That's enough for a leaf device like the HPET. It is
|
||||||
|
not enough for a disk, a keyboard, or a network card, because those hang off a
|
||||||
|
*controller*, on a *bus*, speaking a *protocol*, and no single process should have to
|
||||||
|
know all three.
|
||||||
|
|
||||||
|
Real driver stacks factor into three shapes. This document is about what each one is,
|
||||||
|
what the kernel must give it, how they share code — and precisely which primitive each
|
||||||
|
is still blocked on.
|
||||||
|
|
||||||
|
## Three shapes
|
||||||
|
|
||||||
|
| Shape | Owns | Reaches hardware by | Talks to |
|
||||||
|
|---|---|---|---|
|
||||||
|
| **Host controller driver** (HCD) | a controller — an xHCI PCI function, an AHCI port block | `mmio_map` + `irq_bind` + DMA | the devices behind it, in its bus's language |
|
||||||
|
| **Bus driver** | a bus — a PCI bridge, a USB hub | `device_register`, to publish what it finds | class drivers, over IPC |
|
||||||
|
| **Class / protocol driver** | *nothing* | *nothing* | its bus driver, over IPC |
|
||||||
|
|
||||||
|
The last row is the surprising one and the whole point. A USB keyboard driver touches
|
||||||
|
no registers, takes no interrupts, and maps no memory. It sends HID protocol messages
|
||||||
|
to whatever published the device, and it works identically whether the controller
|
||||||
|
below is xHCI, EHCI, or a Raspberry Pi's DWC2. That is what buys you drivers that
|
||||||
|
outlive the hardware they were written for.
|
||||||
|
|
||||||
|
In practice **HCD and bus driver are usually the same process**. An xHCI driver is a
|
||||||
|
host controller driver (it owns the PCI function, its BARs, its interrupt, its DMA
|
||||||
|
rings) *and* a bus driver (it enumerates USB devices and publishes them). Splitting
|
||||||
|
them is a fiction; what matters is that both *roles* have kernel support, because a
|
||||||
|
plain bus driver with no controller — a USB hub — is also a real thing.
|
||||||
|
|
||||||
|
## The device table is the spine
|
||||||
|
|
||||||
|
danos already has the right central structure. `system/kernel/devices-broker.zig` holds a table of
|
||||||
|
`DeviceDesc`, each with a parent, a class, and a set of resources. Firmware discovery
|
||||||
|
seeds it ([discovery.md](discovery.md)); `device_register` grows it.
|
||||||
|
|
||||||
|
Three invariants make it a capability system rather than a directory:
|
||||||
|
|
||||||
|
1. **A claim is exclusive.** `device_claim(id)` succeeds once. Everything downstream —
|
||||||
|
`mmio_map`, `irq_bind`, `device_register` — checks `devices_broker.ownerOf(id) == me`.
|
||||||
|
2. **A descriptor is a licence to map physical memory.** Whoever claims a device may
|
||||||
|
map its `.memory` resources and bind its `.irq` resources. This is why
|
||||||
|
`device_register` cannot be a free-for-all.
|
||||||
|
3. **Therefore: containment.** Every resource of a registered child must lie inside a
|
||||||
|
resource of the same kind on its parent (`devices_broker.contains`). A bus driver can only
|
||||||
|
ever *subdivide* what it already holds. Without this, `device_register` would be a
|
||||||
|
syscall named "map any physical page you like."
|
||||||
|
|
||||||
|
Containment is transitive by construction: a grandchild is contained in its child,
|
||||||
|
which is contained in the bus. Nothing can be laundered through a chain.
|
||||||
|
|
||||||
|
Note that firmware topology does **not** obey containment, and isn't asked to — a PCI
|
||||||
|
function's BAR is not inside its host bridge's `bus_range`, because a bus-number range
|
||||||
|
is not an address window. Discovery is trusted; user space is not.
|
||||||
|
|
||||||
|
### What a bus driver looks like
|
||||||
|
|
||||||
|
danos ships no demo bus driver — the real ones are `pci-bus`, `ps2-bus`, and
|
||||||
|
`usb-xhci-bus`. The smallest *honest* shape, illustrated here with an HPET register block
|
||||||
|
as the "bus" and its comparators as the "devices", is:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
_ = dev.claim(bus.id); // 1. own the bus
|
||||||
|
const base = dev.mmioMap(bus.id, 0).?; // 2. enumerate it — from the hardware
|
||||||
|
const n = ((cap.* >> 8) & 0x1F) + 1; // GENERAL_CAP says how many children
|
||||||
|
|
||||||
|
for (0..n) |i| { // 3. publish each child
|
||||||
|
var child = std.mem.zeroes(dev.DeviceDesc);
|
||||||
|
child.class = @intFromEnum(dev.DeviceClass.timer);
|
||||||
|
child.resource_count = 1;
|
||||||
|
child.resources[0] = .{ .kind = memory,
|
||||||
|
.start = bus_mmio.start + 0x100 + 0x20 * i,
|
||||||
|
.len = 0x20 };
|
||||||
|
_ = dev.register(bus.id, &child).?; // kernel checks containment
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Each child is left **unclaimed**, which is the handoff: a comparator driver can now
|
||||||
|
`device_claim` one and `mmio_map` it, and will see only its own 0x20-byte window. A child
|
||||||
|
whose window escapes the bus is refused; the in-kernel `containment` test asserts the
|
||||||
|
kernel's table upholds that ([drivers.md](drivers.md)).
|
||||||
|
|
||||||
|
A USB device has *no* resources at all: `resource_count = 0`, because it's addressed
|
||||||
|
through its controller, not by MMIO. That case is allowed and is the common one.
|
||||||
|
|
||||||
|
## Families: sharing code between drivers
|
||||||
|
|
||||||
|
A "family" is two modules, not one:
|
||||||
|
|
||||||
|
- **A logic module** — the parts of the bus that every driver on it re-derives. Config
|
||||||
|
space walking and BAR decode for PCI. Descriptor parsing, control transfers, and hub
|
||||||
|
protocol for USB.
|
||||||
|
- **A protocol module** — the IPC message types that let a class driver talk to
|
||||||
|
*whatever* published its device. This is the part that makes class drivers portable.
|
||||||
|
|
||||||
|
danos already has one of each: `library/runtime/device.zig` is a logic module,
|
||||||
|
[`system/services/vfs/protocol.zig`](system/services/vfs/protocol.zig) is a protocol module shared by `system/services/vfs/vfs.zig`
|
||||||
|
and its clients. The pattern generalises directly:
|
||||||
|
|
||||||
|
```
|
||||||
|
library/
|
||||||
|
runtime/ module "runtime" — syscalls, heap, ipc, device, stdio
|
||||||
|
mmio/ module "mmio" — volatile register access + barriers [M14]
|
||||||
|
bus/
|
||||||
|
pci/ module "pci" — ECAM, BAR decode, capability walk
|
||||||
|
usb/ module "usb" — descriptors, control transfers, hubs
|
||||||
|
proto/
|
||||||
|
vfs/ module "vfs-protocol" (today: system/services/vfs/protocol.zig)
|
||||||
|
block/ module "block-protocol"
|
||||||
|
hid/ module "hid-protocol"
|
||||||
|
|
||||||
|
system/drivers/ one sub-project each → /system/drivers (no `d` suffix)
|
||||||
|
xhci/ HCD + bus driver imports runtime, pci, usb, mmio
|
||||||
|
usb-hid/ class driver imports runtime, usb, hid-protocol
|
||||||
|
block/ class driver imports runtime, block-protocol
|
||||||
|
```
|
||||||
|
|
||||||
|
The only build change needed: [`addUserBinary`](build.zig) currently takes exactly one
|
||||||
|
module (`rt_mod`) and injects it. It should take a slice of modules. That's a
|
||||||
|
five-line change, and it's the *entire* mechanism — Zig modules already give you
|
||||||
|
everything else.
|
||||||
|
|
||||||
|
The discipline that makes this work: **a class driver must not import a bus's logic
|
||||||
|
module.** `usbhid` imports `proto.hid` and `usb` (for descriptor types), never `pci`.
|
||||||
|
If a class driver needs `mmio`, it has become an HCD and should be one.
|
||||||
|
|
||||||
|
## What exists today
|
||||||
|
|
||||||
|
- **M10** — `device_enumerate`, `device_claim`, `mmio_map`. Strong-uncacheable device
|
||||||
|
grants, `device_grant` teardown.
|
||||||
|
- **M11** — `irq_bind` / `irq_ack`. IRQ delivered as an IPC notification; mask before
|
||||||
|
EOI; `irq_ack` is the unmask.
|
||||||
|
- **M12** — `parent` in `DeviceDesc`, `device_register` with resource containment.
|
||||||
|
- **M13** — capability passing. `ipc_call` / `ipc_reply_wait` grew a `send_cap` argument
|
||||||
|
and a `received_cap` return (r8): an endpoint travels with a message, installed into
|
||||||
|
the receiver's handle table (shared, refcount-bumped — a copy, not a move). A full
|
||||||
|
table fails `-ENOSPC` and does not half-deliver. This is the "open" primitive — a bus
|
||||||
|
driver mints a per-device endpoint and hands it to a class driver. The runtime exposes
|
||||||
|
`callCap` and `replyWait(..., send_cap)`; no class driver consumes it yet.
|
||||||
|
- **M14** — DMA memory + the memory-ordering layer. `/lib/mmio` gives drivers typed
|
||||||
|
volatile access and `mb`/`rmb`/`wmb` (per-arch); `dma_alloc`/`dma_free` grant
|
||||||
|
physically-contiguous, pinned, uncacheable, reclaim-on-teardown buffers with the
|
||||||
|
physical address exposed (`pmm.allocContiguous`, a DMA arena, `mapUserDmaInto`).
|
||||||
|
`dma_below_4g` caps the address for legacy engines; `dma_write_combining` is accepted
|
||||||
|
but falls back to coherent until PAT is programmed. The bus drivers use `/lib/mmio`;
|
||||||
|
no DMA driver consumes `dma_alloc` yet.
|
||||||
|
- **M15** — interrupts for PCI devices, the MSI half. Discovery now gives every PCI
|
||||||
|
function its 4 KiB ECAM config space as resource 0 (unblocking the capability walk
|
||||||
|
with no new syscall), and `msi_bind(device_id, endpoint) -> address, data` allocates a
|
||||||
|
per-device edge-triggered vector, delivered as an IPC notification with no mask and no
|
||||||
|
ack cycle. Legacy INTx (`_PRT` parsing + shared lines) is deliberately skipped — MSI
|
||||||
|
is the real answer. QEMU's HPET has no MSI, so delivery is proven with a self-IPI; the
|
||||||
|
first PCI driver is the first real consumer.
|
||||||
|
- **Port I/O** — `io_read`/`io_write(device_id, resource_index, offset, width[, value])`:
|
||||||
|
a claimed device's `io_port` resource lets a driver read/write its ports, gated exactly
|
||||||
|
like `mmio_map` gates memory (direct ring-3 `in`/`out` stays a #GP). This is what makes
|
||||||
|
a PS/2 or 16550 driver possible; the low-rate legacy hardware that needs it is fine with
|
||||||
|
a syscall per access. `io_port` resources were recorded by discovery and ignored — now
|
||||||
|
they're used.
|
||||||
|
- **M16 (detection)** — the IOMMU is now *found*: discovery parses the ACPI DMAR table,
|
||||||
|
maps the first VT-d unit, and reads its version + capabilities (`iommu_present` in the
|
||||||
|
platform info). This is detection only — **no translation domains are programmed, so
|
||||||
|
DMA is still unprotected** (the caveat below). Enforcement lands with the first DMA
|
||||||
|
driver, which is what there is to protect and test against. Proven in the `iommu` test,
|
||||||
|
booted with an emulated `intel-iommu`.
|
||||||
|
- **`system_spawn`** — a user-space supervisor starts a driver:
|
||||||
|
`system_spawn(name, arguments)` loads a binary bundled in the initial-ramdisk as a
|
||||||
|
fresh ring-3 process; `name` becomes the child's argv[0] and the optional
|
||||||
|
NUL-separated `arguments` blob its argv[1..], delivered on a SysV entry stack
|
||||||
|
([sysv.md](sysv.md)). This is what
|
||||||
|
turned the device manager from "log the match" into "run the driver": the kernel now
|
||||||
|
spawns only `init`, `init` spawns the services, and the **device-manager** discovers
|
||||||
|
the hardware and spawns each driver ([drivers.md](drivers.md)). Ungated for now — a
|
||||||
|
spawn capability is future work.
|
||||||
|
|
||||||
|
So: **bus drivers work now, and they're started by the device manager, not the kernel.**
|
||||||
|
HCDs and class drivers do not work yet. Here is exactly why, and exactly what would fix it.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
# Proposed ABI
|
||||||
|
|
||||||
|
## M13 — capability passing, for class drivers ✅ done
|
||||||
|
|
||||||
|
*Implemented as described below (see "What exists today"). The signatures landed
|
||||||
|
verbatim: `send_cap` in r9, `received_cap` returned in r8, `-ENOSPC` on a full receiver
|
||||||
|
table with no delivery. The rest of this section is the original design note.*
|
||||||
|
|
||||||
|
**The blocker.** A class driver has to reach *its* device. Today the only way to find
|
||||||
|
an endpoint is the name registry: `ipc_register(service_id, h)` / `ipc_lookup(id)`,
|
||||||
|
where `ServiceId` is a global integer namespace with `max_services = 8`. You cannot
|
||||||
|
mint one endpoint per USB device that way, and there is no way for a bus driver to
|
||||||
|
*hand* a class driver an endpoint. M7 deferred this deliberately.
|
||||||
|
|
||||||
|
**The fix.** Let a message carry one handle. Sender names a handle in its own table;
|
||||||
|
the kernel installs the endpoint into the receiver's table (bumping `refcount`) and
|
||||||
|
tells the receiver the index it landed at.
|
||||||
|
|
||||||
|
```
|
||||||
|
ipc_call(h, msg, message_len, reply, reply_cap, send_cap) -> reply_len
|
||||||
|
ipc_reply_wait(h, reply, reply_len, recv, recv_cap, send_cap)
|
||||||
|
-> recv_len (rax), badge (rdx), received_cap (r8)
|
||||||
|
```
|
||||||
|
|
||||||
|
`send_cap` is a handle or `no_cap` (`~0`). `received_cap` is the index the transferred
|
||||||
|
endpoint was installed at in the receiver's table, or `no_cap`.
|
||||||
|
|
||||||
|
- Both calls grow from 5 args to 6, which fits: `syscall5` uses `rdi/rsi/rdx/r10/r8`,
|
||||||
|
leaving `r9`. `ipc_reply_wait` already returns two values via `setSyscallResult2`;
|
||||||
|
this needs a third (`setSyscallResult3`).
|
||||||
|
- If the receiver's handle table is full, the call fails `-ENOSPC` and **the message is
|
||||||
|
not delivered** — a half-delivered capability is worse than a failed send.
|
||||||
|
- `closeHandles` already drops references on exit, so the lifetime story is unchanged.
|
||||||
|
|
||||||
|
That single primitive gives you the standard `open` pattern:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
// class driver // bus driver
|
||||||
|
const h = ipc.lookup(.usb).?; const r = ipc.replyWait(ep, ...);
|
||||||
|
const dev_ep = ipc.callCap(h, // ... mint a per-device endpoint,
|
||||||
|
.{ .op = .open, .id = dev_id }); // reply with it as send_cap
|
||||||
|
// now dev_ep is a private channel to that one device
|
||||||
|
```
|
||||||
|
|
||||||
|
## M14 — DMA memory and the memory-ordering contract, for HCDs ✅ done
|
||||||
|
|
||||||
|
*Implemented: `/lib/mmio` (typed volatile access + `mb`/`rmb`/`wmb`, per-arch) and
|
||||||
|
`dma_alloc`/`dma_free` (contiguous, pinned, uncacheable, reclaim-on-teardown, physical
|
||||||
|
address exposed). `dma_write_combining` still falls back to coherent — real WC needs
|
||||||
|
PAT, a small follow-up. The rest of this section is the original design note.*
|
||||||
|
|
||||||
|
**The blocker.** An HCD is a DMA-engine programmer. It needs a descriptor ring the
|
||||||
|
device can read, which means memory that is (a) physically contiguous, (b) at a
|
||||||
|
physical address the driver knows, (c) of the right cacheability, and (d) pinned.
|
||||||
|
[`sysMmap`](system/kernel/process.zig) gives you *none* of the four: it calls `pmm.alloc()`
|
||||||
|
once per page, maps writeback-cached, and never reveals a physical address.
|
||||||
|
|
||||||
|
**The fix.**
|
||||||
|
|
||||||
|
```
|
||||||
|
dma_alloc(len, flags) -> vaddr (rax), paddr (rdx)
|
||||||
|
dma_free(vaddr, len) -> 0
|
||||||
|
|
||||||
|
flags: dma_coherent (1) uncacheable; the default and the only one that's portable
|
||||||
|
dma_wc (2) write-combining — needs PAT programmed; for framebuffers
|
||||||
|
dma_below_4g (4) for devices with 32-bit DMA addressing
|
||||||
|
```
|
||||||
|
|
||||||
|
Guarantees: page-aligned, physically contiguous, zeroed, pinned for the life of the
|
||||||
|
mapping, and the physical address is stable. It needs one thing the kernel lacks —
|
||||||
|
`pmm.allocContiguous(n, max_phys)`; today `pmm.alloc()` hands out one frame at a time
|
||||||
|
with no adjacency guarantee.
|
||||||
|
|
||||||
|
**The memory-ordering contract.** danos has, at the time of writing, **zero memory
|
||||||
|
barriers anywhere in the tree.** That is currently correct-by-accident and won't
|
||||||
|
survive the first DMA driver, or the first ARM boot.
|
||||||
|
|
||||||
|
`volatile` is not a barrier. In Zig it means: don't elide this access, and don't
|
||||||
|
reorder it against *other volatile* accesses. It says nothing about your *ordinary*
|
||||||
|
stores — the descriptor you just filled in normal WB memory — which LLVM may freely
|
||||||
|
sink past a volatile MMIO write. The canonical bug:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
ring[i] = descriptor; // ordinary store to WB RAM
|
||||||
|
doorbell.* = i; // volatile store to UC MMIO
|
||||||
|
// nothing stops the compiler reordering these; the device reads a stale descriptor
|
||||||
|
```
|
||||||
|
|
||||||
|
So the rules, which belong in `library/mmio.zig` and behind `arch`:
|
||||||
|
|
||||||
|
| Situation | Required |
|
||||||
|
|---|---|
|
||||||
|
| MMIO register read/write | `mmio.read` / `mmio.write` (volatile) |
|
||||||
|
| Fill DMA descriptor, then ring doorbell | `wmb()` between them |
|
||||||
|
| Woken by IRQ, then read what the device wrote | `rmb()` before the read |
|
||||||
|
| MMIO write that must complete before the next read | `mb()` |
|
||||||
|
|
||||||
|
And the per-arch lowering — the reason this must be an `arch` primitive and not a
|
||||||
|
sprinkling of `asm volatile`:
|
||||||
|
|
||||||
|
| | x86_64 | aarch64 |
|
||||||
|
|---|---|---|
|
||||||
|
| `mb()` | `mfence` | `dsb sy` |
|
||||||
|
| `rmb()` | `lfence` | `dsb ld` |
|
||||||
|
| `wmb()` | `sfence` | `dsb st` |
|
||||||
|
| DMA cache coherency | coherent; nothing to do | **not guaranteed**; needs non-cacheable buffers or cache maintenance |
|
||||||
|
|
||||||
|
x86 is forgiving here — TSO plus strong-uncacheable MMIO means you usually get away
|
||||||
|
with a compiler barrier alone. ARM is not, and [vision.md](vision.md) makes ARM the win
|
||||||
|
condition. Build the abstraction while there is one caller to fix.
|
||||||
|
|
||||||
|
(Zig note: `@fence` was **removed in 0.16**. Use `@atomicRmw(..., .seq_cst)` for a full
|
||||||
|
barrier, or per-arch inline asm — which is what `library/mmio.zig` should hide.)
|
||||||
|
|
||||||
|
## M15 — interrupts for PCI devices ✅ done (MSI)
|
||||||
|
|
||||||
|
*Implemented the MSI half: ECAM config space per PCI function (resource 0) and
|
||||||
|
`msi_bind` (per-device edge-triggered vector, delivered as a notification). Legacy INTx
|
||||||
|
`_PRT` parsing is skipped on purpose. `msi_bind` returns (address, data) as two values
|
||||||
|
rather than an out-struct. The rest of this section is the original design note.*
|
||||||
|
|
||||||
|
**The blocker, and it's a hard one.** No PCI device can take an interrupt today.
|
||||||
|
[`addBars`](system/devices/acpi.zig) records `.memory` and `.io_port` BARs and never an
|
||||||
|
`.irq`; there is no `_PRT` parsing anywhere in the tree. The HPET is the one exception —
|
||||||
|
it advertises its own interrupt routing in its own registers, a privilege no ordinary
|
||||||
|
device has.
|
||||||
|
|
||||||
|
**The fix, in two halves.**
|
||||||
|
|
||||||
|
*Legacy INTx*: parse `_PRT` from the DSDT to map (device, INTA–D) → GSI, and record it
|
||||||
|
as an `.irq` resource. Then `irq_bind` works unchanged. But INTx lines are **shared**,
|
||||||
|
and `irq.bound[gsi]` holds one endpoint. Sharing needs a list, and every driver on the
|
||||||
|
line must be polled on each interrupt — the reason everyone left INTx behind.
|
||||||
|
|
||||||
|
*MSI/MSI-X*, which is the real answer: per-device vectors, edge-triggered, unshared, no
|
||||||
|
mask/ack cycle, no 24-GSI ceiling. The kernel allocates a vector and hands the driver
|
||||||
|
the (address, data) pair to program into its own MSI capability:
|
||||||
|
|
||||||
|
```
|
||||||
|
msi_bind(dev_id, endpoint, out) -> 0 // out: extern struct { addr: u64, data: u32 }
|
||||||
|
```
|
||||||
|
|
||||||
|
The driver writes those into config space itself — which means it needs config space,
|
||||||
|
which means **discovery should give each `pci_device` a `.memory` resource for its
|
||||||
|
4 KiB ECAM slot**. That's a small change to `parseMcfg` and it unblocks the whole
|
||||||
|
capability walk (MSI, MSI-X, PCIe extended caps) without any new syscall.
|
||||||
|
|
||||||
|
Note QEMU's HPET reports `Tn_FSB_INT_DEL_CAP = 0` — no MSI — so an HPET timer could never
|
||||||
|
exercise this path. The first MSI driver will be the first PCI driver.
|
||||||
|
|
||||||
|
## M16 — the IOMMU, and the honest caveat ◑ detection done, enforcement pending
|
||||||
|
|
||||||
|
*The IOMMU is now detected (DMAR parsed, VT-d unit mapped and read — see the `iommu`
|
||||||
|
test), but **enforcement is not built**: no translation domains are programmed, so the
|
||||||
|
caveat below still holds in full. Detection can't be taken further usefully until there
|
||||||
|
is a DMA driver to protect and QEMU's `intel-iommu` to test the protection against —
|
||||||
|
building the per-device domains alongside that first driver is both the natural order
|
||||||
|
and the only way to verify them. The rest of this section is the original caveat.*
|
||||||
|
|
||||||
|
Everything above is capability-gated at the *CPU*. None of it is gated at the *device*.
|
||||||
|
A driver that can program a bus-mastering engine can make that device write to any
|
||||||
|
physical address, because page tables sit between the CPU and RAM, not between a device
|
||||||
|
and RAM. Until VT-d/DMAR (or SMMU on ARM) is programmed from the DMAR table, **`device_claim`
|
||||||
|
on any DMA-capable device is equivalent to granting ring 0.**
|
||||||
|
|
||||||
|
This does not make the model useless — it's the same position Linux is in with the
|
||||||
|
IOMMU off, and every other guarantee (crash isolation, restart, no shared address
|
||||||
|
space) still holds. But "user-space drivers are memory-safe" is not true yet, and the
|
||||||
|
gap should be named rather than implied.
|
||||||
|
|
||||||
|
## Ordering
|
||||||
|
|
||||||
|
`M13` (capability passing) is independent of `M14`/`M15` and is the cheapest. It
|
||||||
|
unlocks class drivers, which are the shape with no hardware requirements at all — you
|
||||||
|
could write a real one against any device a bus driver publishes tomorrow.
|
||||||
|
|
||||||
|
`M14` and `M15` together unlock the first HCD. `M14`'s barrier layer is worth landing
|
||||||
|
on its own regardless: it's small, obviously correct, and stops every future driver
|
||||||
|
from hand-rolling `*volatile` and getting ARM wrong.
|
||||||
|
|
||||||
|
## See also
|
||||||
|
|
||||||
|
- [drivers.md](drivers.md) — how to write one, concretely.
|
||||||
|
- [discovery.md](discovery.md) / [acpi.md](acpi.md) — where the device table comes from.
|
||||||
|
- [ipc.md](ipc.md) — endpoints, badges, and the notification path an IRQ arrives on.
|
||||||
|
- [resilience.md](resilience.md) — restart, the reason any of this is worth the trouble.
|
||||||
+395
@@ -0,0 +1,395 @@
|
|||||||
|
# Writing a driver
|
||||||
|
|
||||||
|
In a monolithic kernel a driver is a function call away from everything: it runs in
|
||||||
|
ring 0, dereferences any physical address, and its interrupt handler *is* the ISR. In
|
||||||
|
danos a driver is **an ordinary ring-3 process**. It has its own address space, it
|
||||||
|
can crash without taking the kernel with it, and — the point of this document — it
|
||||||
|
can be restarted ([resilience](resilience.md)).
|
||||||
|
|
||||||
|
That leaves three questions the kernel has to answer, because a process can't answer
|
||||||
|
them for itself:
|
||||||
|
|
||||||
|
1. **What hardware exists?** → `device_enumerate`, over the device table discovery built
|
||||||
|
([discovery](discovery.md), [acpi](acpi.md)).
|
||||||
|
2. **How do I touch its registers?** → `device_claim` + `mmio_map`: the kernel maps the
|
||||||
|
device's physical MMIO window into your address space, and from then on it's plain
|
||||||
|
memory. No syscall per register access.
|
||||||
|
3. **How do I find out it wants something?** → `irq_bind`: the interrupt is delivered
|
||||||
|
to you as an IPC notification. You block; the hardware wakes you.
|
||||||
|
|
||||||
|
A driver is, in one sentence, *a process that sleeps until its device has something to
|
||||||
|
say.*
|
||||||
|
|
||||||
|
## How a driver gets started: discover, match, spawn
|
||||||
|
|
||||||
|
Nothing in the kernel decides that the PCI host bridge needs the `pci-bus` driver — that
|
||||||
|
is policy, and policy lives in user space. Boot brings user space up as a three-level
|
||||||
|
supervision hierarchy, each level owning one job:
|
||||||
|
|
||||||
|
```
|
||||||
|
kernel ──spawns──► init (PID 1) ──spawns──► device-manager ──spawns──► pci-bus
|
||||||
|
| | |
|
||||||
|
spawns only init, the service supervisor: the driver supervisor: enumerates
|
||||||
|
publishes the starts the system /system/devices, matches each device
|
||||||
|
initial-ramdisk services (vfs, the to a driver, and system_spawn's it
|
||||||
|
so user space can device-manager). Its
|
||||||
|
system_spawn from it list is init policy.
|
||||||
|
```
|
||||||
|
|
||||||
|
The kernel launches exactly one process — `init` — and hands it nothing but the raw
|
||||||
|
ability to start more (`system_spawn(name, arguments)`, which loads a binary bundled
|
||||||
|
in the initial-ramdisk as a fresh ring-3 process — `name` becoming its argv[0],
|
||||||
|
the optional arguments its argv[1..], on a SysV entry stack, see sysv.md). Everything else is a user-space decision:
|
||||||
|
|
||||||
|
- **init** ([system/services/init](system/services/init/init.zig)) is the **service
|
||||||
|
supervisor**. It spawns the system services danos brings up at boot — today `vfs` and
|
||||||
|
the `device-manager` — from a small list. Drivers are deliberately *not* its job.
|
||||||
|
- **device-manager** ([system/services/device-manager](system/services/device-manager/device-manager.zig))
|
||||||
|
is the **driver supervisor**. It does the three steps a monolithic kernel would do in
|
||||||
|
its probe path, entirely from ring 3:
|
||||||
|
1. **Discover** — `device_enumerate` snapshots the device table the kernel built from
|
||||||
|
ACPI/PCI ([discovery](discovery.md)).
|
||||||
|
2. **Match** — for each device it looks up a driver by `DeviceClass`. The match policy
|
||||||
|
is a table (`driverFor`): today a static `timer → hpet` map; a fuller system reads
|
||||||
|
what each driver *binds* (a manifest under `/system/drivers`, or the driver
|
||||||
|
describing its own match).
|
||||||
|
3. **Spawn** — `system_spawn(driver_name, arguments)` starts the matched driver (the
|
||||||
|
arguments can carry *which* device it matched), which then claims
|
||||||
|
its device and runs the event loop below.
|
||||||
|
|
||||||
|
So "how is a driver discovered and configured" has two halves: **discovery** is the
|
||||||
|
kernel's device table, read by anyone; **configuration** is two user-space policies —
|
||||||
|
init's service list and the device-manager's match table. Both are hardcoded in their
|
||||||
|
respective programs today; the natural next step is to move them into `/etc` (see the
|
||||||
|
milestone notes in [driver-model.md](driver-model.md)). `system_spawn` is currently
|
||||||
|
ungated — any process may spawn any bundled binary — because there is no spawn
|
||||||
|
capability yet.
|
||||||
|
|
||||||
|
## The capability: claim before touch
|
||||||
|
|
||||||
|
The driver syscall numbers (`system/abi.zig`) with the device types they carry
|
||||||
|
(`system/devices/device-abi.zig`), dispatched in `system/kernel/process.zig`:
|
||||||
|
|
||||||
|
| # | Call | Meaning |
|
||||||
|
|---|------|---------|
|
||||||
|
| 11 | `device_enumerate(buf, max) -> total` | Snapshot the device table |
|
||||||
|
| 12 | `device_claim(id) -> ok` | Take **exclusive** ownership |
|
||||||
|
| 13 | `mmio_map(id, res_idx) -> vaddr` | Map a claimed device's register window |
|
||||||
|
| 14 | `irq_bind(id, res_idx, endpoint)` | Deliver that device's IRQ as a notification |
|
||||||
|
| 15 | `irq_ack(id, res_idx)` | Re-arm the IRQ after servicing the device |
|
||||||
|
| 16 | `device_register(parent_id, desc) -> id` | Publish a child of a device you claimed |
|
||||||
|
|
||||||
|
Notice that **nothing takes a physical address or an interrupt number.** Every call
|
||||||
|
names a device by id and a resource by index. That indirection is the entire security
|
||||||
|
model. If `mmio_map` took a physical address, any process could map the kernel's
|
||||||
|
memory; if `irq_bind` took a GSI, any process could bind the keyboard's line and
|
||||||
|
silently intercept it. Instead the kernel checks two things (`process.ownedGsi`, and
|
||||||
|
the same check at the top of `sysMmioMap`):
|
||||||
|
|
||||||
|
- `devices_broker.ownerOf(dev_id) == me` — you claimed it, and claims are exclusive
|
||||||
|
- the resource at `res_idx` is of the right *kind* — `memory` for `mmio_map`, `irq`
|
||||||
|
for `irq_bind`
|
||||||
|
|
||||||
|
The claim is the capability. Everything else follows from it.
|
||||||
|
|
||||||
|
## Registers: `mmio_map`
|
||||||
|
|
||||||
|
`mmio_map` walks the caller's page tables and installs the device's physical frames
|
||||||
|
with `present | user | writable | nx | pcd | pwt`
|
||||||
|
(`arch/x86_64/paging.zig:mapUserDeviceInto`). Two of those bits are load-bearing:
|
||||||
|
|
||||||
|
- **`pcd | pwt`** — strong-uncacheable. A device register is not memory; a cached read
|
||||||
|
would return a stale value and a write might never leave the CPU.
|
||||||
|
- **`device_grant`** (bit 9, one of the PTE's available bits) — marks the leaf as MMIO
|
||||||
|
rather than RAM, so `freeSubtree` skips `pmm.free` on it when the address space is
|
||||||
|
destroyed. Without this, killing a driver would hand the HPET's registers back to
|
||||||
|
the frame allocator as if they were free RAM. The `iopass` test guards it.
|
||||||
|
|
||||||
|
Grants land in their own arena, `0x0000_7100_0000_0000` (PML4[226]), so device pages
|
||||||
|
never widen an existing mapping.
|
||||||
|
|
||||||
|
Then you just… use it:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
const base = dev.mmioMap(dev_id, mmio_res) orelse return;
|
||||||
|
const counter: *volatile u64 = @ptrFromInt(base + 0xF0);
|
||||||
|
const now = counter.*; // a load, straight to the hardware. no kernel involved.
|
||||||
|
```
|
||||||
|
|
||||||
|
## Interrupts: the cycle, and why it has that shape
|
||||||
|
|
||||||
|
An interrupt handler in a microkernel has a problem. The code that knows how to quiet
|
||||||
|
the device is in ring 3, in another address space, and it will not run for
|
||||||
|
microseconds or milliseconds — after a context switch, when the scheduler gets to it.
|
||||||
|
But the CPU wants an EOI *now*, and a **level-triggered** line stays asserted until
|
||||||
|
the device is quieted. EOI a still-asserted line and the I/O APIC redelivers
|
||||||
|
immediately. Forever. The driver never gets to run at all.
|
||||||
|
|
||||||
|
The way out is to mask the line before acknowledging it:
|
||||||
|
|
||||||
|
```
|
||||||
|
kernel ISR irqMask(gsi) // line still asserted; stop it reaching a CPU
|
||||||
|
irqEoi() // now safe to tell the LAPIC we're done
|
||||||
|
notifyFromIsr() // wake the driver — it runs much later
|
||||||
|
|
||||||
|
driver replyWait() -> badge with the notify bit set
|
||||||
|
<clear the device's status register> // NOW the line deasserts
|
||||||
|
irq_ack(dev, res) // kernel unmasks: quiet, so it can't refire
|
||||||
|
|
||||||
|
```
|
||||||
|
|
||||||
|
`irq_ack` is not bookkeeping you could skip. **It is the unmask.** Forget it and the
|
||||||
|
interrupt fires exactly once, ever; call it before the device is quiet and you get an
|
||||||
|
interrupt storm. That single fact explains why `irq_bind` and `irq_ack` are two
|
||||||
|
syscalls and not one.
|
||||||
|
|
||||||
|
This is also why `interruptDispatch` (`arch/x86_64/idt.zig`) no longer issues the EOI
|
||||||
|
itself. It used to, before running the handler — correct for the LAPIC timer, and
|
||||||
|
impossible for a routed device line. Each handler now owns its EOI, because only the
|
||||||
|
handler knows which discipline its source needs.
|
||||||
|
|
||||||
|
### The driver side is an event loop, not a callback
|
||||||
|
|
||||||
|
`IPC_ReplyWait` returns *either* a client request *or* a notification, told apart by
|
||||||
|
the top bit of the badge (`ipc_sync.notify_badge_bit`). So a driver is one
|
||||||
|
single-threaded loop over both of its event sources:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
while (true) {
|
||||||
|
const r = ipc.replyWait(endpoint, reply, &recv);
|
||||||
|
if (r.isNotification()) { // r.source() is the GSI
|
||||||
|
service_device(); // clear the status register
|
||||||
|
_ = dev.irqAck(id, irq_res); // re-arm
|
||||||
|
} else {
|
||||||
|
handle_client_request(recv[0..r.len]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
No reentrancy, no "what am I allowed to call from an interrupt handler", no shared
|
||||||
|
state between ISR and task context. The interrupt is just a message.
|
||||||
|
|
||||||
|
Two properties worth knowing:
|
||||||
|
|
||||||
|
- **An interrupt taken while you're elsewhere is not lost.** If the driver is off in
|
||||||
|
an `ipc_call` to another server when the IRQ fires, `wakeLocked` finds nobody
|
||||||
|
waiting, but the badge is already on the endpoint's notify ring. The next
|
||||||
|
`replyWait` pops it (`ipc_sync.replyWait` checks `popNotify` before the sender FIFO).
|
||||||
|
- **Notifications coalesce, they don't count.** The ring is 8 deep and drops on
|
||||||
|
overflow. That's correct: an IRQ notification is a *level* ("the device wants
|
||||||
|
attention"), not a tally. Re-read the device's status register; never assume one
|
||||||
|
notification means exactly one event. Because the ISR masks the line until you
|
||||||
|
`irq_ack`, at most one badge per GSI can be outstanding — so the ring can only
|
||||||
|
overflow if you bind more than eight GSIs to a single endpoint. Don't.
|
||||||
|
|
||||||
|
## A whole driver
|
||||||
|
|
||||||
|
A minimal leaf driver is only ~150 lines and does all of it. danos ships **no such
|
||||||
|
example binary** — the driver model is proven by the real drivers (`pci-bus`, `ps2-bus`,
|
||||||
|
`usb-xhci-bus`), and a teaching example belongs here, in the docs, rather than as a
|
||||||
|
compiled program nobody runs. Illustrated with a hypothetical HPET timer driver, the
|
||||||
|
shape is:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
const hpet = findHpet(buf) orelse return; // device_enumerate, look for
|
||||||
|
// class=timer with memory + irq
|
||||||
|
_ = dev.claim(hpet.dev_id); // the capability
|
||||||
|
const base = dev.mmioMap(hpet.dev_id, hpet.mmio).?;
|
||||||
|
const endpoint = ipc.createIpcEndpoint().?;
|
||||||
|
|
||||||
|
// program the hardware over the mapping we were just handed
|
||||||
|
reg(base, 0x100).* = level | int_enb | (hpet.gsi << 9); // timer 0 config
|
||||||
|
reg(base, 0x108).* = reg(base, 0xF0).* + period; // comparator
|
||||||
|
reg(base, 0x010).* |= 1; // ENABLE
|
||||||
|
|
||||||
|
_ = dev.irqBind(hpet.dev_id, hpet.irq, endpoint);
|
||||||
|
|
||||||
|
while (...) {
|
||||||
|
const r = ipc.replyWait(endpoint, &.{}, &recv); // blocked. not polling.
|
||||||
|
if (r.badge & notify_bit == 0) continue;
|
||||||
|
reg(base, 0x020).* = 1; // clear status -> deassert
|
||||||
|
reg(base, 0x108).* = reg(base, 0xF0).* + period; // re-arm
|
||||||
|
_ = dev.irqAck(hpet.dev_id, hpet.irq); // unmask
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
The HPET makes a good illustration for a reason that isn't obvious. Its *counter* is a
|
||||||
|
clocksource — the only way to use it is to read it, so it exercises `mmio_map` without
|
||||||
|
needing interrupts at all. Its *comparators* are a clockevent, and can be configured
|
||||||
|
**level-triggered** (`Tn_INT_TYPE_CNF`), which asserts a bit in `GENERAL_INT_STATUS`
|
||||||
|
that the driver must write-1-to-clear. That's a genuine deassert step, so the full
|
||||||
|
mask/ack cycle above is exercised for real rather than being decoration on an
|
||||||
|
edge-triggered line that would have been fine without it.
|
||||||
|
|
||||||
|
One wrinkle it also demonstrates: the ACPI HPET table carries **no interrupt number**.
|
||||||
|
Which I/O APIC inputs a comparator may drive is a bitmask in `Tn_INT_ROUTE_CAP`, in
|
||||||
|
the device's own registers. So discovery (`acpi.parseHpet`) maps the block, reads the
|
||||||
|
mask, and records one concrete GSI as an `irq` resource. The driver then programs
|
||||||
|
`Tn_INT_ROUTE_CNF` to raise exactly that line — and the kernel will only bind the one
|
||||||
|
it recorded. Hardware that describes itself at runtime still has to fit through a
|
||||||
|
static capability.
|
||||||
|
|
||||||
|
## Publishing children: `device_register`
|
||||||
|
|
||||||
|
A device that *contains other devices* — a PCI bridge, a USB hub, or the HPET's block
|
||||||
|
of comparators — needs a driver that enumerates it and tells the kernel what it found.
|
||||||
|
That's `device_register`, and it makes the device table a tree rather than a list
|
||||||
|
(`DeviceDesc.parent`).
|
||||||
|
|
||||||
|
```zig
|
||||||
|
var child = std.mem.zeroes(dev.DeviceDesc);
|
||||||
|
child.class = @intFromEnum(dev.DeviceClass.timer);
|
||||||
|
child.resource_count = 1;
|
||||||
|
child.resources[0] = .{ .kind = memory, .start = bus_base + 0x100, .len = 0x20 };
|
||||||
|
const child_id = dev.register(bus_id, &child).?;
|
||||||
|
```
|
||||||
|
|
||||||
|
The child is left **unclaimed**, which is the whole point: another process claims it and
|
||||||
|
`mmio_map`s it, and sees only that 0x20-byte window.
|
||||||
|
|
||||||
|
The rule the kernel enforces is **containment**: every resource of a child must lie
|
||||||
|
inside a resource of the same kind on its parent. Ranges must nest; an IRQ must match
|
||||||
|
exactly. This isn't bureaucracy — a `DeviceDesc` is a licence to map physical memory, so
|
||||||
|
without containment `device_register` would be a syscall for mapping any page you like. A
|
||||||
|
bus driver may only ever subdivide what it already owns.
|
||||||
|
|
||||||
|
A device with **no resources** is legal and common. A USB device is reached through its
|
||||||
|
controller, not by MMIO, so it gets `resource_count = 0`.
|
||||||
|
|
||||||
|
See [`system/drivers/pci-bus/pci-bus.zig`](../system/drivers/pci-bus/pci-bus.zig) for a
|
||||||
|
real one — it claims a PCI host bridge, maps its ECAM window, and publishes each function
|
||||||
|
it finds as a child — and [driver-model.md](driver-model.md) for how bus drivers, class
|
||||||
|
drivers and host controller drivers fit together.
|
||||||
|
|
||||||
|
## What the kernel does not do for you
|
||||||
|
|
||||||
|
- **It does not quiet your device.** That's the whole reason `irq_ack` exists.
|
||||||
|
- **It does not know your registers.** `mmio_map` hands you a base address; every
|
||||||
|
offset in this document came from the HPET spec, not from danos.
|
||||||
|
- **It does not serialise your driver.** Two clients calling one driver endpoint are
|
||||||
|
serialised by `replyWait`, but nothing stops your driver from being preempted.
|
||||||
|
|
||||||
|
## Limits, today
|
||||||
|
|
||||||
|
Worth knowing before you write the second driver:
|
||||||
|
|
||||||
|
Several things this list used to warn about are now available (see
|
||||||
|
[driver-model.md](driver-model.md)): **port I/O** (`io_read`/`io_write`, claim-gated by
|
||||||
|
the device's `io_port` resource — direct ring-3 `in`/`out` is still a #GP, so a PS/2 or
|
||||||
|
16550 driver goes through these), **DMA memory** (`dma_alloc`: contiguous, pinned,
|
||||||
|
uncacheable, physical address exposed), and **memory barriers** (`/lib/mmio`'s
|
||||||
|
`mb`/`rmb`/`wmb`). What remains:
|
||||||
|
|
||||||
|
- **Page granularity.** `mmio_map` rounds to 4 KiB. Two devices sharing a page means
|
||||||
|
granting one grants the other. A `device_register`ed child's *resource* can be narrower
|
||||||
|
than a page, but its *mapping* can't.
|
||||||
|
- **DMA is not contained.** A driver that can program a bus-mastering device can make
|
||||||
|
that device write to *any* physical address — page tables don't sit between a device
|
||||||
|
and RAM; an IOMMU does. The IOMMU is now *detected* (M16), but no translation domains
|
||||||
|
are programmed, so `device_claim` on a DMA-capable device is still effectively
|
||||||
|
equivalent to granting ring 0. This is the largest gap between the design's promise and
|
||||||
|
what it delivers; enforcement lands with the first DMA driver.
|
||||||
|
- **No `dev_release`.** A claim is never dropped (only IRQ/MSI bindings are, on exit), so
|
||||||
|
a device stays owned for the life of its driver — which blocks restart.
|
||||||
|
- **One endpoint per GSI**, so shared legacy PCI INTx lines can't be split between two
|
||||||
|
drivers. MSI/MSI-X — one vector per device, edge-triggered, unshared — is the real
|
||||||
|
answer, and QEMU's HPET doesn't offer it (`Tn_FSB_INT_DEL_CAP = 0`).
|
||||||
|
- **Polarity is hardcoded** active-high in `irq.bind`. A device whose MADT override
|
||||||
|
says active-low needs that threaded through from discovery.
|
||||||
|
- **14 device vectors** (33–46) and **24 GSIs**, bounded by the stubs `isr.s` emits and
|
||||||
|
by a single I/O APIC.
|
||||||
|
- **Don't bind more than 8 GSIs to one endpoint.** The notify ring is 8 deep and drops
|
||||||
|
on overflow. With one GSI per endpoint that's unreachable — the line is masked from
|
||||||
|
the ISR until `irq_ack`, so at most one badge is ever outstanding. Bind nine devices
|
||||||
|
to one endpoint, though, and a dropped badge leaves that line masked with nobody
|
||||||
|
left to ack it.
|
||||||
|
- **A faulting driver still kills the machine.** There is no per-process kill path: a
|
||||||
|
ring-3 page fault halts the kernel, so `releaseIrqs` runs only on a voluntary
|
||||||
|
`exit`. Fault isolation is the whole premise ([vision](vision.md)) and it is
|
||||||
|
[not built yet](resilience.md).
|
||||||
|
- **A dead driver's device is not reclaimed.** `releaseIrqs` unbinds and masks the
|
||||||
|
line on exit, but the claim is never released — restart is
|
||||||
|
[not built](resilience.md).
|
||||||
|
- **On real hardware, the mask/EOI cycle may need a remote-IRR flush.** Masking a
|
||||||
|
level-triggered redirection entry with remote-IRR set doesn't clear it on some
|
||||||
|
chipsets, and the line never fires again. QEMU clears it on EOI regardless, so the
|
||||||
|
tests can't see this. Linux flushes remote-IRR by toggling the entry to edge and
|
||||||
|
back. See the note at the top of `system/kernel/irq.zig`.
|
||||||
|
|
||||||
|
## Verifying it
|
||||||
|
|
||||||
|
No demo driver ships to prove this end to end; the *real* drivers do, so the tests
|
||||||
|
target them and the kernel primitives directly:
|
||||||
|
|
||||||
|
- **`device-manager`** — boots only the device manager, which discovers the PCI host
|
||||||
|
bridge, matches `pci-bus`, and `system_spawn`s it. The test reads kernel state — the
|
||||||
|
process table and the device tree — to confirm pci-bus came up and registered the
|
||||||
|
functions it enumerated: the whole discover → match → spawn → driver-up chain.
|
||||||
|
- **`acpi-ps2`** — a user-space driver (`ps2-bus`) is woken by its device's IRQ,
|
||||||
|
delivered as an IPC notification, and attaches the keyboard: IRQ-as-IPC, end to end.
|
||||||
|
- **`pci-scan`** — a user-space driver (`pci-bus`) maps its device's MMIO (the ECAM
|
||||||
|
window) and walks it: `mmio_map`, end to end.
|
||||||
|
- **`containment`** — the kernel refuses a `device_register` whose child window escapes
|
||||||
|
the parent's grant (else it would be a syscall for mapping arbitrary memory), while an
|
||||||
|
identical re-register stays idempotent. Asserted in-kernel, straight against the broker.
|
||||||
|
- **`irqfree`** — the teardown path. Binds two owners to one shared endpoint, releases
|
||||||
|
one, and reads the I/O APIC back: the departing owner's line is masked, the sibling's
|
||||||
|
is not. That second half is why bindings are keyed on the owning *task* and not on the
|
||||||
|
endpoint pointer — endpoints are shared, so releasing "everything pointing at this
|
||||||
|
endpoint" would silently mask a live driver's device.
|
||||||
|
- **`iopass`** — the `device_grant` teardown rule, so destroying a driver's address
|
||||||
|
space never returns MMIO frames to the RAM pool.
|
||||||
|
|
||||||
|
```
|
||||||
|
$ python3 test/qemu_test.py device-manager acpi-ps2 pci-scan containment irqfree iopass
|
||||||
|
device-manager ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
|
acpi-ps2 ... PASS
|
||||||
|
pci-scan ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
|
containment ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
|
irqfree ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
|
iopass ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
|
```
|
||||||
|
|
||||||
|
## What's next (not done here)
|
||||||
|
|
||||||
|
The big driver-model pieces — capability passing (class drivers), DMA + barriers, MSI,
|
||||||
|
and IOMMU detection — are **now done** ([driver-model.md](driver-model.md), M13–M16), as
|
||||||
|
is **port I/O** (`io_read`/`io_write`, the claim-gated syscalls that make a PS/2 or 16550
|
||||||
|
driver possible). What's left is IOMMU *enforcement* (per-device domains — it waits on
|
||||||
|
the first DMA driver to protect and test against) and these smaller items:
|
||||||
|
|
||||||
|
- **Releasing a claim.** There is no `dev_release`, and `devices_broker` never drops a claim on
|
||||||
|
exit — only IRQ bindings are released. A dead driver's device stays owned forever,
|
||||||
|
which blocks restart.
|
||||||
|
- **Unregistering children.** `device_register` only appends. A USB device that is
|
||||||
|
unplugged cannot be removed, and a bus driver in a loop can exhaust the 64-entry
|
||||||
|
table.
|
||||||
|
- **Restart.** A supervisor that *spawns* drivers now exists — the device-manager starts
|
||||||
|
them with `system_spawn` — but a supervisor that *restarts* them does not. A driver that
|
||||||
|
dies should release its claim, have its device quiesced, and be respawned; today nothing
|
||||||
|
notices the death. Some pieces (`releaseIrqs`, `device_grant` teardown, the claim table)
|
||||||
|
exist, and `dev_release` (below) is the missing mechanism; the restart policy is the
|
||||||
|
resilience track ([resilience.md](resilience.md)).
|
||||||
|
- **Interrupt priority / threaded IRQ latency.** `notifyFromIsr` enqueues the woken
|
||||||
|
driver but doesn't preempt (`wakeLocked` deliberately leaves that to the caller), so
|
||||||
|
a woken driver waits for the next scheduling point.
|
||||||
|
|
||||||
|
## The driver contract (M17–M18)
|
||||||
|
|
||||||
|
Claiming and mapping is half of being a danos driver; the other half is the
|
||||||
|
**lifecycle and protocol contract**, and the runtime makes it nearly free:
|
||||||
|
|
||||||
|
- Build on `runtime.service.run` — one replyWait loop folding protocol
|
||||||
|
requests, signals, and notifications into callbacks. The harness answers the
|
||||||
|
universal zero-length ping and turns `terminate` into a clean exit for you
|
||||||
|
([process-lifecycle.md](process-lifecycle.md)).
|
||||||
|
- A driver spawned with an assignment (its device id as argv[1]) sends the
|
||||||
|
versioned `hello` to the device manager inside the deadline, and a **bus**
|
||||||
|
driver reports what it discovers with `child_added`
|
||||||
|
([device-manager.md](device-manager.md); usb-xhci-bus is the reference
|
||||||
|
implementation).
|
||||||
|
- Crash freely — that is the design. The kernel releases your claims, IRQ
|
||||||
|
bindings, and MSI vectors at death; the manager reads your exit reason,
|
||||||
|
prunes what you reported, restarts you with backoff, and your fresh instance
|
||||||
|
re-claims and re-reports. Never depend on your own cleanup running
|
||||||
|
(iron rule 1).
|
||||||
+25
-20
@@ -10,7 +10,7 @@ that hands us a working CPU, a memory map, and a screen, and then gets out of th
|
|||||||
way.
|
way.
|
||||||
|
|
||||||
The key thing to understand: **UEFI is not our OS, it's a stepping stone.** It
|
The key thing to understand: **UEFI is not our OS, it's a stepping stone.** It
|
||||||
exists to load *us*. Our `src/boot/efi.zig` is a UEFI *application* — a normal program
|
exists to load *us*. Our `boot/efi.zig` is a UEFI *application* — a normal program
|
||||||
that the firmware runs — and its entire purpose is to gather what the kernel needs
|
that the firmware runs — and its entire purpose is to gather what the kernel needs
|
||||||
and then jump into the kernel.
|
and then jump into the kernel.
|
||||||
|
|
||||||
@@ -20,14 +20,17 @@ UEFI boots by looking for a FAT-formatted partition called the **EFI System
|
|||||||
Partition (ESP)** and running a file at a well-known fallback path:
|
Partition (ESP)** and running a file at a well-known fallback path:
|
||||||
|
|
||||||
```
|
```
|
||||||
esp/EFI/BOOT/BOOTX64.efi <- the "removable media" default for x86-64
|
EFI/BOOT/BOOTX64.efi <- the "removable media" default for x86-64
|
||||||
```
|
```
|
||||||
|
|
||||||
That's exactly the layout `build.zig` assembles. It builds `src/boot/efi.zig` for the
|
The boot volume is the **FHS-shaped `zig-out`** itself (see the repository-layout note
|
||||||
`uefi` target, installs it to `esp/EFI/BOOT/BOOTX64.efi`, and drops the kernel ELF
|
in [README.md](README.md)): `build.zig` installs `boot/efi.zig` (built for the `uefi`
|
||||||
at `esp/kernel`. The `run-x86-64` step then points QEMU at OVMF (UEFI firmware for
|
target) to `zig-out/EFI/BOOT/BOOTX64.efi` — the one path UEFI firmware fixes — and lays
|
||||||
virtual machines) and presents that `esp/` directory to the guest as a FAT drive.
|
the rest out by FHS path: the kernel at `zig-out/system/kernel`, init at
|
||||||
The firmware finds `BOOTX64.efi` and runs it — that's our `main()`.
|
`zig-out/system/services/init`, the initial-ramdisk at `zig-out/boot/`. The
|
||||||
|
`run-x86-64` step points QEMU at OVMF (UEFI firmware for virtual machines) and presents
|
||||||
|
`zig-out` to the guest as a FAT drive. The firmware finds `BOOTX64.efi` and runs it —
|
||||||
|
that's our `main()`, which then loads the kernel and init from their FHS paths.
|
||||||
|
|
||||||
## Boot services: the firmware's API
|
## Boot services: the firmware's API
|
||||||
|
|
||||||
@@ -83,9 +86,9 @@ All of this *must* happen now, because after exit there's no GOP to ask. (See
|
|||||||
|
|
||||||
- Use the **LoadedImage** protocol to discover which device we booted from, then
|
- Use the **LoadedImage** protocol to discover which device we booted from, then
|
||||||
**SimpleFileSystem** to open that volume.
|
**SimpleFileSystem** to open that volume.
|
||||||
- Open the file named `danos`, seek to the end to learn its size, rewind, and read
|
- Open the kernel ELF at its FHS path (`system\kernel`), seek to the end to learn its
|
||||||
the whole ELF into a firmware-allocated pool buffer. (`read` may return short, so
|
size, rewind, and read the whole ELF into a firmware-allocated pool buffer. (`read`
|
||||||
we loop.)
|
may return short, so we loop.)
|
||||||
- Parse the ELF: validate the `\x7fELF` magic and the `x86_64` machine type, then
|
- Parse the ELF: validate the `\x7fELF` magic and the `x86_64` machine type, then
|
||||||
walk the program headers. For every `PT_LOAD` segment we:
|
walk the program headers. For every `PT_LOAD` segment we:
|
||||||
- reserve the exact physical pages it's linked at (`p_paddr`) via
|
- reserve the exact physical pages it's linked at (`p_paddr`) via
|
||||||
@@ -121,7 +124,7 @@ entirely ours.
|
|||||||
### 4. Jump to the kernel
|
### 4. Jump to the kernel
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
const kernel: *const fn (*const BootInfo) callconv(danos.kernel_abi) noreturn =
|
const kernel: *const fn (*const BootInfo) callconv(boot_handoff.kernel_abi) noreturn =
|
||||||
@ptrFromInt(entry);
|
@ptrFromInt(entry);
|
||||||
kernel(&boot_info);
|
kernel(&boot_info);
|
||||||
```
|
```
|
||||||
@@ -139,17 +142,19 @@ kernel is freestanding and uses the **SysV AMD64** convention (first argument in
|
|||||||
read garbage.
|
read garbage.
|
||||||
|
|
||||||
So both sides pin the convention explicitly to SysV via the shared
|
So both sides pin the convention explicitly to SysV via the shared
|
||||||
`danos.kernel_abi` (defined in `src/root.zig`). The loader's function-pointer type
|
`boot_handoff.kernel_abi` (defined in `system/boot-handoff.zig`). The loader's
|
||||||
and the kernel's `_start` both reference it, so the pointer lands in the register
|
function-pointer type and the kernel's `_start` both reference it, so the pointer lands
|
||||||
the kernel expects. This is the whole reason `kernel_abi` lives in the shared
|
in the register the kernel expects. This is the whole reason `kernel_abi` lives in the
|
||||||
`danos` module: it's a contract both binaries must agree on. See
|
shared `boot-handoff` module: it's a contract both binaries must agree on. See
|
||||||
[sysv.md](sysv.md) for what "SysV" means and where else it shows up.
|
[sysv.md](sysv.md) for what "SysV" means and where else it shows up.
|
||||||
|
|
||||||
## The handoff contract
|
## The handoff contract
|
||||||
|
|
||||||
The loader and kernel are two *separate* binaries built for two different targets,
|
The loader and kernel are two *separate* binaries built for two different targets,
|
||||||
so everything they exchange must have an identically-defined memory layout. That's
|
so everything they exchange must have an identically-defined memory layout. That's
|
||||||
what `src/root.zig` provides — imported by both as the `danos` module:
|
what `system/boot-handoff.zig` provides — imported by both as the `boot-handoff` module.
|
||||||
|
It is *only* the handoff: the kernel↔user ABI (`system/abi.zig`) and the device types
|
||||||
|
(`system/devices/device-abi.zig`) are separate contracts the bootloader never sees.
|
||||||
|
|
||||||
- `BootInfo` — the top-level struct passed to the kernel (currently just the
|
- `BootInfo` — the top-level struct passed to the kernel (currently just the
|
||||||
framebuffer; this is where future handoff data like the memory map will go).
|
framebuffer; this is where future handoff data like the memory map will go).
|
||||||
@@ -164,16 +169,16 @@ the loader writes are the bytes the kernel reads.
|
|||||||
```
|
```
|
||||||
power on
|
power on
|
||||||
-> UEFI firmware initialises hardware
|
-> UEFI firmware initialises hardware
|
||||||
-> finds esp/EFI/BOOT/BOOTX64.efi, runs it (our efi.zig main)
|
-> finds EFI/BOOT/BOOTX64.efi on the FHS volume, runs it (our efi.zig main)
|
||||||
-> grab boot services
|
-> grab boot services
|
||||||
-> queryFramebuffer (via GOP: EDID native res, setMode, describe fb)
|
-> queryFramebuffer (via GOP: EDID native res, setMode, describe fb)
|
||||||
-> loadKernel (read danos ELF, load PT_LOAD segments to 0x100000)
|
-> loadKernel (read system/kernel ELF, load PT_LOAD segments to 0x100000)
|
||||||
-> exitBootServices (retry until the memory-map key holds)
|
-> exitBootServices (retry until the memory-map key holds)
|
||||||
-> jump to e_entry, boot_info pointer in RDI
|
-> jump to e_entry, boot_info pointer in RDI
|
||||||
-> kernel _start (src/kernel/main.zig: framebuffer console, then halt)
|
-> kernel _start (system/kernel/kernel.zig: framebuffer console, then halt)
|
||||||
```
|
```
|
||||||
|
|
||||||
Bottom line: **UEFI's job is to give us a CPU, memory, and a framebuffer, then
|
Bottom line: **UEFI's job is to give us a CPU, memory, and a framebuffer, then
|
||||||
disappear.** `src/boot/efi.zig` is the thin bridge that collects those gifts into a
|
disappear.** `boot/efi.zig` is the thin bridge that collects those gifts into a
|
||||||
`BootInfo`, tears down the firmware, and jumps into the kernel — after which we're
|
`BootInfo`, tears down the firmware, and jumps into the kernel — after which we're
|
||||||
on our own.
|
on our own.
|
||||||
|
|||||||
@@ -3,12 +3,12 @@
|
|||||||
Once the kernel knows what RAM exists ([memory-map.md](memory-map.md)), it needs a
|
Once the kernel knows what RAM exists ([memory-map.md](memory-map.md)), it needs a
|
||||||
way to *hand out* that RAM: give me a free page of physical memory, and later,
|
way to *hand out* that RAM: give me a free page of physical memory, and later,
|
||||||
here's one back. That's the **physical frame allocator** (a "physical memory
|
here's one back. That's the **physical frame allocator** (a "physical memory
|
||||||
manager", hence `src/kernel/pmm.zig`). It deals only in fixed 4 KiB **frames** — the
|
manager", hence `system/kernel/pmm.zig`). It deals only in fixed 4 KiB **frames** — the
|
||||||
natural unit because that's the granularity the CPU's paging hardware maps — and
|
natural unit because that's the granularity the CPU's paging hardware maps — and
|
||||||
it is the primitive everything above it stands on: page tables, the kernel heap,
|
it is the primitive everything above it stands on: page tables, the kernel heap,
|
||||||
per-process memory all ultimately ask the frame allocator for pages.
|
per-process memory all ultimately ask the frame allocator for pages.
|
||||||
|
|
||||||
It's **generic kernel code**: it operates on the neutral `danos.MemoryRegion`
|
It's **generic kernel code**: it operates on the neutral `system.MemoryRegion`
|
||||||
array, so there's no UEFI in it and nothing architecture-specific beyond the 4 KiB
|
array, so there's no UEFI in it and nothing architecture-specific beyond the 4 KiB
|
||||||
page. (Contrast [arch.md](arch.md), which is where CPU-specific code lives.)
|
page. (Contrast [arch.md](arch.md), which is where CPU-specific code lives.)
|
||||||
|
|
||||||
@@ -33,7 +33,7 @@ RAM is 32768 frames — a **4 KiB bitmap, a single frame**. Even 64 GiB needs on
|
|||||||
|
|
||||||
## How it works
|
## How it works
|
||||||
|
|
||||||
State lives in `src/kernel/pmm.zig`: the `bitmap` slice, `total_frames`, `used_frames`,
|
State lives in `system/kernel/pmm.zig`: the `bitmap` slice, `total_frames`, `used_frames`,
|
||||||
and a `next_hint` marking where the next allocation scan should start.
|
and a `next_hint` marking where the next allocation scan should start.
|
||||||
|
|
||||||
### init(map) — building it from the memory map
|
### init(map) — building it from the memory map
|
||||||
|
|||||||
+2
-2
@@ -9,10 +9,10 @@ write a 32-bit value to the right address, and a pixel changes color. That's
|
|||||||
exactly what `Console.pixel` does:
|
exactly what `Console.pixel` does:
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
self.rowPtr(y)[x] = color; // src/kernel/console.zig
|
self.rowPtr(y)[x] = color; // system/kernel/console.zig
|
||||||
```
|
```
|
||||||
|
|
||||||
Our `Framebuffer` struct (`src/root.zig`) is the four facts you need to
|
Our `Framebuffer` struct (`system/boot-handoff.zig`) is the four facts you need to
|
||||||
address it:
|
address it:
|
||||||
|
|
||||||
| Field | Meaning |
|
| Field | Meaning |
|
||||||
|
|||||||
+3
-3
@@ -16,7 +16,7 @@ safely, until the machine is reset or powered off.
|
|||||||
## The core of it: `hlt`
|
## The core of it: `hlt`
|
||||||
|
|
||||||
Everything comes down to one x86 instruction. It's CPU-specific, so it lives in
|
Everything comes down to one x86 instruction. It's CPU-specific, so it lives in
|
||||||
the arch module, `src/kernel/arch/x86_64/cpu.zig` (see [arch.md](arch.md)), and the
|
the arch module, `system/kernel/architecture/x86_64/cpu.zig` (see [arch.md](arch.md)), and the
|
||||||
generic kernel calls it as `arch.halt()`:
|
generic kernel calls it as `arch.halt()`:
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
@@ -83,7 +83,7 @@ treats the call:
|
|||||||
signature for a kernel entry point — the bootloader jumps in and nothing ever
|
signature for a kernel entry point — the bootloader jumps in and nothing ever
|
||||||
jumps back out.
|
jumps back out.
|
||||||
|
|
||||||
You can see the chain in `src/kernel/main.zig`: `_start` is `noreturn`, it calls
|
You can see the chain in `system/kernel/kernel.zig`: `_start` is `noreturn`, it calls
|
||||||
`kmain` which is `noreturn`, which ends by calling `arch.halt()` which is
|
`kmain` which is `noreturn`, which ends by calling `arch.halt()` which is
|
||||||
`noreturn`. The "never returns" property is threaded all the way down.
|
`noreturn`. The "never returns" property is threaded all the way down.
|
||||||
|
|
||||||
@@ -106,7 +106,7 @@ There are three halt sites, and they're all the same idea:
|
|||||||
`arch.halt()`. A panic is unrecoverable here, so stopping the machine — rather
|
`arch.halt()`. A panic is unrecoverable here, so stopping the machine — rather
|
||||||
than limping on with corrupted state — is the safe response.
|
than limping on with corrupted state — is the safe response.
|
||||||
|
|
||||||
3. **Bootloader failure** — in `src/boot/efi.zig`, if `boot()` fails *before* handing
|
3. **Bootloader failure** — in `boot/efi.zig`, if `boot()` fails *before* handing
|
||||||
off to the kernel, `main` logs the error and parks the machine with the same
|
off to the kernel, `main` logs the error and parks the machine with the same
|
||||||
loop so the message stays on screen:
|
loop so the message stays on screen:
|
||||||
|
|
||||||
|
|||||||
+1
-1
@@ -7,7 +7,7 @@ top of both to provide what the rest of the kernel actually wants: `alloc(n)` /
|
|||||||
the thing that unlocks dynamic data structures — lists, hash maps, driver state,
|
the thing that unlocks dynamic data structures — lists, hash maps, driver state,
|
||||||
eventually a process table.
|
eventually a process table.
|
||||||
|
|
||||||
It's generic kernel code (`src/kernel/heap.zig`): the allocator logic is
|
It's generic kernel code (`system/kernel/heap.zig`): the allocator logic is
|
||||||
architecture-neutral, using `arch.mapPage` and the frame allocator underneath.
|
architecture-neutral, using `arch.mapPage` and the frame allocator underneath.
|
||||||
|
|
||||||
## A growable free-list allocator
|
## A growable free-list allocator
|
||||||
|
|||||||
+161
@@ -0,0 +1,161 @@
|
|||||||
|
# The input module: broadcasting input events
|
||||||
|
|
||||||
|
A keyboard driver has one keystroke and *many* programs that might want it — a shell, a
|
||||||
|
window server, a logger. None of them owns the hardware, and the driver should not know
|
||||||
|
who is listening. So between the drivers and the listeners sits the **input service**
|
||||||
|
(`system/services/input/`): drivers **publish** events to it, programs **subscribe**, and
|
||||||
|
it fans each event out to every interested subscriber. It is an ordinary ring-3 process
|
||||||
|
reached over IPC, like the [VFS server](../system/services/vfs/vfs.zig) — no kernel knows
|
||||||
|
what a key is.
|
||||||
|
|
||||||
|
## One service, several device classes
|
||||||
|
|
||||||
|
The service carries three device classes today — **keyboard**, **mouse**, and
|
||||||
|
**joystick/gamepad** — and is built to take more
|
||||||
|
([protocol.zig](../system/services/input/protocol.zig)). Each class has its own typed
|
||||||
|
event:
|
||||||
|
|
||||||
|
- `KeyEvent` — `key_down`/`key_up` (physical make/break) and `key_press` (a character was
|
||||||
|
produced, carrying the Unicode scalar); plus a layout-independent `keycode` and a
|
||||||
|
`modifiers` bitmask.
|
||||||
|
- `MouseEvent` — relative `motion` (`dx`/`dy`), `button_down`/`button_up`, and `scroll`.
|
||||||
|
- `JoystickEvent` — `axis` moves (a signed value on a `control` index) and
|
||||||
|
`button_down`/`button_up`.
|
||||||
|
|
||||||
|
All three travel in one **`InputEvent` envelope** tagged with a `DeviceKind`, so the
|
||||||
|
fan-out is a single code path and a subscriber can take a mix of classes on one stream.
|
||||||
|
Decode an envelope with `asKeyboard()` / `asMouse()` / `asJoystick()` (each returns null
|
||||||
|
unless the tag matches). A subscriber names the classes it wants with a **`device_mask`**,
|
||||||
|
and the service routes each event only to subscribers whose mask includes its class — so a
|
||||||
|
mouse-only listener never wakes for keystrokes.
|
||||||
|
|
||||||
|
## Why this needed a new kernel primitive
|
||||||
|
|
||||||
|
The interesting part is delivery, and it runs straight into the shape of danos IPC.
|
||||||
|
[ipc.md](ipc.md) describes a **synchronous rendezvous**: a server holds exactly one
|
||||||
|
pending reply (`Task.ipc_client`) and *must* answer it on its next `replyWait`. Two
|
||||||
|
consequences decide the whole design:
|
||||||
|
|
||||||
|
1. **You cannot block N subscribers waiting for "the next event".** A server can hold only
|
||||||
|
one caller at a time, so the natural "subscriber calls `next_event()` and blocks" API
|
||||||
|
is impossible for more than one subscriber. Delivery therefore has to be **push** — the
|
||||||
|
service reaching out to subscribers — not pull.
|
||||||
|
|
||||||
|
2. **A synchronous push can hang the whole service.** If the service delivered with
|
||||||
|
`ipc_call`, it would block until each subscriber replied. `ipc_call` has no timeout, and
|
||||||
|
the kernel does **not** wake a caller parked on a *dead* peer's endpoint (it only fails a
|
||||||
|
peer that was mid-reply — see [process.zig](../system/kernel/process.zig)
|
||||||
|
`releaseTaskResourcesLocked`). One subscriber that exits mid-delivery would wedge input
|
||||||
|
for everyone. That is the opposite of the resilience the microkernel is for.
|
||||||
|
|
||||||
|
The fix is the asynchronous send that [ipc.md](ipc.md) had already earmarked as future
|
||||||
|
work ("asynchronous / buffered send … for notifications between servers"):
|
||||||
|
|
||||||
|
```
|
||||||
|
ipc_send(handle, message_ptr, message_len) -> 0 / -errno
|
||||||
|
```
|
||||||
|
|
||||||
|
`ipc_send` copies a small payload into the endpoint's **bounded queue** and wakes a
|
||||||
|
receiver, then returns immediately — it never blocks and so can never hang on a dead or
|
||||||
|
slow subscriber. The receiver picks it up through the same `replyWait` it already runs:
|
||||||
|
the wake arrives as a **buffered message** — `notify_badge_bit | notify_message_bit` set in
|
||||||
|
the badge (distinguishing it from a bare IRQ/child-exit notification), the sender's task id
|
||||||
|
in the low bits, and the payload in the receive buffer, with no reply owed. The queue holds
|
||||||
|
16 messages per endpoint; a full queue **drops the oldest**, because a buffered message is
|
||||||
|
discrete data, not a coalescing "level" like an interrupt. See
|
||||||
|
[ipc-synchronous.zig](../system/kernel/ipc-synchronous.zig) (`sendLocked`, `popPost`, and
|
||||||
|
the `replyWait` receive loop).
|
||||||
|
|
||||||
|
This is the async counterpart of `ipc_call`, and the input service is its first consumer.
|
||||||
|
|
||||||
|
## How the pieces fit
|
||||||
|
|
||||||
|
```
|
||||||
|
keyboard/mouse driver, input-source input service subscriber(s)
|
||||||
|
----------------------------------- ------------- -------------
|
||||||
|
connectSource(); loop: replyWait: subscribeKeyboard()/…All:
|
||||||
|
publishKeyboardEvent(k) ─ ipc_call ─▶ publish → broadcast: createIpcEndpoint()
|
||||||
|
publishMouseEvent(m) for each sub whose callCap(subscribe,
|
||||||
|
publishJoystickEvent(j) mask matches event.device: send_cap = ep,
|
||||||
|
ipc_send(sub_ep) ──────▶ device_mask)
|
||||||
|
reply ok loop: next()
|
||||||
|
subscribe → store {ep cap, └─ replyWait(ep)
|
||||||
|
task id, device_mask} → InputEvent
|
||||||
|
```
|
||||||
|
|
||||||
|
- A **subscriber** calls `input.subscribe(mask)` — or a typed helper: `subscribeKeyboard()`,
|
||||||
|
`subscribeMouse()`, `subscribeJoystick()` (one class, `next()` returns the decoded event),
|
||||||
|
or `subscribeAll()` (every class, `next()` returns a tagged `InputEvent`)
|
||||||
|
([library/runtime/input.zig](../library/runtime/input.zig)). It creates its own endpoint
|
||||||
|
and hands it to the service as a **capability** (M13 capability passing — the input
|
||||||
|
service is that feature's first real user), along with its `device_mask`. Then it loops on
|
||||||
|
`next()`, a `replyWait` on that endpoint returning each pushed event.
|
||||||
|
- A **source** (a keyboard, mouse, or joystick driver) calls `input.connectSource()` and the
|
||||||
|
method for its class: `publishKeyboardEvent`, `publishMouseEvent`, or
|
||||||
|
`publishJoystickEvent`. Publishing is a short synchronous `ipc_call` the service answers at
|
||||||
|
once; the service's own fan-out is asynchronous, so publishing never blocks on a slow
|
||||||
|
subscriber.
|
||||||
|
- The **service** ([input.zig](../system/services/input/input.zig)) keeps a small subscriber
|
||||||
|
table (endpoint handle + owning task id + `device_mask`). On `publish` it `ipc_send`s the
|
||||||
|
event to every subscriber whose mask includes the event's device class. On `subscribe` it
|
||||||
|
stores the passed capability and mask and, as housekeeping, prunes any slot whose owning
|
||||||
|
process has exited (checked against `process_enumerate`) — not for correctness (an async
|
||||||
|
send to an orphaned endpoint is harmless) but to reclaim the slot.
|
||||||
|
|
||||||
|
Publisher and subscriber must be **separate processes**: a single thread that both
|
||||||
|
published and serviced its own subscription would deadlock (its `publish` call blocks until
|
||||||
|
the service delivers to its endpoint, which only the same thread could receive).
|
||||||
|
|
||||||
|
## Status and follow-ups
|
||||||
|
|
||||||
|
- **The keyboard is real.** The `ps2-bus` driver owns PNP0303, which carries *both* the
|
||||||
|
0x60/0x64 ports and IRQ1, so reading the hardware lives in the bus, not in
|
||||||
|
[keyboard.zig](../system/drivers/ps2-bus/keyboard.zig): the bus binds IRQ1 and, on each
|
||||||
|
interrupt, drains port 0x60, routing every byte by the status register's
|
||||||
|
auxiliary-output bit to whichever child driver **attached** for that device (an
|
||||||
|
`AttachRequest` to the well-known `ps2_bus` service, carrying the child's endpoint as a
|
||||||
|
capability; the bytes then arrive as asynchronous `ForwardedByte` messages, so the IRQ
|
||||||
|
path never blocks on a child). The keyboard driver decodes the stream — scancode **set 2**,
|
||||||
|
what the keyboard sends with the 8042's legacy translation off, decoded by
|
||||||
|
[scancode.zig](../system/drivers/ps2-bus/scancode.zig) into USB HID usage keycodes with
|
||||||
|
make/break, typematic-repeat, and modifier tracking (host-tested under `zig build test`) —
|
||||||
|
and publishes real `key_down`/`key_press`/`key_up` events.
|
||||||
|
- **Keycode → character** is wired in: the keyboard driver fills a `key_press` event's
|
||||||
|
`character` through [`library/xkeyboard-config`](../library/xkeyboard-config/README.md)
|
||||||
|
(`xkb.map(layout, keycode, mods)` → keysym + Unicode character), synthesizing the ASCII
|
||||||
|
control characters for Enter/Tab/Backspace/Escape, whose keysyms map to no Unicode. The
|
||||||
|
layout defaults to `us`; the bus can pass another as the driver's argv[2] — the seam for
|
||||||
|
a future settings source.
|
||||||
|
- **The mouse is real too.** IRQ12 is enumerated on the auxiliary device's own ACPI node
|
||||||
|
(PNP0F13), so the bus claims that node alongside the controller and routes both IRQs to
|
||||||
|
its one endpoint, acking whichever line the notification's badge names.
|
||||||
|
[mouse.zig](../system/drivers/ps2-bus/mouse.zig) attaches the way the keyboard does and
|
||||||
|
assembles the forwarded bytes with
|
||||||
|
[mouse-packet.zig](../system/drivers/ps2-bus/mouse-packet.zig) (three-byte stream-mode
|
||||||
|
packets: sync/overflow handling, nine-bit movement, screen-convention `dy` — host-tested
|
||||||
|
under `zig build test`) into `button_down`/`button_up` transitions and `motion` events.
|
||||||
|
**Follow-up:** the IntelliMouse magic-knock for a scroll wheel (four-byte packets) and
|
||||||
|
`scroll` events. The hardware-free `input-source` still rotates through all three classes
|
||||||
|
synthetically (including a joystick, which has no driver yet) via the
|
||||||
|
`input.synthetic*Event` helpers.
|
||||||
|
- **Drop-oldest under overflow** is a defined loss; the 16-slot ring absorbs normal bursts.
|
||||||
|
Real backpressure/flow-control is future work.
|
||||||
|
- **`publish` is unauthenticated** — any process may publish, consistent with the current
|
||||||
|
bring-up trust model (see [driver-model.md](driver-model.md)). A source capability is
|
||||||
|
future work.
|
||||||
|
|
||||||
|
## Verifying it
|
||||||
|
|
||||||
|
The `input` case (`python3 test/qemu_test.py input`, in
|
||||||
|
[tests.zig](../system/kernel/tests.zig) `inputTest`) boots the real kernel and spawns the
|
||||||
|
service, the synthetic source (which cycles keyboard, mouse, and joystick events), and a
|
||||||
|
subscriber that took all three classes. It passes only when the subscriber heartbeats
|
||||||
|
`input-test: ok` — proof that an event travelled source → service → subscriber over IPC,
|
||||||
|
exercising `ipc_send`, capability-passing subscription, and per-device routing. Each
|
||||||
|
serial line names the class received, so the log shows all three arriving on one stream.
|
||||||
|
|
||||||
|
## See also
|
||||||
|
|
||||||
|
- [ipc.md](ipc.md) — the synchronous rendezvous and the notification path `ipc_send` extends.
|
||||||
|
- [syscall.md](syscall.md) — the system-call surface, including `ipc_send`.
|
||||||
|
- [driver-model.md](driver-model.md) — class drivers, capability passing (M13), the trust model.
|
||||||
+20
-9
@@ -9,7 +9,7 @@ reboot is miserable.
|
|||||||
|
|
||||||
This is the machinery that catches those faults and prints what happened instead.
|
This is the machinery that catches those faults and prints what happened instead.
|
||||||
It's all x86_64-specific, so it lives behind the [arch](arch.md) boundary in
|
It's all x86_64-specific, so it lives behind the [arch](arch.md) boundary in
|
||||||
`src/kernel/arch/x86_64/`. Only the 32 CPU-defined exception vectors are wired up so far;
|
`system/kernel/architecture/x86_64/`. Only the 32 CPU-defined exception vectors are wired up so far;
|
||||||
device interrupts (timer, keyboard, via the APIC) come later, on the same IDT.
|
device interrupts (timer, keyboard, via the APIC) come later, on the same IDT.
|
||||||
|
|
||||||
## First the GDT
|
## First the GDT
|
||||||
@@ -20,7 +20,7 @@ IDT gate names a code-segment *selector* that must resolve in the current GDT. T
|
|||||||
firmware left a GDT in place, but we don't control it, so we install our own with
|
firmware left a GDT in place, but we don't control it, so we install our own with
|
||||||
known selectors: `0x08` kernel code, `0x10` kernel data.
|
known selectors: `0x08` kernel code, `0x10` kernel data.
|
||||||
|
|
||||||
`src/kernel/arch/x86_64/gdt.zig` holds three flat descriptors — a required null entry,
|
`system/kernel/architecture/x86_64/gdt.zig` holds three flat descriptors — a required null entry,
|
||||||
plus code and data — where the only bits that matter in long mode are the access
|
plus code and data — where the only bits that matter in long mode are the access
|
||||||
byte and the code segment's long-mode (`L`) flag. Loading it (`gdt_flush` in
|
byte and the code segment's long-mode (`L`) flag. Loading it (`gdt_flush` in
|
||||||
`isr.s`) does two things: `lgdt`, then reload the segment registers. The data
|
`isr.s`) does two things: `lgdt`, then reload the segment registers. The data
|
||||||
@@ -33,7 +33,7 @@ into CS:RIP.
|
|||||||
The **Interrupt Descriptor Table** maps each of 256 vectors to a handler. Each
|
The **Interrupt Descriptor Table** maps each of 256 vectors to a handler. Each
|
||||||
entry is a 16-byte *gate* holding the handler's address (split across three
|
entry is a 16-byte *gate* holding the handler's address (split across three
|
||||||
fields, a quirk of the format), the code selector (`0x08`), and flags: `0x8E`
|
fields, a quirk of the format), the code selector (`0x08`), and flags: `0x8E`
|
||||||
means present, ring 0, 64-bit interrupt gate. `src/kernel/arch/x86_64/idt.zig` builds the
|
means present, ring 0, 64-bit interrupt gate. `system/kernel/architecture/x86_64/idt.zig` builds the
|
||||||
table, points the first 32 vectors at their stubs, and loads it with `lidt`
|
table, points the first 32 vectors at their stubs, and loads it with `lidt`
|
||||||
(`idt_flush`).
|
(`idt_flush`).
|
||||||
|
|
||||||
@@ -49,7 +49,7 @@ hit a fault *while trying to deliver another fault* — very often because the
|
|||||||
current stack pointer is bad, so pushing the exception frame itself faulted. If
|
current stack pointer is bad, so pushing the exception frame itself faulted. If
|
||||||
the #DF handler then tried to push onto that same bad stack, it would fault a
|
the #DF handler then tried to push onto that same bad stack, it would fault a
|
||||||
third time and **triple-fault** — an instant reset. So the #DF gate is pointed at
|
third time and **triple-fault** — an instant reset. So the #DF gate is pointed at
|
||||||
**IST1**, a small dedicated stack (`src/kernel/arch/x86_64/tss.zig`) that's always valid.
|
**IST1**, a small dedicated stack (`system/kernel/architecture/x86_64/tss.zig`) that's always valid.
|
||||||
|
|
||||||
Bringing it up: fill in the TSS's IST1 pointer, publish the TSS through a
|
Bringing it up: fill in the TSS's IST1 pointer, publish the TSS through a
|
||||||
descriptor in the GDT (`gdt.setTss`), and load it into the task register with
|
descriptor in the GDT (`gdt.setTss`), and load it into the task register with
|
||||||
@@ -60,7 +60,7 @@ which is why the GDT grew from three entries to five.
|
|||||||
|
|
||||||
On an exception the CPU pushes a small frame (SS, RSP, RFLAGS, CS, RIP) and, for
|
On an exception the CPU pushes a small frame (SS, RSP, RFLAGS, CS, RIP) and, for
|
||||||
*some* vectors, an **error code**. That inconsistency is a nuisance, so each stub
|
*some* vectors, an **error code**. That inconsistency is a nuisance, so each stub
|
||||||
in `src/kernel/arch/x86_64/isr.s` normalises it: vectors that don't get a hardware error
|
in `system/kernel/architecture/x86_64/isr.s` normalises it: vectors that don't get a hardware error
|
||||||
code push a dummy `0`, then every stub pushes its **vector number** and jumps to a
|
code push a dummy `0`, then every stub pushes its **vector number** and jumps to a
|
||||||
shared tail, `isr_common`. The tail pushes all the general registers and calls the
|
shared tail, `isr_common`. The tail pushes all the general registers and calls the
|
||||||
Zig handler with a pointer to the whole thing.
|
Zig handler with a pointer to the whole thing.
|
||||||
@@ -80,11 +80,22 @@ inline). `build.zig` adds `isr.s` to the arch module.
|
|||||||
## Reporting a fault
|
## Reporting a fault
|
||||||
|
|
||||||
`isr_common` calls `exceptionHandler`, which forwards to a swappable `on_fault`
|
`isr_common` calls `exceptionHandler`, which forwards to a swappable `on_fault`
|
||||||
hook. The generic kernel installs a reporter (`onException` in `main.zig`) that
|
hook. The generic kernel installs a reporter (`onException` in `kernel.zig`) that
|
||||||
prints, in red, the exception name and vector, the error code, the faulting RIP
|
prints the exception name and vector, the error code, the faulting RIP
|
||||||
and RSP, and — for a page fault (#PF, vector 14) — the faulting address from
|
and RSP, and — for a page fault (#PF, vector 14) — the faulting address from
|
||||||
**CR2**. Then it halts. There's no fault *recovery* yet, so every exception is
|
**CR2**. What happens next depends on where the fault came from:
|
||||||
terminal; the point is that it's now **visible** instead of a silent reset.
|
|
||||||
|
- **User mode (CPL 3): kill the process, keep the machine.** The kernel is intact
|
||||||
|
(the CPU trapped onto the task's kernel stack), so the faulting process is
|
||||||
|
killed — address space, IRQ bindings, and IPC handles reclaimed; a client it
|
||||||
|
owed a reply to is failed with `-EPEER` — and the core reschedules. A crashing
|
||||||
|
driver takes itself down, never the OS. This is fault recovery step 2 of
|
||||||
|
[resilience.md](resilience.md). NMI, double fault, and machine check are
|
||||||
|
excluded: they report machine trouble regardless of what was running.
|
||||||
|
- **Kernel mode: halt this core.** The trusted base itself is broken, so there is
|
||||||
|
nothing safe to kill; the fault is still *contained* to the core (an
|
||||||
|
application-processor fault leaves the rest of the system running), and the
|
||||||
|
report makes it **visible** instead of a silent reset.
|
||||||
|
|
||||||
The hook is set before `arch.init()` in `kmain`, so a fault during setup is still
|
The hook is set before `arch.init()` in `kmain`, so a fault during setup is still
|
||||||
caught.
|
caught.
|
||||||
|
|||||||
+79
-12
@@ -6,12 +6,20 @@ just call each other — a request becomes a **message**. In a microkernel, what
|
|||||||
was a function call across a monolithic kernel is IPC, so it's a first-class
|
was a function call across a monolithic kernel is IPC, so it's a first-class
|
||||||
concern, not an afterthought.
|
concern, not an afterthought.
|
||||||
|
|
||||||
This first form is a **bounded blocking channel** (`src/kernel/ipc.zig`): a fixed-size
|
There are two layers, built a milestone apart:
|
||||||
ring buffer of messages with a producer/consumer rendezvous, built on the
|
|
||||||
scheduler's [wait queues](scheduling.md).
|
- **`system/kernel/ipc.zig`** — a bounded blocking channel between *kernel threads*,
|
||||||
|
described below. The primitive, and where the blocking discipline was worked out.
|
||||||
|
- **`system/kernel/ipc-synchronous.zig`** — synchronous call/reply between *processes*, across
|
||||||
|
address spaces. What user-space servers and drivers actually talk over. It's the
|
||||||
|
second half of this document.
|
||||||
|
|
||||||
## The channel
|
## The channel
|
||||||
|
|
||||||
|
The first form is a **bounded blocking channel** (`system/kernel/ipc.zig`): a fixed-size
|
||||||
|
ring buffer of messages with a producer/consumer rendezvous, built on the
|
||||||
|
scheduler's [wait queues](scheduling.md).
|
||||||
|
|
||||||
`Channel(T, capacity)` is generic over the message type and buffer size. It holds a
|
`Channel(T, capacity)` is generic over the message type and buffer size. It holds a
|
||||||
ring buffer, a count, and two wait queues:
|
ring buffer, a count, and two wait queues:
|
||||||
|
|
||||||
@@ -42,16 +50,75 @@ full and empty over and over, so both the blocking-send and blocking-recv paths
|
|||||||
exercised heavily. The messages arrive intact and in order (their sum is the
|
exercised heavily. The messages arrive intact and in order (their sum is the
|
||||||
expected `5050`), and neither task busy-waits — they block and wake each other.
|
expected `5050`), and neither task busy-waits — they block and wake each other.
|
||||||
|
|
||||||
|
## Endpoints: call/reply across address spaces
|
||||||
|
|
||||||
|
A channel connects two kernel threads sharing one address space. Real servers are
|
||||||
|
*processes*, so the payload has to cross an address-space boundary. That's
|
||||||
|
`system/kernel/ipc-synchronous.zig`, and its shape is L4's: a synchronous **rendezvous** at an
|
||||||
|
`Endpoint`, with the message copied directly from the sender's pages to the receiver's
|
||||||
|
(`copyAcross` walks both sets of page tables through the physmap — no CR3 switch, no
|
||||||
|
bounce buffer).
|
||||||
|
|
||||||
|
Two syscalls carry it:
|
||||||
|
|
||||||
|
- **`ipc_call(h, msg, reply)`** — copy `msg` to the server, block until it replies.
|
||||||
|
- **`ipc_reply_wait(h, reply, recv)`** — reply to the client you're still holding (if
|
||||||
|
any), then block for the next request. One syscall, because a server's steady state
|
||||||
|
is *always* "finish the last one, wait for the next".
|
||||||
|
|
||||||
|
An endpoint is reached by **handle** — a small integer index into the process's handle
|
||||||
|
table (`Task.handles`), exactly like a file descriptor, and just as unforgeable. The
|
||||||
|
bootstrap problem (how do you get the first handle?) is solved by a tiny name registry:
|
||||||
|
a server calls `ipc_register(service_id, h)` under a well-known small integer, and a
|
||||||
|
client calls `ipc_lookup(service_id)`.
|
||||||
|
|
||||||
|
The server never learns the client's identity beyond a **badge**, delivered alongside
|
||||||
|
the message: the caller's task id.
|
||||||
|
|
||||||
|
### Interrupts are messages too
|
||||||
|
|
||||||
|
`notifyFromIsr` posts an *asynchronous* notification to an endpoint — no payload, no
|
||||||
|
reply owed — and wakes whoever is blocked in `reply_wait`. Its badge has the top bit
|
||||||
|
set (`notify_badge_bit`), which is how a driver's single event loop distinguishes "a
|
||||||
|
client wants something" from "the hardware wants something". Notifications sit in a
|
||||||
|
small coalescing ring on the endpoint, so an interrupt taken while the driver was busy
|
||||||
|
elsewhere is not lost.
|
||||||
|
|
||||||
|
This is what makes a user-space driver possible at all, and it's the subject of
|
||||||
|
[drivers.md](drivers.md).
|
||||||
|
|
||||||
## What's next (not done here)
|
## What's next (not done here)
|
||||||
|
|
||||||
- **Across address spaces.** Today both endpoints are kernel threads sharing the
|
|
||||||
kernel's memory, so the message is copied within one address space. When user
|
|
||||||
mode arrives, the same channel carries messages between *isolated* processes,
|
|
||||||
copying the payload across the boundary — which is where IPC earns its place as
|
|
||||||
the microkernel's backbone.
|
|
||||||
- **Synchronous call/reply.** A request/response pattern (send-and-wait-for-reply)
|
|
||||||
on top of channels, the shape most driver/service calls take.
|
|
||||||
- **Interrupts as messages.** A hardware interrupt delivered to the driver task
|
|
||||||
that owns the device, as an IPC message.
|
|
||||||
- **Priority inheritance** through IPC, so a high-priority client blocked on a
|
- **Priority inheritance** through IPC, so a high-priority client blocked on a
|
||||||
low-priority server doesn't suffer unbounded priority inversion.
|
low-priority server doesn't suffer unbounded priority inversion.
|
||||||
|
- **Handle transfer.** A server can't hand a client a handle to a third endpoint, so
|
||||||
|
every capability is either well-known (the registry) or inherited — there's no way
|
||||||
|
to delegate one.
|
||||||
|
- **Asynchronous / buffered send** for the cases where a rendezvous is the wrong
|
||||||
|
shape (logging, notifications between servers). *Landed as `ipc_send`* — a
|
||||||
|
non-blocking post to an endpoint's bounded payload queue, delivered through
|
||||||
|
`reply_wait` as a buffered message (badge bit `notify_message_bit`). Built for, and
|
||||||
|
first used by, the [input service](input.md)'s keyboard-event broadcast, where a
|
||||||
|
synchronous push would let one dead subscriber hang the fan-out. A full queue drops
|
||||||
|
the oldest (discrete messages, not a coalescing level like the notification ring).
|
||||||
|
- **A bounded reply.** `MSG_MAX` is 256 bytes and the copy runs under the big kernel
|
||||||
|
lock; a bulk transfer wants shared pages, not a copy.
|
||||||
|
|
||||||
|
## Lifecycle conventions over IPC (M17)
|
||||||
|
|
||||||
|
Three conventions from [process-lifecycle.md](process-lifecycle.md) ride the
|
||||||
|
notification mechanism:
|
||||||
|
|
||||||
|
- **Signals** arrive as notifications on the endpoint a process nominated with
|
||||||
|
`signal_bind` (`runtime.process.bindSignals`): badge = the signal bit plus the
|
||||||
|
coalesced pending mask (`runtime.process.signalsFrom` decodes). Statements,
|
||||||
|
never questions; no payload, no reply.
|
||||||
|
- **One-shot timers** (`timer_bind`, `runtime.system.timerOnce`) land as a
|
||||||
|
timer-bit notification — the timed wait: a service arms a deadline and keeps
|
||||||
|
serving, instead of blocking in sleep.
|
||||||
|
- **The universal ping**: a **zero-length request is the liveness probe**,
|
||||||
|
answered with a zero-length reply by the service harness itself
|
||||||
|
(`runtime.service.run`). No protocol's requests start at length zero, so the
|
||||||
|
encoding cannot collide, and a wedged service simply fails to answer — which
|
||||||
|
is the diagnosis. Deep health ("can I reach my hardware?") stays a per-service
|
||||||
|
protocol message.
|
||||||
|
|||||||
+3
-3
@@ -13,7 +13,7 @@ kernel follows.
|
|||||||
|
|
||||||
## The log is multi-sink
|
## The log is multi-sink
|
||||||
|
|
||||||
`src/kernel/log.zig` is the diagnostic log. It fans a message out to a set of
|
`system/kernel/log.zig` is the diagnostic log. It fans a message out to a set of
|
||||||
registered **sinks**, each best-effort and self-guarding:
|
registered **sinks**, each best-effort and self-guarding:
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
@@ -36,7 +36,7 @@ Properties that matter:
|
|||||||
## The framebuffer is *not* a log sink
|
## The framebuffer is *not* a log sink
|
||||||
|
|
||||||
The framebuffer is a general graphics surface, **not inherently a text terminal**.
|
The framebuffer is a general graphics surface, **not inherently a text terminal**.
|
||||||
Today `src/kernel/console.zig` paints a text grid on it as a *bootstrap* console, but
|
Today `system/kernel/console.zig` paints a text grid on it as a *bootstrap* console, but
|
||||||
that's a stop-gap: once the driver machinery exists the framebuffer becomes a proper
|
that's a stop-gap: once the driver machinery exists the framebuffer becomes a proper
|
||||||
**graphics device driver**, and the text crutch goes away. So the log must not assume
|
**graphics device driver**, and the text crutch goes away. So the log must not assume
|
||||||
it — routing the verbose log through a pixel console would bake in "the OS is text".
|
it — routing the verbose log through a pixel console would bake in "the OS is text".
|
||||||
@@ -58,7 +58,7 @@ screen. `console.write` is a no-op when the firmware gave us no framebuffer.
|
|||||||
A framebuffer is not guaranteed — a headless server exposes no UEFI Graphics Output
|
A framebuffer is not guaranteed — a headless server exposes no UEFI Graphics Output
|
||||||
Protocol. That used to be *fatal* (the loader failed the boot). Now the loader hands
|
Protocol. That used to be *fatal* (the loader failed the boot). Now the loader hands
|
||||||
over a "no framebuffer" descriptor (`base == 0`) rather than failing, and
|
over a "no framebuffer" descriptor (`base == 0`) rather than failing, and
|
||||||
`Framebuffer.present()` (in `src/root.zig`) gates every on-screen path. A headless,
|
`Framebuffer.present()` (in `system/boot-handoff.zig`) gates every on-screen path. A headless,
|
||||||
serial-less machine boots and runs correctly — it just goes quiet.
|
serial-less machine boots and runs correctly — it just goes quiet.
|
||||||
|
|
||||||
## Last-resort channels (no text output at all)
|
## Last-resort channels (no text output at all)
|
||||||
|
|||||||
+3
-3
@@ -27,7 +27,7 @@ danos's own neutral format, and the kernel only ever sees that.**
|
|||||||
|
|
||||||
## The neutral format
|
## The neutral format
|
||||||
|
|
||||||
Defined in `src/root.zig`, the shared loader↔kernel contract:
|
Defined in `system/boot-handoff.zig`, the shared loader↔kernel contract:
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
pub const MemoryKind = enum(u32) {
|
pub const MemoryKind = enum(u32) {
|
||||||
@@ -68,7 +68,7 @@ pub const BootInfo = extern struct {
|
|||||||
|
|
||||||
## The loader side (UEFI)
|
## The loader side (UEFI)
|
||||||
|
|
||||||
Two functions in `src/boot/efi.zig`, called from `exitBootServices`:
|
Two functions in `boot/efi.zig`, called from `exitBootServices`:
|
||||||
|
|
||||||
- **`classify`** maps each UEFI descriptor to a `MemoryKind`:
|
- **`classify`** maps each UEFI descriptor to a `MemoryKind`:
|
||||||
`conventional_memory` **and** `boot_services_code`/`boot_services_data → usable`;
|
`conventional_memory` **and** `boot_services_code`/`boot_services_data → usable`;
|
||||||
@@ -121,7 +121,7 @@ The kernel receives a plain array and reads it with zero UEFI knowledge:
|
|||||||
|
|
||||||
```zig
|
```zig
|
||||||
const mm = boot_info.memory_map;
|
const mm = boot_info.memory_map;
|
||||||
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
const regions = @as([*]const system.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
||||||
for (regions) |r| {
|
for (regions) |r| {
|
||||||
if (r.kind == .usable) usable_pages += r.pages;
|
if (r.kind == .usable) usable_pages += r.pages;
|
||||||
}
|
}
|
||||||
|
|||||||
+49
-14
@@ -7,7 +7,7 @@ which live in memory we'd like to reclaim and don't control), switches CR3 onto
|
|||||||
them, and — crucially — maps with **real permissions**.
|
them, and — crucially — maps with **real permissions**.
|
||||||
|
|
||||||
It's x86_64-specific (the 4-level table format is an Intel/AMD thing), so it lives
|
It's x86_64-specific (the 4-level table format is an Intel/AMD thing), so it lives
|
||||||
behind the [arch](arch.md) boundary in `src/kernel/arch/x86_64/paging.zig`.
|
behind the [arch](arch.md) boundary in `system/kernel/architecture/x86_64/paging.zig`.
|
||||||
|
|
||||||
## The format
|
## The format
|
||||||
|
|
||||||
@@ -17,20 +17,54 @@ the final 4 KiB page. Each entry holds a physical address plus flag bits —
|
|||||||
present, writable, and (bit 63) **no-execute**. danos maps everything with 4 KiB
|
present, writable, and (bit 63) **no-execute**. danos maps everything with 4 KiB
|
||||||
pages: precise, and the extra table memory is negligible against available RAM.
|
pages: precise, and the extra table memory is negligible against available RAM.
|
||||||
|
|
||||||
|
## Higher half: the address-space layout
|
||||||
|
|
||||||
|
danos is a **higher-half kernel**. The kernel is linked to run at
|
||||||
|
`0xFFFF_FFFF_8000_0000` but loaded low (the linker script's `AT()` gives each
|
||||||
|
segment a physical load address at 1 MiB up; the bootloader maps the high link
|
||||||
|
address to the low load address in its bootstrap tables and jumps in). The entire
|
||||||
|
**low canonical half is reserved for user space**; the kernel lives in the top half
|
||||||
|
alongside a **physmap** — a straight window onto all of physical memory at
|
||||||
|
`physmap_base + phys`. Wherever the kernel needs to touch a physical address (a
|
||||||
|
page-table frame, an ACPI table, a device register), it adds that constant:
|
||||||
|
`system.physToVirt(phys)`. The layout constants live in `system/boot-handoff.zig`:
|
||||||
|
|
||||||
|
| region | virtual base | PML4 slot |
|
||||||
|
|--------|--------------|-----------|
|
||||||
|
| user image + stack | `0x0000_7000_0000_0000` | 224 (low half) |
|
||||||
|
| kernel heap | `0xFFFF_8000_0000_0000` | 256 |
|
||||||
|
| physmap (all RAM + MMIO windows) | `0xFFFF_8800_0000_0000` + phys | 272 |
|
||||||
|
| kernel image | `0xFFFF_FFFF_8000_0000` | 511 |
|
||||||
|
|
||||||
|
The bootloader builds temporary **bootstrap tables** (identity + a 4 GiB physmap +
|
||||||
|
the high kernel) so it can switch CR3 and jump to the high entry; the kernel then
|
||||||
|
builds its own precise tables below and abandons them. Because both use the same
|
||||||
|
`physmap_base`, any physmap pointer minted before the switch stays valid after it.
|
||||||
|
|
||||||
## What gets mapped, and with what permissions
|
## What gets mapped, and with what permissions
|
||||||
|
|
||||||
The address space is built in three passes (`init`):
|
The address space is built in four passes (`init`):
|
||||||
|
|
||||||
1. **All RAM, identity-mapped RW + NX.** Every non-MMIO region from the
|
1. **All RAM in the physmap, RW + NX.** Every non-MMIO region from the
|
||||||
[memory map](memory-map.md) is mapped virtual == physical, read-write and
|
[memory map](memory-map.md) is mapped at `physToVirt(phys)`, read-write and
|
||||||
*non-executable*. Identity mapping keeps everything already running valid across
|
*non-executable*. There is **no low/identity mapping** — the low half is user
|
||||||
the CR3 switch (the frame allocator addresses frames by physical address, page
|
space. (Frames the kernel touches while still building these tables are reached
|
||||||
tables are reached the same way, the stack stays put).
|
through the loader's bootstrap physmap, which covers the low 4 GiB; both the
|
||||||
2. **The framebuffer and the Local APIC**, the device memory we actually touch,
|
frame allocator and the table builder scan low-address-up, so those frames stay
|
||||||
also RW + NX. Everything else — unbacked address space, other MMIO — is simply
|
under that limit.)
|
||||||
left unmapped, so a stray access faults instead of silently succeeding.
|
2. **The framebuffer and the Local APIC**, the device memory the kernel touches
|
||||||
3. **The kernel's own segments, overlaid with their true ELF permissions.** This is
|
directly, as physmap windows (RW + NX). Other MMIO is mapped on demand by
|
||||||
|
`mapMmio`, also into the physmap; everything else is left unmapped, so a stray
|
||||||
|
access faults instead of silently succeeding.
|
||||||
|
3. **The kernel's own segments, overlaid with their true ELF permissions**, at
|
||||||
|
their high link addresses mapped to their low physical load addresses. This is
|
||||||
the interesting part.
|
the interesting part.
|
||||||
|
4. **Every higher-half PML4 entry pre-created** (an empty PDPT where none exists
|
||||||
|
yet). The kernel half is then a fixed set of top-level slots, so a per-process
|
||||||
|
address space can share it by copying `PML4[256..512)` once — growth beneath
|
||||||
|
those slots (heap, on-demand MMIO) propagates to every address space because
|
||||||
|
they share the PDPTs. `init` asserts no new higher-half PML4 entry appears
|
||||||
|
afterward.
|
||||||
|
|
||||||
### W^X from the ELF program headers
|
### W^X from the ELF program headers
|
||||||
|
|
||||||
@@ -55,9 +89,10 @@ reserved bit and fault.
|
|||||||
|
|
||||||
### The null guard
|
### The null guard
|
||||||
|
|
||||||
Page 0 is deliberately left unmapped. A null (or near-null) pointer dereference now
|
The whole low half is unmapped except for explicit user mappings, so page 0 (and
|
||||||
takes a page fault instead of quietly reading or writing real memory — turning a
|
every near-null address) is unmapped by construction. A null (or near-null) pointer
|
||||||
whole class of silent bugs into an immediate, located crash.
|
dereference in the kernel takes a page fault instead of quietly reading or writing
|
||||||
|
real memory — turning a whole class of silent bugs into an immediate, located crash.
|
||||||
|
|
||||||
## Switching on, and the on-demand API
|
## Switching on, and the on-demand API
|
||||||
|
|
||||||
|
|||||||
+128
@@ -0,0 +1,128 @@
|
|||||||
|
# The power service: events and shutdown
|
||||||
|
|
||||||
|
A laptop lid closes, a battery drains, someone presses the power button — and
|
||||||
|
several parts of the system might care: a session manager dims the screen, a
|
||||||
|
logger notes it, and ultimately *something* has to turn the machine off. None of
|
||||||
|
them owns the hardware that reported the event, and the reporter should not know
|
||||||
|
who is listening. So system power is a **service**: an event source **publishes**
|
||||||
|
button/lid/battery/AC events, interested processes **subscribe**, and one
|
||||||
|
privileged caller — init — can ask it to power the machine off. It is the same
|
||||||
|
publish/subscribe shape as the [input service](input.md), applied to power.
|
||||||
|
|
||||||
|
## Why a service, and why it is named for the domain, not the firmware
|
||||||
|
|
||||||
|
Where the events come from is firmware-specific — on x86 they ride the ACPI SCI
|
||||||
|
([acpi.md](acpi.md)); on a Raspberry Pi they would come from PSCI or a mailbox.
|
||||||
|
What subscribers want is not: *the lid closed* means the same thing regardless of
|
||||||
|
who noticed. So the surface is **domain-named**. There is a `power-protocol`
|
||||||
|
module and a well-known `ServiceId.power = 5`; on x86 the **acpi service**
|
||||||
|
registers it, and on ARM a PSCI/mailbox service will register the *same* id.
|
||||||
|
Subscribers call `runtime.ipc.lookup(.power)` and never learn which firmware they
|
||||||
|
are on — the neutrality the whole [discovery](discovery.md) migration exists to
|
||||||
|
preserve, carried one layer up into a running-system surface.
|
||||||
|
|
||||||
|
This is why the protocol is `power`, not "ACPI events": naming a cross-firmware
|
||||||
|
surface after one firmware would leak x86 into code the ARM port must reuse
|
||||||
|
unchanged.
|
||||||
|
|
||||||
|
## The protocol
|
||||||
|
|
||||||
|
The `power-protocol` module ([system/services/power/protocol.zig](../system/services/power/protocol.zig))
|
||||||
|
follows the vfs-protocol pattern — extern-struct messages, a version, reserved
|
||||||
|
fields. Three operations:
|
||||||
|
|
||||||
|
| Direction | Operation | Purpose |
|
||||||
|
|---|---|---|
|
||||||
|
| subscriber → service | `subscribe` | receive published events; the subscriber's endpoint rides as the call's **capability** (the input/device-manager pattern) |
|
||||||
|
| init → service | `shutdown` | orderly shutdown's last step: enter S5 (soft off) |
|
||||||
|
| service → subscriber | `event` | a published `EventMessage`, delivered as a buffered message (never sent *to* the service) |
|
||||||
|
|
||||||
|
Events are published, not polled: like the input service, the service holds
|
||||||
|
subscriber endpoints as capabilities and `ipc_send`s each event as a buffered
|
||||||
|
message, so a slow or dead subscriber can never wedge the source. The event
|
||||||
|
vocabulary is hardware-neutral:
|
||||||
|
|
||||||
|
- `power_button` — the button was pressed (a fixed ACPI event on x86).
|
||||||
|
- `lid`, `ac`, `battery` — the named GPE-driven events.
|
||||||
|
- `notify` — a device notification that maps to none of the above; its `code`
|
||||||
|
(the ACPI `Notify` argument) and the notifying device's `hid` say which device
|
||||||
|
and what happened.
|
||||||
|
|
||||||
|
An `EventMessage` carries the `event` tag plus `code` and an 8-byte `hid`, so a
|
||||||
|
generic `notify` is fully described without a second round trip.
|
||||||
|
|
||||||
|
**`shutdown` is authority, not information.** It is the only operation that
|
||||||
|
*does* something irreversible, so it is gated: the contract is that only init
|
||||||
|
(PID 1) may request it, because init is the process that has already run the stop
|
||||||
|
sequence over everything else. The acpi service implements this as a **soft
|
||||||
|
gate** — it honors `shutdown` only from a process that is a *subscriber*, and
|
||||||
|
init is the one subscriber. That stands in for "only the system supervisor may
|
||||||
|
power off" without hard-coding a pid, so it still holds under tests where PID 1
|
||||||
|
is not init.
|
||||||
|
|
||||||
|
## Orderly shutdown
|
||||||
|
|
||||||
|
Powering off cleanly is where the power service, the [process
|
||||||
|
lifecycle](process-lifecycle.md), and [ACPI events](acpi.md) compose. init
|
||||||
|
already supervises the services it starts; for shutdown it runs **one event loop
|
||||||
|
over one endpoint** that carries three things at once: its children's exit
|
||||||
|
notifications, the lifecycle **signals** it can receive (`terminate`), and the
|
||||||
|
**power events** it subscribes to — plus a re-arming heartbeat timer proving PID
|
||||||
|
1 is alive. (init subscribes with retries, because the power service registers
|
||||||
|
`.power` well after init starts; a missing power service is not fatal — a
|
||||||
|
`terminate` signal drives the same path.)
|
||||||
|
|
||||||
|
On a `power_button` event or a `terminate` signal, init:
|
||||||
|
|
||||||
|
1. logs that it is shutting down,
|
||||||
|
2. runs the standard stop sequence — `runtime.process.stop(child, deadline,
|
||||||
|
endpoint)` — over its children **in reverse spawn order**, so the VFS stops
|
||||||
|
last (other services may flush through it), each child getting the
|
||||||
|
*terminate → deadline → kill* escalation from
|
||||||
|
[process-lifecycle.md](process-lifecycle.md), and
|
||||||
|
3. requests `.power` `shutdown`.
|
||||||
|
|
||||||
|
The service then enters **S5** (soft off) by writing `SLP_TYP | SLP_EN` to the
|
||||||
|
PM1 control register(s) from ring 3, mirroring the kernel's own
|
||||||
|
`system/devices/power.zig` `sleepValue`. If the write returns instead of powering
|
||||||
|
the machine off, it logs loudly so a test fails rather than hangs.
|
||||||
|
|
||||||
|
**No new system call was needed for S5.** The broad io_port grant on the
|
||||||
|
`acpi-tables` node ([discovery.md](discovery.md)) already put the PM1 control
|
||||||
|
ports in the acpi service's hands, so writing S5 from ring 3 is something it
|
||||||
|
could physically already do; formalizing it as a protocol operation added a
|
||||||
|
contract, not authority. The kernel keeps `power.zig` for its own test paths and
|
||||||
|
panic-time poweroff, where no user space is available to ask.
|
||||||
|
|
||||||
|
## Verifying it
|
||||||
|
|
||||||
|
Two QEMU scenarios exercise the path, both injecting a real ACPI power-button
|
||||||
|
press via QMP `system_powerdown` (there is no other deterministic power event on
|
||||||
|
this config):
|
||||||
|
|
||||||
|
- `power-button` proves the source: the acpi service's SCI handler logs the
|
||||||
|
press and publishes `power_button` (the ACPI half is in [acpi.md](acpi.md)).
|
||||||
|
- `orderly-shutdown` proves the whole composition: button → init logs shutting
|
||||||
|
down → children stopped → the service enters S5 → QEMU exits. The ordered
|
||||||
|
regex is the proof, and QEMU's self-exit through S5 is the pass.
|
||||||
|
|
||||||
|
## Scope
|
||||||
|
|
||||||
|
Interface-complete but validated on real hardware (the author's laptop) later,
|
||||||
|
because QEMU does not emulate them: battery `_BST`/`_BIF` evaluation beyond the
|
||||||
|
interface stubs, lid and AC events, and the embedded controller's `_Qxx`
|
||||||
|
queries. Deliberately out of scope for now: reboot over the power protocol, S3
|
||||||
|
sleep, per-device D-states (a future lifecycle-vocabulary extension, since
|
||||||
|
"suspend" has the shape of a signal every driver must answer and has no consumer
|
||||||
|
until laptop sleep), and thermal zones.
|
||||||
|
|
||||||
|
## See also
|
||||||
|
|
||||||
|
- [acpi.md](acpi.md) — where the events come from on x86: the SCI, the power
|
||||||
|
button fixed event, and GPE/Notify dispatch in the acpi service.
|
||||||
|
- [discovery.md](discovery.md) — why the surface is domain-named, and the
|
||||||
|
firmware neutrality that makes a PSCI backend drop-in on ARM.
|
||||||
|
- [process-lifecycle.md](process-lifecycle.md) — the stop sequence
|
||||||
|
(`terminate → deadline → kill`) and signals init composes into shutdown.
|
||||||
|
- [device-manager.md](device-manager.md) — the supervision model init mirrors for
|
||||||
|
its own children.
|
||||||
@@ -0,0 +1,327 @@
|
|||||||
|
# Process lifecycle: signals over IPC
|
||||||
|
|
||||||
|
**Status: increments 1–4 built** (2026-07-12): claim release on death, exit
|
||||||
|
reasons, published exit events, and signals + one-shot timers + the service
|
||||||
|
harness are all in — the interface below is as-built. The primitives underneath
|
||||||
|
predate this design ([process-management.md](process-management.md):
|
||||||
|
spawn, the supervision link, kill, child-exit notifications); this document designs
|
||||||
|
the layer above them — the standard vocabulary a danos process speaks about its own
|
||||||
|
life, and the stable `runtime.process` interface that carries it. Nothing here is
|
||||||
|
device- or driver-specific: a driver, the VFS, and a user application all stop,
|
||||||
|
reload, and die the same way. The device manager is simply this design's first
|
||||||
|
serious customer ([device-manager.md](device-manager.md)).
|
||||||
|
|
||||||
|
**"POSIX" in this document means the concepts, never the letter of the standard.**
|
||||||
|
danos borrows the ideas and the hard-won lessons (what SIGTERM *means*, why SIGPIPE
|
||||||
|
was a mistake) without inheriting the mechanism, the API, or the names. The naming
|
||||||
|
rule is danos's own and it is strict: plain words that communicate intent
|
||||||
|
(`terminate`, `reload`, `exited`) and the IPC vocabulary the system already speaks
|
||||||
|
(`bind`, `subscribe`, `publish`, `endpoint`) — never `SIG*`, never a second word for
|
||||||
|
a concept that already has one. Literal POSIX arrives later and lives elsewhere: the
|
||||||
|
`std.os.danos` seam that makes danos a Zig target, and eventually a **musl-based C
|
||||||
|
layer** on the same native surface (see [zig-self-hosting.md](zig-self-hosting.md)) —
|
||||||
|
musl's syscall surface retargeted at danos system calls and IPC protocols (files onto
|
||||||
|
the VFS protocol, `sigaction`/`wait` onto this lifecycle, sockets onto whatever
|
||||||
|
networking becomes). Ported programs see POSIX; the system underneath never does.
|
||||||
|
|
||||||
|
## Why a standard vocabulary
|
||||||
|
|
||||||
|
A supervisor can only manage processes it has never heard of if "please exit" means
|
||||||
|
the same thing to all of them. That is the one thing POSIX signals got deeply right:
|
||||||
|
`SIGTERM` means the same thing to nginx and to a five-line script, which is why
|
||||||
|
process supervision on Unix (init systems, container runtimes) is possible at all.
|
||||||
|
danos wants that property from day one, because supervision-and-restart is the
|
||||||
|
system's core motivation ([resilience.md](resilience.md)).
|
||||||
|
|
||||||
|
What POSIX got wrong — for a system like this — is the **delivery mechanism**:
|
||||||
|
asynchronous control-flow hijack. A Unix handler runs on a stolen stack at an
|
||||||
|
arbitrary instruction boundary, which is why the async-signal-safe function list
|
||||||
|
exists, why `errno` must be saved, and why the canonical signal bug is a SIGTERM
|
||||||
|
handler innocently calling `printf` mid-`malloc`. That entire bug class comes from
|
||||||
|
the mechanism, not the vocabulary, and none of it is worth importing.
|
||||||
|
|
||||||
|
A microkernel already has the right channel: **a signal is a message.** QNX delivers
|
||||||
|
POSIX signals over its message passing; seL4 has notification objects; Erlang turned
|
||||||
|
"death is a message to whoever linked" into a reliability philosophy. danos has
|
||||||
|
already done it once without naming it: a child's death arrives as a notification
|
||||||
|
badge on the supervisor's endpoint — the microkernel's SIGCHLD, the IRQ-as-IPC
|
||||||
|
pattern reused. Signals are the same pattern reused a third time.
|
||||||
|
|
||||||
|
## The mechanism
|
||||||
|
|
||||||
|
- **`signal_bind(endpoint)`** — a process nominates the endpoint its signals arrive
|
||||||
|
on, exactly as `irq_bind` nominates where a device's interrupts land. The runtime
|
||||||
|
does this at startup for any program that opts in.
|
||||||
|
- **`process_signal(id, signal)`** — posts the signal as an asynchronous
|
||||||
|
notification to the target's bound endpoint: badge = `notify_badge_bit |
|
||||||
|
notify_signal_bit | pending signals`. Non-blocking for the sender, always.
|
||||||
|
- **Pending signals coalesce** in a per-process bitmask until the target next waits
|
||||||
|
— exactly like interrupt notifications, and exactly POSIX's own semantics for
|
||||||
|
non-realtime signals (two pending SIGTERMs are one SIGTERM). The bitmask *is* the
|
||||||
|
design: signals carry no payload. Anything with a payload is a protocol message.
|
||||||
|
- **Authority**: the supervisor may signal its children — the same link that is
|
||||||
|
already the kill authority. A process may signal itself. Anything broader waits
|
||||||
|
for transferable process handles.
|
||||||
|
- **No binding, no problem**: a process that never calls `signal_bind` is not
|
||||||
|
broken — its signals pend unread and only `process_kill` works on it. Simple
|
||||||
|
programs stay simple; the vocabulary is opt-in, the kill authority is not.
|
||||||
|
|
||||||
|
Because delivery is a message into the process's own event loop, there is no
|
||||||
|
async-signal-safe list in danos: a handler is ordinary code running at a point the
|
||||||
|
process chose. The bug class is gone by construction, not by discipline.
|
||||||
|
|
||||||
|
## The vocabulary: POSIX.1-1990, sorted honestly
|
||||||
|
|
||||||
|
The full 1990 set, and what each becomes. Two intrinsically problematic cases get a
|
||||||
|
defense below the table.
|
||||||
|
|
||||||
|
| POSIX.1-1990 | danos disposition | Notes |
|
||||||
|
|---|---|---|
|
||||||
|
| SIGTERM | signal `terminate` | finish up and exit; the supervisor's polite half |
|
||||||
|
| SIGHUP | signal `reload` | re-read configuration / re-scan |
|
||||||
|
| SIGINT | signal `interrupt` | interactive interrupt; meaningful once a console can send it, in the vocabulary now so numbering is stable |
|
||||||
|
| SIGQUIT | signal `quit` | as SIGINT, without the core-dump baggage |
|
||||||
|
| SIGALRM | signal `alarm` | timer expiry as a message; the Unix SIGALRM+`longjmp` timeout hacks are impossible here. In the vocabulary, unbuilt: no consumer yet, and when one appears it is runtime sugar over the existing timer — zero kernel work |
|
||||||
|
| SIGUSR1, SIGUSR2 | signals `user_1`, `user_2` | service-defined |
|
||||||
|
| SIGCHLD | **already exists** — the exit notification | the badge carries the child id, dodging the classic coalescing bug (Unix code must loop `waitpid`) |
|
||||||
|
| SIGKILL | `process_kill` — kernel mechanism | its definition is "cannot be handled"; it was never really a signal |
|
||||||
|
| SIGABRT | exit reason `abort` | `abort()` is synchronous self-termination, not an event |
|
||||||
|
| SIGSEGV, SIGILL, SIGFPE | exit reasons, **never delivered** | see below |
|
||||||
|
| SIGPIPE | **an error return**, not a signal | see below |
|
||||||
|
| SIGSTOP, SIGTSTP, SIGTTIN, SIGTTOU, SIGCONT | deferred | job control needs terminals, sessions, and process groups; stop/continue is scheduler territory |
|
||||||
|
|
||||||
|
**The fault signals (SIGSEGV, SIGILL, SIGFPE) are intrinsically wrong for messages.**
|
||||||
|
They are *synchronous* — raised at a specific faulting instruction, not "sometime
|
||||||
|
soon". A message cannot be delivered to a process whose next instruction re-faults;
|
||||||
|
it never reaches its event loop to read it. POSIX only makes fault handlers "work"
|
||||||
|
via the async hijack (run the handler *instead of* the instruction), and even there,
|
||||||
|
returning from a SIGSEGV handler without curing the cause is undefined behavior.
|
||||||
|
danos's architecture already has the better answer: fault → the kernel kills the
|
||||||
|
process ([resilience.md](resilience.md) step 2, built) → the supervisor reads the
|
||||||
|
reason → restart. Recovery is restart, not a handler. This is also truer to the 1990
|
||||||
|
standard than handling is: the standard's default action for all three was
|
||||||
|
"terminate the process".
|
||||||
|
|
||||||
|
**SIGPIPE deserves special contempt.** Its default kills a process that writes to a
|
||||||
|
closed pipe — which is why "the whole server died because one client disconnected"
|
||||||
|
is roughly every network daemon's first production bug, and why every mature codebase
|
||||||
|
contains the same fix: ignore SIGPIPE, handle the `EPIPE` error return. danos made
|
||||||
|
the right choice natively already — a reply owed to a dead peer fails with `-EPEER`.
|
||||||
|
Errors from operations are error returns from those operations. The posix layer can
|
||||||
|
synthesize SIGPIPE for ported code that expects it.
|
||||||
|
|
||||||
|
### Statements, not questions
|
||||||
|
|
||||||
|
A signal and a protocol message both travel over IPC — the difference is the
|
||||||
|
**contract**, not the transport. danos IPC has two primitives, both already in
|
||||||
|
daily use: the **asynchronous notification** (a badge — bits that coalesce into a
|
||||||
|
pending mask; the sender never blocks; no payload, *no reply path*; how IRQs and
|
||||||
|
exit events arrive) and the **synchronous call** (a rendezvous — payload both
|
||||||
|
ways, the caller waits for the reply; how VFS requests work). A signal is the
|
||||||
|
first kind: a *statement*. `terminate` wants no reply — the exit notification is
|
||||||
|
its acknowledgement.
|
||||||
|
|
||||||
|
A health probe is the second kind: a *question*, worthless without its answer —
|
||||||
|
and the answer's absence within a deadline is the very thing being measured.
|
||||||
|
Asked as a signal it has no reply channel (a coalescing bit can't carry an answer,
|
||||||
|
and the authority rule forbids a child signalling its supervisor back); asked as a
|
||||||
|
call, the timeout-is-the-diagnosis semantics come free. So there is no `health`
|
||||||
|
signal. Liveness is the common **`ping`**: a reserved request every harness-run
|
||||||
|
service answers automatically on its main endpoint — still free for the service
|
||||||
|
author, still one obvious way — and a supervisor's probe is a `ping` call with a
|
||||||
|
deadline.
|
||||||
|
|
||||||
|
## The two iron rules
|
||||||
|
|
||||||
|
1. **Cleanup is the kernel's job.** A process can die with no warning — fault,
|
||||||
|
kill, power. Correctness must never depend on a `terminate` handler running. On
|
||||||
|
any death the kernel releases the address space, IPC handles, IRQ bindings, and
|
||||||
|
owed replies (built), and must also release **device, I/O-port, and interrupt
|
||||||
|
claims and MSI vectors** (the known gap in
|
||||||
|
[process-management.md](process-management.md); increment 1). A signal handler is
|
||||||
|
for *graceful* work — flushing, deregistering, saving — never for *necessary*
|
||||||
|
work.
|
||||||
|
2. **Kill is not a signal, and exit reasons are load-bearing.** The standard stop
|
||||||
|
sequence is *terminate → deadline → `process_kill`*; the unhandleable kill stays
|
||||||
|
a kernel mechanism. And a supervisor deciding whether to restart must know *how*
|
||||||
|
the child died: clean exit (meant to — don't restart), fault (restart with
|
||||||
|
backoff), killed (the supervisor did it). The exit notification today carries
|
||||||
|
only the id; it grows a reason. Restart policy cannot be written without it.
|
||||||
|
|
||||||
|
## Who learns of a death
|
||||||
|
|
||||||
|
A death has three audiences, and conflating them is how systems end up with either
|
||||||
|
zombie state or privileged snooping:
|
||||||
|
|
||||||
|
1. **The supervisor** — gets the exit notification on the endpoint it gave at spawn
|
||||||
|
(built), which grows the `ExitReason` (increment 2). The supervisor is the only
|
||||||
|
audience that needs the *reason*, because it is the only one deciding whether to
|
||||||
|
restart.
|
||||||
|
2. **The peer owed a reply** — already built: a client that dies mid-request fails
|
||||||
|
the server's reply with `-EPEER`; a server that dies fails its waiting clients
|
||||||
|
the same way. This covers the *synchronous* case only.
|
||||||
|
3. **The subscribers** — the new piece, and it is the input service's
|
||||||
|
publish/subscribe shape ([input.md](input.md)) applied to exits. A stateful
|
||||||
|
service accumulates per-client state across many requests: the VFS holds a dead
|
||||||
|
client's open file handles, the input service holds its subscriptions, a future
|
||||||
|
network stack holds its sockets. None of these are the client's supervisor, and
|
||||||
|
none learn anything from a failed reply if the client simply never calls again.
|
||||||
|
So the kernel **publishes every exit** to whoever subscribed:
|
||||||
|
`process_subscribe(endpoint)` adds a subscriber, and each death posts a
|
||||||
|
notification to every subscriber (badge = `notify_exit_bit | process id` — the
|
||||||
|
same encoding supervisors already decode, the IRQ-as-IPC pattern once more). The
|
||||||
|
subscriber filters for ids it holds state for and releases what the dead client
|
||||||
|
held. Correlating is free of bookkeeping: an IPC sender's badge already *is* its
|
||||||
|
task id (`runtime.ipc.Received`), so the id a service has been keying client
|
||||||
|
state by all along is the id the exit event carries.
|
||||||
|
|
||||||
|
Subscription, not broadcast-to-everyone: only processes that asked receive
|
||||||
|
events, the kernel keeps a bounded subscriber table, and delivery is the same
|
||||||
|
non-blocking coalescing notification as everything else — a dying process never
|
||||||
|
waits on its mourners. Subscribing is ungated, like `process_enumerate`: what is
|
||||||
|
running (and dying) is not a secret between cooperating processes. Subscribers
|
||||||
|
do not receive the exit reason — the VFS does not care *why* the client died.
|
||||||
|
|
||||||
|
This is the service-side mirror of iron rule 1: **a service must never depend on
|
||||||
|
its clients cleaning up after themselves.** Handle release on client death is the
|
||||||
|
service's job, triggered by the published exit event — never by a courtesy
|
||||||
|
"closing now" message that a crashed client will never send.
|
||||||
|
|
||||||
|
## The stable interface: `runtime.process`
|
||||||
|
|
||||||
|
`runtime.process` already owns what a process receives at birth (`Init`, the
|
||||||
|
argv contract). It grows to own the other end of life.
|
||||||
|
|
||||||
|
**The runtime is the stable interface; the numbers are not.** danos applications do
|
||||||
|
not make system calls — they call the runtime library, and the system-call numbers,
|
||||||
|
notification bits, and signal bit positions beneath it are a **private kernel ↔
|
||||||
|
runtime contract** that may change at any time (settled 2026-07-12). This is why
|
||||||
|
the runtime exists. Today kernel and runtime ship from one tree in one image, so
|
||||||
|
"stability" is simply building them together. When driver binaries start shipping
|
||||||
|
as separately-versioned applications — the whole point of the restart design — the
|
||||||
|
binary's embedded runtime version becomes compatibility metadata (the same idea as
|
||||||
|
the protocol version in the device manager's `hello`), and the kernel refuses what
|
||||||
|
it cannot serve. Signals therefore need no reserved numbering scheme: the enum
|
||||||
|
below is vocabulary, not ABI.
|
||||||
|
|
||||||
|
```zig
|
||||||
|
/// The signal vocabulary. The value is the bit position in the pending mask — a
|
||||||
|
/// private kernel/runtime detail, free to change while they ship together.
|
||||||
|
pub const Signal = enum(u5) {
|
||||||
|
terminate = 0, // SIGTERM: finish up and exit
|
||||||
|
reload = 1, // SIGHUP: re-read configuration
|
||||||
|
interrupt = 2, // SIGINT
|
||||||
|
quit = 3, // SIGQUIT
|
||||||
|
alarm = 4, // SIGALRM
|
||||||
|
user_1 = 5, // SIGUSR1
|
||||||
|
user_2 = 6, // SIGUSR2
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A decoded pending mask: the coalesced set of signals a notification delivered.
|
||||||
|
pub const SignalSet = struct {
|
||||||
|
pending: u32,
|
||||||
|
pub fn has(set: SignalSet, signal: Signal) bool { ... }
|
||||||
|
pub fn iterate(set: SignalSet) Iterator { ... }
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Nominate `endpoint` as this process's signal endpoint (signal_bind). The
|
||||||
|
/// runtime's service harness calls this; a bare program may call it directly and
|
||||||
|
/// fold signals into its own replyWait loop.
|
||||||
|
pub fn bindSignals(endpoint: usize) bool { ... }
|
||||||
|
|
||||||
|
/// Decode a received badge into signals, or null if the badge is not a signal
|
||||||
|
/// notification (mirrors ipc.Received.isChildExit).
|
||||||
|
pub fn signalsFrom(badge: usize) ?SignalSet { ... }
|
||||||
|
|
||||||
|
/// Send `signal` to process `id`. Supervisor-gated, like kill; non-blocking.
|
||||||
|
pub fn sendSignal(id: u32, signal: Signal) bool { ... }
|
||||||
|
|
||||||
|
/// The standard stop sequence: terminate, wait up to `deadline_ms` for the exit
|
||||||
|
/// notification, then process_kill. The one call a supervisor needs.
|
||||||
|
pub fn stop(id: u32, deadline_ms: u64) void { ... }
|
||||||
|
|
||||||
|
/// Subscribe `endpoint` to published exit events (process_subscribe). Every
|
||||||
|
/// process death posts an asynchronous notification: badge = notify_exit_bit |
|
||||||
|
/// process id — the same encoding a supervisor's exit notification uses, decoded
|
||||||
|
/// by the same ipc.Received helpers. For stateful services: release what the dead
|
||||||
|
/// client held (file handles, subscriptions, sockets). Ungated, like
|
||||||
|
/// process_enumerate.
|
||||||
|
pub fn subscribeExits(endpoint: usize) bool { ... }
|
||||||
|
|
||||||
|
/// How a process ended — queried after the exit notification (the kernel records
|
||||||
|
/// it first, so the two never race). What restart policy reads. (Built in M17.2.)
|
||||||
|
pub const ExitReason = enum(u8) {
|
||||||
|
exited, // returned from main / clean exit
|
||||||
|
aborted, // abort() — deliberate self-termination (SIGABRT's ghost; reserved)
|
||||||
|
segmentation_fault, // SIGSEGV's ghost
|
||||||
|
illegal_instruction, // SIGILL's ghost
|
||||||
|
arithmetic_fault, // SIGFPE's ghost
|
||||||
|
protection_fault, // general protection fault
|
||||||
|
fault, // any other CPU exception
|
||||||
|
killed, // process_kill
|
||||||
|
};
|
||||||
|
```
|
||||||
|
|
||||||
|
Two deliberate absences. There is no `mask`/`block` API — a process that is not
|
||||||
|
ready for a signal simply has not waited on its endpoint yet; the pending mask *is*
|
||||||
|
the blocked set. And there is no per-signal handler registration at this layer —
|
||||||
|
dispatch is the process's own `switch` over `SignalSet`, or the service harness's
|
||||||
|
callbacks (`on_terminate`, `on_reload`) for programs that want defaults.
|
||||||
|
|
||||||
|
### The service harness
|
||||||
|
|
||||||
|
`runtime.service` owns the `replyWait` loop and folds every event source — signals,
|
||||||
|
child exits, protocol messages — into callbacks, with the vocabulary's defaults:
|
||||||
|
`terminate` returns from the loop (clean exit), the common `ping` is answered automatically,
|
||||||
|
`reload` is ignored unless overridden. One loop, no locking, nothing reentrant. A
|
||||||
|
service author writes domain logic; the lifecycle contract is satisfied by the
|
||||||
|
harness. A process that bypasses the harness and ignores its signals meets the
|
||||||
|
deadline-then-kill escalation — you cannot force a process to implement an
|
||||||
|
interface, but you can make compliance free and non-compliance fatal.
|
||||||
|
|
||||||
|
### The musl layer later
|
||||||
|
|
||||||
|
The POSIX C layer is a **musl port**: musl's arch/syscall layer retargeted so that
|
||||||
|
what musl believes are kernel syscalls become danos runtime calls and IPC — `open`
|
||||||
|
and `read` onto the VFS protocol, `kill`/`sigaction`/`waitpid` onto this document's
|
||||||
|
vocabulary, `exit` onto the runtime's exit path. `sigaction` handlers registered
|
||||||
|
through it are invoked by the runtime's loop when the signal message arrives —
|
||||||
|
synchronous underneath, async-looking to ported code, delivered at wait boundaries
|
||||||
|
the way most Unix programs already experience signals (at syscalls). No stack hijack
|
||||||
|
ever happens, `SA_RESTART` semantics come free because nothing was interrupted, and
|
||||||
|
SIGPIPE can be synthesized from `-EPEER` for the programs that expect it. C programs
|
||||||
|
get POSIX; danos-native programs never pay for it.
|
||||||
|
|
||||||
|
## Increments
|
||||||
|
|
||||||
|
1. **Kernel: release device/port/IRQ claims and MSI vectors on death** — the
|
||||||
|
cleanup half of iron rule 1, and the prerequisite for any restart story. Test:
|
||||||
|
kill a claiming driver, spawn it again, the claim succeeds.
|
||||||
|
2. **Exit reason in the death notification** (`ExitReason` above).
|
||||||
|
3. **Exit events**: `process_subscribe` in the kernel (bounded subscriber table,
|
||||||
|
publishes on every death), `runtime.process.subscribeExits`; the VFS becomes the
|
||||||
|
first subscriber — releasing a dead client's handles is its proof test.
|
||||||
|
4. **Signals**: `signal_bind` + `process_signal` + the pending mask in the kernel;
|
||||||
|
`runtime.process` grows the interface above; the service harness handles
|
||||||
|
`terminate` and answers the common `ping`; `stop()` for supervisors.
|
||||||
|
|
||||||
|
[device-manager.md](device-manager.md) builds directly on all four.
|
||||||
|
|
||||||
|
## Settled questions (2026-07-12)
|
||||||
|
|
||||||
|
- **Signal numbering is not ABI**: the runtime is the stable interface; the numbers
|
||||||
|
beneath it are a private kernel ↔ runtime contract (see "The stable interface").
|
||||||
|
- **Liveness is a `ping` call, not a signal**: signals are statements, questions
|
||||||
|
are synchronous calls (see "Statements, not questions"). A service wanting *deep*
|
||||||
|
health ("can I reach my hardware?") defines its own protocol message on top.
|
||||||
|
- **Process handles: deferred.** Pids + the supervisor gate cover everything
|
||||||
|
planned; transferable handles (Fuchsia-style, delegating signalling without
|
||||||
|
delegating kill) wait for the capability table to grow types beyond endpoints.
|
||||||
|
- **`alarm`: in the vocabulary, unbuilt.** No consumer yet; when one appears it is
|
||||||
|
runtime sugar over the existing timer (arm a timer that posts your own signal) —
|
||||||
|
zero kernel work, so deferring costs nothing.
|
||||||
|
- **Subscription granularity: all exits**, subscriber-side filtering — one
|
||||||
|
subscription per service, a bounded kernel table. Per-id subscriptions only if
|
||||||
|
event volume ever matters (hundreds of processes, not before).
|
||||||
|
- **Client identity across the exit boundary: no convention needed** — an IPC
|
||||||
|
sender's badge already is its task id (see "Who learns of a death").
|
||||||
@@ -0,0 +1,120 @@
|
|||||||
|
# Process Management
|
||||||
|
|
||||||
|
How danos lists, supervises, and kills processes — the microkernel answer to
|
||||||
|
`ps`, `kill`, and `SIGCHLD`/`wait`.
|
||||||
|
|
||||||
|
## Why system calls, not `/proc`
|
||||||
|
|
||||||
|
Unix systems sit on a spectrum. Classic BSD/macOS list processes through
|
||||||
|
syscalls (`sysctl(KERN_PROC)`) and kill through `kill(2)`; Linux renders the
|
||||||
|
process table as `/proc` for *reading* but still kills through a syscall; Plan 9
|
||||||
|
made the file tree the whole interface (`echo kill > /proc/n/ctl`). Microkernels
|
||||||
|
mostly abandon ambient PIDs: Minix and QNX route everything through a user-space
|
||||||
|
process-manager server, and Fuchsia/seL4 control processes only through handles.
|
||||||
|
|
||||||
|
danos rules out `/proc` **as the primitive**: here a `/proc` would be served by
|
||||||
|
the VFS server — a user process — which would put the VFS in the path of process
|
||||||
|
control. If the VFS (or anything under it) hangs, nothing could be listed or
|
||||||
|
killed, *including the hung VFS*. The control plane for processes must not
|
||||||
|
depend on a process. So the primitives are kernel system calls; a read-only
|
||||||
|
`/proc` rendering can be layered on later, and a POSIX-style process-manager
|
||||||
|
server can be built *from* these primitives when one is needed.
|
||||||
|
|
||||||
|
## The three primitives
|
||||||
|
|
||||||
|
### `process_enumerate(buffer, maximum) -> total`
|
||||||
|
|
||||||
|
A snapshot of the task table into a caller buffer of `abi.ProcessDescriptor`
|
||||||
|
(id, supervisor, state, priority, name) — the exact shape of
|
||||||
|
`device_enumerate`, so `ps` is a user program over a snapshot, not a kernel
|
||||||
|
service. The total may exceed what fit; call again with a larger buffer. Kernel
|
||||||
|
tasks are included with an empty name — an honest listing shows the idle tasks
|
||||||
|
too. Ungated and read-only: what is running is not a secret between cooperating
|
||||||
|
bring-up processes.
|
||||||
|
|
||||||
|
### `system_spawn(..., exit_endpoint) -> child id`, and the supervision link
|
||||||
|
|
||||||
|
`system_spawn` records the caller as the child's **supervisor** and returns the
|
||||||
|
child's process id (ids are monotonic, never reused — a stale id can only miss).
|
||||||
|
That link is the kill authority: it answers "who may kill process 7?" without
|
||||||
|
inventing users or permissions, the same way a device *claim* is the capability
|
||||||
|
for `mmio_map`. It composes with the supervision hierarchy the device manager
|
||||||
|
already forms: init supervises the services it starts, the device manager
|
||||||
|
supervises the drivers it matches. (A transferable process *handle* — Fuchsia
|
||||||
|
style — can replace the id once the handle table grows types beyond endpoints.)
|
||||||
|
|
||||||
|
`exit_endpoint` (a handle, or `abi.no_cap`) is the supervisor's death-watch: when
|
||||||
|
the child ends — clean exit, CPU fault, or `process_kill` — the kernel posts an
|
||||||
|
asynchronous notification to that endpoint, exactly like a bound IRQ. The badge
|
||||||
|
carries `abi.notify_badge_bit | abi.notify_exit_bit | child_id`, so one endpoint
|
||||||
|
supervises many children and can even share with IRQ notifications. This is the
|
||||||
|
microkernel's SIGCHLD: no new mechanism, just the IRQ-as-IPC pattern reused, and
|
||||||
|
a supervisor's event loop (`ipc.replyWait`) already knows how to receive it. The
|
||||||
|
child holds a reference to the endpoint from birth, so the notification cannot
|
||||||
|
dangle even if the supervisor dies first.
|
||||||
|
|
||||||
|
### `process_kill(id) -> 0 / -ESRCH / -EPERM`
|
||||||
|
|
||||||
|
Only the supervisor may kill; kernel tasks are not killable processes. Like a
|
||||||
|
signal, delivery is prompt but asynchronous — 0 means the kill is accepted and
|
||||||
|
irrevocable; the exit notification confirms completion.
|
||||||
|
|
||||||
|
## How a kill lands (the kernel mechanics)
|
||||||
|
|
||||||
|
Everything below runs under the big kernel lock, where task states cannot move.
|
||||||
|
|
||||||
|
- **Target ready or blocked** (not on any core): reaped on the killer's own
|
||||||
|
call. The reap releases what death always releases (IRQ bindings first, then
|
||||||
|
a client the target still owed a reply to is failed with `-EPEER`, IPC handles
|
||||||
|
closed, the exit notification posted last) — plus the unlinking only a
|
||||||
|
*remote* death needs: out of the ready queue, out of an endpoint's sender FIFO
|
||||||
|
(`Task.ipc_wait_endpoint`), out of a receive wait queue (`Task.wait_queue`),
|
||||||
|
and out of any server's owed-reply slot, so nothing ever dequeues a dangling
|
||||||
|
pointer. Destroying the address space is safe because no core can have it
|
||||||
|
loaded: every switch away from a task loads the next task's tables.
|
||||||
|
- **Target running on another core**: it cannot be torn down mid-instruction,
|
||||||
|
so it is condemned (`Task.kill_pending`) and dies at whichever comes first:
|
||||||
|
- its next **system_call entry** — checked before dispatch, so a condemned
|
||||||
|
process cannot spawn, claim, or message anything on its way out;
|
||||||
|
- its core's next **timer tick** — but only when the task is not inside one
|
||||||
|
of its own system calls (`Task.in_system_call`): the tick may have
|
||||||
|
interrupted kernel code mid-operation, where teardown would leak whatever
|
||||||
|
the operation held. User-mode execution is always a safe kill point. The
|
||||||
|
tick-time terminate abandons the interrupt frame exactly like the fault
|
||||||
|
path (the LAPIC is acknowledged before the tick hook runs);
|
||||||
|
- any core's tick finding it **blocked or ready** (it entered a syscall and
|
||||||
|
parked after being condemned) — reaped by the same remote-reap path.
|
||||||
|
|
||||||
|
A pure user-mode spin loop that never makes a system call therefore dies
|
||||||
|
within one tick; nothing a process does can outrun the kill.
|
||||||
|
|
||||||
|
The scheduler stays below the process layer: finishing a kill (IRQ bindings,
|
||||||
|
handles, the notification) is called *up* through two hooks process.zig
|
||||||
|
registers at boot (`terminate_current_hook`, `reap_task_hook`), mirroring how
|
||||||
|
the architecture layer calls up into `tick`.
|
||||||
|
|
||||||
|
## Known gaps (bring-up honesty)
|
||||||
|
|
||||||
|
- ~~Device claims are not released on death~~ Closed (M17.1): every path out of a
|
||||||
|
process releases its device claims alongside its IRQ and MSI bindings
|
||||||
|
(`releaseTaskResourcesLocked`), so a restarted driver can claim its hardware
|
||||||
|
again — the cleanup half of [process-lifecycle.md](process-lifecycle.md)'s iron
|
||||||
|
rule 1. The `claim-release` test proves the kill → release → re-claim cycle.
|
||||||
|
- Kernel stacks of dead tasks are leaked, as on every exit path (no reaper yet).
|
||||||
|
- ~~There is no exit status in the notification~~ Closed (M17.2): the kernel
|
||||||
|
records how every process ends — exited, a fault class, or killed — before it
|
||||||
|
posts the exit notification, and the supervisor reads it with
|
||||||
|
`process_exit_reason` (`runtime.process.exitReason`). This is the input to
|
||||||
|
restart policy ([process-lifecycle.md](process-lifecycle.md)); an exit *code*
|
||||||
|
for the clean case can still ride alongside later.
|
||||||
|
- Enumerate writes through the caller's raw pointer under the bring-up trust
|
||||||
|
model, like `device_enumerate` (an unmapped page is a self-DoS, not an
|
||||||
|
isolation break).
|
||||||
|
|
||||||
|
## Tests
|
||||||
|
|
||||||
|
`process-list` (enumerate), `process-kill` (kernel-level kill paths, refusals,
|
||||||
|
notifications), `supervision` (the whole user-side surface via the process-test
|
||||||
|
service: spawn supervised → enumerate → kill blocked and spinning children →
|
||||||
|
notifications → gone), `claim-release` (a killed claim-holder's device is
|
||||||
|
claimable again). See test/qemu_test.py.
|
||||||
+16
-3
@@ -1,6 +1,17 @@
|
|||||||
# Resilience: fault isolation and live restart
|
# Resilience: fault isolation and live restart
|
||||||
|
|
||||||
A design/research note, not built yet. This is the property danos is really chasing:
|
Steps 1–4 of the ordering below are **built** (M17–M18, 2026-07-13): user-mode
|
||||||
|
isolation; fault → kill the process → keep the core (`onException`; the
|
||||||
|
`fault-recovery` test); the supervisor notification **with exit reasons**
|
||||||
|
([process-lifecycle.md](process-lifecycle.md) — clean exit, fault class, or
|
||||||
|
killed, recorded before the notice posts); and the **restart policy itself**
|
||||||
|
([device-manager.md](device-manager.md)): the device manager supervises every
|
||||||
|
driver, restarts crashes with backoff, caps crash loops, and re-claims work
|
||||||
|
because the kernel releases a dead process's claims. The `driver-restart` and
|
||||||
|
`usb-report` scenarios prove kill → release → respawn → re-claim → re-report
|
||||||
|
end to end. What remains of this document's ladder is scope, not mechanism:
|
||||||
|
more of the system moved into restartable processes (the discovery migration,
|
||||||
|
[discovery.md](discovery.md), is the next rung). This is the property danos is really chasing:
|
||||||
**if a part of the OS breaks, isolate it, and re-initialise it — without rebooting.**
|
**if a part of the OS breaks, isolate it, and re-initialise it — without rebooting.**
|
||||||
A crashed driver gets restarted; a wedged service gets killed and brought back. It's
|
A crashed driver gets restarted; a wedged service gets killed and brought back. It's
|
||||||
the reason the [microkernel](vision.md) shape was chosen, and it's a *separate* goal
|
the reason the [microkernel](vision.md) shape was chosen, and it's a *separate* goal
|
||||||
@@ -111,9 +122,11 @@ Honest boundaries:
|
|||||||
## Suggested ordering
|
## Suggested ordering
|
||||||
|
|
||||||
1. **User mode + address-space isolation** — the shared prerequisite (also on the
|
1. **User mode + address-space isolation** — the shared prerequisite (also on the
|
||||||
path for everything else).
|
path for everything else). **Done.**
|
||||||
2. **Kernel: fault → kill process → notify.** Turn today's "halt on fault" into
|
2. **Kernel: fault → kill process → notify.** Turn today's "halt on fault" into
|
||||||
"confine to the process and report it."
|
"confine to the process and report it." **Done** (the kill and reclaim; the
|
||||||
|
supervisor notification waits for step 3's supervisor). A killed server's
|
||||||
|
pending client is unblocked with `-EPEER` rather than hung.
|
||||||
3. **A minimal supervisor server** that can (re)start a process.
|
3. **A minimal supervisor server** that can (re)start a process.
|
||||||
4. **Resource cleanup on death** — reclaim memory/MMIO/IPC/IRQ, via caps or a grant
|
4. **Resource cleanup on death** — reclaim memory/MMIO/IPC/IRQ, via caps or a grant
|
||||||
table.
|
table.
|
||||||
|
|||||||
+23
-2
@@ -6,8 +6,8 @@ ready task always runs, and tasks at the same priority take turns. That model is
|
|||||||
chosen for [real-time](vision.md) — it's predictable (you can reason about which
|
chosen for [real-time](vision.md) — it's predictable (you can reason about which
|
||||||
task runs when) and its decisions are O(1), unlike a fair-share scheduler.
|
task runs when) and its decisions are O(1), unlike a fair-share scheduler.
|
||||||
|
|
||||||
The scheduler proper (`src/kernel/sched.zig`) is generic; the context switch and new-task
|
The scheduler proper (`system/kernel/sched.zig`) is generic; the context switch and new-task
|
||||||
stack setup are architecture-specific (`src/kernel/arch/x86_64/`, see [arch](arch.md)).
|
stack setup are architecture-specific (`system/kernel/architecture/x86_64/`, see [arch](arch.md)).
|
||||||
|
|
||||||
## Tasks
|
## Tasks
|
||||||
|
|
||||||
@@ -67,6 +67,27 @@ exist, which is what a real-time scheduler needs.
|
|||||||
- **Round-robin within a level.** When a task is descheduled it goes to the *back*
|
- **Round-robin within a level.** When a task is descheduled it goes to the *back*
|
||||||
of its level's queue, so equal-priority tasks share the CPU fairly.
|
of its level's queue, so equal-priority tasks share the CPU fairly.
|
||||||
|
|
||||||
|
## Affinity: pinning a task to a core
|
||||||
|
|
||||||
|
By default a task runs on **any** core — the ready queue above is global, and any
|
||||||
|
idle core pulls the highest-priority task from it (work-conserving; see
|
||||||
|
[smp.md](smp.md)). A task can instead be **pinned** to one core with
|
||||||
|
`spawnOn(entry, priority, cpu)`, giving it an *affinity*: it will only ever run
|
||||||
|
there, never migrating.
|
||||||
|
|
||||||
|
Mechanically, each core has its **own** pinned queue (same 8-level FIFO + bitmap)
|
||||||
|
alongside the global one. A pinned task is enqueued only into its core's pinned
|
||||||
|
queue; selection compares the top of the global queue and the running core's pinned
|
||||||
|
queue and takes the higher priority (still O(1) — two bit-scans and a compare), with
|
||||||
|
a pinned task winning an equal-priority tie so it can't be starved by global work.
|
||||||
|
Because every queue is mutated under the [big kernel lock](smp.md), one core enqueuing
|
||||||
|
into another core's pinned queue is safe.
|
||||||
|
|
||||||
|
This is the *explicit-affinity* model (no surprise migration mid-deadline), which is
|
||||||
|
the more real-time-predictable direction. `spawnOn` refuses to pin to an offline or
|
||||||
|
out-of-range core — it creates the task unpinned instead, so it still runs somewhere
|
||||||
|
rather than stranding in a queue no core services, and returns whether the pin took.
|
||||||
|
|
||||||
## Sleeping and the idle task
|
## Sleeping and the idle task
|
||||||
|
|
||||||
A task can **block** — give up the CPU until an event, rather than busy-wait
|
A task can **block** — give up the CPU until an event, rather than busy-wait
|
||||||
|
|||||||
+114
-7
@@ -16,10 +16,14 @@ danos specifics):
|
|||||||
- **Real-time** — whether timing is *predictable*. Comes from bounded operations
|
- **Real-time** — whether timing is *predictable*. Comes from bounded operations
|
||||||
(our O(1) scheduler), not from core count.
|
(our O(1) scheduler), not from core count.
|
||||||
|
|
||||||
danos is uniprocessor today: one global `current` task, one set of ready queues, one
|
danos now runs on multiple cores. The firmware starts only the **bootstrap processor
|
||||||
timer. Even on an 8-core CPU, the firmware starts only the **bootstrap processor
|
(BSP)**; the kernel wakes the other cores (**application processors**, APs) with
|
||||||
(BSP)**; the other cores (**application processors**, APs) sit parked until the
|
INIT–SIPI–SIPI, brings each up into 64-bit long mode with its own descriptor tables,
|
||||||
kernel wakes them, which it doesn't yet.
|
LAPIC and timer, and drops it into the scheduler. Tasks run **genuinely in parallel** —
|
||||||
|
the `smp` self-test confirms worker tasks executing on all four cores at once under
|
||||||
|
QEMU `-smp 4`. Shared kernel state (scheduler queues, IPC) is serialised behind a big
|
||||||
|
kernel lock. What's left is refinement, not first-light: per-core run queues, IPIs,
|
||||||
|
and thread-to-core affinity (see [Implementation status](#implementation-status)).
|
||||||
|
|
||||||
## The common microkernel instinct: don't share kernel state
|
## The common microkernel instinct: don't share kernel state
|
||||||
|
|
||||||
@@ -114,14 +118,20 @@ active reconsideration in favour of resilience — see [vision.md](vision.md).)
|
|||||||
Whatever the top goal, the *sequence* is the same and seL4 validates starting simple:
|
Whatever the top goal, the *sequence* is the same and seL4 validates starting simple:
|
||||||
|
|
||||||
1. **Enumerate cores** — needs [device discovery](discovery.md) (ACPI MADT on x86,
|
1. **Enumerate cores** — needs [device discovery](discovery.md) (ACPI MADT on x86,
|
||||||
device tree on ARM). SMP is a concrete consumer of that work.
|
device tree on ARM). SMP is a concrete consumer of that work. **Done on x86:** the
|
||||||
|
MADT parse records every usable Local APIC — with the `apic_id` an AP wake targets —
|
||||||
|
and `platform.cpus()` returns the list (see [discovery.md](discovery.md)). The boot
|
||||||
|
log reports the count; the ARM (device-tree) path still needs it.
|
||||||
2. **Wake the APs** — INIT–SIPI–SIPI on x86; PSCI/spin-tables on ARM. Each core brings
|
2. **Wake the APs** — INIT–SIPI–SIPI on x86; PSCI/spin-tables on ARM. Each core brings
|
||||||
up its own tables, timer, and idle task.
|
up its own tables, timer, and idle task. **Done on x86** — cores climb to long mode,
|
||||||
|
set up their own GDT/TSS, and enter the scheduler; tasks run in parallel across all
|
||||||
|
cores ([status](#implementation-status)).
|
||||||
3. **Start with a big kernel lock.** It's a legitimate first design, not a shortcut —
|
3. **Start with a big kernel lock.** It's a legitimate first design, not a shortcut —
|
||||||
philosophically aligned with a tiny kernel, and it lets the single-core correctness
|
philosophically aligned with a tiny kernel, and it lets the single-core correctness
|
||||||
model you already have (the interrupt-flag discipline in
|
model you already have (the interrupt-flag discipline in
|
||||||
[scheduling.md](scheduling.md)) stay largely intact: one lock around kernel entry
|
[scheduling.md](scheduling.md)) stay largely intact: one lock around kernel entry
|
||||||
instead of rethinking every critical section.
|
instead of rethinking every critical section. **Done** — see
|
||||||
|
`system/kernel/sync.zig`.
|
||||||
4. **Later, if contention bites,** evolve toward **per-core run queues + explicit
|
4. **Later, if contention bites,** evolve toward **per-core run queues + explicit
|
||||||
affinity** (the Fiasco.OC direction) — also the more real-time-predictable model.
|
affinity** (the Fiasco.OC direction) — also the more real-time-predictable model.
|
||||||
5. **Placement stays a user-space policy** — the kernel runs a thread on the core it's
|
5. **Placement stays a user-space policy** — the kernel runs a thread on the core it's
|
||||||
@@ -130,6 +140,103 @@ Whatever the top goal, the *sequence* is the same and seL4 validates starting si
|
|||||||
Big-lock-first → per-core-later. The affinity/MCS depth is only worth it if real-time
|
Big-lock-first → per-core-later. The affinity/MCS depth is only worth it if real-time
|
||||||
turns out to be the actual goal.
|
turns out to be the actual goal.
|
||||||
|
|
||||||
|
## Implementation status
|
||||||
|
|
||||||
|
The "wake + schedule" build (real parallel task execution) is going in as a sequence
|
||||||
|
of green checkpoints — each step keeps the single-core test suite passing before the
|
||||||
|
next lands.
|
||||||
|
|
||||||
|
**Done:**
|
||||||
|
|
||||||
|
- **Core enumeration** — the MADT parse records every usable Local APIC (with its
|
||||||
|
`apic_id`, which an AP wake targets); `platform.cpus()` returns the list. See
|
||||||
|
[discovery.md](discovery.md).
|
||||||
|
- **The big kernel lock** (`system/kernel/sync.zig`) — one coarse spinlock guarding the
|
||||||
|
scheduler queues and IPC, always held with local interrupts disabled. It is held
|
||||||
|
*across* a context switch and released by whichever task resumes (the hand-off
|
||||||
|
rule); `task_trampoline` releases it for a freshly-spawned task. `scheduler.zig` and
|
||||||
|
`ipc.zig` run every critical section under it. Uncontended on one core, so behaviour
|
||||||
|
is identical to the old interrupt-flag model.
|
||||||
|
- **Per-CPU state** — a `PerCpu` struct (running task, idle task, APIC id) per core,
|
||||||
|
its pointer kept in the x86 **GS base** (`IA32_GS_BASE`; no `swapgs`, since there's
|
||||||
|
no user mode yet). The old global `current` is now `thisCpu().current`. The ready
|
||||||
|
queues stay **global** under the lock — work-conserving, so any idle core will pull
|
||||||
|
the highest-priority ready task; per-core queues are a later optimisation.
|
||||||
|
- **AP wake to long mode** — `arch.startSecondary` drives INIT–SIPI–SIPI (via the
|
||||||
|
LAPIC ICR) to wake each parked core one at a time. A woken core starts in 16-bit
|
||||||
|
real mode at a low page and runs the [trampoline](../system/kernel/architecture/x86_64/trampoline.s)
|
||||||
|
up through protected mode into 64-bit long mode, then lands in `smp.zig:apEntry`,
|
||||||
|
publishes its per-CPU pointer, and reports in. Verified in QEMU with `-smp 4`:
|
||||||
|
all four cores report `online`.
|
||||||
|
|
||||||
|
The trampoline earns its complexity from four hardware facts:
|
||||||
|
- a STARTUP IPI vectors a core to physical `vector << 12` (a *byte* vector), so the
|
||||||
|
trampoline must live **below 1 MiB** — the kernel reserves that page from the frame
|
||||||
|
allocator at boot, before paging/heap draw down the scarce low frames;
|
||||||
|
- the blanket RAM identity map is **NX** (W^X), but the AP fetches the trampoline
|
||||||
|
from it under paging, so that one page is made executable for bring-up;
|
||||||
|
- the blob is copied to a page whose address isn't known at link time, so it is
|
||||||
|
**position-independent**: it derives its own base from `CS` and, crucially,
|
||||||
|
addresses data *segment-relative in real mode* (where the segment base already
|
||||||
|
supplies the page base) but *base-register-relative in protected/long mode* (flat
|
||||||
|
segments, base 0). Getting that distinction wrong was the first bug found;
|
||||||
|
- an AP starts with a bare `CR0`/`CR4`, but the kernel is built **with SSE** (the
|
||||||
|
x86_64 baseline) and the compiler emits SSE for things as ordinary as a struct
|
||||||
|
copy — so the trampoline must set `CR4.OSFXSR`/`OSXMMEXCPT` and fix `CR0.EM`/`MP`,
|
||||||
|
or the first SSE instruction on the AP `#UD`s. The BSP inherited those bits from
|
||||||
|
UEFI; the AP has to set them itself. This was the second bug — it masqueraded as a
|
||||||
|
fault in `lgdt` (the first kernel code after entry that the compiler vectorised).
|
||||||
|
|
||||||
|
- **Per-core tables + scheduler entry** — each AP loads **its own GDT** (with its own
|
||||||
|
TSS descriptor) and **its own TSS** (its own IST/`rsp0` stack), loads the shared
|
||||||
|
IDT, enables its LAPIC and timer, then calls the generic `secondaryMain`: it turns
|
||||||
|
its bring-up context into the core's idle task (as task 0 is for the BSP), marks the
|
||||||
|
core online, and enters the run loop. With interrupts on, each core's own timer tick
|
||||||
|
preempts its idle context into whatever the global ready queue offers — so all cores
|
||||||
|
pull real work in parallel. The `smp` test spawns CPU-bound workers and confirms they
|
||||||
|
execute on all four cores at once, and `fault-ap-df` pins a #DF to an AP and checks
|
||||||
|
that core catches it on **its own** IST (a broken per-core TSS would triple-fault) —
|
||||||
|
reported as "core N: …", so a fault is always attributed to the core it happened on,
|
||||||
|
and is contained to that core (the rest of the system keeps running).
|
||||||
|
- **Thread affinity** — `spawnOn(entry, priority, cpu)` pins a task to a core (its own
|
||||||
|
per-core pinned queue, merged with the global queue at selection; see
|
||||||
|
[scheduling.md](scheduling.md#affinity-pinning-a-task-to-a-core)). The `affinity`
|
||||||
|
test confirms a pinned task never migrates. This is the mechanism the fault-on-AP
|
||||||
|
test rides on, and the *explicit-affinity* real-time-predictable model.
|
||||||
|
- **Right-sized footprint** — the per-CPU ceiling (`system.max_cpus`, one constant
|
||||||
|
shared by discovery, the scheduler, and the per-core GDT/TSS) is generous (128), but
|
||||||
|
the *large* per-core resources — the kernel and IST (double-fault) stacks — are
|
||||||
|
**heap-allocated at bring-up**, only for cores that actually come online. Only the
|
||||||
|
BSP's IST stack is static, because it must exist before the frame allocator does.
|
||||||
|
This kept the kernel image small (a static `[128][16 KiB]` IST array would have been
|
||||||
|
2 MiB of `.bss`); it's a few tens of KiB instead.
|
||||||
|
- **`single_threaded` off** — the kernel was built `single_threaded = true`, which
|
||||||
|
compiles `std.atomic` down to plain non-atomic ops. Harmless on one core, but it
|
||||||
|
quietly breaks the big kernel lock across cores; it's now `false`.
|
||||||
|
- **Re-armable wake + retry** — the trampoline frame is reserved for the system's
|
||||||
|
life, but kept **inert between wakes**: zeroed and non-executable, armed (blob
|
||||||
|
copied in, page made executable) only for the moment a core is actually climbing,
|
||||||
|
then disarmed again. So there's never a dormant executable page, and a core can be
|
||||||
|
(re)woken at any time — `arch.startSecondary` is one self-contained attempt (arm →
|
||||||
|
INIT–SIPI–SIPI → disarm), and its `INIT` resets a wedged core, so retrying just
|
||||||
|
works. Boot retries a non-responding core up to three times; the same primitive is
|
||||||
|
the groundwork a future **power manager** would drive to bring cores up (and,
|
||||||
|
eventually, its counterpart to take them offline — which additionally needs the
|
||||||
|
core's tasks migrated off first).
|
||||||
|
|
||||||
|
**Next (refinement, not first-light):**
|
||||||
|
|
||||||
|
- **IPIs** — cross-core wake/preempt. Not needed for correctness: an idle core wakes
|
||||||
|
on its own timer tick and pulls ready work then; IPIs only cut that latency from
|
||||||
|
≤1 ms to near-instant.
|
||||||
|
- **Per-core run queues** — the Fiasco.OC direction, if the single global queue's lock
|
||||||
|
contention ever bites. (Thread *affinity* already exists — see above; this is the
|
||||||
|
further step of giving each core its own primary run queue for load distribution.)
|
||||||
|
- **Fault recovery** — today a fault halts (only) the faulting core. Turning that into
|
||||||
|
"kill the task, keep the core running" is the [resilience](resilience.md) track (it
|
||||||
|
needs the task's lock/resource state handled), and for taking a core fully offline,
|
||||||
|
its tasks migrated first.
|
||||||
|
|
||||||
## Further reading
|
## Further reading
|
||||||
|
|
||||||
**Microkernel SMP & scheduling**
|
**Microkernel SMP & scheduling**
|
||||||
|
|||||||
+13
-1
@@ -1,5 +1,15 @@
|
|||||||
# System Calls
|
# System Calls
|
||||||
System calls (syscalls) are the bridge between your programs and the operating system's restricted core (kernel).
|
System calls (syscalls) are the bridge between your programs and the operating system's restricted core (kernel).
|
||||||
|
|
||||||
|
> **Status:** danos has real user processes (M3). User programs enter the kernel
|
||||||
|
> via the `syscall` instruction (STAR/LSTAR/SFMASK set per core; the entry stub in
|
||||||
|
> `isr.s` does the `swapgs` + kernel-stack switch and reuses the interrupt
|
||||||
|
> dispatcher). The `int 0x80` gate is kept alongside as a minimal test path. The
|
||||||
|
> current call set is still a placeholder — `0 = exit(code)`, `1 = ping`,
|
||||||
|
> `2 = write(ptr, len)`, `3 = sleep(ms)` (see `system/kernel/process.zig`); the
|
||||||
|
> handler dispatches on whether the caller is a scheduled process (its own address
|
||||||
|
> space) or a borrowed test thread. The microkernel set below (IPC_Call /
|
||||||
|
> IPC_ReplyWait / Yield) replaces it once a second user server exists.
|
||||||
|
|
||||||
## The Mechanism of a Syscall
|
## The Mechanism of a Syscall
|
||||||
|
|
||||||
@@ -37,6 +47,8 @@ Everything else---including`read()`,`write()`,`malloc()`, and`fork()`---will run
|
|||||||
- **What it does:**Used strictly by your background user-space servers (like your disk driver or filesystem). It sends a reply to the last client that called it, and immediately puts the server to sleep until the next request arrives.[[1](https://news.ycombinator.com/item?id=33078441)]
|
- **What it does:**Used strictly by your background user-space servers (like your disk driver or filesystem). It sends a reply to the last client that called it, and immediately puts the server to sleep until the next request arrives.[[1](https://news.ycombinator.com/item?id=33078441)]
|
||||||
3. **`Yield()`/`Thread_Ctrl()`**
|
3. **`Yield()`/`Thread_Ctrl()`**
|
||||||
- **What it does:**Allows a thread to voluntarily give up its CPU time slice, or allows a root task to spawn/kill threads.
|
- **What it does:**Allows a thread to voluntarily give up its CPU time slice, or allows a root task to spawn/kill threads.
|
||||||
|
4. **`ipc_send(endpoint, message_buffer)`(Asynchronous Send)**
|
||||||
|
- **What it does:**Posts a small payload to an endpoint's bounded queue and returns *without* blocking — no rendezvous, no reply. The receiver picks it up through the same `IPC_ReplyWait`, as a buffered message. It is the async counterpart of `IPC_Call`, for one-to-many broadcasts where a synchronous rendezvous would let one dead or slow receiver hang the sender. The [input service](input.md) — keyboard-event fan-out — is its first user. A full queue drops the oldest message (a buffered message is discrete data, unlike a coalescing interrupt notification).
|
||||||
|
|
||||||
* * * * *
|
* * * * *
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,227 @@
|
|||||||
|
# System Requirements
|
||||||
|
|
||||||
|
Minimum and recommended hardware for running danos. Every requirement below is
|
||||||
|
grounded in what the current code actually assumes at boot — this is a
|
||||||
|
description of the real target, not an aspirational one.
|
||||||
|
|
||||||
|
## Summary
|
||||||
|
|
||||||
|
danos targets a **modern UEFI x86-64 PC with ACPI and PCIe**. The practical
|
||||||
|
minimum is:
|
||||||
|
|
||||||
|
- 64-bit x86-64 CPU with SSE2, APIC, and `syscall`/`sysret`
|
||||||
|
- UEFI firmware (no BIOS / legacy boot)
|
||||||
|
- ACPI tables: MADT, MCFG, FADT
|
||||||
|
- PCIe with an ECAM (MMConfig) window
|
||||||
|
- **128 MiB RAM** (target); see [Memory](#memory) for the breakdown
|
||||||
|
- USB via **xHCI only**
|
||||||
|
|
||||||
|
There is no support for legacy BIOS boot, x2APIC, port-IO PCI configuration, or
|
||||||
|
any USB host controller other than xHCI.
|
||||||
|
|
||||||
|
## Plain-language hardware guide
|
||||||
|
|
||||||
|
If you don't want to cross-reference chipset datasheets, here's roughly what era
|
||||||
|
of PC works. These are **guidance based on when the required features became
|
||||||
|
standard**, not a list of tested machines — the authoritative rules are in the
|
||||||
|
technical sections below.
|
||||||
|
|
||||||
|
The feature that sets the floor is **built-in xHCI USB** (danos supports no other
|
||||||
|
USB controller) combined with **UEFI firmware**. Both became standard on
|
||||||
|
mainstream desktops and laptops around **2012**.
|
||||||
|
|
||||||
|
| | Known-good baseline | Comfortable recommendation |
|
||||||
|
|---|---|---|
|
||||||
|
| **Intel** | 3rd-gen Core "Ivy Bridge" (2012) with a 7-series "Panther Point" chipset — Intel's first chipset with xHCI built in | 6th-gen Core "Skylake" (2015) or newer |
|
||||||
|
| **AMD** | A-series "Llano" APU with an A75 FCH (2011) — the industry's first chipset with built-in xHCI | Any AM4 platform, i.e. Ryzen (2017) or newer |
|
||||||
|
|
||||||
|
**AMD is not behind Intel here — it was first.** AMD's A75 FCH shipped with
|
||||||
|
native xHCI in April 2011, about a year *ahead* of Intel's 7-series (2012); AMD
|
||||||
|
was the first vendor to earn USB-IF certification for chipset-level USB 3.0. The
|
||||||
|
two "comfortable recommendation" dates differ only because they name convenient,
|
||||||
|
long-supported product lines (Skylake, Ryzen) — not because of any USB
|
||||||
|
capability gap. Every AMD desktop platform from the A75 FCH (2011) and FM2/AM3+
|
||||||
|
era onward has built-in xHCI, and any of them qualifies as a baseline.
|
||||||
|
|
||||||
|
Older 64-bit machines (e.g. Intel Core 2, Nehalem, Sandy Bridge) meet the CPU
|
||||||
|
requirements but typically **lack built-in xHCI and/or ship with BIOS instead of
|
||||||
|
UEFI**, so they are not supported.
|
||||||
|
|
||||||
|
### Matching your CPU by name
|
||||||
|
|
||||||
|
If you know your chip's marketing name or codename, find it here. Everything from
|
||||||
|
the **Supported** rows down works; the **Too old** row does not.
|
||||||
|
|
||||||
|
**Intel Core** (the "-lake"/"-bridge"/"-well" codenames):
|
||||||
|
|
||||||
|
| Status | Generation | Codename(s) | Year |
|
||||||
|
|---|---|---|---|
|
||||||
|
| Too old | 2nd gen | Sandy Bridge | 2011 |
|
||||||
|
| Supported (baseline) | 3rd gen | Ivy Bridge | 2012 |
|
||||||
|
| Supported | 4th–5th gen | Haswell, Broadwell | 2013–2014 |
|
||||||
|
| **Recommended** | 6th–9th gen | **Skylake**, Kaby Lake, Coffee Lake | 2015–2018 |
|
||||||
|
| Recommended | 10th–11th gen | Comet Lake, Ice Lake, Tiger Lake, Rocket Lake | 2019–2021 |
|
||||||
|
| Recommended | 12th gen+ | Alder Lake, Raptor Lake | 2021–2023 |
|
||||||
|
| Recommended | Core Ultra | Meteor Lake, Arrow Lake, Lunar Lake | 2023+ |
|
||||||
|
|
||||||
|
**AMD:**
|
||||||
|
|
||||||
|
| Status | Family | Codename(s) | Year |
|
||||||
|
|---|---|---|---|
|
||||||
|
| Supported (baseline) | A-series APU (A75/A85 FCH) | Llano, Trinity, Richland, Kaveri | 2011–2014 |
|
||||||
|
| Supported | FX (AM3+) | Bulldozer, Piledriver | 2011–2012 |
|
||||||
|
| **Recommended** | **Ryzen** 1000–5000 (AM4) | Summit/Pinnacle Ridge, Matisse, Vermeer (Zen–Zen 3) | 2017–2020 |
|
||||||
|
| Recommended | Ryzen 7000+ (AM5) | Raphael, Granite Ridge (Zen 4 / Zen 5) | 2022+ |
|
||||||
|
| Recommended | Threadripper / EPYC | Zen and later | 2017+ |
|
||||||
|
|
||||||
|
(These map generations to the era their platforms shipped built-in xHCI + UEFI;
|
||||||
|
they are guidance, not a tested-hardware list.)
|
||||||
|
|
||||||
|
**Two caveats that matter regardless of CPU:**
|
||||||
|
|
||||||
|
- **Firmware must be UEFI.** Many 2011-era machines could do either UEFI or
|
||||||
|
legacy BIOS — danos needs it set to UEFI. There is no BIOS boot path.
|
||||||
|
- **Input is PS/2 only, for now.** danos does not yet support USB
|
||||||
|
keyboards/mice. This is fine on most **laptops** (their built-in keyboards are
|
||||||
|
wired to a PS/2-style i8042 controller) but means a **desktop with only USB
|
||||||
|
ports** currently has no usable keyboard. USB HID input is planned.
|
||||||
|
|
||||||
|
Virtual machines are the easiest way to meet every requirement: QEMU (with OVMF/
|
||||||
|
UEFI, a `qemu-xhci` controller, and the default Q35 machine type), or any
|
||||||
|
hypervisor configured for UEFI firmware and an xHCI USB controller.
|
||||||
|
|
||||||
|
## CPU / architecture
|
||||||
|
|
||||||
|
| Requirement | Detail | Source |
|
||||||
|
|---|---|---|
|
||||||
|
| **x86-64, 64-bit only** | Kernel and loader are built exclusively for `x86_64`; the loader rejects any non-x86-64 kernel ELF (`error.WrongArchitecture`). | `build.zig:285`, `boot/efi.zig:418` |
|
||||||
|
| **Long mode + PAE + NX** | AP trampoline sets `CR4.PAE`, `EFER.LME`, `EFER.NXE`; NX is used in kernel page-table entries. | `system/kernel/architecture/x86_64/trampoline.s:62` |
|
||||||
|
| **SSE / SSE2** | Baseline: the compiler emits SSE for ordinary struct copies. Trampoline enables `CR4.OSFXSR` + `OSXMMEXCPT` and clears `CR0.EM`. | `build.zig:282`, `trampoline.s:62` |
|
||||||
|
| **`syscall` / `sysret`** | Primary user↔kernel entry path. `EFER.SCE` enabled; `STAR`/`LSTAR`/`SFMASK` programmed per core. (`int 0x80` exists as a parallel gate.) | `architecture/x86_64/per-cpu.zig:59`, `isr.s:169` |
|
||||||
|
| **Local APIC (xAPIC)** | LAPIC accessed via MMIO at `0xFEE00000`. LAPIC ID read as a `u8` — classic xAPIC. **x2APIC is not supported** (no MSR path). | `apic.zig:62`, `apic.zig:414` |
|
||||||
|
| **CPUID + RDTSC** | CPUID leaf `0x15` for TSC frequency; RDTSC is the monotonic clock. | `apic.zig:279`, `apic.zig:84` |
|
||||||
|
| **SMP (optional)** | Multi-core supported via INIT–SIPI–SIPI; ceiling `maximum_cpus = 128`. Single core is fine. Cores beyond the ceiling are parked. | `system/parameters.zig:16`, `apic.zig:144` |
|
||||||
|
|
||||||
|
## Firmware / boot
|
||||||
|
|
||||||
|
- **UEFI only.** A custom UEFI application loader is installed to
|
||||||
|
`\EFI\BOOT\BOOTX64.efi`. There is **no BIOS, multiboot, or limine** path. The
|
||||||
|
loader tolerates UEFI Class-3 machines with no legacy PIC/PIT.
|
||||||
|
(`build.zig:464`, `boot/efi.zig`)
|
||||||
|
- **ACPI is the hardware-discovery mechanism.** The RSDP is taken from the UEFI
|
||||||
|
configuration table (ACPI 2.0 GUID preferred, 1.0 fallback). Without a valid
|
||||||
|
RSDP there is **no device discovery** — no SMP, no IOAPIC routing, no PCI/USB.
|
||||||
|
(`efi.zig:578`, `boot-handoff.zig:144`)
|
||||||
|
- **Required ACPI tables:** MADT (interrupt topology), MCFG (PCIe ECAM base),
|
||||||
|
FADT (power / PM timer). Optionally consumed: HPET, DMAR, SPCR.
|
||||||
|
(`system/devices/acpi.zig:3`)
|
||||||
|
- The loader reads `/system/kernel`, `/system/services/init`, and
|
||||||
|
`/boot/initial-ramdisk.img` off the FAT boot volume. The kernel can boot
|
||||||
|
"kernel-only" without init or the ramdisk. (`efi.zig:14`, `efi.zig:66`)
|
||||||
|
|
||||||
|
## Interrupt controller
|
||||||
|
|
||||||
|
- **Local APIC + I/O APIC required.** I/O APIC base, GSI base, and MADT
|
||||||
|
interrupt-source overrides come from ACPI. (`cpu.zig:365`, `apic.zig:119`)
|
||||||
|
- **MSI supported** — edge-triggered, keyed by vector, no I/O APIC mask cycle.
|
||||||
|
Vector window 33–46, timer on 32, spurious on 47. (`system/kernel/irq.zig:70`,
|
||||||
|
`cpu.zig:397`)
|
||||||
|
- The legacy 8259 PIC is remapped and masked **only if present** (MADT
|
||||||
|
`PCAT_COMPAT`); it is not required. (`apic.zig:103`)
|
||||||
|
|
||||||
|
## PCI / PCIe
|
||||||
|
|
||||||
|
- **PCIe with ECAM (MMConfig) required.** The PCI bus driver maps the host
|
||||||
|
bridge's ECAM window (1 MiB config space per bus) and computes config
|
||||||
|
addresses directly. **There is no legacy CF8/CFC port-IO config path** — the
|
||||||
|
driver bails if the bridge exposes no ECAM window. The ECAM base comes from
|
||||||
|
the ACPI MCFG table. (`system/drivers/pci-bus/pci-bus.zig:41`, `acpi.zig:6`)
|
||||||
|
|
||||||
|
## USB
|
||||||
|
|
||||||
|
- **xHCI only.** The sole USB driver is `usb-xhci-bus`, and the device manager
|
||||||
|
binds it strictly to PCI prog-IF `0x30` (xHCI). UHCI / OHCI / EHCI exist only
|
||||||
|
as report strings with no driver behind them — **USB 1.x/2.0-only controllers
|
||||||
|
are not supported.** (`system/drivers/usb-xhci-bus/`,
|
||||||
|
`system/services/device-manager/device-manager.zig:34`)
|
||||||
|
- USB input (keyboard/mouse over HID) is future work; the current input stack is
|
||||||
|
PS/2. See [Buses & devices](#buses--devices).
|
||||||
|
|
||||||
|
## Timers
|
||||||
|
|
||||||
|
Calibration prefers, in order: (1) CPUID leaf `0x15` TSC frequency, (2) HPET,
|
||||||
|
(3) ACPI PM timer (3.579545 MHz, from FADT), (4) legacy PIT. Any one suffices —
|
||||||
|
HPET/PM-timer/PIT are optional fallbacks when CPUID `0x15` is absent.
|
||||||
|
(`apic.zig:180`)
|
||||||
|
|
||||||
|
- **TSC** — monotonic high-resolution clock.
|
||||||
|
- **LAPIC timer** — scheduler heartbeat, periodic at `timer_hz = 1000 Hz`.
|
||||||
|
(`parameters.zig:39`)
|
||||||
|
|
||||||
|
## Memory
|
||||||
|
|
||||||
|
**Target: 128 MiB RAM.** The system uses 4 KiB pages and a bitmap physical-frame
|
||||||
|
allocator built from the firmware memory map. There is no hardcoded minimum-RAM
|
||||||
|
constant — the allocator only panics if there is no usable region, or none large
|
||||||
|
enough to hold its own bitmap. (`system/kernel/pmm.zig:13`, `pmm.zig:77`)
|
||||||
|
|
||||||
|
Where the budget goes:
|
||||||
|
|
||||||
|
| Consumer | Size | Source |
|
||||||
|
|---|---|---|
|
||||||
|
| Kernel heap (cap, grown one page at a time) | up to **64 MiB** | `system/kernel/heap.zig:26` |
|
||||||
|
| Kernel stack, per CPU | 16 KiB | `parameters.zig:26` |
|
||||||
|
| IST stack, per CPU | 16 KiB | `parameters.zig:36` |
|
||||||
|
| User stack, per task | 8 pages / 32 KiB | `parameters.zig:32` |
|
||||||
|
| Max concurrent tasks | 32 | `parameters.zig:23` |
|
||||||
|
| Boot page-table pool | 64 frames / 256 KiB | `efi.zig:299` |
|
||||||
|
|
||||||
|
The 64 MiB heap cap plus kernel image, per-CPU stacks, task stacks, the frame
|
||||||
|
bitmap, and DMA-contiguous allocations fit comfortably within 128 MiB on a
|
||||||
|
single- or low-core-count machine. Very high core counts (toward the 128-CPU
|
||||||
|
ceiling) add per-CPU stack overhead and push toward more RAM.
|
||||||
|
|
||||||
|
**Note on the 4 GiB physmap:** the loader identity-maps and physmaps the low
|
||||||
|
4 GiB of address space with 2 MiB leaves. This is *virtual address* reach, not a
|
||||||
|
RAM requirement — RAM above 4 GiB simply needs an extra mapping window and is not
|
||||||
|
needed to boot. (`efi.zig:305`)
|
||||||
|
|
||||||
|
Virtual-memory layout (`boot-handoff.zig:47`):
|
||||||
|
|
||||||
|
| Region | Base |
|
||||||
|
|---|---|
|
||||||
|
| User space | `0x0000_7000_0000_0000` |
|
||||||
|
| Kernel heap | `0xFFFF_8000_0000_0000` |
|
||||||
|
| Physmap | `0xFFFF_8800_0000_0000` |
|
||||||
|
| Kernel image | `0xFFFF_FFFF_8000_0000` |
|
||||||
|
|
||||||
|
## Buses & devices
|
||||||
|
|
||||||
|
Buses with real drivers today:
|
||||||
|
|
||||||
|
- **PCIe** via ECAM (`pci-bus`)
|
||||||
|
- **xHCI USB** (`usb-xhci-bus`)
|
||||||
|
- **PS/2** keyboard + mouse (`ps2-bus`) — the current input stack
|
||||||
|
- **Serial UART** (16550/16450), configured from the ACPI SPCR table
|
||||||
|
|
||||||
|
**No storage driver exists yet.** AHCI / NVMe / IDE are named for reporting only;
|
||||||
|
there is no block-device driver. Persistent storage is future work.
|
||||||
|
|
||||||
|
## IOMMU
|
||||||
|
|
||||||
|
**Detection only; enforcement deferred.** The ACPI DMAR table is parsed for the
|
||||||
|
first VT-d DRHD unit and its capabilities are exposed via `PlatformInfo`
|
||||||
|
(`iommu_present`, `iommu_base`, `iommu_version`). No DMA-remapping tables are
|
||||||
|
programmed and no translation is enforced. An IOMMU is therefore **not required**
|
||||||
|
and does not currently constrain devices. (`system/devices/acpi.zig:96`)
|
||||||
|
|
||||||
|
## What is explicitly NOT supported
|
||||||
|
|
||||||
|
- Legacy BIOS / multiboot / limine boot
|
||||||
|
- 32-bit x86
|
||||||
|
- x2APIC
|
||||||
|
- Legacy port-IO (CF8/CFC) PCI configuration
|
||||||
|
- Non-xHCI USB (UHCI / OHCI / EHCI)
|
||||||
|
- Machines without ACPI (no device discovery)
|
||||||
|
- Persistent storage (no AHCI / NVMe / IDE driver yet)
|
||||||
|
- USB HID input (PS/2 only for now)
|
||||||
+36
-4
@@ -1,6 +1,6 @@
|
|||||||
# SysV: the kernel's calling convention
|
# SysV: the kernel's calling convention
|
||||||
|
|
||||||
Several places in danos say "the kernel is SysV" — most visibly `src/root.zig`:
|
Several places in danos say "the kernel is SysV" — most visibly `system/boot-handoff.zig`:
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
pub const kernel_abi: std.builtin.CallingConvention = .{ .x86_64_sysv = .{} };
|
pub const kernel_abi: std.builtin.CallingConvention = .{ .x86_64_sysv = .{} };
|
||||||
@@ -52,19 +52,51 @@ argument arrives in **RCX**, not RDI.
|
|||||||
|
|
||||||
danos's two binaries default to different conventions:
|
danos's two binaries default to different conventions:
|
||||||
|
|
||||||
- `src/boot/efi.zig` is built for the UEFI target, so its default C convention is
|
- `boot/efi.zig` is built for the UEFI target, so its default C convention is
|
||||||
Microsoft x64 (first argument → RCX).
|
Microsoft x64 (first argument → RCX).
|
||||||
- The kernel is freestanding, so its convention is SysV (first argument → RDI).
|
- The kernel is freestanding, so its convention is SysV (first argument → RDI).
|
||||||
|
|
||||||
When the loader jumps to the kernel passing the `BootInfo` pointer, both sides have
|
When the loader jumps to the kernel passing the `BootInfo` pointer, both sides have
|
||||||
to agree *which register that pointer lands in*. Left to their defaults, the loader
|
to agree *which register that pointer lands in*. Left to their defaults, the loader
|
||||||
would place it in RCX while the kernel looked in RDI — and the kernel would read
|
would place it in RCX while the kernel looked in RDI — and the kernel would read
|
||||||
garbage. So both sides reference the same `danos.kernel_abi` (SysV): the loader's
|
garbage. So both sides reference the same `system.kernel_abi` (SysV): the loader's
|
||||||
function-pointer type and the kernel's `_start` both carry
|
function-pointer type and the kernel's `_start` both carry
|
||||||
`callconv(danos.kernel_abi)`, and the pointer reliably arrives in RDI. That is the
|
`callconv(system.kernel_abi)`, and the pointer reliably arrives in RDI. That is the
|
||||||
whole reason `kernel_abi` lives in the shared contract — see [efi.md](efi.md) for
|
whole reason `kernel_abi` lives in the shared contract — see [efi.md](efi.md) for
|
||||||
the handoff it governs.
|
the handoff it governs.
|
||||||
|
|
||||||
|
## The process-entry stack (argc/argv)
|
||||||
|
|
||||||
|
The SysV ABI also fixes what a *fresh process* finds on its stack — and danos
|
||||||
|
follows it, so its own runtime and any future C libc read arguments the same way.
|
||||||
|
At the first user instruction, `rsp` is 16-byte aligned and points at (addresses
|
||||||
|
growing upward):
|
||||||
|
|
||||||
|
```
|
||||||
|
rsp → argc u64
|
||||||
|
argv[0] … argv[argc-1] pointers into the strings area below
|
||||||
|
NULL argv terminator
|
||||||
|
NULL envp terminator (no environment yet)
|
||||||
|
{AT_PAGESZ, page size} auxiliary vector
|
||||||
|
{AT_NULL, 0} auxiliary-vector terminator
|
||||||
|
argv string bytes NUL-terminated
|
||||||
|
───────────────────────── stack top (stack_top_virtual)
|
||||||
|
```
|
||||||
|
|
||||||
|
The kernel builds this block at the top of the process's stack — 8 pages (32 KiB,
|
||||||
|
`parameters.user_stack_pages`) mapped RW+NX below a fixed top, with the page below
|
||||||
|
them left unmapped as a **guard**, so a stack overflow faults (killing only that
|
||||||
|
process) instead of silently corrupting the image
|
||||||
|
(`buildEntryStack` in `system/kernel/process.zig`); `argv[0]` is always the path
|
||||||
|
or initial-ramdisk name the process was spawned as, and `system_spawn`'s optional
|
||||||
|
argument blob becomes `argv[1..]`. The runtime's `_start`
|
||||||
|
(`library/runtime/start.zig`) hands the block to `rt_start`, which builds a
|
||||||
|
`runtime.process.Init` from it and passes that to the program's `main`
|
||||||
|
(`pub fn main(init: runtime.process.Init)`; a parameterless `main()` is also
|
||||||
|
accepted). A C runtime's `crt0` would walk
|
||||||
|
the identical layout unmodified — that's the compatibility being bought. The
|
||||||
|
`args` test proves the round trip.
|
||||||
|
|
||||||
## Where else it surfaces
|
## Where else it surfaces
|
||||||
|
|
||||||
- **The red zone → `red_zone = false`.** `build.zig` disables the red zone for the
|
- **The red zone → `red_zone = false`.** `build.zig` disables the red zone for the
|
||||||
|
|||||||
+7
-6
@@ -8,8 +8,9 @@ without a human staring at the screen.
|
|||||||
There are two layers:
|
There are two layers:
|
||||||
|
|
||||||
- **Host unit tests** (`zig build test`) — for pure, platform-independent logic in
|
- **Host unit tests** (`zig build test`) — for pure, platform-independent logic in
|
||||||
the shared `danos` module (the handoff layout in `src/root.zig`). These compile
|
the shared contracts (`system/boot-handoff.zig`, `system/abi.zig`,
|
||||||
for the host and run natively.
|
`system/devices/device-abi.zig`), which also compile-checks the three-way split
|
||||||
|
stays self-consistent. These compile for the host and run natively.
|
||||||
- **QEMU integration tests** (`python3 test/qemu_test.py`) — boot the real kernel
|
- **QEMU integration tests** (`python3 test/qemu_test.py`) — boot the real kernel
|
||||||
and check its behaviour. This is the interesting part.
|
and check its behaviour. This is the interesting part.
|
||||||
|
|
||||||
@@ -17,7 +18,7 @@ There are two layers:
|
|||||||
|
|
||||||
The framebuffer console draws pixels, which a test can't read without
|
The framebuffer console draws pixels, which a test can't read without
|
||||||
screen-scraping. So the kernel also writes everything to a **serial port**
|
screen-scraping. So the kernel also writes everything to a **serial port**
|
||||||
(`src/kernel/arch/x86_64/serial.zig`, a 16550 UART on COM1). `Console.write` mirrors every
|
(`system/kernel/architecture/x86_64/serial.zig`, a 16550 UART on COM1). `Console.write` mirrors every
|
||||||
byte to it, so all kernel output — boot log, memory summary, exception reports —
|
byte to it, so all kernel output — boot log, memory summary, exception reports —
|
||||||
appears on serial as plain text.
|
appears on serial as plain text.
|
||||||
|
|
||||||
@@ -29,7 +30,7 @@ a new architecture's UART is what makes the same tests run there.
|
|||||||
## In-kernel test cases
|
## In-kernel test cases
|
||||||
|
|
||||||
Building with `-Dtest-case=<name>` makes the kernel, after normal bring-up, run one
|
Building with `-Dtest-case=<name>` makes the kernel, after normal bring-up, run one
|
||||||
self-test from `src/kernel/tests.zig` instead of idling. Each case writes structured
|
self-test from `system/kernel/tests.zig` instead of idling. Each case writes structured
|
||||||
markers to serial:
|
markers to serial:
|
||||||
|
|
||||||
```
|
```
|
||||||
@@ -108,7 +109,7 @@ firmware, boot method, serial device). The cases are architecture-neutral —
|
|||||||
So bringing up a second architecture — an AArch64 Raspberry Pi is the motivating
|
So bringing up a second architecture — an AArch64 Raspberry Pi is the motivating
|
||||||
one — means:
|
one — means:
|
||||||
|
|
||||||
1. implement `src/kernel/arch/aarch64/` (CPU ops, its UART, exception vectors, page
|
1. implement `system/kernel/arch/aarch64/` (CPU ops, its UART, exception vectors, page
|
||||||
tables) behind the same `arch` interface,
|
tables) behind the same `arch` interface,
|
||||||
2. add an `aarch64` entry to `ARCHES` with its `qemu-system-aarch64` invocation,
|
2. add an `aarch64` entry to `ARCHES` with its `qemu-system-aarch64` invocation,
|
||||||
|
|
||||||
@@ -118,7 +119,7 @@ architectures".
|
|||||||
|
|
||||||
## Writing a new case
|
## Writing a new case
|
||||||
|
|
||||||
1. Add a function to `src/kernel/tests.zig` and dispatch it in `run` on its name.
|
1. Add a function to `system/kernel/tests.zig` and dispatch it in `run` on its name.
|
||||||
2. Emit `[PASS]/[FAIL]` lines and a `DANOS-TEST-RESULT:` line (non-faulting cases),
|
2. Emit `[PASS]/[FAIL]` lines and a `DANOS-TEST-RESULT:` line (non-faulting cases),
|
||||||
or trigger the condition and rely on the handler's output (faulting cases).
|
or trigger the condition and rely on the handler's output (faulting cases).
|
||||||
3. Add an entry to `CASES` in `test/qemu_test.py` with the regex that proves it.
|
3. Add an entry to `CASES` in `test/qemu_test.py` with the regex that proves it.
|
||||||
|
|||||||
+117
@@ -0,0 +1,117 @@
|
|||||||
|
# Timers and time
|
||||||
|
|
||||||
|
Two different needs hide under the word "timer", and danos keeps them apart:
|
||||||
|
|
||||||
|
- **Reading the clock** — *what time is it?* A read of a free-running counter.
|
||||||
|
- **Waiting** — *wake me in N milliseconds*, or *notify me when a deadline passes.*
|
||||||
|
|
||||||
|
Both are answered by the **kernel**, because the kernel already owns a timer: it has
|
||||||
|
to, to preempt tasks. The LAPIC heartbeat and the calibrated TSC that back all of this
|
||||||
|
are built in [device-interrupts.md](device-interrupts.md); the scheduler's blocking and
|
||||||
|
wait queues are in [scheduling.md](scheduling.md). This page is about the surface a
|
||||||
|
ring-3 program actually uses, and one deliberate absence: **there is no user-space time
|
||||||
|
service.**
|
||||||
|
|
||||||
|
## Why time is a syscall, not a service
|
||||||
|
|
||||||
|
The tempting microkernel move is to put a timer *driver* in user space and have
|
||||||
|
applications ask it for the time over IPC. For a **monotonic clock that is wrong** —
|
||||||
|
reading `now()` should never cost an IPC round trip. The kernel is already holding the
|
||||||
|
answer: it computes the current time every time it schedules, from the TSC, in a couple
|
||||||
|
of instructions. Surfacing that as a system call is pure mechanism; routing it through a
|
||||||
|
message to another process would be slower *and* redundant, and a device like the HPET
|
||||||
|
(uncacheable MMIO reads) is a particularly bad thing to read on every `now()`.
|
||||||
|
|
||||||
|
This is the same conclusion every serious system reaches: Linux and Zircon read the
|
||||||
|
counter in the vDSO, L4 exposes a clock field in a shared kernel page, seL4 reads the
|
||||||
|
cycle counter directly. None of them make a clock read an IPC. danos makes it a syscall.
|
||||||
|
|
||||||
|
That "from the TSC" hides a portability question, because the TSC is only a valid clock
|
||||||
|
when the CPU guarantees it is *invariant* and when every core's TSC is *synchronized*.
|
||||||
|
danos checks both — the invariant-TSC CPUID bit (`0x80000007` EDX[8], set on Intel and
|
||||||
|
AMD), and a cross-core "warp" check as the cores come up — and falls back to the HPET
|
||||||
|
counter when either fails. So `now()` stays accurate on a real Intel box, a real AMD box,
|
||||||
|
and inside a VM alike; only the source behind it differs. The mechanism is in
|
||||||
|
[device-interrupts.md](device-interrupts.md).
|
||||||
|
|
||||||
|
So the timer hardware lives in the kernel, and there is **no `hpet` driver and no time
|
||||||
|
server** to consume. (An earlier HPET driver existed only to *demonstrate* the driver
|
||||||
|
model; that role now lives in [drivers.md](drivers.md), as documentation.) The one place
|
||||||
|
a user-space time service *is* justified — **wall-clock / calendar time** — is discussed
|
||||||
|
at the end; it is deliberately not built yet.
|
||||||
|
|
||||||
|
## The three system calls
|
||||||
|
|
||||||
|
Time and waiting are three entries in the small syscall table ([syscall.md](syscall.md)):
|
||||||
|
|
||||||
|
- **`clock` (#23)** → monotonic nanoseconds since boot. It only moves forward. Not
|
||||||
|
wall-clock: no date, no timezone. Backed by `architecture.nanos()` (TSC, scaled with a
|
||||||
|
128-bit intermediate so a long uptime can't overflow) — a few nanoseconds of
|
||||||
|
resolution, and just an `rdtsc` plus a multiply.
|
||||||
|
- **`sleep` (#3)** → block the caller for N milliseconds. The scheduler records a wake
|
||||||
|
deadline and the tick sweep wakes it (`scheduler.sleep`).
|
||||||
|
- **`timer_bind` (#31)** → arm a one-shot timer that, after N milliseconds, posts a
|
||||||
|
**timer notification** to an IPC endpoint. Unlike `sleep` it does **not** block: a
|
||||||
|
service can keep answering messages on the same endpoint while a deadline is pending.
|
||||||
|
This is the timed wait that stop-sequence escalation, hello deadlines, and restart
|
||||||
|
backoff are built from ([process-lifecycle.md](process-lifecycle.md),
|
||||||
|
[device-manager.md](device-manager.md)).
|
||||||
|
|
||||||
|
The kernel's own scheduling timer (the LAPIC, vector 32) is never exposed to user space;
|
||||||
|
programs read the TSC through `clock` and get timed wakeups through `sleep`/`timer_bind`,
|
||||||
|
both riding the scheduler tick.
|
||||||
|
|
||||||
|
## `runtime.time` — the generic interface
|
||||||
|
|
||||||
|
Applications don't call the syscalls directly; they use `runtime.time`
|
||||||
|
(`library/runtime/time.zig`), a thin `Instant`/`Duration` layer over them — an ergonomic
|
||||||
|
front door, not new mechanism.
|
||||||
|
|
||||||
|
```zig
|
||||||
|
const time = @import("runtime").time;
|
||||||
|
|
||||||
|
const start = time.now(); // Instant — monotonic
|
||||||
|
doWork();
|
||||||
|
const took = start.elapsed(); // Duration
|
||||||
|
time.sleep(time.Duration.fromMillis(5)); // block ~5 ms
|
||||||
|
|
||||||
|
// A deadline delivered as a notification, so a service keeps serving meanwhile:
|
||||||
|
_ = time.after(endpoint, time.Duration.fromMillis(200));
|
||||||
|
```
|
||||||
|
|
||||||
|
- `Duration` is nanoseconds under the hood, with `fromNanos/fromMicros/fromMillis/
|
||||||
|
fromSeconds` and `asNanos/asMillis`. `ceilMillis` rounds *up* to the kernel's
|
||||||
|
millisecond granularity, so a sub-millisecond `sleep` never rounds down to zero and
|
||||||
|
returns early. All arithmetic saturates rather than wraps.
|
||||||
|
- `Instant` is a point on the monotonic clock: `since`, `elapsed`, `plus`, `reached` —
|
||||||
|
built for deadline loops (`while (!deadline.reached()) …`).
|
||||||
|
- `now()` / `monotonicNanos()` wrap `clock`. `available()` reports whether the clock is
|
||||||
|
calibrated at all (the kernel returns 0 until the TSC frequency is known, so a caller
|
||||||
|
that needs real time can treat 0 as "unavailable" rather than assume it advances).
|
||||||
|
- `sleep(d)` wraps `sleep`; `spin(d)` busy-polls `now()` for the sub-millisecond delays
|
||||||
|
the millisecond tick can't express; `after(endpoint, d)` wraps `timer_bind`.
|
||||||
|
|
||||||
|
The raw wrappers (`system.clock`, `system.sleep`, `system.timerOnce`) stay in
|
||||||
|
`library/runtime/system.zig`; `runtime.time` is the layer meant for everyday use.
|
||||||
|
|
||||||
|
## Wall-clock time (not built)
|
||||||
|
|
||||||
|
Everything above is **monotonic**: elapsed time since boot, perfect for timeouts and
|
||||||
|
measurement, useless for "what is the date?" Calendar time — a real-time clock, time
|
||||||
|
zones, leap seconds — is genuinely a **user-space** concern, and it *is* the case a time
|
||||||
|
service is for. It would be backed by an **RTC** driver (the CMOS real-time clock), not
|
||||||
|
the HPET, and exposed as a `CLOCK_REALTIME`-style service alongside the monotonic
|
||||||
|
syscall. It is deferred until something needs it; the monotonic clock the kernel already
|
||||||
|
owns covers every current use.
|
||||||
|
|
||||||
|
## Verifying it
|
||||||
|
|
||||||
|
`runtime.time`'s `Instant`/`Duration` arithmetic has unit tests that run on the host:
|
||||||
|
|
||||||
|
```
|
||||||
|
$ zig build test # includes library/runtime/time.zig
|
||||||
|
```
|
||||||
|
|
||||||
|
End to end, the proof the clock is real is that it *advances*: read `now()`, `sleep` a
|
||||||
|
`Duration`, read `now()` again, and the second reading is later — the kernel's timer
|
||||||
|
driving a ring-3 program with no service in between.
|
||||||
+13
-4
@@ -81,11 +81,20 @@ prerequisites.
|
|||||||
(with boot-services memory reclaimed), [paging](paging.md) with W^X, [exceptions and
|
(with boot-services memory reclaimed), [paging](paging.md) with W^X, [exceptions and
|
||||||
interrupts](interrupts.md), a [calibrated timer + ns clock](device-interrupts.md), a
|
interrupts](interrupts.md), a [calibrated timer + ns clock](device-interrupts.md), a
|
||||||
[heap](heap.md), a [fixed-priority preemptive scheduler](scheduling.md) with blocking,
|
[heap](heap.md), a [fixed-priority preemptive scheduler](scheduling.md) with blocking,
|
||||||
and in-kernel [IPC channels](ipc.md) — plus a [test harness](testing.md).
|
in-kernel [IPC channels](ipc.md), SMP (all cores scheduling, with affinity), a
|
||||||
|
**higher-half kernel** with a physmap, and **user space**: per-process address
|
||||||
|
spaces, `syscall`/`sysret` with the `swapgs` discipline, a user-ELF loader, and
|
||||||
|
`/system/services/init` — a real user ELF built from `system/services/init/`, running at CPL 3 as PID 1 on its
|
||||||
|
own page tables — plus a [test harness](testing.md).
|
||||||
|
|
||||||
- **Isolation track** — **user mode + address-space isolation** (higher-half kernel,
|
- **Isolation track** — **user mode + address-space isolation**. *Done: a
|
||||||
ring 3, per-process page tables). The substrate everything else needs. *Next, and a
|
higher-half kernel with a physmap (the low half is user space), per-process
|
||||||
prerequisite for the resilience and driver tracks.*
|
address spaces with CR3 switched on context switch, the `swapgs` discipline,
|
||||||
|
`syscall`/`sysret`, a user-ELF loader, and `/system/services/init` running as a real
|
||||||
|
preemptive ring-3 process (PID 1). Remaining polish: an address-space/stack
|
||||||
|
reaper for exited tasks, SMAP + fault-recovering copy-in/out, the real IPC
|
||||||
|
syscalls (IPC_Call/IPC_ReplyWait — they arrive with the second user server),
|
||||||
|
and TLB shootdown once a process has more than one thread.*
|
||||||
- **Resilience track** — fault → kill → notify, a supervisor/reincarnation server,
|
- **Resilience track** — fault → kill → notify, a supervisor/reincarnation server,
|
||||||
resource cleanup on death, then a restartable driver as proof. Needs isolation.
|
resource cleanup on death, then a restartable driver as proof. Needs isolation.
|
||||||
See [resilience.md](resilience.md).
|
See [resilience.md](resilience.md).
|
||||||
|
|||||||
@@ -0,0 +1,349 @@
|
|||||||
|
# Running Zig on danos: the self-hosting roadmap
|
||||||
|
|
||||||
|
A design note (not built yet) on the path to making danos a **real Zig target** — a
|
||||||
|
target you can name (`-target x86_64-danos`) and, eventually, run the Zig compiler
|
||||||
|
itself on. It is forward-looking, like [vision.md](vision.md): it sets a direction
|
||||||
|
and the decisions that follow from it, so the code we write now bends toward it
|
||||||
|
instead of away.
|
||||||
|
|
||||||
|
This note deliberately does **not** cover a text editor or terminal. Those are
|
||||||
|
easier (single-process, I/O-bound) and fall out of the early phases here almost for
|
||||||
|
free; the hard, shaping problem is the standard-library surface, so that is what
|
||||||
|
this roadmap is about.
|
||||||
|
|
||||||
|
The analysis behind it was done against **Zig 0.16** (the pinned toolchain). Zig's
|
||||||
|
standard library moves between releases — especially the parts described here — so
|
||||||
|
treat upstream references as "the shape in 0.16.x," and expect to re-check them on a
|
||||||
|
toolchain bump.
|
||||||
|
|
||||||
|
## The win condition
|
||||||
|
|
||||||
|
danos runs the Zig compiler when a bare
|
||||||
|
|
||||||
|
```
|
||||||
|
zig build-exe hello.zig
|
||||||
|
```
|
||||||
|
|
||||||
|
completes **on danos** and produces a runnable danos binary. Note the milestone is
|
||||||
|
`build-exe`, not `zig build`: the `zig build` runner spawns child processes (the
|
||||||
|
build steps), which needs a whole process-control surface danos does not have yet.
|
||||||
|
A single `build-exe` needs none of that (see Phase 3). Reaching `build-exe` is
|
||||||
|
"self-hosting"; reaching `zig build` is a later, separate lift.
|
||||||
|
|
||||||
|
### Non-goals
|
||||||
|
|
||||||
|
- **No Linux syscall/ABI emulation.** danos will not implement the Linux `syscall`
|
||||||
|
interface so that stock `x86_64-linux` binaries run. That is a permanent
|
||||||
|
compatibility treadmill and it inverts the microkernel design — explicitly out.
|
||||||
|
- **No musl port yet.** A musl libc port is a reasonable *later* effort (it unlocks
|
||||||
|
the C ecosystem), but it is not on the critical path to Zig-on-danos, and it is
|
||||||
|
deferred. The roadmap below is arranged so the work still pays off if musl ever
|
||||||
|
happens (see "The same surface, twice").
|
||||||
|
- **Editor/terminal are out of scope for this note** (they are downstream of Phase 1).
|
||||||
|
|
||||||
|
**On FFI.** Foreign-function interop splits the same way as the doors below. Zig-level
|
||||||
|
and C-ABI-*exposing* FFI (`extern`, `callconv(.c)`, C-ABI structs) work on a real target
|
||||||
|
immediately — and the `std.os.danos` seam is C-ABI-shaped by construction, so it is
|
||||||
|
FFI-friendly from the start. *Consuming* C libraries (`@cImport`, linking archives) is
|
||||||
|
the part that needs a libc + headers, i.e. the deferred musl door. So an eventual FFI
|
||||||
|
need reinforces keeping that door open; it does not change the plan.
|
||||||
|
|
||||||
|
## The realization that shapes everything: 0.16 gives us *one* seam
|
||||||
|
|
||||||
|
The instinct "to target Zig we'd have to reimplement all the `std` namespaces" was
|
||||||
|
how older Zig worked. Zig 0.16 (post-"writergate") is far kinder:
|
||||||
|
|
||||||
|
- **`std.fs` is essentially gone.** It is now path helpers plus deprecated aliases;
|
||||||
|
there is no `std.fs.File`, `std.fs.Dir`, or `std.fs.cwd()`. File and directory
|
||||||
|
work goes through **`std.Io`** — a single runtime **vtable** (`Io.zig`) of
|
||||||
|
function pointers handed to `main` as `std.process.Init.io`. `std.Io.File` and
|
||||||
|
`std.Io.Dir` are thin forwarders to that vtable. `Io.zig` and the `fs` shim carry
|
||||||
|
**zero** per-OS branches.
|
||||||
|
- **`std.posix` is one generic body** parameterised over a single `system` module.
|
||||||
|
With no libc, `system` resolves **per target OS**: `.linux => std.os.linux`,
|
||||||
|
`.plan9 => std.os.plan9`, and so on. The generic `std.posix.read`/`write`/`open`
|
||||||
|
bodies are just `system.read(...)` plus an errno switch — *identical for every
|
||||||
|
OS*. The only variable is what `system` binds to.
|
||||||
|
- **`std.os.<tag>`** (e.g. `std/os/linux.zig`) is therefore the real porting seam: a
|
||||||
|
low-level, C-ABI-shaped module of `read/write/open/close/lseek/mmap/clock/exit/…`
|
||||||
|
plus an `errno` enum and the constant tables (`O_*`, `CLOCK_*`, `S_*`).
|
||||||
|
|
||||||
|
Put together: **to port danos we write `std.os.danos` once** — the ~30-operation
|
||||||
|
seam — and the whole `std.posix` / `std.fs` / `std.Io` tower above it lights up
|
||||||
|
generically, because none of it branches on the OS. That is a dramatically smaller
|
||||||
|
and more contained target than "reimplement the namespaces."
|
||||||
|
|
||||||
|
## Three doors, and why we take the first
|
||||||
|
|
||||||
|
| Door | What it is | Verdict |
|
||||||
|
|------|-----------|---------|
|
||||||
|
| **1. Implement the std seam** (`std.os.danos`) | Write the ~30-op `system` module over danos's native ABI + VFS; the generic std tower lights up. | **Take this.** The only door that touches neither C nor the Linux ABI. |
|
||||||
|
| **2. Port musl** | Port musl libc to danos, link Zig against it. | Defer. Good later for the *C* ecosystem; barely helps *Zig* (std only uses libc on the libc-linked path). |
|
||||||
|
| **3. Emulate the Linux ABI** | Implement Linux syscalls so stock linux binaries run. | Reject. Bottomless compatibility treadmill; against the design. |
|
||||||
|
|
||||||
|
### The same surface, twice
|
||||||
|
|
||||||
|
Doors 1 and 2 are the **same native surface at different layers**. `std.posix.read`
|
||||||
|
is `system.read(...)` + an errno switch *regardless of OS* — the only question is
|
||||||
|
whether `system` is **`std.os.danos` (Zig)** or **musl (C)**. Either way, the set of
|
||||||
|
danos-facing operations you must implement is the *same* ~30 ops, all bottoming out
|
||||||
|
in danos's native syscalls + the VFS/FAT server.
|
||||||
|
|
||||||
|
So the runtime work below is **not throwaway** if musl ever happens: you are building
|
||||||
|
the danos-native implementations of that surface either way. Door 1 just packages
|
||||||
|
them as Zig; a future musl re-uses the identical kernel/VFS operations underneath. The
|
||||||
|
two symmetries worth keeping in mind: doors 1 and 2 converge at the **top** (identical
|
||||||
|
POSIX surface); doors 2 and 3 converge at the **bottom** (unmodified musl needs the
|
||||||
|
Linux syscall ABI). Door 1 is the only one that avoids both C and Linux.
|
||||||
|
|
||||||
|
### A fork is table stakes — for any door
|
||||||
|
|
||||||
|
`std.Target.Os.Tag` is a **closed enum** baked into the compiler binary *and* into
|
||||||
|
the `std` linked with every program; `-target x86_64-danos` resolves through it. So
|
||||||
|
adding `danos` as a name requires patching and rebuilding the compiler — even the
|
||||||
|
musl door needs this. "Fork Zig" is therefore not an extra cost unique to door 1; it
|
||||||
|
is the price of admission for *any* real target. What door 1 adds on top is small and
|
||||||
|
localised (below).
|
||||||
|
|
||||||
|
## The architecture decision: `runtime.os` + `runtime.fs`, and retire `posix`
|
||||||
|
|
||||||
|
danos already has the right split ([the private-ABI boundary](../README.md)): the
|
||||||
|
kernel exposes a minimal syscall ABI ([syscall.md](syscall.md)); the **`runtime`**
|
||||||
|
library is the stable, danos-native application ABI. What this roadmap adds:
|
||||||
|
|
||||||
|
- **`runtime.os` — the seam.** A C-ABI-shaped module of the ~30 operations
|
||||||
|
(`read/write/open/close/lseek/mmap/munmap/clock/exit/…`) + an errno enum + the
|
||||||
|
constant tables, each backed by danos's native syscalls and the VFS. **Structure it
|
||||||
|
to mirror `std/os/linux.zig`.** This is the load-bearing, *non-throwaway* artifact:
|
||||||
|
when we fork Zig, `runtime.os` is copy-pasted (near-verbatim) into `std.os.danos`.
|
||||||
|
- **`runtime.fs` — the thin native file API** danos programs use *today*, layered
|
||||||
|
over `runtime.os`. It is also the concrete backing for the `std.Io` vtable's
|
||||||
|
file-write entry once we're a real target, which is why program stdout, diagnostics,
|
||||||
|
and file writes should all be *decided once at that seam* rather than as bespoke
|
||||||
|
per-call helpers (see "How this informs decisions now").
|
||||||
|
|
||||||
|
**Do not hand-mirror the high-level std namespaces.** `std.fs`/`std.Io`/`std.process`
|
||||||
|
are generic and OS-agnostic; once `std.os.danos` exists and we fork, upstream *gives*
|
||||||
|
them to danos for free. Hand-writing `runtime.std.fs` to imitate them would be
|
||||||
|
redundant the day the fork works, and it would chase a moving target (0.16's `std.Io`
|
||||||
|
is large and still shifting). Build the seam well; take the tower for free.
|
||||||
|
|
||||||
|
**Why not a library called `std`?** Because `@import("std")` resolves to the
|
||||||
|
compiler-provided standard library; a user module named `std` would *shadow* it for
|
||||||
|
anything that imports it that way. That is the real reason the seam lives *inside* a
|
||||||
|
forked std as `std/os/danos.zig`, not as a `runtime.std` library — and why danos's end
|
||||||
|
state (`@import("std")` just working, and knowing danos) is the most natively Zig it can
|
||||||
|
be. `runtime.os` is only the interim staging ground: developed against the stock
|
||||||
|
toolchain so Phase 1 need not wait on the fork, then promoted near-verbatim into the
|
||||||
|
fork's `std/os/danos.zig`.
|
||||||
|
|
||||||
|
### Retire `library/posix`
|
||||||
|
|
||||||
|
The `posix` compatibility layer (`unistd`, `stdio`) was the right instinct too early.
|
||||||
|
Its whole value is POSIX *spellings* for POSIX software — and danos has no POSIX
|
||||||
|
software; every current caller is danos-native code that could use `runtime.fs`
|
||||||
|
directly. The real POSIX story arrives later and from elsewhere (musl, or upstream
|
||||||
|
`std`'s own posix over `std.os.danos`), which supersedes a hand-rolled shim. So it is
|
||||||
|
premature abstraction that adds a "which layer do I use?" fork with no payoff yet.
|
||||||
|
|
||||||
|
Its footprint is tiny: **five** call sites, all `unistd` file operations —
|
||||||
|
`system/services/fat/fat.zig` (`mount`), the `vfs-test` and `fat-test` clients, and
|
||||||
|
(from the boot-log work) `init.zig` and `log-flush.zig`. `stdio.zig` is dead — nothing
|
||||||
|
imports it. The plan: build `runtime.fs`, migrate those five to it, delete
|
||||||
|
`library/posix/`, and drop the `posix` module from `build.zig`'s `addUserBinary`.
|
||||||
|
|
||||||
|
## Where danos stands: coverage vs. the gaps
|
||||||
|
|
||||||
|
What the seam needs, and what danos already provides:
|
||||||
|
|
||||||
|
| std need | danos today | Gap |
|
||||||
|
|----------|-------------|-----|
|
||||||
|
| open / read / write / close / lseek | VFS (via the current `unistd`, → `runtime.fs`) | none — repackage |
|
||||||
|
| directory read (`getdents`) | VFS `readdir` | none — repackage |
|
||||||
|
| mmap / munmap | native syscalls ([abi.zig](../system/abi.zig)) | none |
|
||||||
|
| page allocator | over `mmap`, via `root.os.heap.page_allocator` override | ~30-line hook |
|
||||||
|
| monotonic clock | `clock` syscall | none |
|
||||||
|
| args / argv | SysV entry stack ([sysv.md](sysv.md)), `runtime.process.Init` | none |
|
||||||
|
| stdout / stderr | `debug_write` today | wire fd 1/2 to a console **byte** stream |
|
||||||
|
| mkdir / unlink / rename / truncate | done — engine + VFS + `runtime.fs` (Phase 2) | — |
|
||||||
|
| stat fields | `{size, kind, mtime}` | **mode / inode** still missing (cache validity) |
|
||||||
|
| wall-clock / realtime | done — `wall_clock` syscall (CMOS RTC, Phase 2d) | — |
|
||||||
|
| **environment variables** | `Init` has no env field | missing (can start empty) |
|
||||||
|
| **cwd / chdir** | paths are absolute or bare | missing (no cwd anchor) |
|
||||||
|
| **entropy / random** | — | missing (needed behind `vtable.random`) |
|
||||||
|
| process spawn + exit status | `system_spawn` starts a *named ramdisk binary*; `ExitReason` is a *category* | no exec-of-path, no numeric `WEXITSTATUS` |
|
||||||
|
| threads | one thread per process | avoided via `-fsingle-threaded` (below) |
|
||||||
|
| symlinks | `NodeKind` has the tag; unimplemented | low priority |
|
||||||
|
|
||||||
|
The clustering is clear: reads and memory are basically done; the real work is
|
||||||
|
**filesystem mutation + richer stat + wall-clock**, and a few small seam pieces
|
||||||
|
(page-allocator hook, stdio bytes, entropy). Process spawning and threads are
|
||||||
|
side-stepped entirely for a single `build-exe`.
|
||||||
|
|
||||||
|
## The roadmap
|
||||||
|
|
||||||
|
### Phase 0 — Make `danos` a real target
|
||||||
|
|
||||||
|
**Host, target, self-host — keep the three roles straight.** The *host* is where the
|
||||||
|
compiler runs (your mac + linux dev machines); the *target* is what it emits (`danos`);
|
||||||
|
and eventually danos becomes a host too (self-hosting — the win condition). So the move
|
||||||
|
is: fork the compiler, build it **for** your dev hosts, and teach it to **cross-compile
|
||||||
|
to** danos. You already do this — danos is cross-compiled `freestanding` from your dev
|
||||||
|
host today; Phase 0 swaps that `freestanding` target for a real `x86_64-danos` one, which
|
||||||
|
is what unlocks the native `std`.
|
||||||
|
|
||||||
|
**Why a compiler fork, not just a `--zig-lib-dir` override.** `std.Target.Os.Tag` is a
|
||||||
|
*closed enum compiled into the compiler binary*, so `-target x86_64-danos` will not even
|
||||||
|
parse unless the compiler itself knows the tag. Overriding the std lib directory alone
|
||||||
|
cannot add a target — and there is no libc-only shortcut (a future musl needs the same
|
||||||
|
patch). The only alternative, staying on `freestanding` + hand-shims, is exactly the
|
||||||
|
non-native feel we are leaving: `@import("std")` there is stubbed, not real.
|
||||||
|
|
||||||
|
**The fork.** Clone `ziglang/zig` at the pinned 0.16 tag; build it with a stock
|
||||||
|
same-version `zig` (`zig build` in the tree — a standard, LLVM-pulling, roughly one-time
|
||||||
|
build); point danos's `build.zig`/CI at the resulting binary. Four localised patches:
|
||||||
|
|
||||||
|
- add `danos` to `std.Target.Os.Tag`, in the "no version range" group alongside
|
||||||
|
plan9/serenity;
|
||||||
|
- add `danos` to the freestanding/other **no-op `_start` list** in `std`'s `start.zig`,
|
||||||
|
so std does *not* emit its own System-V `_start` — danos keeps owning the entry shim
|
||||||
|
and `Init`/argv construction it already builds ([sysv.md](sysv.md));
|
||||||
|
- wire the `system` selector `.danos => std.os.danos` in `std.posix`;
|
||||||
|
- add `std/os/danos.zig` — **the seam itself**, promoted near-verbatim from the
|
||||||
|
`runtime.os` developed first in Phase 1 (against the stock toolchain, so the fork is
|
||||||
|
not a prerequisite for starting).
|
||||||
|
|
||||||
|
This is the fork treadmill we accept once. Keep the patch set tiny and `else`-friendly,
|
||||||
|
pin to one 0.16.x, and rebase on point releases.
|
||||||
|
|
||||||
|
### Phase 1 — `runtime.os` read-side + allocator + stdio + cwd; retire `posix`
|
||||||
|
|
||||||
|
Author `runtime.os` (→ `std.os.danos`): the `errno` enum, the constant tables, and
|
||||||
|
the C-convention `read / write / open / openat / close / lseek / mmap / munmap /
|
||||||
|
exit`, each returning result-or-`-errno`. Most backing already exists (VFS + native
|
||||||
|
mmap + clock).
|
||||||
|
|
||||||
|
- Provide `page_allocator` via `root.os.heap.page_allocator` (a thin override over
|
||||||
|
danos `mmap`). This sits **outside** the `std.Io` vtable, so it is wired separately.
|
||||||
|
- Wire fd 0/1/2 to a console **byte** stream (today output only reaches `debug_write`;
|
||||||
|
input is structured `InputEvent` IPC — a byte tty is a new, small thing in both
|
||||||
|
directions).
|
||||||
|
- Add a `getcwd`/`chdir` anchor so `std.fs.cwd()`-style resolution has something to
|
||||||
|
resolve against.
|
||||||
|
- Build `runtime.fs` over `runtime.os`; migrate the five `posix` callers to it; delete
|
||||||
|
`library/posix/` and drop its build module.
|
||||||
|
|
||||||
|
After Phase 1, the surface an editor or terminal needs (open/read/write/close/lseek/
|
||||||
|
readdir/isatty/args/exit) exists. Those are downstream and out of scope here.
|
||||||
|
|
||||||
|
### Phase 2 — Filesystem mutation + real stat (the compiler's cache tower)
|
||||||
|
|
||||||
|
danos's biggest genuine gap, and the correctness-critical one:
|
||||||
|
|
||||||
|
- Add **mkdir / unlink / rename / truncate** to *both* the VFS wire protocol
|
||||||
|
([protocol.zig](../system/services/vfs/protocol.zig)) and the FAT engine
|
||||||
|
([engine.zig](../system/services/fat/engine.zig)), then expose them via `runtime.os`.
|
||||||
|
- Extend `stat` beyond `{size, kind}` to carry **mtime + inode + mode** — `std`'s file
|
||||||
|
stat needs them for build-cache validity — which in turn needs **wall-clock** time
|
||||||
|
(danos is monotonic-only today; an RTC/time service is the dependency).
|
||||||
|
|
||||||
|
Because `std.fs`/`std.Io` have no per-OS branches, finishing this in `runtime.os`
|
||||||
|
lights up the whole file tower for the compiler at once. Environment can stay an empty
|
||||||
|
map until the kernel populates a non-empty `envp`.
|
||||||
|
|
||||||
|
**Status — Phase 2 complete.** `truncate` (O_TRUNC, closing the boot-log stale-tail
|
||||||
|
bug), `mkdir`, `unlink`, and `rename` are all wired through the FAT engine, the VFS
|
||||||
|
protocol + router, and `runtime.fs` (`makeDirectory` / `remove` / `rename`) —
|
||||||
|
host-tested and QEMU-tested (`fat-mutations` + `fat-rename` make a directory, write+read
|
||||||
|
a file in it, rename it, then remove it through the mount). `removeFile` and `rename`
|
||||||
|
are LFN-aware; `rename` is same-directory + 8.3 (cross-directory and long-name-
|
||||||
|
preserving rename are noted limitations). Wall-clock is now a kernel syscall
|
||||||
|
(`wall_clock`, a CMOS-RTC read anchored to the monotonic clock), and the FAT engine
|
||||||
|
stamps and reports **mtime** — `stat` / `runtime.fs.Attributes` carry a real
|
||||||
|
modification time (the `fat-mtime` case reads it back within seconds of the host clock).
|
||||||
|
The remaining `stat` fields, `mode`/`inode`, are deferred (not needed until the
|
||||||
|
compiler's cache layer wants them). **Everything past here is gated on Phase 0 (the
|
||||||
|
fork):** the `runtime.os` seam, `cwd`, stdio-as-fds, and the compiler bring-up.
|
||||||
|
|
||||||
|
### Phase 3 — Single-threaded, self-linked compiler bring-up
|
||||||
|
|
||||||
|
Build the compiler with **two load-bearing flags**:
|
||||||
|
|
||||||
|
- **`-fsingle-threaded`** removes `std.Thread` entirely — `Thread.spawn` is a hard
|
||||||
|
compile error under it, and `std.Io`'s threaded backend runs inline. danos being
|
||||||
|
one-thread-per-process is therefore **not** a blocker. Parallel codegen is a
|
||||||
|
throughput optimisation, not a correctness requirement.
|
||||||
|
- **`-fno-llvm -fno-lld`** keeps codegen and linking **in-process** (the self-hosted
|
||||||
|
x86-64 backend + self-linker), so a single `build-exe` **never forks a child**. That
|
||||||
|
is what lets us defer the entire spawn/exec/wait surface.
|
||||||
|
|
||||||
|
Then supply the few remaining seam pieces: `now` (wrap the danos clock), an entropy
|
||||||
|
source behind `vtable.random` (`randomSecure` can alias it initially — low volume, for
|
||||||
|
temp-file names and hashmap seeds), and the Phase-2 mkdir/rename/unlink for cache dir
|
||||||
|
trees and atomic temp-then-rename output.
|
||||||
|
|
||||||
|
**Explicitly deferred** (not on the `build-exe` path): child-process spawn/exec (only
|
||||||
|
`zig build` and external tools need it), `std.Thread`, `fsync` (FAT is write-through
|
||||||
|
today), symlinks, and musl.
|
||||||
|
|
||||||
|
## Risks and gotchas
|
||||||
|
|
||||||
|
- **The std-fork rebase treadmill is the main ongoing cost.** A new OS tag touches the
|
||||||
|
same broad file set plan9/serenity touch (hundreds of `native_os` sites, plus
|
||||||
|
"unsupported OS" `@compileError` dead-ends a new tag must be routed around), and the
|
||||||
|
entire `std.Io` layer is new in 0.16 and still moving. Stay pinned to one 0.16.x,
|
||||||
|
keep additions localised and `else`-friendly. Watch the closed-enum gotcha: adding
|
||||||
|
`danos` to `Os.Tag` can break existing *exhaustive* switches that lack an `else`, so
|
||||||
|
expect to touch switch sites beyond the ones you implement.
|
||||||
|
- **Single-threaded is load-bearing.** The "no `std.Thread`" simplification rests
|
||||||
|
entirely on `-fsingle-threaded`. If a dependency or flag flips threading back on, you
|
||||||
|
inherit an unescapable compile error (no root-hook exists) — the only outs are a full
|
||||||
|
thread-impl fork or linking libc for pthreads. Keep `single_threaded` asserted end to
|
||||||
|
end.
|
||||||
|
- **In-process linking is load-bearing.** Reaching the compiler without fork/exec
|
||||||
|
depends on `-fno-llvm -fno-lld`. The moment you shell out to LLD/`ld`, you need the
|
||||||
|
full `spawn`/`wait` surface — the hardest microkernel piece — and danos's
|
||||||
|
`system_spawn` only starts a *named ramdisk binary*, not exec of an arbitrary path.
|
||||||
|
Verify the self-hosted backend covers the target output before assuming child
|
||||||
|
processes are optional.
|
||||||
|
- **The shim cannot host the compiler.** danos's current `runtime`/`posix` is fine for
|
||||||
|
danos's *own* native programs, but the compiler `import`s *upstream* `std`, which on
|
||||||
|
a non-target hits the void `system` stub. So the compiler forces the real target
|
||||||
|
(Phase 0's fork). Do not over-invest in extending the hand-shim for compiler
|
||||||
|
purposes; put that effort into `runtime.os` + the VFS/FAT operations, which both the
|
||||||
|
fork *and* a future musl consume.
|
||||||
|
- **`"w"`/`O_CREAT` does not truncate — a silent-corruption bug on this road.** The FAT
|
||||||
|
engine's `writeFile` only *grows* `node.size`, so overwriting a shorter file leaves
|
||||||
|
trailing garbage. Harmless for the boot log today, but for a compiler it means
|
||||||
|
**corrupt `.o`/cache files that look like nondeterministic compiler bugs.** Land
|
||||||
|
`truncate` (Phase 2) before the compiler ever writes cache.
|
||||||
|
- **Exit status is categorical, not numeric.** `process_exit_reason` returns an
|
||||||
|
`ExitReason` *category*, not a numeric code (`WEXITSTATUS`). Fine while spawn is
|
||||||
|
stubbed; the day `zig build` or external tools arrive, plan a kernel exit-record
|
||||||
|
extension — do not let it surprise you.
|
||||||
|
|
||||||
|
## How this informs decisions now
|
||||||
|
|
||||||
|
Two current decisions fall out of this roadmap:
|
||||||
|
|
||||||
|
1. **The `runtime.fs` / `std.Io` question resolves at the vtable seam.** Because 0.16
|
||||||
|
routes *all* output through the `std.Io` vtable's file-write entry, and stdout/stderr
|
||||||
|
are just `File`s with well-known handles, build `runtime.fs` (and the console stdout)
|
||||||
|
as the concrete backing for that entry — not as a bespoke `std.Io.Writer`-only shim.
|
||||||
|
Decide it once, at the seam, and program stdout, diagnostics, and file writes all
|
||||||
|
flow through the same danos VFS/console path.
|
||||||
|
2. **The boot-log `truncate` caveat is now fixed** (Phase 2a). It was the same
|
||||||
|
`writeFile`-only-grows gap that on the self-hosting road would corrupt build output;
|
||||||
|
`engine.truncate` + an O_TRUNC open flag now free the old chain so a shorter rewrite
|
||||||
|
leaves no stale tail, and the boot-log flush opens with it.
|
||||||
|
|
||||||
|
## Related
|
||||||
|
|
||||||
|
- [vision.md](vision.md) — the north star this serves.
|
||||||
|
- [syscall.md](syscall.md) — the kernel↔runtime ABI `runtime.os` is built on.
|
||||||
|
- [sysv.md](sysv.md) — the entry stack (`argc/argv/envp/auxv`) danos already constructs.
|
||||||
|
- [ipc.md](ipc.md) — the IPC the VFS/FAT operations travel over.
|
||||||
|
- [danos-file-system-hierarchy-FSH.md](danos-file-system-hierarchy-FSH.md) — the
|
||||||
|
filesystem layout the file surface serves.
|
||||||
|
- [coding-standards.md](coding-standards.md) — danos naming (why the compat spellings
|
||||||
|
are confined, and now retired).
|
||||||
@@ -0,0 +1,80 @@
|
|||||||
|
//! /lib/mmio — typed volatile MMIO register access, plus the memory-ordering
|
||||||
|
//! barriers a device driver needs. Used by drivers on top of an `mmio_map` grant.
|
||||||
|
//!
|
||||||
|
//! **`volatile` is not a barrier.** In Zig it means only: don't elide this access, and
|
||||||
|
//! don't reorder it against *other volatile* accesses. It says nothing about ordinary
|
||||||
|
//! stores — the DMA descriptor you just filled in write-back RAM — which the compiler
|
||||||
|
//! (and, on weakly-ordered hardware, the CPU) may freely move past a volatile MMIO
|
||||||
|
//! write. The canonical bug:
|
||||||
|
//!
|
||||||
|
//! ring[i] = descriptor; // ordinary store to WB RAM
|
||||||
|
//! doorbell.* = i; // volatile store to UC MMIO
|
||||||
|
//! // nothing orders these; the device can read a stale descriptor
|
||||||
|
//!
|
||||||
|
//! Put a `wmb()` between them. The barriers lower per-architecture — which is the whole
|
||||||
|
//! reason they are a named primitive and not scattered `asm volatile`:
|
||||||
|
//!
|
||||||
|
//! x86_64 aarch64
|
||||||
|
//! mb() mfence dsb sy
|
||||||
|
//! rmb() lfence dsb ld
|
||||||
|
//! wmb() sfence dsb st
|
||||||
|
//!
|
||||||
|
//! x86 is forgiving (TSO + strong-uncacheable MMIO), so a compiler barrier usually
|
||||||
|
//! suffices; ARM is not, and ARM is the win condition (docs/vision.md) — so the
|
||||||
|
//! abstraction exists now, while there is one caller (hpet) to get right. See
|
||||||
|
//! docs/driver-model.md (M14) for the full ordering contract.
|
||||||
|
|
||||||
|
const builtin = @import("builtin");
|
||||||
|
|
||||||
|
/// Read a register of type `T` at absolute virtual address `addr` — a location inside
|
||||||
|
/// a device's `mmio_map` grant. `volatile`: never elided, never reordered against
|
||||||
|
/// another volatile access.
|
||||||
|
pub inline fn read(comptime T: type, addr: usize) T {
|
||||||
|
return @as(*const volatile T, @ptrFromInt(addr)).*;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write `value` of type `T` to the register at absolute virtual address `addr`.
|
||||||
|
pub inline fn write(comptime T: type, addr: usize, value: T) void {
|
||||||
|
@as(*volatile T, @ptrFromInt(addr)).* = value;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Full barrier: all loads and stores before it are globally visible before any after
|
||||||
|
/// it. Use when an MMIO write must complete before a following read.
|
||||||
|
pub inline fn mb() void {
|
||||||
|
switch (builtin.target.cpu.arch) {
|
||||||
|
.x86_64 => asm volatile ("mfence" ::: .{ .memory = true }),
|
||||||
|
.aarch64 => asm volatile ("dsb sy" ::: .{ .memory = true }),
|
||||||
|
else => @compileError("mmio.mb: unsupported architecture"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read barrier: loads before it complete before loads after it. Use after an IRQ
|
||||||
|
/// wake, before reading what the device wrote to shared memory.
|
||||||
|
pub inline fn rmb() void {
|
||||||
|
switch (builtin.target.cpu.arch) {
|
||||||
|
.x86_64 => asm volatile ("lfence" ::: .{ .memory = true }),
|
||||||
|
.aarch64 => asm volatile ("dsb ld" ::: .{ .memory = true }),
|
||||||
|
else => @compileError("mmio.rmb: unsupported architecture"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write barrier: stores before it become visible before stores after it. Use between
|
||||||
|
/// filling a DMA descriptor in RAM and ringing the device's doorbell.
|
||||||
|
pub inline fn wmb() void {
|
||||||
|
switch (builtin.target.cpu.arch) {
|
||||||
|
.x86_64 => asm volatile ("sfence" ::: .{ .memory = true }),
|
||||||
|
.aarch64 => asm volatile ("dsb st" ::: .{ .memory = true }),
|
||||||
|
else => @compileError("mmio.wmb: unsupported architecture"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
test "barriers emit and registers round-trip through a RAM cell" {
|
||||||
|
// The barriers must at least assemble for the host arch; ordering can't be unit
|
||||||
|
// tested, but a missing/mistyped mnemonic is caught here.
|
||||||
|
wmb();
|
||||||
|
rmb();
|
||||||
|
mb();
|
||||||
|
var cell: u64 = 0;
|
||||||
|
write(u64, @intFromPtr(&cell), 0xDEAD_BEEF);
|
||||||
|
try @import("std").testing.expectEqual(@as(u64, 0xDEAD_BEEF), read(u64, @intFromPtr(&cell)));
|
||||||
|
}
|
||||||
@@ -0,0 +1,62 @@
|
|||||||
|
//! Block-device client: the helper a filesystem uses to read and write a block
|
||||||
|
//! device (a USB stick, via usb-storage) without hand-rolling the block-protocol
|
||||||
|
//! IPC. Layered over `ipc` and the shared `block-protocol` wire format, like
|
||||||
|
//! `runtime.usb` over the transfer protocol.
|
||||||
|
//!
|
||||||
|
//! Transfers name a caller-owned DMA buffer by physical address (from
|
||||||
|
//! `runtime.dma.alloc`), so whole sectors move without crossing the IPC size
|
||||||
|
//! limit — the same handoff usb-storage uses toward the controller.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const ipc = @import("ipc.zig");
|
||||||
|
const system = @import("system.zig");
|
||||||
|
const protocol = @import("block-protocol");
|
||||||
|
|
||||||
|
pub const Geometry = struct { block_size: u32, block_count: u64 };
|
||||||
|
|
||||||
|
pub const Device = struct {
|
||||||
|
endpoint: ipc.Handle,
|
||||||
|
|
||||||
|
/// The device's block size and total block count.
|
||||||
|
pub fn geometry(self: Device) ?Geometry {
|
||||||
|
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.geometry), .lba = 0, .count = 0, .physical = 0 };
|
||||||
|
var reply: [protocol.reply_size]u8 = undefined;
|
||||||
|
const n = ipc.call(self.endpoint, std.mem.asBytes(&request), &reply) catch return null;
|
||||||
|
if (n < protocol.reply_size) return null;
|
||||||
|
const result = std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]);
|
||||||
|
if (result.status != 0) return null;
|
||||||
|
return .{ .block_size = result.block_size, .block_count = result.block_count };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read `count` blocks starting at `lba` into the DMA buffer at `physical`.
|
||||||
|
pub fn read(self: Device, lba: u64, count: u32, physical: u64) bool {
|
||||||
|
return self.transfer(.read, lba, count, physical);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write `count` blocks starting at `lba` from the DMA buffer at `physical`.
|
||||||
|
pub fn write(self: Device, lba: u64, count: u32, physical: u64) bool {
|
||||||
|
return self.transfer(.write, lba, count, physical);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn transfer(self: Device, operation: protocol.Operation, lba: u64, count: u32, physical: u64) bool {
|
||||||
|
var request = protocol.Request{ .operation = @intFromEnum(operation), .lba = lba, .count = count, .physical = physical };
|
||||||
|
var reply: [protocol.reply_size]u8 = undefined;
|
||||||
|
const n = ipc.call(self.endpoint, std.mem.asBytes(&request), &reply) catch return false;
|
||||||
|
if (n < protocol.reply_size) return false;
|
||||||
|
return std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status == 0;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Look up the block device, retrying generously while the USB storage chain
|
||||||
|
/// (controller reset, enumeration, mass-storage bring-up) comes up.
|
||||||
|
pub fn open() ?Device {
|
||||||
|
// Patient: the whole USB storage chain (firmware discovery, xHCI reset and
|
||||||
|
// enumeration, mass-storage bring-up) must complete first, which can take
|
||||||
|
// tens of seconds under emulation.
|
||||||
|
var attempts: usize = 0;
|
||||||
|
while (attempts < 1200) : (attempts += 1) {
|
||||||
|
if (ipc.lookup(.block)) |handle| return .{ .endpoint = handle };
|
||||||
|
system.sleep(50);
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
@@ -0,0 +1,131 @@
|
|||||||
|
//! User-space device access: enumerate the kernel's device table, claim a device,
|
||||||
|
//! map its MMIO, and bind its interrupt. A driver uses these to find and take
|
||||||
|
//! ownership of its hardware; the claim is the capability the kernel checks before
|
||||||
|
//! mapping registers or routing an IRQ.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const device_abi = @import("device-abi");
|
||||||
|
const sc = @import("system-call.zig");
|
||||||
|
|
||||||
|
pub const DeviceDescriptor = device_abi.DeviceDescriptor;
|
||||||
|
pub const ResourceDescriptor = device_abi.ResourceDescriptor;
|
||||||
|
pub const DeviceClass = device_abi.DeviceClass;
|
||||||
|
pub const ResourceKind = device_abi.ResourceKind;
|
||||||
|
|
||||||
|
inline fn failed(r: usize) bool {
|
||||||
|
return r > ~@as(usize, 0) - 4095;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Copy up to `buffer.len` device descriptors into `buffer`; returns the total count.
|
||||||
|
pub fn enumerate(buffer: []DeviceDescriptor) usize {
|
||||||
|
return sc.systemCall2(.device_enumerate, @intFromPtr(buffer.ptr), buffer.len);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Take exclusive ownership of device `id`. Returns false if taken or invalid.
|
||||||
|
pub fn claim(id: u64) bool {
|
||||||
|
return !failed(sc.systemCall1(.device_claim, id));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Map resource `resource_index` (which must be an MMIO window) of claimed device
|
||||||
|
/// `device_id` into this address space; returns the register base virtual address.
|
||||||
|
pub fn mmioMap(device_id: u64, resource_index: u64) ?usize {
|
||||||
|
const r = sc.systemCall2(.mmio_map, device_id, resource_index);
|
||||||
|
return if (failed(r)) null else r;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `DeviceDescriptor.parent` for a device with no parent.
|
||||||
|
pub const no_parent = device_abi.no_parent;
|
||||||
|
|
||||||
|
/// `DeviceDescriptor.pci_class` for a device that is not a PCI function. Set this on
|
||||||
|
/// descriptors passed to `register` unless the child really is one.
|
||||||
|
pub const no_pci_class = device_abi.no_pci_class;
|
||||||
|
|
||||||
|
/// Publish `descriptor` as a child of `parent_id`, which this process must have claimed.
|
||||||
|
/// Returns the new device id. The child is left unclaimed, so whichever driver owns
|
||||||
|
/// that class of device can `claim` it — that is how a bus hands off a device.
|
||||||
|
///
|
||||||
|
/// Every resource in `descriptor` must be **contained** in a parent resource of the same
|
||||||
|
/// kind: a sub-window of the parent's MMIO, or one of its IRQs. The kernel refuses
|
||||||
|
/// anything else, because a device descriptor is a licence to map physical memory and
|
||||||
|
/// a bus driver may only subdivide what it already owns. `descriptor.id` and `descriptor.parent`
|
||||||
|
/// are ignored. A device with no resources at all is fine — a USB device is reached
|
||||||
|
/// through its controller, not by MMIO.
|
||||||
|
pub fn register(parent_id: u64, descriptor: *const DeviceDescriptor) ?u64 {
|
||||||
|
const r = sc.systemCall2(.device_register, parent_id, @intFromPtr(descriptor));
|
||||||
|
return if (failed(r)) null else r;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Bind resource `resource_index` (which must be an IRQ) of claimed device `device_id` to
|
||||||
|
/// `endpoint`. From then on the interrupt arrives as an asynchronous notification:
|
||||||
|
/// `ipc.replyWait` on that endpoint returns with the high bit set in `badge` and the
|
||||||
|
/// low bits carrying the GSI. The kernel masks the line before waking you.
|
||||||
|
pub fn irqBind(device_id: u64, resource_index: u64, endpoint: usize) bool {
|
||||||
|
return !failed(sc.systemCall3(.irq_bind, device_id, resource_index, endpoint));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Re-arm a bound IRQ. Call this **after** quieting the device (clearing whatever
|
||||||
|
/// status register holds its line asserted) — the kernel left the line masked
|
||||||
|
/// precisely because it could not do that for you. Skip it and the interrupt never
|
||||||
|
/// fires again; call it before the device is quiet and a level-triggered line storms.
|
||||||
|
pub fn irqAck(device_id: u64, resource_index: u64) bool {
|
||||||
|
return !failed(sc.systemCall2(.irq_ack, device_id, resource_index));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The Message-Signalled Interrupt address/data a driver programs into its device's
|
||||||
|
/// MSI capability. The device raises the interrupt by writing `data` to `address`.
|
||||||
|
pub const Msi = struct { address: u64, data: u32 };
|
||||||
|
|
||||||
|
/// Set up MSI for a claimed device: the kernel allocates a per-device edge-triggered
|
||||||
|
/// vector, binds it to `endpoint` (delivered like `irqBind`, but with no mask and no
|
||||||
|
/// `irqAck` cycle), and returns the (address, data) to write into the device's MSI
|
||||||
|
/// capability — found by mmio_mapping the device's ECAM config space (resource 0) and
|
||||||
|
/// walking its capability list. Returns null on failure. Two return values (address in
|
||||||
|
/// rax, data in rdx), so a hand-written stub.
|
||||||
|
pub fn msiBind(device_id: u64, endpoint: usize) ?Msi {
|
||||||
|
var rax: usize = undefined;
|
||||||
|
var rdx: usize = undefined;
|
||||||
|
asm volatile ("syscall"
|
||||||
|
: [rax] "={rax}" (rax),
|
||||||
|
[rdx] "={rdx}" (rdx),
|
||||||
|
: [n] "{rax}" (@intFromEnum(abi.SystemCall.msi_bind)),
|
||||||
|
[a0] "{rdi}" (device_id),
|
||||||
|
[a1] "{rsi}" (endpoint),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
if (failed(rax)) return null;
|
||||||
|
return .{ .address = rax, .data = @intCast(rdx) };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read `width` bytes (1, 2, or 4) from a port in a claimed device's `io_port`
|
||||||
|
/// resource, at byte `offset` within it. Ring 3 has no direct `in`/`out`, so a legacy
|
||||||
|
/// driver (PS/2, 16550 UART) reaches its ports through this claim-gated call — each
|
||||||
|
/// access is a syscall, which is fine for the low-rate hardware that needs it. Returns
|
||||||
|
/// null if the capability check fails (device not claimed, wrong resource, out of
|
||||||
|
/// range). A device that decodes no data returns all-ones, which is a valid value, not
|
||||||
|
/// a failure.
|
||||||
|
pub fn ioRead(device_id: u64, resource_index: u64, offset: u64, width: u8) ?u32 {
|
||||||
|
const r = sc.systemCall4(.io_read, device_id, resource_index, offset, width);
|
||||||
|
return if (failed(r)) null else @intCast(r);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write `value` (its low `width` bytes, 1/2/4) to a port in a claimed device's
|
||||||
|
/// `io_port` resource, at byte `offset`. Same capability gate as `ioRead`.
|
||||||
|
pub fn ioWrite(device_id: u64, resource_index: u64, offset: u64, width: u8, value: u32) bool {
|
||||||
|
return !failed(sc.systemCall5(.io_write, device_id, resource_index, offset, width, value));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Find DeviceDescription by hid
|
||||||
|
///
|
||||||
|
/// Utility function for driver development
|
||||||
|
pub fn findDeviceDescriptorByHid(buffer: []DeviceDescriptor, hid_needle: []const u8) ?DeviceDescriptor {
|
||||||
|
const total = enumerate(buffer);
|
||||||
|
const n = @min(total, buffer.len);
|
||||||
|
for (@as([]DeviceDescriptor, buffer[0..n])) |d| {
|
||||||
|
const hid_haystack = d.hid[0..@intCast(d.hid_len)];
|
||||||
|
if (std.mem.eql(u8, hid_haystack, hid_needle)) {
|
||||||
|
return d;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return null;
|
||||||
|
}
|
||||||
@@ -0,0 +1,48 @@
|
|||||||
|
//! User-space DMA memory: `dma_alloc` / `dma_free`. A driver that programs a
|
||||||
|
//! bus-mastering engine needs a descriptor ring the device can read — memory that is
|
||||||
|
//! physically contiguous, at a physical address the driver knows, uncacheable, and
|
||||||
|
//! pinned. `mmap` gives none of those; this does. Pair it with the barriers in
|
||||||
|
//! `/lib/mmio` (fill the ring, `wmb()`, ring the doorbell). See docs/driver-model.md.
|
||||||
|
|
||||||
|
const abi = @import("abi");
|
||||||
|
const sc = @import("system-call.zig");
|
||||||
|
|
||||||
|
/// Allocation flags. `coherent` (uncacheable) is the portable default; the rest are
|
||||||
|
/// opt-in for specific hardware — see `abi`.
|
||||||
|
pub const coherent: usize = abi.dma_coherent;
|
||||||
|
pub const write_combining: usize = abi.dma_write_combining;
|
||||||
|
pub const below_4g: usize = abi.dma_below_4g;
|
||||||
|
|
||||||
|
/// A DMA allocation: the `virtual` address the CPU touches, and the `physical` address
|
||||||
|
/// to program into the device's descriptor-ring / base registers.
|
||||||
|
pub const Region = struct {
|
||||||
|
virtual: usize,
|
||||||
|
physical: usize,
|
||||||
|
};
|
||||||
|
|
||||||
|
inline fn failed(r: usize) bool {
|
||||||
|
return r > ~@as(usize, 0) - 4095;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Allocate `len` bytes of DMA-capable memory with `flags` (e.g. `coherent`, or
|
||||||
|
/// `coherent | below_4g`). Returns the virtual/physical pair, or null on failure. Two
|
||||||
|
/// return values — the virtual address in rax, the physical address in rdx — so it
|
||||||
|
/// needs a hand-written stub.
|
||||||
|
pub fn alloc(len: usize, flags: usize) ?Region {
|
||||||
|
var rax: usize = undefined;
|
||||||
|
var rdx: usize = undefined; // out: physical address
|
||||||
|
asm volatile ("syscall"
|
||||||
|
: [rax] "={rax}" (rax),
|
||||||
|
[rdx] "={rdx}" (rdx),
|
||||||
|
: [n] "{rax}" (@intFromEnum(abi.SystemCall.dma_alloc)),
|
||||||
|
[a0] "{rdi}" (len),
|
||||||
|
[a1] "{rsi}" (flags),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
if (failed(rax)) return null;
|
||||||
|
return .{ .virtual = rax, .physical = rdx };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Release a region from a prior `alloc` (`virtual` and the same `len`).
|
||||||
|
pub fn free(virtual: usize, len: usize) void {
|
||||||
|
_ = sc.systemCall2(.dma_free, virtual, len);
|
||||||
|
}
|
||||||
@@ -0,0 +1,274 @@
|
|||||||
|
//! runtime.fs — the danos-native file API. A program opens, reads, writes, and
|
||||||
|
//! lists files served by the user-space VFS (system/services/vfs), each call
|
||||||
|
//! marshalling a vfs-protocol request over IPC. This is the danos-native layer
|
||||||
|
//! danos programs use directly; it is also where the file operations that later
|
||||||
|
//! become `std.os.danos` are staged (see docs/zig-self-hosting.md). It replaces
|
||||||
|
//! the old POSIX `unistd` shim — a compatibility spelling danos does not need yet.
|
||||||
|
//!
|
||||||
|
//! Handles are *values*, not entries in a global descriptor table: a `File` /
|
||||||
|
//! `Directory` owns its VFS node id and (for files) a byte offset. So there is no
|
||||||
|
//! per-process fd limit and no shared table to synchronise — the danos-native
|
||||||
|
//! shape, unlike the POSIX fd model the old shim emulated.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const ipc = @import("ipc.zig");
|
||||||
|
const protocol = @import("vfs-protocol");
|
||||||
|
|
||||||
|
/// The kind of a filesystem node — re-exported so a caller need not import the
|
||||||
|
/// wire protocol.
|
||||||
|
pub const Kind = protocol.NodeKind;
|
||||||
|
|
||||||
|
/// A node's metadata (the answer to a status request).
|
||||||
|
pub const Attributes = struct {
|
||||||
|
size: u64,
|
||||||
|
kind: Kind,
|
||||||
|
/// Modification time — Unix epoch seconds, UTC. 0 if the filesystem has none.
|
||||||
|
mtime: u64 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
// Map a wire `NodeKind` value to the enum, defaulting anything unrecognised to
|
||||||
|
// `.regular` (the server is trusted, but a value outside the enum would be
|
||||||
|
// illegal to `@enumFromInt` directly).
|
||||||
|
fn kindFromWire(value: u32) Kind {
|
||||||
|
return switch (value) {
|
||||||
|
@intFromEnum(Kind.directory) => .directory,
|
||||||
|
@intFromEnum(Kind.character_device) => .character_device,
|
||||||
|
@intFromEnum(Kind.block_device) => .block_device,
|
||||||
|
@intFromEnum(Kind.symbolic_link) => .symbolic_link,
|
||||||
|
@intFromEnum(Kind.fifo) => .fifo,
|
||||||
|
@intFromEnum(Kind.socket) => .socket,
|
||||||
|
else => .regular,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// How to open a path.
|
||||||
|
pub const OpenOptions = struct {
|
||||||
|
/// Create the file if it does not exist.
|
||||||
|
create: bool = false,
|
||||||
|
/// Open a directory node (for listing) rather than a file.
|
||||||
|
directory: bool = false,
|
||||||
|
/// Truncate an existing file to zero length on open (O_TRUNC) — replace its
|
||||||
|
/// contents rather than overwriting in place.
|
||||||
|
truncate: bool = false,
|
||||||
|
|
||||||
|
fn wireFlags(self: OpenOptions) u32 {
|
||||||
|
var f: u32 = 0;
|
||||||
|
if (self.create) f |= protocol.create;
|
||||||
|
if (self.directory) f |= protocol.directory;
|
||||||
|
if (self.truncate) f |= protocol.truncate;
|
||||||
|
return f;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
// The VFS server endpoint, looked up once by well-known id and cached.
|
||||||
|
var vfs_handle: ipc.Handle = 0;
|
||||||
|
var vfs_resolved = false;
|
||||||
|
fn vfs() ?ipc.Handle {
|
||||||
|
if (!vfs_resolved) {
|
||||||
|
vfs_handle = ipc.lookup(.vfs) orelse return null;
|
||||||
|
vfs_resolved = true;
|
||||||
|
}
|
||||||
|
return vfs_handle;
|
||||||
|
}
|
||||||
|
|
||||||
|
const Result = struct { reply: protocol.Reply, payload: []u8 };
|
||||||
|
|
||||||
|
// One request/reply round trip: [Request header][send payload] -> VFS ->
|
||||||
|
// [Reply header][receive payload]. The receive payload lands in `out`.
|
||||||
|
fn transact(request: protocol.Request, send: []const u8, out: []u8) ?Result {
|
||||||
|
const h = vfs() orelse return null;
|
||||||
|
var message: [protocol.message_maximum]u8 = undefined;
|
||||||
|
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||||
|
const slen = @min(send.len, protocol.maximum_payload);
|
||||||
|
@memcpy(message[protocol.request_size..][0..slen], send[0..slen]);
|
||||||
|
|
||||||
|
var rbuf: [protocol.message_maximum]u8 = undefined;
|
||||||
|
const n = ipc.call(h, message[0 .. protocol.request_size + slen], &rbuf) catch return null;
|
||||||
|
if (n < protocol.reply_size) return null;
|
||||||
|
const reply = std.mem.bytesToValue(protocol.Reply, rbuf[0..protocol.reply_size]);
|
||||||
|
const rpl = @min(n - protocol.reply_size, out.len);
|
||||||
|
@memcpy(out[0..rpl], rbuf[protocol.reply_size..][0..rpl]);
|
||||||
|
return .{ .reply = reply, .payload = out[0..rpl] };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An open file: a VFS node plus a byte cursor. Read and write advance the cursor.
|
||||||
|
pub const File = struct {
|
||||||
|
node: u64,
|
||||||
|
offset: u64 = 0,
|
||||||
|
|
||||||
|
/// Read up to `buffer.len` bytes at the current offset; returns the count, or
|
||||||
|
/// null on error.
|
||||||
|
pub fn read(self: *File, buffer: []u8) ?usize {
|
||||||
|
const want: u32 = @intCast(@min(buffer.len, protocol.maximum_payload));
|
||||||
|
const request = protocol.Request{ .operation = .read, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
|
||||||
|
const r = transact(request, &.{}, buffer) orelse return null;
|
||||||
|
if (r.reply.status != 0) return null;
|
||||||
|
self.offset += r.reply.len;
|
||||||
|
return r.reply.len;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write `data` at the current offset; returns the count written. A single
|
||||||
|
/// call is capped at the VFS payload size, so the return may be short — use
|
||||||
|
/// `writeAll` to write the whole slice. Null on error.
|
||||||
|
pub fn write(self: *File, data: []const u8) ?usize {
|
||||||
|
const want: u32 = @intCast(@min(data.len, protocol.maximum_payload));
|
||||||
|
const request = protocol.Request{ .operation = .write, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
|
||||||
|
const r = transact(request, data[0..want], &.{}) orelse return null;
|
||||||
|
if (r.reply.status != 0) return null;
|
||||||
|
self.offset += r.reply.len;
|
||||||
|
return r.reply.len;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write all of `data`, looping past the per-call payload cap. Returns the
|
||||||
|
/// total written, or null if a write failed before any progress.
|
||||||
|
pub fn writeAll(self: *File, data: []const u8) ?usize {
|
||||||
|
var written: usize = 0;
|
||||||
|
while (written < data.len) {
|
||||||
|
const n = self.write(data[written..]) orelse return if (written == 0) null else written;
|
||||||
|
if (n == 0) return written; // no forward progress; stop rather than spin
|
||||||
|
written += n;
|
||||||
|
}
|
||||||
|
return written;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Move the read/write cursor to an absolute byte position.
|
||||||
|
pub fn seekTo(self: *File, position: u64) void {
|
||||||
|
self.offset = position;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// This file's metadata.
|
||||||
|
pub fn attributes(self: *File) ?Attributes {
|
||||||
|
const request = protocol.Request{ .operation = .status, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
|
||||||
|
var buffer: [@sizeOf(protocol.FileStatus)]u8 = undefined;
|
||||||
|
const r = transact(request, &.{}, &buffer) orelse return null;
|
||||||
|
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.FileStatus)) return null;
|
||||||
|
const status = std.mem.bytesToValue(protocol.FileStatus, buffer[0..@sizeOf(protocol.FileStatus)]);
|
||||||
|
return .{ .size = status.size, .kind = kindFromWire(status.kind), .mtime = status.mtime };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Release the VFS's open handle for this file.
|
||||||
|
pub fn close(self: *File) void {
|
||||||
|
const request = protocol.Request{ .operation = .close, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
|
||||||
|
_ = transact(request, &.{}, &.{});
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Open (or create, with `.create`) `path`. Returns the open file, or null.
|
||||||
|
pub fn open(path: []const u8, options: OpenOptions) ?File {
|
||||||
|
const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = options.wireFlags() };
|
||||||
|
const r = transact(request, path, &.{}) orelse return null;
|
||||||
|
if (r.reply.status != 0) return null;
|
||||||
|
return .{ .node = r.reply.node };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A path's metadata without keeping it open (open -> status -> close).
|
||||||
|
pub fn attributes(path: []const u8) ?Attributes {
|
||||||
|
var file = open(path, .{}) orelse return null;
|
||||||
|
defer file.close();
|
||||||
|
return file.attributes();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether `path` resolves — handy as a readiness check (e.g. waiting for a mount
|
||||||
|
/// to come up before writing to it).
|
||||||
|
pub fn exists(path: []const u8) bool {
|
||||||
|
return attributes(path) != null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One entry returned by `Directory.next`.
|
||||||
|
pub const Entry = struct {
|
||||||
|
kind: Kind = .regular,
|
||||||
|
size: u64 = 0,
|
||||||
|
name_buffer: [64]u8 = undefined,
|
||||||
|
name_len: usize = 0,
|
||||||
|
|
||||||
|
pub fn name(self: *const Entry) []const u8 {
|
||||||
|
return self.name_buffer[0..self.name_len];
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// An open directory being listed, cursor-advanced by `next`.
|
||||||
|
pub const Directory = struct {
|
||||||
|
node: u64,
|
||||||
|
cursor: u64 = 0,
|
||||||
|
|
||||||
|
/// Fill `entry` with the next directory entry; false at end of directory or
|
||||||
|
/// on error.
|
||||||
|
pub fn next(self: *Directory, entry: *Entry) bool {
|
||||||
|
const request = protocol.Request{ .operation = .readdir, .node = self.node, .offset = self.cursor, .len = 0, .flags = 0 };
|
||||||
|
var buffer: [protocol.message_maximum]u8 = undefined;
|
||||||
|
const r = transact(request, &.{}, &buffer) orelse return false;
|
||||||
|
if (r.reply.status != 0 or r.reply.len == 0) return false; // error or EOF
|
||||||
|
if (r.payload.len < protocol.directory_entry_size) return false;
|
||||||
|
const header = std.mem.bytesToValue(protocol.DirectoryEntry, r.payload[0..protocol.directory_entry_size]);
|
||||||
|
entry.kind = kindFromWire(header.kind);
|
||||||
|
entry.size = header.size;
|
||||||
|
const source = r.payload[protocol.directory_entry_size..];
|
||||||
|
const nlen = @min(@min(@as(usize, header.name_len), source.len), entry.name_buffer.len);
|
||||||
|
@memcpy(entry.name_buffer[0..nlen], source[0..nlen]);
|
||||||
|
entry.name_len = nlen;
|
||||||
|
self.cursor += 1;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Release the VFS's open handle for this directory.
|
||||||
|
pub fn close(self: *Directory) void {
|
||||||
|
var f = File{ .node = self.node };
|
||||||
|
f.close();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Open `path` as a directory for listing. Returns null if it isn't one / on error.
|
||||||
|
pub fn openDirectory(path: []const u8) ?Directory {
|
||||||
|
const file = open(path, .{ .directory = true }) orelse return null;
|
||||||
|
return .{ .node = file.node };
|
||||||
|
}
|
||||||
|
|
||||||
|
// A path-based request that returns only a status (mkdir, unlink).
|
||||||
|
fn pathOperation(operation: protocol.Operation, path: []const u8) bool {
|
||||||
|
const request = protocol.Request{ .operation = operation, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = 0 };
|
||||||
|
const r = transact(request, path, &.{}) orelse return false;
|
||||||
|
return r.reply.status == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Create a directory at `path` (its parent must already exist). Returns true on
|
||||||
|
/// success. Only works under a mounted filesystem that supports directories.
|
||||||
|
pub fn makeDirectory(path: []const u8) bool {
|
||||||
|
return pathOperation(.mkdir, path);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Remove the file at `path`. Returns true on success. Directories are refused
|
||||||
|
/// (a separate directory-removal would have to check emptiness).
|
||||||
|
pub fn remove(path: []const u8) bool {
|
||||||
|
return pathOperation(.unlink, path);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Rename `old_path` to `new_path`. Both must be in the same directory (same-
|
||||||
|
/// directory, 8.3-name rename only for now). Returns true on success.
|
||||||
|
pub fn rename(old_path: []const u8, new_path: []const u8) bool {
|
||||||
|
const total = old_path.len + 1 + new_path.len;
|
||||||
|
if (total > protocol.maximum_payload) return false;
|
||||||
|
var payload: [protocol.maximum_payload]u8 = undefined;
|
||||||
|
@memcpy(payload[0..old_path.len], old_path);
|
||||||
|
payload[old_path.len] = 0;
|
||||||
|
@memcpy(payload[old_path.len + 1 ..][0..new_path.len], new_path);
|
||||||
|
const request = protocol.Request{ .operation = .rename, .node = 0, .offset = 0, .len = @intCast(total), .flags = 0 };
|
||||||
|
const r = transact(request, payload[0..total], &.{}) orelse return false;
|
||||||
|
return r.reply.status == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Mount a filesystem backend (its server endpoint) at absolute path `target`;
|
||||||
|
/// the VFS then routes everything under `target` to that backend. This is the one
|
||||||
|
/// call that hands the VFS a capability (the backend endpoint). Returns true on
|
||||||
|
/// success.
|
||||||
|
pub fn mount(target: []const u8, backend: ipc.Handle) bool {
|
||||||
|
const h = vfs() orelse return false;
|
||||||
|
const request = protocol.Request{ .operation = .mount, .node = 0, .offset = 0, .len = @intCast(target.len), .flags = 0 };
|
||||||
|
var message: [protocol.message_maximum]u8 = undefined;
|
||||||
|
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||||
|
const tlen = @min(target.len, protocol.maximum_payload);
|
||||||
|
@memcpy(message[protocol.request_size..][0..tlen], target[0..tlen]);
|
||||||
|
var rbuf: [protocol.message_maximum]u8 = undefined;
|
||||||
|
const result = ipc.callCap(h, message[0 .. protocol.request_size + tlen], &rbuf, backend) catch return false;
|
||||||
|
if (result.len < protocol.reply_size) return false;
|
||||||
|
return std.mem.bytesToValue(protocol.Reply, rbuf[0..protocol.reply_size]).status == 0;
|
||||||
|
}
|
||||||
@@ -0,0 +1,193 @@
|
|||||||
|
//! The user-space heap: C-convention dynamic allocation (`malloc`/`free`/…) plus
|
||||||
|
//! a `std.mem.Allocator` adapter over the same free list, so both C-style code
|
||||||
|
//! and Zig `std` containers share one heap.
|
||||||
|
//!
|
||||||
|
//! The algorithm is a straight port of the kernel's first-fit free list
|
||||||
|
//! (system/kernel/heap.zig): an address-ordered singly linked list of free blocks,
|
||||||
|
//! split on allocation and coalesced with neighbours on free. The only thing
|
||||||
|
//! that changes on this side of the system_call boundary is where memory comes from
|
||||||
|
//! — `grow` asks the kernel for pages via `mmap` instead of mapping frames
|
||||||
|
//! itself, and the kernel picks the base address.
|
||||||
|
//!
|
||||||
|
//! Single-threaded and 16-byte maximum alignment, exactly like the kernel heap; a
|
||||||
|
//! lock and larger alignments come when user programs gain threads.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const system_calls = @import("system.zig");
|
||||||
|
|
||||||
|
const page_size = abi.page_size;
|
||||||
|
|
||||||
|
/// A block header, at the start of every block; while free it also links the
|
||||||
|
/// free list via `next`.
|
||||||
|
const Block = extern struct {
|
||||||
|
size: usize, // total block size in bytes, including this header; a multiple of 16
|
||||||
|
next: ?*Block, // free-list link (only meaningful while free)
|
||||||
|
};
|
||||||
|
|
||||||
|
const header_size = @sizeOf(Block); // 16
|
||||||
|
const minimum_block = header_size + 16; // smallest block worth splitting off
|
||||||
|
/// Grow granularity: one `mmap` per 64 KiB amortises the system_call.
|
||||||
|
const chunk = 64 * 1024;
|
||||||
|
|
||||||
|
var free_list: ?*Block = null;
|
||||||
|
|
||||||
|
fn alignUp(value: usize, alignment: usize) usize {
|
||||||
|
return (value + alignment - 1) & ~(alignment - 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn payloadOf(block: *Block) [*]u8 {
|
||||||
|
return @ptrFromInt(@intFromPtr(block) + header_size);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Ask the kernel for more pages and add them as a free block. Because each
|
||||||
|
/// `mmap` is an independent grant, cross-grant coalescing happens only when the
|
||||||
|
/// kernel returns adjacent bases (its arena is a bump allocator, so consecutive
|
||||||
|
/// grants usually are adjacent). Returns false if the kernel is out of memory.
|
||||||
|
fn grow(minimum_bytes: usize) bool {
|
||||||
|
const bytes = alignUp(@max(minimum_bytes, chunk), page_size);
|
||||||
|
const ret = system_calls.mmap(bytes, system_calls.PROT_READ | system_calls.PROT_WRITE);
|
||||||
|
if (system_calls.mmapFailed(ret)) return false;
|
||||||
|
|
||||||
|
const block: *Block = @ptrFromInt(ret);
|
||||||
|
block.size = bytes;
|
||||||
|
insertFree(block); // coalesces if this grant is adjacent to a prior one
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Insert a block into the address-ordered free list, coalescing with the
|
||||||
|
/// physically adjacent free blocks on either side.
|
||||||
|
fn insertFree(block: *Block) void {
|
||||||
|
var previous: ?*Block = null;
|
||||||
|
var current = free_list;
|
||||||
|
while (current) |c| : (current = c.next) {
|
||||||
|
if (@intFromPtr(c) > @intFromPtr(block)) break;
|
||||||
|
previous = c;
|
||||||
|
}
|
||||||
|
|
||||||
|
block.next = current;
|
||||||
|
if (previous) |p| p.next = block else free_list = block;
|
||||||
|
|
||||||
|
// Merge forward into `current` if they're contiguous.
|
||||||
|
if (current) |c| {
|
||||||
|
if (@intFromPtr(block) + block.size == @intFromPtr(c)) {
|
||||||
|
block.size += c.size;
|
||||||
|
block.next = c.next;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Merge `previous` forward into `block` if they're contiguous.
|
||||||
|
if (previous) |p| {
|
||||||
|
if (@intFromPtr(p) + p.size == @intFromPtr(block)) {
|
||||||
|
p.size += block.size;
|
||||||
|
p.next = block.next;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Allocate `len` bytes (16-byte aligned), or null if out of memory.
|
||||||
|
fn rawAlloc(len: usize) ?[*]u8 {
|
||||||
|
const need = alignUp(header_size + len, 16);
|
||||||
|
|
||||||
|
var attempts: u32 = 0;
|
||||||
|
while (attempts < 2) : (attempts += 1) {
|
||||||
|
var previous: ?*Block = null;
|
||||||
|
var current = free_list;
|
||||||
|
while (current) |block| : ({
|
||||||
|
previous = block;
|
||||||
|
current = block.next;
|
||||||
|
}) {
|
||||||
|
if (block.size < need) continue;
|
||||||
|
|
||||||
|
if (block.size >= need + minimum_block) {
|
||||||
|
// Split: carve `need` off the front, leave the rest free.
|
||||||
|
const rest: *Block = @ptrFromInt(@intFromPtr(block) + need);
|
||||||
|
rest.size = block.size - need;
|
||||||
|
rest.next = block.next;
|
||||||
|
if (previous) |p| p.next = rest else free_list = rest;
|
||||||
|
block.size = need;
|
||||||
|
} else {
|
||||||
|
// Take the whole block.
|
||||||
|
if (previous) |p| p.next = block.next else free_list = block.next;
|
||||||
|
}
|
||||||
|
return payloadOf(block);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Nothing fit: grow and try once more.
|
||||||
|
if (!grow(need)) return null;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn rawFree(ptr: [*]u8) void {
|
||||||
|
const block: *Block = @ptrFromInt(@intFromPtr(ptr) - header_size);
|
||||||
|
insertFree(block);
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- C ABI: the global implicit heap ---------------------------------------
|
||||||
|
// `extern "C"` symbols so future C code links the same malloc/free directly.
|
||||||
|
|
||||||
|
export fn malloc(size: usize) callconv(.c) ?*anyopaque {
|
||||||
|
if (size == 0) return null;
|
||||||
|
const p = rawAlloc(size) orelse return null;
|
||||||
|
return @ptrCast(p);
|
||||||
|
}
|
||||||
|
|
||||||
|
export fn free(ptr: ?*anyopaque) callconv(.c) void {
|
||||||
|
const p = ptr orelse return;
|
||||||
|
rawFree(@ptrCast(p));
|
||||||
|
}
|
||||||
|
|
||||||
|
export fn calloc(nmemb: usize, size: usize) callconv(.c) ?*anyopaque {
|
||||||
|
const total = std.math.mul(usize, nmemb, size) catch return null; // overflow-safe
|
||||||
|
if (total == 0) return null;
|
||||||
|
const p = rawAlloc(total) orelse return null;
|
||||||
|
@memset(p[0..total], 0);
|
||||||
|
return @ptrCast(p);
|
||||||
|
}
|
||||||
|
|
||||||
|
export fn realloc(ptr: ?*anyopaque, size: usize) callconv(.c) ?*anyopaque {
|
||||||
|
const p = ptr orelse return malloc(size);
|
||||||
|
if (size == 0) {
|
||||||
|
rawFree(@ptrCast(p));
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
const block: *Block = @ptrFromInt(@intFromPtr(p) - header_size);
|
||||||
|
const old_payload = block.size - header_size;
|
||||||
|
if (size <= old_payload) return p; // shrink/same: keep the block
|
||||||
|
const np = rawAlloc(size) orelse return null; // grow: alloc + copy + free
|
||||||
|
@memcpy(np[0..old_payload], @as([*]u8, @ptrCast(p))[0..old_payload]);
|
||||||
|
rawFree(@ptrCast(p));
|
||||||
|
return @ptrCast(np);
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- std.mem.Allocator interface (same free list) --------------------------
|
||||||
|
|
||||||
|
pub fn allocator() std.mem.Allocator {
|
||||||
|
return .{ .ptr = undefined, .vtable = &vtable };
|
||||||
|
}
|
||||||
|
|
||||||
|
const vtable = std.mem.Allocator.VTable{
|
||||||
|
.alloc = allocImpl,
|
||||||
|
.resize = resizeImpl,
|
||||||
|
.remap = remapImpl,
|
||||||
|
.free = freeImpl,
|
||||||
|
};
|
||||||
|
|
||||||
|
fn allocImpl(_: *anyopaque, len: usize, alignment: std.mem.Alignment, _: usize) ?[*]u8 {
|
||||||
|
if (alignment.toByteUnits() > 16) return null; // blocks are 16-byte aligned
|
||||||
|
return rawAlloc(len);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn resizeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, new_len: usize, _: usize) bool {
|
||||||
|
// In-place iff the new payload still fits the current block.
|
||||||
|
const block: *Block = @ptrFromInt(@intFromPtr(memory.ptr) - header_size);
|
||||||
|
return new_len + header_size <= block.size;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn remapImpl(_: *anyopaque, _: []u8, _: std.mem.Alignment, _: usize, _: usize) ?[*]u8 {
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn freeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, _: usize) void {
|
||||||
|
rawFree(memory.ptr);
|
||||||
|
}
|
||||||
@@ -0,0 +1,220 @@
|
|||||||
|
//! User-space input helpers: the client and publisher sides of the input service, so a
|
||||||
|
//! program listening for input events — or a driver broadcasting them — doesn't hand-roll
|
||||||
|
//! the IPC. Layered over `ipc` (endpoints, capability passing, `send`) and the shared
|
||||||
|
//! `input-protocol` wire format, the way `device.zig` layers over the raw `device_*` calls.
|
||||||
|
//! See system/services/input/input.zig.
|
||||||
|
//!
|
||||||
|
//! The service carries several device classes (keyboard, mouse, joystick/gamepad). A
|
||||||
|
//! **source** publishes its class with the matching method:
|
||||||
|
//! var source = input.connectSource() orelse return;
|
||||||
|
//! _ = source.publishKeyboardEvent(.{ .kind = ..., .keycode = ..., ... });
|
||||||
|
//! _ = source.publishMouseEvent(.{ ... });
|
||||||
|
//! _ = source.publishJoystickEvent(.{ ... });
|
||||||
|
//!
|
||||||
|
//! A **subscriber** either takes one class with a typed helper —
|
||||||
|
//! var keys = input.subscribeKeyboard() orelse return;
|
||||||
|
//! while (true) { const key = keys.next() orelse continue; ... }
|
||||||
|
//! — or takes several at once and inspects the tagged envelope:
|
||||||
|
//! var listener = input.subscribeAll() orelse return;
|
||||||
|
//! while (true) {
|
||||||
|
//! const event = listener.next() orelse continue;
|
||||||
|
//! if (event.asKeyboard()) |k| { ... } else if (event.asMouse()) |m| { ... }
|
||||||
|
//! }
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const ipc = @import("ipc.zig");
|
||||||
|
const system = @import("system.zig");
|
||||||
|
const protocol = @import("input-protocol");
|
||||||
|
|
||||||
|
pub const DeviceKind = protocol.DeviceKind;
|
||||||
|
pub const InputEvent = protocol.InputEvent;
|
||||||
|
pub const KeyEvent = protocol.KeyEvent;
|
||||||
|
pub const MouseEvent = protocol.MouseEvent;
|
||||||
|
pub const JoystickEvent = protocol.JoystickEvent;
|
||||||
|
pub const EventKind = protocol.EventKind;
|
||||||
|
pub const MouseEventKind = protocol.MouseEventKind;
|
||||||
|
pub const JoystickEventKind = protocol.JoystickEventKind;
|
||||||
|
pub const Keycode = protocol.Keycode;
|
||||||
|
|
||||||
|
/// Interest masks re-exported so a caller can `subscribe(input.device_keyboard |
|
||||||
|
/// input.device_mouse)`.
|
||||||
|
pub const device_keyboard = protocol.device_keyboard;
|
||||||
|
pub const device_mouse = protocol.device_mouse;
|
||||||
|
pub const device_joystick = protocol.device_joystick;
|
||||||
|
pub const device_all = protocol.device_all;
|
||||||
|
|
||||||
|
/// Look up the input service, retrying while it is still coming up. Both a subscriber and
|
||||||
|
/// a source race the service's registration at boot, so both wait for it here rather than
|
||||||
|
/// failing. Returns the service endpoint handle, or null if it never appears.
|
||||||
|
fn lookupService() ?ipc.Handle {
|
||||||
|
var attempts: usize = 0;
|
||||||
|
while (attempts < 100) : (attempts += 1) {
|
||||||
|
if (ipc.lookup(.input)) |handle| return handle;
|
||||||
|
system.sleep(50);
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- subscribing ------------------------------------------------------------
|
||||||
|
|
||||||
|
/// A subscription to the input service: our own endpoint, which the service pushes events
|
||||||
|
/// to. `next` returns each event as a tagged `InputEvent`; use `asKeyboard`/`asMouse`/
|
||||||
|
/// `asJoystick` to decode. Created with `subscribe`/`subscribeAll`; for a single device
|
||||||
|
/// class prefer the typed helpers (`subscribeKeyboard`, ...), which return decoded events.
|
||||||
|
pub const Subscriber = struct {
|
||||||
|
/// The endpoint the service delivers events to (created and owned by us; its handle
|
||||||
|
/// was handed to the service as a capability at subscribe time).
|
||||||
|
endpoint: ipc.Handle,
|
||||||
|
receive: [protocol.event_size]u8 = undefined,
|
||||||
|
|
||||||
|
/// Block until the next event is pushed, and return it. Events arrive as asynchronous
|
||||||
|
/// buffered messages (`ipc_send` from the service), so nothing is owed in reply — the
|
||||||
|
/// empty reply this issues is a harmless no-op. Returns null for any non-event wake-up
|
||||||
|
/// (there should be none), so callers can loop.
|
||||||
|
pub fn next(self: *Subscriber) ?InputEvent {
|
||||||
|
const got = ipc.replyWait(self.endpoint, &.{}, &self.receive, null);
|
||||||
|
if (!got.isMessage() or got.len < protocol.event_size) return null;
|
||||||
|
return std.mem.bytesToValue(InputEvent, self.receive[0..protocol.event_size]);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Subscribe to the input classes named in `device_mask` (an OR of `device_*`, or
|
||||||
|
/// `device_all`). Creates an endpoint for the service to push to and hands it over as a
|
||||||
|
/// capability. Returns a `Subscriber` to loop `next` on, or null on failure.
|
||||||
|
pub fn subscribe(device_mask: u32) ?Subscriber {
|
||||||
|
const service = lookupService() orelse return null;
|
||||||
|
const endpoint = ipc.createIpcEndpoint() orelse return null;
|
||||||
|
|
||||||
|
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.subscribe), .device_mask = device_mask };
|
||||||
|
var reply: [protocol.reply_size]u8 = undefined;
|
||||||
|
const result = ipc.callCap(service, std.mem.asBytes(&request), &reply, endpoint) catch return null;
|
||||||
|
if (result.len < protocol.reply_size) return null;
|
||||||
|
if (std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status != 0) return null;
|
||||||
|
return .{ .endpoint = endpoint };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Subscribe to every input class (keyboard, mouse, joystick) on one stream.
|
||||||
|
pub fn subscribeAll() ?Subscriber {
|
||||||
|
return subscribe(device_all);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A subscriber filtered to keyboard events, whose `next` returns a decoded `KeyEvent`.
|
||||||
|
pub const KeyboardSubscriber = struct {
|
||||||
|
inner: Subscriber,
|
||||||
|
pub fn next(self: *KeyboardSubscriber) ?KeyEvent {
|
||||||
|
return (self.inner.next() orelse return null).asKeyboard();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A subscriber filtered to mouse events, whose `next` returns a decoded `MouseEvent`.
|
||||||
|
pub const MouseSubscriber = struct {
|
||||||
|
inner: Subscriber,
|
||||||
|
pub fn next(self: *MouseSubscriber) ?MouseEvent {
|
||||||
|
return (self.inner.next() orelse return null).asMouse();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A subscriber filtered to joystick/gamepad events, whose `next` returns a decoded
|
||||||
|
/// `JoystickEvent`.
|
||||||
|
pub const JoystickSubscriber = struct {
|
||||||
|
inner: Subscriber,
|
||||||
|
pub fn next(self: *JoystickSubscriber) ?JoystickEvent {
|
||||||
|
return (self.inner.next() orelse return null).asJoystick();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Subscribe to keyboard events only; `next` returns decoded `KeyEvent`s.
|
||||||
|
pub fn subscribeKeyboard() ?KeyboardSubscriber {
|
||||||
|
return .{ .inner = subscribe(device_keyboard) orelse return null };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Subscribe to mouse events only; `next` returns decoded `MouseEvent`s.
|
||||||
|
pub fn subscribeMouse() ?MouseSubscriber {
|
||||||
|
return .{ .inner = subscribe(device_mouse) orelse return null };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Subscribe to joystick/gamepad events only; `next` returns decoded `JoystickEvent`s.
|
||||||
|
pub fn subscribeJoystick() ?JoystickSubscriber {
|
||||||
|
return .{ .inner = subscribe(device_joystick) orelse return null };
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- publishing -------------------------------------------------------------
|
||||||
|
|
||||||
|
/// A connection to the input service for a source (a keyboard/mouse/joystick driver) that
|
||||||
|
/// publishes events. Each `publish*Event` is a short synchronous call the service answers
|
||||||
|
/// at once; its own fan-out to subscribers is asynchronous, so publishing never blocks on
|
||||||
|
/// a slow subscriber.
|
||||||
|
pub const Publisher = struct {
|
||||||
|
service: ipc.Handle,
|
||||||
|
|
||||||
|
fn publish(self: Publisher, event: InputEvent) bool {
|
||||||
|
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.publish), .event = event };
|
||||||
|
var reply: [protocol.reply_size]u8 = undefined;
|
||||||
|
const len = ipc.call(self.service, std.mem.asBytes(&request), &reply) catch return false;
|
||||||
|
if (len < protocol.reply_size) return false;
|
||||||
|
return std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Broadcast a keyboard event to every subscriber that took keyboard events.
|
||||||
|
pub fn publishKeyboardEvent(self: Publisher, event: KeyEvent) bool {
|
||||||
|
return self.publish(InputEvent.fromKeyboard(event));
|
||||||
|
}
|
||||||
|
/// Broadcast a mouse event to every subscriber that took mouse events.
|
||||||
|
pub fn publishMouseEvent(self: Publisher, event: MouseEvent) bool {
|
||||||
|
return self.publish(InputEvent.fromMouse(event));
|
||||||
|
}
|
||||||
|
/// Broadcast a joystick/gamepad event to every subscriber that took joystick events.
|
||||||
|
pub fn publishJoystickEvent(self: Publisher, event: JoystickEvent) bool {
|
||||||
|
return self.publish(InputEvent.fromJoystick(event));
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Connect to the input service as an event source, waiting for it to come up. Returns a
|
||||||
|
/// `Publisher`, or null if the service never registered.
|
||||||
|
pub fn connectSource() ?Publisher {
|
||||||
|
return .{ .service = lookupService() orelse return null };
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- synthetic scaffolding --------------------------------------------------
|
||||||
|
|
||||||
|
/// Synthetic key events, shared by the demo source and the keyboard driver's placeholder
|
||||||
|
/// stream while real scancode decoding is still a follow-up. `step` rolls through A..E,
|
||||||
|
/// emitting for each key a `key_down`, then a `key_press` carrying the character, then a
|
||||||
|
/// `key_up`. Scaffolding, not wire protocol — hence it lives with the helpers.
|
||||||
|
pub fn syntheticKeyEvent(step: usize) KeyEvent {
|
||||||
|
const Key = struct { code: Keycode, character: u32 };
|
||||||
|
const keys = [_]Key{
|
||||||
|
.{ .code = .a, .character = 'A' },
|
||||||
|
.{ .code = .b, .character = 'B' },
|
||||||
|
.{ .code = .c, .character = 'C' },
|
||||||
|
.{ .code = .d, .character = 'D' },
|
||||||
|
.{ .code = .e, .character = 'E' },
|
||||||
|
};
|
||||||
|
const key = keys[(step / 3) % keys.len];
|
||||||
|
return switch (step % 3) {
|
||||||
|
0 => .{ .kind = @intFromEnum(EventKind.key_down), .keycode = @intFromEnum(key.code), .character = 0, .modifiers = 0 },
|
||||||
|
1 => .{ .kind = @intFromEnum(EventKind.key_press), .keycode = @intFromEnum(key.code), .character = key.character, .modifiers = 0 },
|
||||||
|
else => .{ .kind = @intFromEnum(EventKind.key_up), .keycode = @intFromEnum(key.code), .character = 0, .modifiers = 0 },
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Synthetic mouse events (placeholder until real PS/2 packet decoding). `step` alternates
|
||||||
|
/// a small diagonal motion with a left-button click.
|
||||||
|
pub fn syntheticMouseEvent(step: usize) MouseEvent {
|
||||||
|
return switch (step % 3) {
|
||||||
|
0 => .{ .kind = @intFromEnum(MouseEventKind.motion), .button = 0, .dx = 1, .dy = 1, .scroll_x = 0, .scroll_y = 0, .buttons = 0 },
|
||||||
|
1 => .{ .kind = @intFromEnum(MouseEventKind.button_down), .button = protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = protocol.mouse_button_left },
|
||||||
|
else => .{ .kind = @intFromEnum(MouseEventKind.button_up), .button = protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = 0 },
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Synthetic joystick/gamepad events (placeholder until a real controller driver). `step`
|
||||||
|
/// sweeps axis 0 and toggles button 0.
|
||||||
|
pub fn syntheticJoystickEvent(step: usize) JoystickEvent {
|
||||||
|
return switch (step % 3) {
|
||||||
|
0 => .{ .kind = @intFromEnum(JoystickEventKind.axis), .control = 0, .value = 16384, .buttons = 0 },
|
||||||
|
1 => .{ .kind = @intFromEnum(JoystickEventKind.button_down), .control = 0, .value = 0, .buttons = 1 },
|
||||||
|
else => .{ .kind = @intFromEnum(JoystickEventKind.button_up), .control = 0, .value = 0, .buttons = 0 },
|
||||||
|
};
|
||||||
|
}
|
||||||
@@ -0,0 +1,194 @@
|
|||||||
|
//! User-space IPC helpers over the kernel's synchronous IPC syscalls. A client
|
||||||
|
//! `call`s an endpoint (send + block for reply); the VFS server and drivers are
|
||||||
|
//! reached this way. The server side (`replyWait`, which returns two values) is
|
||||||
|
//! added with the first server binary.
|
||||||
|
|
||||||
|
const abi = @import("abi");
|
||||||
|
const sc = @import("system-call.zig");
|
||||||
|
|
||||||
|
/// A small-int handle into the calling process's handle table.
|
||||||
|
pub const Handle = usize;
|
||||||
|
|
||||||
|
/// A fixed-size, register-friendly message payload. Server protocols (VFS, driver)
|
||||||
|
/// layer their own wire format on top of the bytes a call carries.
|
||||||
|
pub const Message = extern struct {
|
||||||
|
tag: u64 = 0,
|
||||||
|
a: u64 = 0,
|
||||||
|
b: u64 = 0,
|
||||||
|
c: u64 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Whether a system_call return value is a wrapped -errno (lands in the top page).
|
||||||
|
inline fn failed(r: usize) bool {
|
||||||
|
return r > ~@as(usize, 0) - 4095;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Create a new endpoint owned by this process; returns its handle.
|
||||||
|
pub fn createIpcEndpoint() ?Handle {
|
||||||
|
const r = sc.systemCall0(.create_ipc_endpoint);
|
||||||
|
return if (failed(r)) null else r;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Publish endpoint `h` under a well-known service id so other processes find it.
|
||||||
|
pub fn register(id: abi.ServiceId, h: Handle) bool {
|
||||||
|
return !failed(sc.systemCall2(.ipc_register, @intFromEnum(id), h));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Find the endpoint published under `id`, installing a handle to it in this
|
||||||
|
/// process.
|
||||||
|
pub fn lookup(id: abi.ServiceId) ?Handle {
|
||||||
|
const r = sc.systemCall1(.ipc_lookup, @intFromEnum(id));
|
||||||
|
return if (failed(r)) null else r;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const CallError = error{Failed};
|
||||||
|
|
||||||
|
/// The result of a capability-passing `callCap`: the reply length, and the handle of
|
||||||
|
/// an endpoint the server sent back (e.g. a per-device channel), or null.
|
||||||
|
pub const Reply = struct {
|
||||||
|
len: usize,
|
||||||
|
cap: ?Handle,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Send `message` to endpoint `h` and block until the server replies into `reply`,
|
||||||
|
/// optionally handing the server a capability (`send_cap`) and receiving one back.
|
||||||
|
/// This is the class-driver "open" primitive: call a bus with `send_cap = null`, get a
|
||||||
|
/// private per-device endpoint back in `.cap`. Two return values (reply length in rax,
|
||||||
|
/// received handle in r8) need a hand-written stub — r8 is read-write (in: reply
|
||||||
|
/// capacity, arg #4; out: the received handle).
|
||||||
|
pub fn callCap(h: Handle, message: []const u8, reply: []u8, send_cap: ?Handle) CallError!Reply {
|
||||||
|
var rax: usize = undefined;
|
||||||
|
var r8: usize = reply.len; // in: reply capacity (arg #4); out: received capability handle
|
||||||
|
asm volatile ("syscall"
|
||||||
|
: [rax] "={rax}" (rax),
|
||||||
|
[r8] "+{r8}" (r8),
|
||||||
|
: [n] "{rax}" (@intFromEnum(abi.SystemCall.ipc_call)),
|
||||||
|
[a0] "{rdi}" (h),
|
||||||
|
[a1] "{rsi}" (@intFromPtr(message.ptr)),
|
||||||
|
[a2] "{rdx}" (message.len),
|
||||||
|
[a3] "{r10}" (@intFromPtr(reply.ptr)),
|
||||||
|
[a5] "{r9}" (send_cap orelse abi.no_cap),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
if (failed(rax)) return error.Failed;
|
||||||
|
return .{ .len = rax, .cap = if (r8 == abi.no_cap) null else r8 };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Send `message` to endpoint `h` and block until the server replies into `reply`.
|
||||||
|
/// Returns the reply length. The common case: no capability passed either way.
|
||||||
|
pub fn call(h: Handle, message: []const u8, reply: []u8) CallError!usize {
|
||||||
|
return (try callCap(h, message, reply, null)).len;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Post `message` to endpoint `h`'s asynchronous queue and return immediately — no
|
||||||
|
/// rendezvous, no reply, no blocking. The receiver picks it up through `replyWait` as a
|
||||||
|
/// buffered message (`Received.isMessage`). Unlike `call`, this **cannot hang on a dead
|
||||||
|
/// or slow peer**, which is why a broadcaster (the input service) delivers events this
|
||||||
|
/// way. The payload must fit an endpoint slot (64 bytes); a full queue drops the oldest
|
||||||
|
/// message. Returns false on failure (bad handle, oversized payload, bad buffer).
|
||||||
|
pub fn send(h: Handle, message: []const u8) bool {
|
||||||
|
return !failed(sc.systemCall3(.ipc_send, h, @intFromPtr(message.ptr), message.len));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Set in `Received.badge` when what arrived is an asynchronous notification — a
|
||||||
|
/// bound device interrupt — rather than a client's message. The low bits carry the
|
||||||
|
/// GSI. See `isNotification`.
|
||||||
|
pub const notify_badge_bit: u64 = abi.notify_badge_bit;
|
||||||
|
|
||||||
|
/// Set alongside `notify_badge_bit` when the notification is a **signal** — the
|
||||||
|
/// lifecycle vocabulary of docs/process-lifecycle.md, delivered to the endpoint
|
||||||
|
/// nominated with `process.bindSignals`. Decode with `process.signalsFrom`.
|
||||||
|
pub const notify_signal_bit: u64 = abi.notify_signal_bit;
|
||||||
|
|
||||||
|
/// Set alongside `notify_badge_bit` when the notification is a **one-shot timer**
|
||||||
|
/// landing (`system.timerOnce`).
|
||||||
|
pub const notify_timer_bit: u64 = abi.notify_timer_bit;
|
||||||
|
|
||||||
|
/// Set alongside `notify_badge_bit` when the notification is a **child-exit
|
||||||
|
/// notice** — a process this one spawned (with an exit endpoint) has ended —
|
||||||
|
/// rather than a device interrupt. The low bits carry the child's process id.
|
||||||
|
pub const notify_exit_bit: u64 = abi.notify_exit_bit;
|
||||||
|
|
||||||
|
/// Set alongside `notify_badge_bit` when the wake-up is a **buffered message** — a payload
|
||||||
|
/// posted with `send` (`ipc_send`) — rather than a bare device interrupt or child-exit
|
||||||
|
/// notice. The payload is in the `replyWait` receive buffer (`Received.len` bytes); the
|
||||||
|
/// low bits of the badge carry the sender's task id. See `Received.isMessage`.
|
||||||
|
pub const notify_message_bit: u64 = abi.notify_message_bit;
|
||||||
|
|
||||||
|
/// The result of a `replyWait`: the request length, the sender's badge (a task id, or
|
||||||
|
/// an IRQ notification if the high bit is set), and any capability the request carried.
|
||||||
|
pub const Received = struct {
|
||||||
|
len: usize,
|
||||||
|
badge: u64,
|
||||||
|
cap: ?Handle,
|
||||||
|
|
||||||
|
/// True if this wake-up was an asynchronous notification (a device interrupt
|
||||||
|
/// or a child-exit notice), not a client request. An event loop branches on
|
||||||
|
/// this; there is no reply owed on the notification path.
|
||||||
|
pub fn isNotification(self: Received) bool {
|
||||||
|
return self.badge & notify_badge_bit != 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// True if this wake-up tells of a supervised child's end — the notification
|
||||||
|
/// requested by passing an exit endpoint to `system.spawnSupervised`.
|
||||||
|
pub fn isChildExit(self: Received) bool {
|
||||||
|
return self.isNotification() and self.badge & notify_exit_bit != 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// True if this wake-up is a **buffered message** posted with `send` (`ipc_send`):
|
||||||
|
/// there is a payload in the receive buffer (`self.len` bytes) and no reply is owed.
|
||||||
|
/// The subscriber side of a broadcast branches on this.
|
||||||
|
pub fn isMessage(self: Received) bool {
|
||||||
|
return self.isNotification() and self.badge & notify_message_bit != 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The task id of whoever posted a buffered message, meaningful only when
|
||||||
|
/// Whether this arrival is a signal notification — decode the set with
|
||||||
|
/// `process.signalsFrom(badge)`.
|
||||||
|
pub fn isSignal(self: Received) bool {
|
||||||
|
return self.isNotification() and self.badge & notify_signal_bit != 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether this arrival is a one-shot timer landing (`system.timerOnce`).
|
||||||
|
pub fn isTimer(self: Received) bool {
|
||||||
|
return self.isNotification() and self.badge & notify_timer_bit != 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `isMessage`. (The badge's low bits, with the three high marker bits masked off.)
|
||||||
|
pub fn senderTaskId(self: Received) u32 {
|
||||||
|
return @intCast(self.badge & ~(notify_badge_bit | notify_exit_bit | notify_message_bit));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The interrupt source (a GSI), meaningful only when `isNotification` and
|
||||||
|
/// not `isChildExit`.
|
||||||
|
pub fn source(self: Received) u64 {
|
||||||
|
return self.badge & ~notify_badge_bit;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The ended child's process id, meaningful only when `isChildExit`.
|
||||||
|
pub fn childProcessId(self: Received) u32 {
|
||||||
|
return @intCast(self.badge & ~(notify_badge_bit | notify_exit_bit));
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Server side of IPC_ReplyWait: deliver `reply` to the client last received (if any,
|
||||||
|
/// optionally handing it `send_cap`), then block until the next request arrives in
|
||||||
|
/// `receive`. Returns its length, the sender badge, and any capability the request
|
||||||
|
/// carried (in `.cap`). Three return values — length in rax, badge in rdx, received
|
||||||
|
/// handle in r8 — so it needs a hand-written stub: rdx is read-write (in: reply length,
|
||||||
|
/// arg #3; out: badge) and r8 is read-write (in: receive capacity, arg #4; out: handle).
|
||||||
|
pub fn replyWait(h: Handle, reply: []const u8, receive: []u8, send_cap: ?Handle) Received {
|
||||||
|
var rax: usize = undefined;
|
||||||
|
var rdx: usize = reply.len; // in: reply_len (arg #3); out: badge
|
||||||
|
var r8: usize = receive.len; // in: receive capacity (arg #4); out: received capability handle
|
||||||
|
asm volatile ("syscall"
|
||||||
|
: [rax] "={rax}" (rax),
|
||||||
|
[rdx] "+{rdx}" (rdx),
|
||||||
|
[r8] "+{r8}" (r8),
|
||||||
|
: [n] "{rax}" (@intFromEnum(abi.SystemCall.ipc_reply_wait)),
|
||||||
|
[a0] "{rdi}" (h),
|
||||||
|
[a1] "{rsi}" (@intFromPtr(reply.ptr)),
|
||||||
|
[a3] "{r10}" (@intFromPtr(receive.ptr)),
|
||||||
|
[a5] "{r9}" (send_cap orelse abi.no_cap),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
return .{ .len = rax, .badge = rdx, .cap = if (r8 == abi.no_cap) null else r8 };
|
||||||
|
}
|
||||||
@@ -0,0 +1,137 @@
|
|||||||
|
//! Process-level runtime types: what a user program receives at entry (`Init`,
|
||||||
|
//! the argv contract) and the process end of the lifecycle
|
||||||
|
//! (docs/process-lifecycle.md) — today the exit reason a supervisor reads to
|
||||||
|
//! decide restart; signals and the stop sequence land here with M17.4. Mirrors
|
||||||
|
//! the spirit of `std.process.Init.Minimal` in danos terms — std's `Args` holds
|
||||||
|
//! no data on freestanding targets, so the type is danos's own.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const sc = @import("system-call.zig");
|
||||||
|
const ipc = @import("ipc.zig");
|
||||||
|
const system = @import("system.zig");
|
||||||
|
|
||||||
|
/// Everything a program receives at entry. Passed to
|
||||||
|
/// `pub fn main(init: runtime.process.Init)`; programs that need nothing keep
|
||||||
|
/// `pub fn main() void`. An `environment` field is added here once the kernel
|
||||||
|
/// passes a non-empty envp (today it is always empty — see docs/sysv.md).
|
||||||
|
pub const Init = struct {
|
||||||
|
arguments: Arguments,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The process arguments (argc/argv), parsed from the kernel-built System V
|
||||||
|
/// entry block. The bytes live in the entry block at the top of the stack page,
|
||||||
|
/// NUL-terminated, valid for the process's lifetime.
|
||||||
|
pub const Arguments = struct {
|
||||||
|
/// argc — at least 1: argument 0 is the path or name this binary was
|
||||||
|
/// spawned as.
|
||||||
|
count: usize,
|
||||||
|
/// The argv pointers in the entry block (NULL-terminated after `count`
|
||||||
|
/// entries).
|
||||||
|
vector: [*]const [*:0]const u8,
|
||||||
|
|
||||||
|
/// Argument `index` (0 = the program's own path/name), or null if out of
|
||||||
|
/// range.
|
||||||
|
pub fn get(arguments: Arguments, index: usize) ?[:0]const u8 {
|
||||||
|
if (index >= arguments.count) return null;
|
||||||
|
return std.mem.span(arguments.vector[index]);
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn iterate(arguments: Arguments) Iterator {
|
||||||
|
return .{ .arguments = arguments };
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const Iterator = struct {
|
||||||
|
arguments: Arguments,
|
||||||
|
index: usize = 0,
|
||||||
|
|
||||||
|
pub fn next(iterator: *Iterator) ?[:0]const u8 {
|
||||||
|
const argument = iterator.arguments.get(iterator.index) orelse return null;
|
||||||
|
iterator.index += 1;
|
||||||
|
return argument;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
/// How a process ended — what a supervisor's restart policy reads: a clean exit
|
||||||
|
/// meant to stop, a fault wants a restart with backoff, killed means the
|
||||||
|
/// supervisor did it itself (docs/process-lifecycle.md).
|
||||||
|
pub const ExitReason = abi.ExitReason;
|
||||||
|
|
||||||
|
/// How dead child `id` ended. Ask after the exit notification arrives — the
|
||||||
|
/// kernel records the reason before it posts the notification, so this never
|
||||||
|
/// races it. Returns null for an id that never lived, is still alive, was
|
||||||
|
/// evicted from the kernel's bounded record, or is not this process's child
|
||||||
|
/// (the same authority gate as `kill`).
|
||||||
|
pub fn exitReason(id: u32) ?ExitReason {
|
||||||
|
const r = sc.systemCall1(.process_exit_reason, id);
|
||||||
|
if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno
|
||||||
|
return @enumFromInt(r);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The signal vocabulary (docs/process-lifecycle.md): POSIX's concepts, danos's
|
||||||
|
/// names, message delivery. A signal is a one-way coalescing statement — never a
|
||||||
|
/// question (liveness is the zero-length ping call) and never kill (that is
|
||||||
|
/// `system.kill`, unhandleable by definition).
|
||||||
|
pub const Signal = abi.Signal;
|
||||||
|
|
||||||
|
/// The coalesced set of signals one notification delivered: two pending
|
||||||
|
/// terminates arrive as one. Decode a received badge with `signalsFrom`.
|
||||||
|
pub const SignalSet = struct {
|
||||||
|
pending: u32,
|
||||||
|
|
||||||
|
pub fn has(set: SignalSet, signal: Signal) bool {
|
||||||
|
return set.pending & (@as(u32, 1) << @intFromEnum(signal)) != 0;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Nominate `endpoint` as this process's signal endpoint. Signals posted while
|
||||||
|
/// unbound have pended; they are delivered immediately on bind, coalesced.
|
||||||
|
pub fn bindSignals(endpoint: usize) bool {
|
||||||
|
return sc.systemCall1(.signal_bind, endpoint) == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode a received badge into the signals it delivered, or null if it is not
|
||||||
|
/// a signal notification.
|
||||||
|
pub fn signalsFrom(badge: u64) ?SignalSet {
|
||||||
|
if (badge & abi.notify_badge_bit == 0 or badge & abi.notify_signal_bit == 0) return null;
|
||||||
|
return .{ .pending = @truncate(badge & ~(abi.notify_badge_bit | abi.notify_signal_bit)) };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Post `signal` to child `id` (or to yourself). Supervisor-gated, like kill;
|
||||||
|
/// non-blocking, always — a statement, not a conversation.
|
||||||
|
pub fn sendSignal(id: u32, signal: Signal) bool {
|
||||||
|
return sc.systemCall2(.process_signal, id, @intFromEnum(signal)) == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The standard stop sequence (docs/process-lifecycle.md): terminate, wait up to
|
||||||
|
/// `deadline_ms` for the exit notification on `exit_endpoint` (the endpoint the
|
||||||
|
/// child was spawned with), then kill. Any *other* notifications arriving on
|
||||||
|
/// that endpoint while stopping are consumed and dropped — a supervisor with
|
||||||
|
/// concurrent traffic implements the same sequence inside its own event loop
|
||||||
|
/// (arm `system.timerOnce`, keep serving) instead of calling this.
|
||||||
|
pub fn stop(id: u32, deadline_ms: u64, exit_endpoint: usize) void {
|
||||||
|
_ = sendSignal(id, .terminate);
|
||||||
|
_ = system.timerOnce(exit_endpoint, deadline_ms);
|
||||||
|
var receive: [8]u8 = undefined;
|
||||||
|
while (true) {
|
||||||
|
const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null);
|
||||||
|
if (got.isChildExit() and got.childProcessId() == id) return;
|
||||||
|
if (got.isTimer()) break; // the deadline passed first — escalate
|
||||||
|
}
|
||||||
|
_ = system.kill(id);
|
||||||
|
while (true) {
|
||||||
|
const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null);
|
||||||
|
if (got.isChildExit() and got.childProcessId() == id) return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Subscribe `endpoint` to published exit events: every process death posts an
|
||||||
|
/// asynchronous notification with the same badge encoding as a supervisor's exit
|
||||||
|
/// notice (decode with `ipc.Received.isChildExit`/`childProcessId`). For stateful
|
||||||
|
/// services: release what the dead client held — file handles, subscriptions —
|
||||||
|
/// because a service must never depend on clients cleaning up after themselves
|
||||||
|
/// (docs/process-lifecycle.md). Ungated, like `system.processes`.
|
||||||
|
pub fn subscribeExits(endpoint: usize) bool {
|
||||||
|
return sc.systemCall1(.process_subscribe, endpoint) == 0;
|
||||||
|
}
|
||||||
@@ -0,0 +1,65 @@
|
|||||||
|
//! danos user-space runtime library — a nascent libc. Every user binary (init,
|
||||||
|
//! and later the VFS server + device drivers) imports this as `@import("runtime")`:
|
||||||
|
//! system_call wrappers, the C-convention heap, IPC helpers, and the process start
|
||||||
|
//! shim. It is compiled into each binary (inheriting its `.large` code model and
|
||||||
|
//! freestanding target), so all user programs share one implementation.
|
||||||
|
//!
|
||||||
|
//! A user binary needs three lines:
|
||||||
|
//! const runtime = @import("runtime");
|
||||||
|
//! pub const panic = runtime.panic;
|
||||||
|
//! comptime { _ = &runtime.start._start; } // pull the entry shim in
|
||||||
|
//! and a `pub fn main() void` or `pub fn main(init: runtime.process.Init) void`
|
||||||
|
//! (arguments arrive via `init`).
|
||||||
|
|
||||||
|
pub const system = @import("system.zig");
|
||||||
|
/// Monotonic time, delays, and deadlines over the kernel clock/sleep/timer syscalls
|
||||||
|
/// — an `Instant`/`Duration` front door, no time service (docs/timers.md).
|
||||||
|
pub const time = @import("time.zig");
|
||||||
|
pub const heap = @import("heap.zig");
|
||||||
|
pub const ipc = @import("ipc.zig");
|
||||||
|
pub const start = @import("start.zig");
|
||||||
|
/// The VFS wire protocol (shared with the VFS server).
|
||||||
|
pub const vfs_protocol = @import("vfs-protocol");
|
||||||
|
|
||||||
|
/// The device-manager protocol: hello + tree reports (docs/device-manager.md).
|
||||||
|
pub const device_manager_protocol = @import("device-manager-protocol");
|
||||||
|
|
||||||
|
/// The power protocol: events (button, lid, battery) + shutdown (docs/power.md).
|
||||||
|
pub const power_protocol = @import("power-protocol");
|
||||||
|
/// Keyboard-event listening (subscribe/next) and broadcasting (publish), over the input
|
||||||
|
/// service. See library/runtime/input.zig and system/services/input/.
|
||||||
|
pub const input = @import("input.zig");
|
||||||
|
/// The input wire protocol (shared with the input service and its clients).
|
||||||
|
pub const input_protocol = @import("input-protocol");
|
||||||
|
/// POSIX-style file API: open/read/write/lseek/stat/close.
|
||||||
|
/// C stdio: fopen/fread/fwrite/fseek/ftell/fclose over unistd.
|
||||||
|
/// Device access for drivers: enumerate/claim/mmioMap.
|
||||||
|
pub const device = @import("device.zig");
|
||||||
|
/// DMA-capable memory for drivers: contiguous, pinned, uncacheable buffers.
|
||||||
|
pub const dma = @import("dma.zig");
|
||||||
|
|
||||||
|
/// USB class-driver client: open a device on the xHCI bus and drive it
|
||||||
|
/// (control / interrupt / bulk transfers). See library/runtime/usb.zig.
|
||||||
|
pub const usb = @import("usb.zig");
|
||||||
|
|
||||||
|
/// Block-device client: read/write a block device (a USB stick, via
|
||||||
|
/// usb-storage). See library/runtime/block.zig.
|
||||||
|
pub const block = @import("block.zig");
|
||||||
|
|
||||||
|
/// The danos-native file API (open/read/write/list over the user-space VFS) — the
|
||||||
|
/// layer danos programs use directly, and where the operations that later become
|
||||||
|
/// `std.os.danos` are staged. See docs/zig-self-hosting.md.
|
||||||
|
pub const fs = @import("fs.zig");
|
||||||
|
|
||||||
|
/// Re-exported so a user binary can `pub const panic = runtime.panic;`.
|
||||||
|
pub const panic = start.panic;
|
||||||
|
|
||||||
|
/// Process entry types: the `Init` handed to `main`, and its `Arguments`.
|
||||||
|
pub const process = @import("process.zig");
|
||||||
|
|
||||||
|
/// The service harness: one replyWait loop folding requests, signals, and
|
||||||
|
/// notifications into callbacks (docs/process-lifecycle.md).
|
||||||
|
pub const service = @import("service.zig");
|
||||||
|
|
||||||
|
/// The heap as a `std.mem.Allocator`, for Zig `std` containers in user code.
|
||||||
|
pub const allocator = heap.allocator;
|
||||||
@@ -0,0 +1,83 @@
|
|||||||
|
//! The service harness (docs/process-lifecycle.md): one replyWait loop that
|
||||||
|
//! folds protocol requests, signals, and subscribed notifications into
|
||||||
|
//! callbacks — so the lifecycle contract ("answers ping, exits on terminate")
|
||||||
|
//! is satisfied by construction and a service author writes domain logic only.
|
||||||
|
//! Nothing is asynchronous inside the process: a callback runs at a point the
|
||||||
|
//! loop chose, never on a hijacked stack — the whole reason signals are
|
||||||
|
//! messages.
|
||||||
|
//!
|
||||||
|
//! The liveness probe: a **zero-length request is the universal ping**, answered
|
||||||
|
//! with a zero-length reply by the harness itself. No protocol's requests start
|
||||||
|
//! at length zero, so the encoding cannot collide, and there is nothing for a
|
||||||
|
//! service author to implement — a wedged service simply fails to answer, which
|
||||||
|
//! is the diagnosis (see docs/ipc.md).
|
||||||
|
|
||||||
|
const abi = @import("abi");
|
||||||
|
const ipc = @import("ipc.zig");
|
||||||
|
const process = @import("process.zig");
|
||||||
|
|
||||||
|
pub const Callbacks = struct {
|
||||||
|
/// Called once with the service's endpoint before the loop starts — the
|
||||||
|
/// place to subscribe to exit events, bind IRQs, or announce readiness.
|
||||||
|
/// Return false to abort startup (the process exits).
|
||||||
|
init: ?*const fn (endpoint: ipc.Handle) bool = null,
|
||||||
|
/// One protocol request from `sender` (a task id): write the reply into
|
||||||
|
/// `reply`, return its length. `capability` is the handle the request
|
||||||
|
/// carried, if any (M13 cap passing — how a subscriber hands over its
|
||||||
|
/// endpoint). The zero-length ping never reaches this.
|
||||||
|
on_message: *const fn (message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize,
|
||||||
|
/// A notification that is not a signal — a subscribed exit event, a bound
|
||||||
|
/// IRQ, a timer landing. The raw badge; decode with the ipc helpers.
|
||||||
|
on_notification: ?*const fn (badge: u64) void = null,
|
||||||
|
/// The reload signal. Default: ignored.
|
||||||
|
on_reload: ?*const fn () void = null,
|
||||||
|
/// The terminate signal, called before the loop returns. The clean exit is
|
||||||
|
/// the return itself — never put *necessary* work here (iron rule 1: a kill
|
||||||
|
/// arrives with no warning; this is for graceful extras only).
|
||||||
|
on_terminate: ?*const fn () void = null,
|
||||||
|
/// Publish the endpoint under a well-known service id at startup.
|
||||||
|
service: ?abi.ServiceId = null,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Run the service: create and (optionally) register the endpoint, bind signals
|
||||||
|
/// to it, call `init`, then serve until `terminate` arrives — at which point the
|
||||||
|
/// loop returns and main's return is the clean exit the supervisor reads as
|
||||||
|
/// `ExitReason.exited`. `maximum_message` sizes the receive and reply buffers
|
||||||
|
/// (a service passes its protocol's message maximum).
|
||||||
|
pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
|
||||||
|
const endpoint = ipc.createIpcEndpoint() orelse return;
|
||||||
|
if (callbacks.service) |id| {
|
||||||
|
if (!ipc.register(id, endpoint)) return;
|
||||||
|
}
|
||||||
|
_ = process.bindSignals(endpoint);
|
||||||
|
if (callbacks.init) |initialise| {
|
||||||
|
if (!initialise(endpoint)) return;
|
||||||
|
}
|
||||||
|
|
||||||
|
var reply_buffer: [maximum_message]u8 = undefined;
|
||||||
|
var reply_len: usize = 0;
|
||||||
|
var receive: [maximum_message]u8 = undefined;
|
||||||
|
while (true) {
|
||||||
|
const got = ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive, null);
|
||||||
|
if (got.isNotification()) {
|
||||||
|
reply_len = 0; // nothing owed for a notification
|
||||||
|
if (process.signalsFrom(got.badge)) |signals| {
|
||||||
|
if (signals.has(.reload)) {
|
||||||
|
if (callbacks.on_reload) |onReload| onReload();
|
||||||
|
}
|
||||||
|
if (signals.has(.terminate)) {
|
||||||
|
if (callbacks.on_terminate) |onTerminate| onTerminate();
|
||||||
|
return; // the loop's return IS the clean exit
|
||||||
|
}
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (callbacks.on_notification) |onNotification| onNotification(got.badge);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (got.len == 0) {
|
||||||
|
reply_len = 0; // the universal ping: a zero-length reply, from the harness
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
reply_len = callbacks.on_message(receive[0..got.len], &reply_buffer, got.senderTaskId(), got.cap);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,86 @@
|
|||||||
|
//! The user-space process entry shim. Every user binary roots `_start` here (via
|
||||||
|
//! `entry = _start` in build.zig) and forces this file to be analysed with
|
||||||
|
//! `comptime { _ = &runtime.start._start; }`, so the whole runtime is linked in.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const system = @import("system.zig");
|
||||||
|
const process = @import("process.zig");
|
||||||
|
|
||||||
|
/// The kernel enters at `_start` with rsp 16-aligned, pointing at the System V
|
||||||
|
/// process-entry block it built: argc, argv pointers, NULL, envp terminator, the
|
||||||
|
/// auxiliary vector, then the strings (see system/kernel/process.zig,
|
||||||
|
/// `buildEntryStack`). Capture that address in rdi — the first SysV argument —
|
||||||
|
/// before `call` disturbs the stack; the call's pushed return address also puts
|
||||||
|
/// rsp ≡ 8 (mod 16), satisfying the ABI before any Zig frame runs. The `ud2` is a
|
||||||
|
/// safety net if `rt_start` ever returns.
|
||||||
|
pub export fn _start() callconv(.naked) noreturn {
|
||||||
|
asm volatile (
|
||||||
|
\\mov %%rsp, %%rdi
|
||||||
|
\\call rt_start
|
||||||
|
\\ud2
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The first Zig frame, entered with `stack` pointing at the kernel-built entry
|
||||||
|
/// block. Build the `process.Init` from it and dispatch to the program's `main`,
|
||||||
|
/// whose signature is inspected at comptime. The heap is lazy (first alloc grows
|
||||||
|
/// it), so there is no other runtime init to order here.
|
||||||
|
export fn rt_start(stack: [*]const u64) callconv(.c) noreturn {
|
||||||
|
const init: process.Init = .{ .arguments = .{
|
||||||
|
.count = stack[0],
|
||||||
|
.vector = @ptrCast(stack + 1),
|
||||||
|
} };
|
||||||
|
system.exit(callMain(init));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Comptime-dispatch on root.main's signature, in the spirit of std's start.zig:
|
||||||
|
/// zero parameters or one `process.Init`; returns void, noreturn, u8, !void, or !u8.
|
||||||
|
fn callMain(init: process.Init) u8 {
|
||||||
|
const root = @import("root"); // the user binary's root source file
|
||||||
|
const main_information = @typeInfo(@TypeOf(root.main)).@"fn";
|
||||||
|
|
||||||
|
const call_arguments = switch (main_information.params.len) {
|
||||||
|
0 => .{},
|
||||||
|
1 => arguments: {
|
||||||
|
const Parameter = main_information.params[0].type orelse
|
||||||
|
@compileError("main's parameter must be runtime.process.Init (not anytype)");
|
||||||
|
if (Parameter != process.Init)
|
||||||
|
@compileError("main's parameter must be runtime.process.Init, found " ++ @typeName(Parameter));
|
||||||
|
break :arguments .{init};
|
||||||
|
},
|
||||||
|
else => @compileError("main takes no parameters or a single runtime.process.Init"),
|
||||||
|
};
|
||||||
|
|
||||||
|
const ReturnType = main_information.return_type.?;
|
||||||
|
switch (@typeInfo(ReturnType)) {
|
||||||
|
.noreturn => @call(.auto, root.main, call_arguments),
|
||||||
|
.void => {
|
||||||
|
@call(.auto, root.main, call_arguments);
|
||||||
|
return 0;
|
||||||
|
},
|
||||||
|
.int => {
|
||||||
|
if (ReturnType != u8)
|
||||||
|
@compileError("main's integer return type must be u8, found " ++ @typeName(ReturnType));
|
||||||
|
return @call(.auto, root.main, call_arguments);
|
||||||
|
},
|
||||||
|
.error_union => {
|
||||||
|
const payload = @call(.auto, root.main, call_arguments) catch |err| {
|
||||||
|
var buffer: [128]u8 = undefined;
|
||||||
|
const line = std.fmt.bufPrint(&buffer, "main returned error: {s}\n", .{@errorName(err)}) catch "main returned an error\n";
|
||||||
|
_ = system.write(line);
|
||||||
|
return 1; // distinct from panic's 127
|
||||||
|
};
|
||||||
|
if (@TypeOf(payload) == void) return 0;
|
||||||
|
if (@TypeOf(payload) == u8) return payload;
|
||||||
|
@compileError("main's error-union payload must be void or u8, found " ++ @typeName(@TypeOf(payload)));
|
||||||
|
},
|
||||||
|
else => @compileError("main must return void, noreturn, u8, !void, or !u8, found " ++ @typeName(ReturnType)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// No runtime to unwind into — report a panic as a nonzero exit code.
|
||||||
|
pub const panic = std.debug.FullPanic(struct {
|
||||||
|
fn panic(_: []const u8, _: ?usize) noreturn {
|
||||||
|
system.exit(127);
|
||||||
|
}
|
||||||
|
}.panic);
|
||||||
@@ -0,0 +1,67 @@
|
|||||||
|
//! Raw `system_call` instruction wrappers for user space — one per arity.
|
||||||
|
//!
|
||||||
|
//! ABI: number in rax, arguments in rdi, rsi, rdx, r10, r8, r9, result in rax.
|
||||||
|
//! The `system_call` instruction itself clobbers rcx (it holds the return rip) and
|
||||||
|
//! r11 (the saved rflags); the kernel entry stub preserves everything else.
|
||||||
|
//! Note argument #3 goes in **r10, not rcx** — rcx is unavailable across the
|
||||||
|
//! instruction, so the kernel reads the 4th argument from r10.
|
||||||
|
|
||||||
|
const abi = @import("abi");
|
||||||
|
const SystemCall = abi.SystemCall;
|
||||||
|
|
||||||
|
pub inline fn systemCall0(n: SystemCall) usize {
|
||||||
|
return asm volatile ("syscall"
|
||||||
|
: [ret] "={rax}" (-> usize),
|
||||||
|
: [n] "{rax}" (@intFromEnum(n)),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
}
|
||||||
|
|
||||||
|
pub inline fn systemCall1(n: SystemCall, a0: usize) usize {
|
||||||
|
return asm volatile ("syscall"
|
||||||
|
: [ret] "={rax}" (-> usize),
|
||||||
|
: [n] "{rax}" (@intFromEnum(n)),
|
||||||
|
[a0] "{rdi}" (a0),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
}
|
||||||
|
|
||||||
|
pub inline fn systemCall2(n: SystemCall, a0: usize, a1: usize) usize {
|
||||||
|
return asm volatile ("syscall"
|
||||||
|
: [ret] "={rax}" (-> usize),
|
||||||
|
: [n] "{rax}" (@intFromEnum(n)),
|
||||||
|
[a0] "{rdi}" (a0),
|
||||||
|
[a1] "{rsi}" (a1),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
}
|
||||||
|
|
||||||
|
pub inline fn systemCall3(n: SystemCall, a0: usize, a1: usize, a2: usize) usize {
|
||||||
|
return asm volatile ("syscall"
|
||||||
|
: [ret] "={rax}" (-> usize),
|
||||||
|
: [n] "{rax}" (@intFromEnum(n)),
|
||||||
|
[a0] "{rdi}" (a0),
|
||||||
|
[a1] "{rsi}" (a1),
|
||||||
|
[a2] "{rdx}" (a2),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
}
|
||||||
|
|
||||||
|
pub inline fn systemCall4(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: usize) usize {
|
||||||
|
return asm volatile ("syscall"
|
||||||
|
: [ret] "={rax}" (-> usize),
|
||||||
|
: [n] "{rax}" (@intFromEnum(n)),
|
||||||
|
[a0] "{rdi}" (a0),
|
||||||
|
[a1] "{rsi}" (a1),
|
||||||
|
[a2] "{rdx}" (a2),
|
||||||
|
[a3] "{r10}" (a3),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
}
|
||||||
|
|
||||||
|
pub inline fn systemCall5(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: usize, a4: usize) usize {
|
||||||
|
return asm volatile ("syscall"
|
||||||
|
: [ret] "={rax}" (-> usize),
|
||||||
|
: [n] "{rax}" (@intFromEnum(n)),
|
||||||
|
[a0] "{rdi}" (a0),
|
||||||
|
[a1] "{rsi}" (a1),
|
||||||
|
[a2] "{rdx}" (a2),
|
||||||
|
[a3] "{r10}" (a3),
|
||||||
|
[a4] "{r8}" (a4),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
}
|
||||||
@@ -0,0 +1,165 @@
|
|||||||
|
//! Typed system_call surface for user space — thin wrappers over the raw `system_call`
|
||||||
|
//! stubs, one per kernel call. Numbers come from `abi.SystemCall`, the single
|
||||||
|
//! source of truth shared with the kernel dispatcher.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const sc = @import("system-call.zig");
|
||||||
|
|
||||||
|
/// `mmap` protection flags (matching the usual C bit values). Grants are always
|
||||||
|
/// readable+writable today; the kernel does not yet honour finer prot.
|
||||||
|
pub const PROT_READ: usize = abi.prot_read;
|
||||||
|
pub const PROT_WRITE: usize = abi.prot_write;
|
||||||
|
pub const PROT_EXEC: usize = abi.prot_exec;
|
||||||
|
|
||||||
|
/// One `processes` entry — re-exported from the shared ABI so a user program can
|
||||||
|
/// declare its snapshot buffer without importing `abi` itself.
|
||||||
|
pub const ProcessDescriptor = abi.ProcessDescriptor;
|
||||||
|
|
||||||
|
/// Give up the rest of this quantum.
|
||||||
|
pub fn yield() void {
|
||||||
|
_ = sc.systemCall0(.yield);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write raw bytes to the kernel log (a bring-up diagnostic; real output goes
|
||||||
|
/// through the console/VFS later). Returns the byte count, or a wrapped -1.
|
||||||
|
pub fn write(message: []const u8) usize {
|
||||||
|
return sc.systemCall2(.debug_write, @intFromPtr(message.ptr), message.len);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Block the caller for `ms` milliseconds.
|
||||||
|
pub fn sleep(ms: usize) void {
|
||||||
|
_ = sc.systemCall1(.sleep, ms);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Arm a one-shot timer: after `ms` milliseconds the kernel posts a timer
|
||||||
|
/// notification (`ipc.Received.isTimer`) to `endpoint`. The timed wait of
|
||||||
|
/// docs/process-lifecycle.md — a service arms a deadline and keeps serving,
|
||||||
|
/// instead of blocking in sleep; what stop-sequence escalation, hello deadlines,
|
||||||
|
/// and restart backoff are built from.
|
||||||
|
pub fn timerOnce(endpoint: usize, ms: u64) bool {
|
||||||
|
return sc.systemCall2(.timer_bind, endpoint, ms) == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Monotonic nanoseconds since boot — a time source for timeouts and short delays. It
|
||||||
|
/// only ever moves forward. This is *not* wall-clock time (no date, no timezone — that
|
||||||
|
/// is a user-space service layered on top). Deadline pattern for a bounded poll loop:
|
||||||
|
///
|
||||||
|
/// const deadline = clock() + timeout_ns;
|
||||||
|
/// while (clock() < deadline) { ... }
|
||||||
|
pub fn clock() u64 {
|
||||||
|
return @intCast(sc.systemCall0(.clock));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Wall-clock time in Unix epoch seconds (UTC) — the real date/time, from the RTC.
|
||||||
|
/// Unlike `clock` (monotonic since boot), this tracks calendar time, so it is what a
|
||||||
|
/// filesystem stamps as a file's modification time. Formatting it into a calendar
|
||||||
|
/// date/timezone is user-space policy layered on top.
|
||||||
|
pub fn wallClock() u64 {
|
||||||
|
return @intCast(sc.systemCall0(.wall_clock));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Copy bytes out of the kernel's in-memory diagnostic log — the accumulated
|
||||||
|
/// stream of everything `write` (and the kernel itself) has emitted — starting at
|
||||||
|
/// `offset`, into `out`. Returns the number of bytes copied (0 at end of buffer).
|
||||||
|
/// A program reads the whole log by looping from offset 0, advancing by the return
|
||||||
|
/// value, until it gets 0. This is how the boot log is persisted to disk on a
|
||||||
|
/// headless/real machine where serial output is otherwise lost.
|
||||||
|
pub fn klogRead(offset: usize, out: []u8) usize {
|
||||||
|
return sc.systemCall3(.klog_read, offset, @intFromPtr(out.ptr), out.len);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// End the process. Never returns.
|
||||||
|
pub fn exit(code: usize) noreturn {
|
||||||
|
_ = sc.systemCall1(.exit, code);
|
||||||
|
unreachable; // the kernel never returns from exit
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Start the binary bundled in the initial-ramdisk under `name` as a new ring-3
|
||||||
|
/// process, returning the child's process id (or null on failure). The child's
|
||||||
|
/// argv[0] is `name`, and the caller becomes its **supervisor** — the only process
|
||||||
|
/// allowed to `kill` it. This is how a supervisor (the device manager) launches a
|
||||||
|
/// driver it matched — danos-native, not POSIX (a spawn/exec family comes with the
|
||||||
|
/// POSIX layer later).
|
||||||
|
pub fn spawn(name: []const u8) ?u32 {
|
||||||
|
return spawnSupervised(name, &.{}, null);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Like `spawn`, but hands the child command-line arguments: they arrive as
|
||||||
|
/// argv[1..] on its System V entry stack (argv[0] is still `name`).
|
||||||
|
pub fn spawnWithArguments(name: []const u8, arguments: []const []const u8) ?u32 {
|
||||||
|
return spawnSupervised(name, arguments, null);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The full spawn: command-line arguments for the child, and an optional endpoint
|
||||||
|
/// (a handle from `ipc.createIpcEndpoint`) the kernel notifies when the child ends
|
||||||
|
/// — any way it ends: clean exit, fault, or `kill`. The notification arrives via
|
||||||
|
/// `ipc.replyWait` as a badge with the child-exit bit set and the child's id in
|
||||||
|
/// the low bits (`ipc.Received.isChildExit`/`childProcessId`), so one endpoint can
|
||||||
|
/// supervise many children. Arguments are marshalled to the kernel as one
|
||||||
|
/// NUL-separated blob; the combined arguments must fit `blob` (the kernel caps the
|
||||||
|
/// blob at 256 bytes and argc at 8 anyway). Returns the child's process id, or
|
||||||
|
/// null on failure.
|
||||||
|
pub fn spawnSupervised(name: []const u8, arguments: []const []const u8, exit_endpoint: ?usize) ?u32 {
|
||||||
|
var blob: [256]u8 = undefined;
|
||||||
|
var len: usize = 0;
|
||||||
|
for (arguments, 0..) |argument, i| {
|
||||||
|
if (i != 0) {
|
||||||
|
if (len >= blob.len) return null;
|
||||||
|
blob[len] = 0;
|
||||||
|
len += 1;
|
||||||
|
}
|
||||||
|
if (len + argument.len > blob.len) return null;
|
||||||
|
@memcpy(blob[len..][0..argument.len], argument);
|
||||||
|
len += argument.len;
|
||||||
|
}
|
||||||
|
const r = sc.systemCall5(.system_spawn, @intFromPtr(name.ptr), name.len, if (len == 0) 0 else @intFromPtr(&blob), len, exit_endpoint orelse abi.no_cap);
|
||||||
|
if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno
|
||||||
|
return @intCast(r);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Snapshot the process table into `out` (up to its length) and return the total
|
||||||
|
/// number of live processes — which may exceed `out.len`; call again with a larger
|
||||||
|
/// buffer for the full listing. Kernel tasks are included, with an empty name.
|
||||||
|
/// The primitive `ps` is built on.
|
||||||
|
pub fn processes(out: []abi.ProcessDescriptor) usize {
|
||||||
|
return sc.systemCall2(.process_enumerate, @intFromPtr(out.ptr), out.len);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether a process spawned under `name` (its argv[0]) is currently alive.
|
||||||
|
pub fn isProcessRunning(name: []const u8) bool {
|
||||||
|
var table: [32]ProcessDescriptor = undefined;
|
||||||
|
const total = processes(&table);
|
||||||
|
for (table[0..@min(total, table.len)]) |descriptor| {
|
||||||
|
if (std.mem.eql(u8, descriptor.name[0..descriptor.name_length], name)) return true;
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// End process `id`. Only its supervisor — the process that spawned it — may;
|
||||||
|
/// anyone else gets false, as does a stale or unknown id (ids are never reused).
|
||||||
|
/// Delivery is prompt but asynchronous, like a signal: a target caught running on
|
||||||
|
/// another core dies at its next system call or timer tick. True means the kill
|
||||||
|
/// is accepted and irrevocable; the exit notification (if an endpoint was given
|
||||||
|
/// at spawn) confirms completion.
|
||||||
|
pub fn kill(id: u32) bool {
|
||||||
|
return sc.systemCall1(.process_kill, id) == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable
|
||||||
|
/// memory and return the base virtual address. On failure returns a value in the
|
||||||
|
/// top page (see `mmapFailed`). The user heap grows through this call.
|
||||||
|
pub fn mmap(len: usize, prot: usize) usize {
|
||||||
|
return sc.systemCall2(.mmap, len, prot);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Release a range previously handed out by `mmap`.
|
||||||
|
pub fn munmap(base: usize, len: usize) usize {
|
||||||
|
return sc.systemCall2(.munmap, base, len);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether an `mmap` return value is an error (the kernel returns a wrapped
|
||||||
|
/// -errno, which lands in the top page — no real grant base is ever that high).
|
||||||
|
pub inline fn mmapFailed(ret: usize) bool {
|
||||||
|
return ret > ~@as(usize, 0) - 4095;
|
||||||
|
}
|
||||||
@@ -0,0 +1,169 @@
|
|||||||
|
//! The danos time interface — monotonic time, delays, and deadlines for user space.
|
||||||
|
//!
|
||||||
|
//! There is no time *service*: the kernel already owns the scheduling timer and
|
||||||
|
//! surfaces it directly, so reading the clock is one system call (an `rdtsc` and a
|
||||||
|
//! scale), never an IPC round trip (docs/timers.md explains why). This module is a
|
||||||
|
//! thin, generic layer over the `clock`/`sleep`/`timer_bind` wrappers in `system.zig`
|
||||||
|
//! — an ergonomic `Instant`/`Duration` front door, not new mechanism.
|
||||||
|
//!
|
||||||
|
//! It is **monotonic** time only: nanoseconds since boot, moving forward, no date or
|
||||||
|
//! timezone. Wall-clock/calendar time is a separate user-space service (an RTC-backed
|
||||||
|
//! CLOCK_REALTIME) layered on top later.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const system = @import("system.zig");
|
||||||
|
|
||||||
|
const nanos_per_micro: u64 = 1_000;
|
||||||
|
const nanos_per_milli: u64 = 1_000_000;
|
||||||
|
const nanos_per_second: u64 = 1_000_000_000;
|
||||||
|
|
||||||
|
/// A span of time, held as nanoseconds. Constructors name their unit; accessors
|
||||||
|
/// truncate toward zero. `ceilMillis` rounds *up*, since `sleep`/`after` land on the
|
||||||
|
/// kernel's millisecond granularity and rounding down could return early.
|
||||||
|
pub const Duration = struct {
|
||||||
|
ns: u64,
|
||||||
|
|
||||||
|
pub fn fromNanos(n: u64) Duration {
|
||||||
|
return .{ .ns = n };
|
||||||
|
}
|
||||||
|
pub fn fromMicros(n: u64) Duration {
|
||||||
|
return .{ .ns = n *| nanos_per_micro };
|
||||||
|
}
|
||||||
|
pub fn fromMillis(n: u64) Duration {
|
||||||
|
return .{ .ns = n *| nanos_per_milli };
|
||||||
|
}
|
||||||
|
pub fn fromSeconds(n: u64) Duration {
|
||||||
|
return .{ .ns = n *| nanos_per_second };
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn asNanos(d: Duration) u64 {
|
||||||
|
return d.ns;
|
||||||
|
}
|
||||||
|
pub fn asMicros(d: Duration) u64 {
|
||||||
|
return d.ns / nanos_per_micro;
|
||||||
|
}
|
||||||
|
pub fn asMillis(d: Duration) u64 {
|
||||||
|
return d.ns / nanos_per_milli;
|
||||||
|
}
|
||||||
|
pub fn asSeconds(d: Duration) u64 {
|
||||||
|
return d.ns / nanos_per_second;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whole milliseconds, rounded up — the argument `sleep`/`after` pass the kernel.
|
||||||
|
/// A non-zero sub-millisecond duration becomes 1 ms rather than 0.
|
||||||
|
pub fn ceilMillis(d: Duration) u64 {
|
||||||
|
return (d.ns +| (nanos_per_milli - 1)) / nanos_per_milli;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn plus(a: Duration, b: Duration) Duration {
|
||||||
|
return .{ .ns = a.ns +| b.ns };
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A point on the monotonic clock — nanoseconds since boot. Compare and subtract
|
||||||
|
/// instants to measure elapsed time; it never runs backward, so `since` is safe to
|
||||||
|
/// saturate at zero rather than wrap.
|
||||||
|
pub const Instant = struct {
|
||||||
|
ns: u64,
|
||||||
|
|
||||||
|
/// The span from `earlier` to `self`, saturating at zero if `earlier` is later
|
||||||
|
/// (which the monotonic clock should never produce, but callers may pass any pair).
|
||||||
|
pub fn since(self: Instant, earlier: Instant) Duration {
|
||||||
|
return .{ .ns = self.ns -| earlier.ns };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// How long since this instant, sampled now.
|
||||||
|
pub fn elapsed(self: Instant) Duration {
|
||||||
|
return now().since(self);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// This instant advanced by `d` (a deadline, `d` from here).
|
||||||
|
pub fn plus(self: Instant, d: Duration) Instant {
|
||||||
|
return .{ .ns = self.ns +| d.ns };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether the monotonic clock has reached this instant (used as a deadline).
|
||||||
|
pub fn reached(deadline: Instant) bool {
|
||||||
|
return now().ns >= deadline.ns;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The current monotonic time.
|
||||||
|
pub fn now() Instant {
|
||||||
|
return .{ .ns = system.clock() };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Monotonic nanoseconds since boot — the raw `clock()` reading, for callers that
|
||||||
|
/// want a plain integer instead of an `Instant`.
|
||||||
|
pub fn monotonicNanos() u64 {
|
||||||
|
return system.clock();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether the monotonic clock is usable. The kernel returns 0 until the TSC is
|
||||||
|
/// calibrated (`tsc_hz == 0`); a caller that needs real time can treat that as
|
||||||
|
/// "unavailable" instead of assuming the clock advances.
|
||||||
|
pub fn available() bool {
|
||||||
|
return system.clock() != 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Block the caller for at least `d`, rounded up to the kernel's millisecond
|
||||||
|
/// granularity. For sub-millisecond precision the scheduler cannot express, use
|
||||||
|
/// `spin`.
|
||||||
|
pub fn sleep(d: Duration) void {
|
||||||
|
system.sleep(d.ceilMillis());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Block the caller for `ms` milliseconds — the coarse, allocation-free form.
|
||||||
|
pub fn sleepMillis(ms: u64) void {
|
||||||
|
system.sleep(ms);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Busy-wait until `d` has elapsed, polling the monotonic clock. This burns the CPU
|
||||||
|
/// on purpose, to hit sub-millisecond delays the scheduler's millisecond tick cannot.
|
||||||
|
/// Prefer `sleep` for anything at or above a millisecond.
|
||||||
|
pub fn spin(d: Duration) void {
|
||||||
|
const deadline = now().plus(d);
|
||||||
|
while (!deadline.reached()) {}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Arm a one-shot timer against `endpoint` (a handle from `ipc.createIpcEndpoint`):
|
||||||
|
/// after `d` the kernel posts a timer notification (`ipc.Received.isTimer`) there.
|
||||||
|
/// Unlike `sleep`, this does not block — a service can keep serving IPC on the same
|
||||||
|
/// endpoint while the deadline is pending. Rounds `d` up to milliseconds; returns
|
||||||
|
/// false if the timer could not be armed. See `system.timerOnce`.
|
||||||
|
pub fn after(endpoint: usize, d: Duration) bool {
|
||||||
|
return system.timerOnce(endpoint, d.ceilMillis());
|
||||||
|
}
|
||||||
|
|
||||||
|
test "Duration unit conversions round toward zero" {
|
||||||
|
try std.testing.expectEqual(@as(u64, 1_000_000_000), Duration.fromSeconds(1).asNanos());
|
||||||
|
try std.testing.expectEqual(@as(u64, 1_500), Duration.fromNanos(1_500).asNanos());
|
||||||
|
try std.testing.expectEqual(@as(u64, 2), Duration.fromMillis(2).asMillis());
|
||||||
|
try std.testing.expectEqual(@as(u64, 1), Duration.fromNanos(1_999_999).asMillis());
|
||||||
|
try std.testing.expectEqual(@as(u64, 250), Duration.fromMicros(250).asMicros());
|
||||||
|
}
|
||||||
|
|
||||||
|
test "ceilMillis rounds up, and never turns a nonzero span into zero" {
|
||||||
|
try std.testing.expectEqual(@as(u64, 0), Duration.fromNanos(0).ceilMillis());
|
||||||
|
try std.testing.expectEqual(@as(u64, 1), Duration.fromNanos(1).ceilMillis());
|
||||||
|
try std.testing.expectEqual(@as(u64, 1), Duration.fromMillis(1).ceilMillis());
|
||||||
|
try std.testing.expectEqual(@as(u64, 2), Duration.fromNanos(nanos_per_milli + 1).ceilMillis());
|
||||||
|
try std.testing.expectEqual(@as(u64, 5), Duration.fromMillis(5).ceilMillis());
|
||||||
|
}
|
||||||
|
|
||||||
|
test "Instant arithmetic: since saturates, plus/reached form deadlines" {
|
||||||
|
const t0 = Instant{ .ns = 1_000 };
|
||||||
|
const t1 = Instant{ .ns = 4_000 };
|
||||||
|
try std.testing.expectEqual(@as(u64, 3_000), t1.since(t0).asNanos());
|
||||||
|
// earlier-than-self can't happen on a monotonic clock; saturate rather than wrap.
|
||||||
|
try std.testing.expectEqual(@as(u64, 0), t0.since(t1).asNanos());
|
||||||
|
const deadline = t0.plus(Duration.fromNanos(2_500));
|
||||||
|
try std.testing.expectEqual(@as(u64, 3_500), deadline.ns);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "saturating arithmetic does not overflow at the u64 ceiling" {
|
||||||
|
const big = Duration.fromSeconds(std.math.maxInt(u64));
|
||||||
|
try std.testing.expectEqual(@as(u64, std.math.maxInt(u64)), big.asNanos());
|
||||||
|
const late = Instant{ .ns = std.math.maxInt(u64) };
|
||||||
|
try std.testing.expectEqual(@as(u64, std.math.maxInt(u64)), late.plus(Duration.fromSeconds(10)).ns);
|
||||||
|
}
|
||||||
@@ -0,0 +1,162 @@
|
|||||||
|
//! USB class-driver client: the helper a keyboard, mouse, or mass-storage driver
|
||||||
|
//! uses to reach its device through the xHCI bus driver, so it never hand-rolls
|
||||||
|
//! the transfer-protocol IPC. Layered over `ipc` and the shared
|
||||||
|
//! `usb-transfer-protocol` wire format, the way `input.zig` layers over the input
|
||||||
|
//! service and `device.zig` over the raw device calls.
|
||||||
|
//!
|
||||||
|
//! A class driver, spawned with its interface's assigned device id as argv[1]:
|
||||||
|
//! if (!usb.helloManager(id)) return; // meet the spawn deadline
|
||||||
|
//! var device = usb.open(id) orelse return; // open + get its endpoints
|
||||||
|
//! _ = device.controlOut(usb_abi.setProtocol(...));// class requests, descriptors
|
||||||
|
//! _ = device.subscribeInterrupt(address, length); // reports arrive asynchronously
|
||||||
|
//! while (true) { ... ipc.replyWait(device.endpoint, ...) ... } // its own loop
|
||||||
|
//!
|
||||||
|
//! Reports are delivered to `device.endpoint` as asynchronous `InterruptReport`
|
||||||
|
//! messages (the class driver runs a bare `replyWait` loop to read them, because
|
||||||
|
//! the service harness drops buffered-message payloads — see service.zig).
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const ipc = @import("ipc.zig");
|
||||||
|
const system = @import("system.zig");
|
||||||
|
const protocol = @import("usb-transfer-protocol");
|
||||||
|
const device_manager = @import("device-manager-protocol");
|
||||||
|
|
||||||
|
pub const Endpoint = protocol.Endpoint;
|
||||||
|
pub const InterruptReport = protocol.InterruptReport;
|
||||||
|
pub const max_report_data = protocol.max_report_data;
|
||||||
|
|
||||||
|
// Endpoint transfer types (EndpointDescriptor attributes), for `findEndpoint`.
|
||||||
|
pub const transfer_type_bulk: u8 = 2;
|
||||||
|
pub const transfer_type_interrupt: u8 = 3;
|
||||||
|
|
||||||
|
/// An opened USB device: the bus endpoint to send requests to, this driver's own
|
||||||
|
/// endpoint that reports arrive on, the device token, and the interface's
|
||||||
|
/// endpoints (so a driver need not re-read the configuration descriptor).
|
||||||
|
pub const Device = struct {
|
||||||
|
bus: ipc.Handle,
|
||||||
|
endpoint: ipc.Handle,
|
||||||
|
token: u64,
|
||||||
|
class: u8,
|
||||||
|
subclass: u8,
|
||||||
|
protocol_code: u8,
|
||||||
|
interface_number: u8,
|
||||||
|
endpoint_count: usize = 0,
|
||||||
|
endpoints: [protocol.max_reported_endpoints]Endpoint = undefined,
|
||||||
|
|
||||||
|
/// The interface's first endpoint of the given transfer type and direction
|
||||||
|
/// (`transfer_type_bulk` / `transfer_type_interrupt`), or null.
|
||||||
|
pub fn findEndpoint(self: *const Device, transfer_type: u8, direction_in: bool) ?Endpoint {
|
||||||
|
for (self.endpoints[0..self.endpoint_count]) |endpoint| {
|
||||||
|
if (endpoint.transfer_type == transfer_type and (endpoint.address & 0x80 != 0) == direction_in) return endpoint;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn controlTransfer(self: *Device, setup: [8]u8, direction_in: bool, data: []u8) ?usize {
|
||||||
|
var request = protocol.ControlRequest{
|
||||||
|
.device_token = self.token,
|
||||||
|
.setup = setup,
|
||||||
|
.direction_in = @intFromBool(direction_in),
|
||||||
|
.data_length = @intCast(data.len),
|
||||||
|
};
|
||||||
|
if (!direction_in and data.len > 0) @memcpy(request.data[0..data.len], data);
|
||||||
|
var reply: [@sizeOf(protocol.ControlReply)]u8 = undefined;
|
||||||
|
const length = ipc.call(self.bus, std.mem.asBytes(&request), &reply) catch return null;
|
||||||
|
if (length < @sizeOf(protocol.ControlReply)) return null;
|
||||||
|
const control_reply = std.mem.bytesToValue(protocol.ControlReply, reply[0..@sizeOf(protocol.ControlReply)]);
|
||||||
|
if (control_reply.status != 0) return null;
|
||||||
|
const actual = @min(control_reply.actual_length, data.len);
|
||||||
|
if (direction_in and actual > 0) @memcpy(data[0..actual], control_reply.data[0..actual]);
|
||||||
|
return actual;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A control transfer with no data stage (SET_PROTOCOL, SET_IDLE, ...). The
|
||||||
|
/// `setup` is a bit-cast `usb_abi.Request`.
|
||||||
|
pub fn controlOut(self: *Device, setup: [8]u8) bool {
|
||||||
|
return self.controlTransfer(setup, false, &.{}) != null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A device-to-host control transfer, returning the bytes read into `out`.
|
||||||
|
pub fn controlIn(self: *Device, setup: [8]u8, out: []u8) ?usize {
|
||||||
|
return self.controlTransfer(setup, true, out);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Begin periodic IN polling of an interrupt endpoint; reports flow back to
|
||||||
|
/// `self.endpoint` as asynchronous `InterruptReport` messages.
|
||||||
|
pub fn subscribeInterrupt(self: *Device, endpoint_address: u8, max_length: u16) bool {
|
||||||
|
var request = protocol.InterruptSubscribeRequest{
|
||||||
|
.device_token = self.token,
|
||||||
|
.endpoint_address = endpoint_address,
|
||||||
|
.max_length = max_length,
|
||||||
|
};
|
||||||
|
var reply: [@sizeOf(protocol.InterruptSubscribeReply)]u8 = undefined;
|
||||||
|
const length = ipc.call(self.bus, std.mem.asBytes(&request), &reply) catch return false;
|
||||||
|
if (length < @sizeOf(protocol.InterruptSubscribeReply)) return false;
|
||||||
|
return std.mem.bytesToValue(protocol.InterruptSubscribeReply, reply[0..@sizeOf(protocol.InterruptSubscribeReply)]).status == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One bulk transfer (IN or OUT per `endpoint_address`'s direction bit) to or
|
||||||
|
/// from the caller's own DMA buffer at `physical`. Returns the bytes moved.
|
||||||
|
pub fn bulk(self: *Device, endpoint_address: u8, physical: u64, length: u32) ?u32 {
|
||||||
|
var request = protocol.BulkRequest{
|
||||||
|
.device_token = self.token,
|
||||||
|
.physical_address = physical,
|
||||||
|
.length = length,
|
||||||
|
.endpoint_address = endpoint_address,
|
||||||
|
};
|
||||||
|
var reply: [@sizeOf(protocol.BulkReply)]u8 = undefined;
|
||||||
|
const replied = ipc.call(self.bus, std.mem.asBytes(&request), &reply) catch return null;
|
||||||
|
if (replied < @sizeOf(protocol.BulkReply)) return null;
|
||||||
|
const bulk_reply = std.mem.bytesToValue(protocol.BulkReply, reply[0..@sizeOf(protocol.BulkReply)]);
|
||||||
|
if (bulk_reply.status != 0) return null;
|
||||||
|
return bulk_reply.actual_length;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Look up the USB bus and open the device with the assigned id, handing over a
|
||||||
|
/// freshly created endpoint for asynchronous interrupt reports. Retries while the
|
||||||
|
/// bus is still coming up (a class driver races the bus driver at boot).
|
||||||
|
pub fn open(device_id: u64) ?Device {
|
||||||
|
var attempts: usize = 0;
|
||||||
|
const bus = while (attempts < 100) : (attempts += 1) {
|
||||||
|
if (ipc.lookup(.usb_bus)) |handle| break handle;
|
||||||
|
system.sleep(20);
|
||||||
|
} else return null;
|
||||||
|
|
||||||
|
const endpoint = ipc.createIpcEndpoint() orelse return null;
|
||||||
|
var request = protocol.OpenRequest{ .device_id = device_id };
|
||||||
|
var reply: [@sizeOf(protocol.OpenReply)]u8 = undefined;
|
||||||
|
const result = ipc.callCap(bus, std.mem.asBytes(&request), &reply, endpoint) catch return null;
|
||||||
|
if (result.len < @sizeOf(protocol.OpenReply)) return null;
|
||||||
|
const open_reply = std.mem.bytesToValue(protocol.OpenReply, reply[0..@sizeOf(protocol.OpenReply)]);
|
||||||
|
if (open_reply.status != 0) return null;
|
||||||
|
|
||||||
|
var device = Device{
|
||||||
|
.bus = bus,
|
||||||
|
.endpoint = endpoint,
|
||||||
|
.token = open_reply.device_token,
|
||||||
|
.class = open_reply.interface_class,
|
||||||
|
.subclass = open_reply.interface_subclass,
|
||||||
|
.protocol_code = open_reply.interface_protocol,
|
||||||
|
.interface_number = open_reply.interface_number,
|
||||||
|
.endpoint_count = @min(open_reply.endpoint_count, protocol.max_reported_endpoints),
|
||||||
|
};
|
||||||
|
for (0..device.endpoint_count) |index| device.endpoints[index] = open_reply.endpoints[index];
|
||||||
|
return device;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Hello the device manager as a class driver (Role.device) so a supervised
|
||||||
|
/// spawn meets its hello deadline. Retries while the manager comes up.
|
||||||
|
pub fn helloManager(device_id: u64) bool {
|
||||||
|
var attempts: usize = 0;
|
||||||
|
const manager = while (attempts < 100) : (attempts += 1) {
|
||||||
|
if (ipc.lookup(.device_manager)) |handle| break handle;
|
||||||
|
system.sleep(20);
|
||||||
|
} else return false;
|
||||||
|
|
||||||
|
const hello = device_manager.Hello{ .role = @intFromEnum(device_manager.Role.device), .device_id = device_id };
|
||||||
|
var reply: [device_manager.message_maximum]u8 = undefined;
|
||||||
|
const length = ipc.call(manager, std.mem.asBytes(&hello), &reply) catch return false;
|
||||||
|
if (length < device_manager.reply_size) return false;
|
||||||
|
return std.mem.bytesToValue(device_manager.HelloReply, reply[0..device_manager.reply_size]).status == 0;
|
||||||
|
}
|
||||||
@@ -0,0 +1,52 @@
|
|||||||
|
/* Shared link layout for every user binary (init, servers, drivers).
|
||||||
|
*
|
||||||
|
* Linked at a fixed user-space virtual base (set by `image_base` in build.zig,
|
||||||
|
* inside the kernel's user region). Same discipline as the kernel's script:
|
||||||
|
* one PT_LOAD per permission set, every section page-aligned, so the kernel's
|
||||||
|
* user-ELF loader can map each segment with exact W^X permissions. Note the
|
||||||
|
* linker also emits a read-only PT_LOAD covering the ELF headers at the image
|
||||||
|
* base, so the entry point comes from e_entry, not the base address.
|
||||||
|
*/
|
||||||
|
|
||||||
|
ENTRY(_start)
|
||||||
|
|
||||||
|
/* FLAGS bits: 1=X, 2=W, 4=R. */
|
||||||
|
PHDRS {
|
||||||
|
text PT_LOAD FLAGS(5); /* R + X */
|
||||||
|
rodata PT_LOAD FLAGS(4); /* R */
|
||||||
|
data PT_LOAD FLAGS(6); /* R + W */
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTIONS {
|
||||||
|
/* The `.large` code model (needed for the >4 GiB image base) emits code and
|
||||||
|
* data into .ltext/.lrodata/.ldata/.lbss; fold those into the matching
|
||||||
|
* permission segment alongside the normal names. */
|
||||||
|
.text ALIGN(4K) : {
|
||||||
|
*(.text .text.*)
|
||||||
|
*(.ltext .ltext.*)
|
||||||
|
} :text
|
||||||
|
|
||||||
|
.rodata ALIGN(4K) : {
|
||||||
|
*(.rodata .rodata.*)
|
||||||
|
*(.lrodata .lrodata.*)
|
||||||
|
} :rodata
|
||||||
|
|
||||||
|
.data ALIGN(4K) : {
|
||||||
|
*(.data .data.*)
|
||||||
|
*(.ldata .ldata.*)
|
||||||
|
} :data
|
||||||
|
|
||||||
|
/* .bss occupies memory but not file space; the loader zeroes the
|
||||||
|
* filesz..memsz gap. */
|
||||||
|
.bss ALIGN(4K) : {
|
||||||
|
*(.bss .bss.*)
|
||||||
|
*(.lbss .lbss.*)
|
||||||
|
*(COMMON)
|
||||||
|
} :data
|
||||||
|
|
||||||
|
/DISCARD/ : {
|
||||||
|
*(.comment)
|
||||||
|
*(.note .note.*)
|
||||||
|
*(.eh_frame .eh_frame_hdr)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,65 @@
|
|||||||
|
# xkeyboard-config — X11 keyboard layouts, compiled to Zig
|
||||||
|
|
||||||
|
This module turns a physical key (a **USB HID usage**, as the [input module](../../docs/input.md)
|
||||||
|
delivers in `KeyEvent.keycode`) plus a modifier state into a **keysym** and, when the key
|
||||||
|
produces one, a **character** (a Unicode scalar). It is what lets a `keycode` become a
|
||||||
|
`character` — a keymap — without danos shipping an X11 runtime.
|
||||||
|
|
||||||
|
The layout data comes from the X11 [xkeyboard-config](https://gitlab.freedesktop.org/xkeyboard-config/xkeyboard-config)
|
||||||
|
database, but it is **compiled to native Zig at build time** rather than parsed at runtime.
|
||||||
|
`tools/make-xkeyboard-config.py` reads the vendored xkb data and emits pure-data tables into
|
||||||
|
`generated/layouts.zig`; `xkeyboard-config.zig` is the hand-written API over them. This is
|
||||||
|
the same build-time-codegen pattern as `tools/make-initial-ramdisk.py`.
|
||||||
|
|
||||||
|
## Using it
|
||||||
|
|
||||||
|
```zig
|
||||||
|
const xkb = @import("xkeyboard-config");
|
||||||
|
|
||||||
|
const m = xkb.map(xkb.us, key_event.keycode, .{ .shift = shift_held, .caps_lock = caps });
|
||||||
|
if (m.character) |ch| { /* a printable Unicode scalar */ }
|
||||||
|
// m.keysym is always set (e.g. an X11 keysym for Return / F1 / a dead key).
|
||||||
|
|
||||||
|
const layout = xkb.byName("gb") orelse xkb.us; // choose a layout by name
|
||||||
|
for (xkb.all) |l| { /* enumerate available layouts */ }
|
||||||
|
```
|
||||||
|
|
||||||
|
`Modifiers` carries `shift`, `caps_lock`, `level3` (AltGr), and `control`. `map` selects the
|
||||||
|
level from the key's XKB *type* (the generated data) and those modifiers (the policy, in
|
||||||
|
`xkeyboard-config.zig`), so data and semantics stay separable.
|
||||||
|
|
||||||
|
Layouts: **us, gb, de, fr, es, dvorak**.
|
||||||
|
|
||||||
|
## Regenerating
|
||||||
|
|
||||||
|
```sh
|
||||||
|
python3 tools/make-xkeyboard-config.py fetch # network: download + vendor the data subset
|
||||||
|
python3 tools/make-xkeyboard-config.py generate # offline: emit generated/layouts.zig
|
||||||
|
# or, from the build:
|
||||||
|
zig build gen-xkeyboard-config
|
||||||
|
```
|
||||||
|
|
||||||
|
- **`fetch`** downloads the pinned xkeyboard-config release (version + sha256 in the script),
|
||||||
|
resolves the `include` graph for the configured layouts, and vendors *only* the symbols
|
||||||
|
files actually reached (plus `keysymdef.h`, `COPYING`, and `PROVENANCE.md`) into `vendor/`.
|
||||||
|
Run it when bumping the version or adding a layout.
|
||||||
|
- **`generate`** is deterministic and offline — same vendored input produces byte-identical
|
||||||
|
output. To add a layout, extend `TARGETS` (and `HID_TO_NAME` if a new physical key is
|
||||||
|
involved), then re-run `fetch` (to vendor any new includes) and `generate`.
|
||||||
|
|
||||||
|
## Scope
|
||||||
|
|
||||||
|
A pragmatic subset, enough for real Latin-script typing:
|
||||||
|
|
||||||
|
- **Group 1 only** — no multi-layout group switching.
|
||||||
|
- **No dead-key / compose composition** — a dead key returns its keysym with no `character`
|
||||||
|
(composing `´` + `e` → `é` is a higher layer's job).
|
||||||
|
- **Curated key types** — the common XKB types (one/two-level, alphabetic, four-level, …);
|
||||||
|
unmapped keys and unknown types fall back to level-by-shift.
|
||||||
|
- **6 layouts** — extend via `TARGETS` as above.
|
||||||
|
|
||||||
|
## Licensing
|
||||||
|
|
||||||
|
xkeyboard-config and `keysymdef.h` (xorgproto) are MIT/X11 licensed. The vendored data
|
||||||
|
subset carries the upstream `vendor/COPYING`, and `vendor/PROVENANCE.md` records the exact
|
||||||
|
version, source URL, and sha256. The generated tables are a derived work under the same terms.
|
||||||
File diff suppressed because it is too large
Load Diff
+190
@@ -0,0 +1,190 @@
|
|||||||
|
Copyright 1996 by Joseph Moss
|
||||||
|
Copyright (C) 2002-2007 Free Software Foundation, Inc.
|
||||||
|
Copyright (C) Dmitry Golubev <lastguru@mail.ru>, 2003-2004
|
||||||
|
Copyright (C) 2004, Gregory Mokhin <mokhin@bog.msu.ru>
|
||||||
|
Copyright (C) 2006 Erdal Ronahî
|
||||||
|
|
||||||
|
Permission to use, copy, modify, distribute, and sell this software and its
|
||||||
|
documentation for any purpose is hereby granted without fee, provided that
|
||||||
|
the above copyright notice appear in all copies and that both that
|
||||||
|
copyright notice and this permission notice appear in supporting
|
||||||
|
documentation, and that the name of the copyright holder(s) not be used in
|
||||||
|
advertising or publicity pertaining to distribution of the software without
|
||||||
|
specific, written prior permission. The copyright holder(s) makes no
|
||||||
|
representations about the suitability of this software for any purpose. It
|
||||||
|
is provided "as is" without express or implied warranty.
|
||||||
|
|
||||||
|
THE COPYRIGHT HOLDER(S) DISCLAIMS ALL WARRANTIES WITH REGARD TO THIS SOFTWARE,
|
||||||
|
INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS, IN NO
|
||||||
|
EVENT SHALL THE COPYRIGHT HOLDER(S) BE LIABLE FOR ANY SPECIAL, INDIRECT OR
|
||||||
|
CONSEQUENTIAL DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE,
|
||||||
|
DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER
|
||||||
|
TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR
|
||||||
|
PERFORMANCE OF THIS SOFTWARE.
|
||||||
|
|
||||||
|
|
||||||
|
Copyright (c) 1996 Digital Equipment Corporation
|
||||||
|
|
||||||
|
Permission is hereby granted, free of charge, to any person obtaining
|
||||||
|
a copy of this software and associated documentation files (the
|
||||||
|
"Software"), to deal in the Software without restriction, including
|
||||||
|
without limitation the rights to use, copy, modify, merge, publish,
|
||||||
|
distribute, sublicense, and sell copies of the Software, and to
|
||||||
|
permit persons to whom the Software is furnished to do so, subject to
|
||||||
|
the following conditions:
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice shall be included
|
||||||
|
in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS
|
||||||
|
OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||||
|
IN NO EVENT SHALL DIGITAL EQUIPMENT CORPORATION BE LIABLE FOR ANY CLAIM,
|
||||||
|
DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR
|
||||||
|
OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR
|
||||||
|
THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||||
|
|
||||||
|
Except as contained in this notice, the name of the Digital Equipment
|
||||||
|
Corporation shall not be used in advertising or otherwise to promote
|
||||||
|
the sale, use or other dealings in this Software without prior written
|
||||||
|
authorization from Digital Equipment Corporation.
|
||||||
|
|
||||||
|
|
||||||
|
Copyright 1996, 1998 The Open Group
|
||||||
|
|
||||||
|
Permission to use, copy, modify, distribute, and sell this software and its
|
||||||
|
documentation for any purpose is hereby granted without fee, provided that
|
||||||
|
the above copyright notice appear in all copies and that both that
|
||||||
|
copyright notice and this permission notice appear in supporting
|
||||||
|
documentation.
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice shall be
|
||||||
|
included in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||||
|
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||||
|
IN NO EVENT SHALL THE OPEN GROUP BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||||
|
OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||||
|
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||||
|
OTHER DEALINGS IN THE SOFTWARE.
|
||||||
|
|
||||||
|
Except as contained in this notice, the name of The Open Group shall
|
||||||
|
not be used in advertising or otherwise to promote the sale, use or
|
||||||
|
other dealings in this Software without prior written authorization
|
||||||
|
from The Open Group.
|
||||||
|
|
||||||
|
|
||||||
|
Copyright 2004-2005 Sun Microsystems, Inc. All rights reserved.
|
||||||
|
|
||||||
|
Permission is hereby granted, free of charge, to any person obtaining a
|
||||||
|
copy of this software and associated documentation files (the "Software"),
|
||||||
|
to deal in the Software without restriction, including without limitation
|
||||||
|
the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||||
|
and/or sell copies of the Software, and to permit persons to whom the
|
||||||
|
Software is furnished to do so, subject to the following conditions:
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice (including the next
|
||||||
|
paragraph) shall be included in all copies or substantial portions of the
|
||||||
|
Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||||
|
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||||
|
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||||
|
THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||||
|
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
|
||||||
|
FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||||
|
DEALINGS IN THE SOFTWARE.
|
||||||
|
|
||||||
|
|
||||||
|
Copyright (c) 1996 by Silicon Graphics Computer Systems, Inc.
|
||||||
|
|
||||||
|
Permission to use, copy, modify, and distribute this
|
||||||
|
software and its documentation for any purpose and without
|
||||||
|
fee is hereby granted, provided that the above copyright
|
||||||
|
notice appear in all copies and that both that copyright
|
||||||
|
notice and this permission notice appear in supporting
|
||||||
|
documentation, and that the name of Silicon Graphics not be
|
||||||
|
used in advertising or publicity pertaining to distribution
|
||||||
|
of the software without specific prior written permission.
|
||||||
|
Silicon Graphics makes no representation about the suitability
|
||||||
|
of this software for any purpose. It is provided "as is"
|
||||||
|
without any express or implied warranty.
|
||||||
|
|
||||||
|
SILICON GRAPHICS DISCLAIMS ALL WARRANTIES WITH REGARD TO THIS
|
||||||
|
SOFTWARE, INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY
|
||||||
|
AND FITNESS FOR A PARTICULAR PURPOSE. IN NO EVENT SHALL SILICON
|
||||||
|
GRAPHICS BE LIABLE FOR ANY SPECIAL, INDIRECT OR CONSEQUENTIAL
|
||||||
|
DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE,
|
||||||
|
DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE
|
||||||
|
OR OTHER TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH
|
||||||
|
THE USE OR PERFORMANCE OF THIS SOFTWARE.
|
||||||
|
|
||||||
|
|
||||||
|
Copyright (c) 1996 X Consortium
|
||||||
|
|
||||||
|
Permission is hereby granted, free of charge, to any person obtaining
|
||||||
|
a copy of this software and associated documentation files (the
|
||||||
|
"Software"), to deal in the Software without restriction, including
|
||||||
|
without limitation the rights to use, copy, modify, merge, publish,
|
||||||
|
distribute, sublicense, and/or sell copies of the Software, and to
|
||||||
|
permit persons to whom the Software is furnished to do so, subject to
|
||||||
|
the following conditions:
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice shall be
|
||||||
|
included in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||||
|
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||||
|
IN NO EVENT SHALL THE X CONSORTIUM BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||||
|
OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||||
|
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||||
|
OTHER DEALINGS IN THE SOFTWARE.
|
||||||
|
|
||||||
|
Except as contained in this notice, the name of the X Consortium shall
|
||||||
|
not be used in advertising or otherwise to promote the sale, use or
|
||||||
|
other dealings in this Software without prior written authorization
|
||||||
|
from the X Consortium.
|
||||||
|
|
||||||
|
|
||||||
|
Copyright (C) 2004, 2006 Ævar Arnfjörð Bjarmason <avarab@gmail.com>
|
||||||
|
|
||||||
|
Permission to use, copy, modify, distribute, and sell this software and its
|
||||||
|
documentation for any purpose is hereby granted without fee, provided that
|
||||||
|
the above copyright notice appear in all copies and that both that
|
||||||
|
copyright notice and this permission notice appear in supporting
|
||||||
|
documentation.
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice shall be
|
||||||
|
included in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||||
|
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||||
|
IN NO EVENT SHALL THE OPEN GROUP BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||||
|
OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||||
|
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||||
|
OTHER DEALINGS IN THE SOFTWARE.
|
||||||
|
|
||||||
|
Except as contained in this notice, the name of a copyright holder shall
|
||||||
|
not be used in advertising or otherwise to promote the sale, use or
|
||||||
|
other dealings in this Software without prior written authorization of
|
||||||
|
the copyright holder.
|
||||||
|
|
||||||
|
|
||||||
|
Copyright (C) 1999, 2000 by Anton Zinoviev <anton@lml.bas.bg>
|
||||||
|
|
||||||
|
This software may be used, modified, copied, distributed, and sold,
|
||||||
|
in both source and binary form provided that the above copyright
|
||||||
|
and these terms are retained. Under no circumstances is the author
|
||||||
|
responsible for the proper functioning of this software, nor does
|
||||||
|
the author assume any responsibility for damages incurred with its
|
||||||
|
use.
|
||||||
|
|
||||||
|
Permission is granted to anyone to use, distribute and modify
|
||||||
|
this file in any way, provided that the above copyright notice
|
||||||
|
is left intact and the author of the modification summarizes
|
||||||
|
the changes in this header.
|
||||||
|
|
||||||
|
This file is distributed without any expressed or implied warranty.
|
||||||
+21
@@ -0,0 +1,21 @@
|
|||||||
|
# Vendored xkeyboard-config subset
|
||||||
|
|
||||||
|
- **Package**: xkeyboard-config 2.44
|
||||||
|
- **Source**: https://gitlab.freedesktop.org/xkeyboard-config/xkeyboard-config/-/archive/xkeyboard-config-2.44/xkeyboard-config-2.44.tar.gz
|
||||||
|
- **sha256**: `35e34edeaf4e8da8d0696ff6b241ee11ddb1b8c6730bac7252d4d0a88ea5f05b`
|
||||||
|
- **keysymdef.h**: xorgproto, copied from `/opt/homebrew/include/X11/keysymdef.h`
|
||||||
|
- **License**: MIT/X11 (see COPYING)
|
||||||
|
|
||||||
|
Only the symbols files reachable from the generated layouts (tools/make-xkeyboard-config.py `TARGETS`) are vendored; regenerate with
|
||||||
|
`python3 tools/make-xkeyboard-config.py fetch` then `... generate`.
|
||||||
|
|
||||||
|
Vendored symbols files:
|
||||||
|
|
||||||
|
- `symbols/de`
|
||||||
|
- `symbols/es`
|
||||||
|
- `symbols/fr`
|
||||||
|
- `symbols/gb`
|
||||||
|
- `symbols/kpdl`
|
||||||
|
- `symbols/latin`
|
||||||
|
- `symbols/level3`
|
||||||
|
- `symbols/us`
|
||||||
+2584
File diff suppressed because it is too large
Load Diff
+1232
File diff suppressed because it is too large
Load Diff
+250
@@ -0,0 +1,250 @@
|
|||||||
|
// Keyboard layouts for Spain.
|
||||||
|
|
||||||
|
// Modified for a real Spanish keyboard by Jon Tombs.
|
||||||
|
default partial alphanumeric_keys
|
||||||
|
xkb_symbols "basic" {
|
||||||
|
|
||||||
|
include "latin(type4)"
|
||||||
|
|
||||||
|
name[Group1]="Spanish";
|
||||||
|
|
||||||
|
key <TLDE> { [ masculine, ordfeminine, backslash, backslash ] };
|
||||||
|
key <AE01> { [ 1, exclam, bar, exclamdown ] };
|
||||||
|
key <AE03> { [ 3, periodcentered, numbersign, sterling ] };
|
||||||
|
key <AE04> { [ 4, dollar, asciitilde, dollar ] };
|
||||||
|
key <AE11> { [apostrophe, question, backslash, questiondown ] };
|
||||||
|
key <AE12> { [exclamdown, questiondown, dead_cedilla, dead_ogonek] };
|
||||||
|
|
||||||
|
key <AD11> { [dead_grave, dead_circumflex, bracketleft, dead_abovering ] };
|
||||||
|
key <AD12> { [ plus, asterisk, bracketright, dead_macron ] };
|
||||||
|
|
||||||
|
key <AC10> { [ ntilde, Ntilde, dead_tilde, dead_doubleacute ] };
|
||||||
|
key <AC11> { [dead_acute, dead_diaeresis, braceleft, dead_caron ] };
|
||||||
|
key <BKSL> { [ ccedilla, Ccedilla, braceright, dead_breve ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "winkeys" {
|
||||||
|
|
||||||
|
include "es(basic)"
|
||||||
|
name[Group1]="Spanish (Windows)";
|
||||||
|
include "eurosign(5)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "nodeadkeys" {
|
||||||
|
|
||||||
|
include "es(basic)"
|
||||||
|
|
||||||
|
name[Group1]="Spanish (no dead keys)";
|
||||||
|
|
||||||
|
key <AE12> { [exclamdown, questiondown, cedilla, ogonek ] };
|
||||||
|
key <AD11> { [ grave, asciicircum, bracketleft, degree ] };
|
||||||
|
key <AD12> { [ plus, asterisk, bracketright, macron ] };
|
||||||
|
key <AC07> { [ j, J, ezh, EZH ] };
|
||||||
|
key <AC10> { [ ntilde, Ntilde, asciitilde, doubleacute ] };
|
||||||
|
key <AC11> { [ acute, diaeresis, braceleft, caron ] };
|
||||||
|
key <BKSL> { [ ccedilla, Ccedilla, braceright, breve ] };
|
||||||
|
key <AB10> { [ minus, underscore, ellipsis, abovedot ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Spanish Dvorak mapping (note R-H exchange)
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "dvorak" {
|
||||||
|
|
||||||
|
name[Group1]="Spanish (Dvorak)";
|
||||||
|
|
||||||
|
key <TLDE> {[ masculine, ordfeminine, backslash, degree ]};
|
||||||
|
key <AE01> {[ 1, exclam, bar, onesuperior ]};
|
||||||
|
key <AE02> {[ 2, quotedbl, at, twosuperior ]};
|
||||||
|
key <AE03> {[ 3, periodcentered, numbersign, threesuperior ]};
|
||||||
|
key <AE04> {[ 4, dollar, asciitilde, onequarter ]};
|
||||||
|
key <AE05> {[ 5, percent, brokenbar, fiveeighths ]};
|
||||||
|
key <AE06> {[ 6, ampersand, notsign, threequarters ]};
|
||||||
|
key <AE07> {[ 7, slash, onehalf, seveneighths ]};
|
||||||
|
key <AE08> {[ 8, parenleft, oneeighth, threeeighths ]};
|
||||||
|
key <AE09> {[ 9, parenright, asciicircum ]};
|
||||||
|
key <AE10> {[ 0, equal, grave, dead_doubleacute ]};
|
||||||
|
key <AE11> {[ apostrophe, question, dead_macron, dead_ogonek ]};
|
||||||
|
key <AE12> {[ exclamdown, questiondown, dead_breve, dead_abovedot ]};
|
||||||
|
|
||||||
|
key <AD01> {[ period, colon, less, guillemotleft ]};
|
||||||
|
key <AD02> {[ comma, semicolon, greater, guillemotright ]};
|
||||||
|
key <AD03> {[ ntilde, Ntilde, lstroke, Lstroke ]};
|
||||||
|
key <AD04> {[ p, P, paragraph ]};
|
||||||
|
key <AD05> {[ y, Y, yen ]};
|
||||||
|
key <AD06> {[ f, F, tslash, Tslash ]};
|
||||||
|
key <AD07> {[ g, G, dstroke, Dstroke ]};
|
||||||
|
key <AD08> {[ c, C, cent, copyright ]};
|
||||||
|
key <AD09> {[ h, H, hstroke, Hstroke ]};
|
||||||
|
key <AD10> {[ l, L, sterling ]};
|
||||||
|
key <AD11> {[ dead_grave, dead_circumflex, bracketleft, dead_caron ]};
|
||||||
|
key <AD12> {[ plus, asterisk, bracketright, plusminus ]};
|
||||||
|
|
||||||
|
key <AC01> {[ a, A, ae, AE ]};
|
||||||
|
key <AC02> {[ o, O, oslash, Oslash ]};
|
||||||
|
key <AC03> {[ e, E, EuroSign ]};
|
||||||
|
key <AC04> {[ u, U, aring, Aring ]};
|
||||||
|
key <AC05> {[ i, I, oe, OE ]};
|
||||||
|
key <AC06> {[ d, D, eth, ETH ]};
|
||||||
|
key <AC07> {[ r, R, registered, trademark ]};
|
||||||
|
key <AC08> {[ t, T, thorn, THORN ]};
|
||||||
|
key <AC09> {[ n, N, eng, ENG ]};
|
||||||
|
key <AC10> {[ s, S, ssharp, section ]};
|
||||||
|
key <AC11> {[ dead_acute, dead_diaeresis, braceleft, dead_tilde ]};
|
||||||
|
key <BKSL> {[ ccedilla, Ccedilla, braceright, dead_cedilla ]};
|
||||||
|
|
||||||
|
key <LSGT> {[ less, greater, guillemotleft, guillemotright ]};
|
||||||
|
key <AB01> {[ minus, underscore, hyphen, macron ]};
|
||||||
|
key <AB02> {[ q, Q, currency ]};
|
||||||
|
key <AB03> {[ j, J ]};
|
||||||
|
key <AB04> {[ k, K, kra ]};
|
||||||
|
key <AB05> {[ x, X, multiply, division ]};
|
||||||
|
key <AB06> {[ b, B ]};
|
||||||
|
key <AB07> {[ m, M, mu ]};
|
||||||
|
key <AB08> {[ w, W ]};
|
||||||
|
key <AB09> {[ v, V ]};
|
||||||
|
key <AB10> {[ z, Z ]};
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "cat" {
|
||||||
|
|
||||||
|
include "es(basic)"
|
||||||
|
|
||||||
|
name[Group1]="Catalan (Spain, with middle-dot L)";
|
||||||
|
|
||||||
|
key <AC09> { [ l, L, 0x1000140, 0x100013F ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "ast" {
|
||||||
|
|
||||||
|
include "es(basic)"
|
||||||
|
|
||||||
|
name[Group1]="Asturian (Spain, with bottom-dot H and L)";
|
||||||
|
|
||||||
|
key <AC06> { [ h, H, 0x1001E25, 0x1001E24 ] };
|
||||||
|
key <AC09> { [ l, L, 0x1001E37, 0x1001E36 ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "olpc" {
|
||||||
|
|
||||||
|
// #HW-SPECIFIC
|
||||||
|
|
||||||
|
// http://wiki.laptop.org/go/OLPC_Spanish_Keyboard
|
||||||
|
|
||||||
|
include "us(basic)"
|
||||||
|
name[Group1]="Spanish";
|
||||||
|
|
||||||
|
key <AE00> { [ masculine, ordfeminine ] };
|
||||||
|
key <AE01> { [ 1, exclam, bar ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, at ] };
|
||||||
|
key <AE03> { [ 3, dead_grave, numbersign, grave ] };
|
||||||
|
key <AE05> { [ 5, percent, asciicircum, dead_circumflex ] };
|
||||||
|
key <AE06> { [ 6, ampersand, notsign ] };
|
||||||
|
key <AE07> { [ 7, slash, backslash ] };
|
||||||
|
key <AE08> { [ 8, parenleft ] };
|
||||||
|
key <AE09> { [ 9, parenright ] };
|
||||||
|
key <AE10> { [ 0, equal ] };
|
||||||
|
key <AE11> { [ apostrophe, question ] };
|
||||||
|
key <AE12> { [ exclamdown, questiondown ] };
|
||||||
|
|
||||||
|
key <AD03> { [ e, E, EuroSign ] };
|
||||||
|
key <AD11> { [ dead_acute, dead_diaeresis, acute, dead_abovering ] };
|
||||||
|
key <AD12> { [ bracketleft, braceleft ] };
|
||||||
|
|
||||||
|
key <AC10> { [ ntilde, Ntilde ] };
|
||||||
|
key <AC11> { [ plus, asterisk, dead_tilde ] };
|
||||||
|
key <AC12> { [ bracketright, braceright, section ] };
|
||||||
|
|
||||||
|
key <AB08> { [ comma, semicolon ] };
|
||||||
|
key <AB09> { [ period, colon ] };
|
||||||
|
key <AB10> { [ minus, underscore ] };
|
||||||
|
|
||||||
|
key <I219> { [ less, greater, ISO_Next_Group ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "olpcm" {
|
||||||
|
|
||||||
|
// #HW-SPECIFIC
|
||||||
|
|
||||||
|
// Mechanical (non-membrane) OLPC Spanish keyboard layout.
|
||||||
|
// See: http://wiki.laptop.org/go/OLPC_Spanish_Non-membrane_Keyboard
|
||||||
|
|
||||||
|
include "us(basic)"
|
||||||
|
name[Group1]="Spanish";
|
||||||
|
|
||||||
|
key <AE00> { [ questiondown, exclamdown, backslash ] };
|
||||||
|
key <AE01> { [ 1, exclam, bar ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, at ] };
|
||||||
|
key <AE03> { [ 3, dead_grave, numbersign, grave ] };
|
||||||
|
key <AE04> { [ 4, dollar, asciitilde, dead_tilde ] };
|
||||||
|
key <AE05> { [ 5, percent, asciicircum, dead_circumflex ] };
|
||||||
|
key <AE06> { [ 6, ampersand, notsign ] };
|
||||||
|
key <AE07> { [ 7, slash, backslash ] }; // no '\' label on olpcm, leave for compatibility
|
||||||
|
key <AE08> { [ 8, parenleft, masculine ] };
|
||||||
|
key <AE09> { [ 9, parenright, ordfeminine ] };
|
||||||
|
key <AE10> { [ 0, equal ] };
|
||||||
|
key <AE11> { [ apostrophe, question ] };
|
||||||
|
|
||||||
|
key <AD03> { [ e, E, EuroSign ] };
|
||||||
|
key <AD11> { [ dead_acute, dead_diaeresis, dead_abovering, acute ] };
|
||||||
|
key <AD12> { [ plus, asterisk ] };
|
||||||
|
|
||||||
|
key <AC10> { [ ntilde, Ntilde ] };
|
||||||
|
// no AC11 or AC12 on olpcm
|
||||||
|
|
||||||
|
key <AB08> { [ comma, semicolon ] };
|
||||||
|
key <AB09> { [ period, colon ] };
|
||||||
|
key <AB10> { [ minus, underscore ] };
|
||||||
|
|
||||||
|
key <AA02> { [ less, greater ] };
|
||||||
|
key <AA06> { [ bracketleft, braceleft, ccedilla, Ccedilla ] };
|
||||||
|
key <AA07> { [ bracketright, braceright ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "deadtilde" {
|
||||||
|
|
||||||
|
include "es(basic)"
|
||||||
|
|
||||||
|
name[Group1]="Spanish (dead tilde)";
|
||||||
|
|
||||||
|
key <AE04> { [ 4, dollar, dead_tilde, dollar ] };
|
||||||
|
key <AC10> { [ ntilde, Ntilde, asciitilde, dead_doubleacute ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "olpc2" {
|
||||||
|
// #HW-SPECIFIC
|
||||||
|
|
||||||
|
// Modified variant of US International layout, specifically for Peru
|
||||||
|
// Contact: Sayamindu Dasgupta <sayamindu@laptop.org>
|
||||||
|
|
||||||
|
include "us(olpc)"
|
||||||
|
name[Group1]="Spanish";
|
||||||
|
|
||||||
|
key <AE03> { [ 3, numbersign, dead_grave, dead_grave] }; // combining grave
|
||||||
|
key <I236> { [ XF86Start ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
// EXTRAS:
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "sun_type6" {
|
||||||
|
include "sun_vndr/es(sun_type6)"
|
||||||
|
};
|
||||||
+1404
File diff suppressed because it is too large
Load Diff
+249
@@ -0,0 +1,249 @@
|
|||||||
|
// Keyboard layouts for Great Britain.
|
||||||
|
|
||||||
|
default partial alphanumeric_keys
|
||||||
|
xkb_symbols "basic" {
|
||||||
|
|
||||||
|
// The basic UK layout, also known as the IBM 166 layout,
|
||||||
|
// but with the useless brokenbar pushed two levels up.
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
name[Group1]="English (UK)";
|
||||||
|
|
||||||
|
key <TLDE> { [ grave, notsign, bar, bar ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, twosuperior, oneeighth ] };
|
||||||
|
key <AE03> { [ 3, sterling, threesuperior, sterling ] };
|
||||||
|
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||||
|
|
||||||
|
key <AC11> { [apostrophe, at, dead_circumflex, dead_caron] };
|
||||||
|
key <BKSL> { [numbersign, asciitilde, dead_grave, dead_breve ] };
|
||||||
|
|
||||||
|
key <LSGT> { [ backslash, bar, bar, brokenbar ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "intl" {
|
||||||
|
|
||||||
|
// A UK layout but with five accents made into dead keys:
|
||||||
|
// grave, diaeresis, circumflex, acute, and tilde.
|
||||||
|
// By Phil Jones <philjones1 at blueyonder.co.uk>.
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
name[Group1]="English (UK, intl., with dead keys)";
|
||||||
|
|
||||||
|
key <TLDE> { [ dead_grave, notsign, bar, bar ] };
|
||||||
|
key <AE02> { [ 2, dead_diaeresis, twosuperior, onehalf ] };
|
||||||
|
key <AE03> { [ 3, sterling, threesuperior, onethird ] };
|
||||||
|
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||||
|
key <AE06> { [ 6, dead_circumflex, threequarters, onesixth ] };
|
||||||
|
|
||||||
|
key <AC11> { [ dead_acute, at, apostrophe, bar ] };
|
||||||
|
key <BKSL> { [ numbersign, dead_tilde, bar, bar ] };
|
||||||
|
|
||||||
|
key <LSGT> { [ backslash, bar, bar, bar ] };
|
||||||
|
key <AB08> { [ comma, less, ccedilla, Ccedilla ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "extd" {
|
||||||
|
// Clone of the Microsoft "United Kingdom Extended" layout, which
|
||||||
|
// includes dead keys for: grave; diaeresis; circumflex; tilde; and
|
||||||
|
// accute. It also enables direct access to accute characters using
|
||||||
|
// the Multi_key (Alt Gr).
|
||||||
|
//
|
||||||
|
// Taken from...
|
||||||
|
// "Windows Keyboard Layouts"
|
||||||
|
// https://docs.microsoft.com/en-gb/globalization/windows-keyboard-layouts#U
|
||||||
|
//
|
||||||
|
// -- Jonathan Miles <jon@cybah.co.uk>
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
name[Group1]="English (UK, extended, Windows)";
|
||||||
|
|
||||||
|
key <TLDE> { [ dead_grave, notsign, brokenbar, NoSymbol ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, dead_diaeresis, onehalf ] };
|
||||||
|
key <AE03> { [ 3, sterling, threesuperior, onethird ] };
|
||||||
|
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||||
|
key <AE06> { [ 6, asciicircum, dead_circumflex, NoSymbol ] };
|
||||||
|
|
||||||
|
key <AD02> { [ w, W, wacute, Wacute ] };
|
||||||
|
key <AD03> { [ e, E, eacute, Eacute ] };
|
||||||
|
key <AD06> { [ y, Y, yacute, Yacute ] };
|
||||||
|
key <AD07> { [ u, U, uacute, Uacute ] };
|
||||||
|
key <AD08> { [ i, I, iacute, Iacute ] };
|
||||||
|
key <AD09> { [ o, O, oacute, Oacute ] };
|
||||||
|
key <AD12> { [ bracketright, braceright, NoSymbol, bar ] };
|
||||||
|
|
||||||
|
key <AC01> { [ a, A, aacute, Aacute ] };
|
||||||
|
key <AC11> { [ apostrophe, at, dead_acute, grave ] };
|
||||||
|
key <BKSL> { [ numbersign, asciitilde, dead_tilde, backslash ] };
|
||||||
|
|
||||||
|
key <LSGT> { [ backslash, bar, NoSymbol, NoSymbol ] };
|
||||||
|
key <AB03> { [ c, C, ccedilla, Ccedilla ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
// Describe the differences between the US Colemak layout
|
||||||
|
// and a UK variant. By Andy Buckley (andy@insectnation.org)
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "colemak" {
|
||||||
|
include "us(colemak)"
|
||||||
|
|
||||||
|
name[Group1]="English (UK, Colemak)";
|
||||||
|
|
||||||
|
key <TLDE> { [ grave, notsign, bar, asciitilde ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, twosuperior, oneeighth ] };
|
||||||
|
key <AE03> { [ 3, sterling, threesuperior, sterling ] };
|
||||||
|
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||||
|
|
||||||
|
key <AC11> { [apostrophe, at, dead_circumflex, dead_caron] };
|
||||||
|
key <BKSL> { [numbersign, asciitilde, dead_grave, dead_breve ] };
|
||||||
|
|
||||||
|
key <LSGT> { [ backslash, bar, asciitilde, brokenbar ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Colemak-DH (ISO) layout, UK Variant, https://colemakmods.github.io/mod-dh/
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "colemak_dh" {
|
||||||
|
include "us(colemak_dh)"
|
||||||
|
|
||||||
|
name[Group1]="English (UK, Colemak-DH)";
|
||||||
|
|
||||||
|
key <TLDE> { [ grave, notsign, bar, asciitilde ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, twosuperior, oneeighth ] };
|
||||||
|
key <AE03> { [ 3, sterling, threesuperior, sterling ] };
|
||||||
|
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||||
|
|
||||||
|
key <AC11> { [apostrophe, at, dead_circumflex, dead_caron] };
|
||||||
|
key <BKSL> { [numbersign, asciitilde, dead_grave, dead_breve ] };
|
||||||
|
|
||||||
|
key <AB05> { [ backslash, bar, asciitilde, brokenbar ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
// Dvorak (UK) keymap (by odaen) allowing the usage of
|
||||||
|
// the £ and ? key and swapping the @ and " keys.
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "dvorak" {
|
||||||
|
include "us(dvorak-alt-intl)"
|
||||||
|
|
||||||
|
name[Group1]="English (UK, Dvorak)";
|
||||||
|
|
||||||
|
key <TLDE> { [ grave, notsign, bar, bar ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, twosuperior, NoSymbol ] };
|
||||||
|
key <AE03> { [ 3, sterling, threesuperior, NoSymbol ] };
|
||||||
|
key <AD01> { [ apostrophe, at ] };
|
||||||
|
key <BKSL> { [ numbersign, asciitilde ] };
|
||||||
|
key <LSGT> { [ backslash, bar ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Dvorak letter positions, but punctuation all in the normal UK positions.
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "dvorakukp" {
|
||||||
|
include "gb(dvorak)"
|
||||||
|
|
||||||
|
name[Group1]="English (UK, Dvorak, with UK punctuation)";
|
||||||
|
|
||||||
|
key <AE11> { [ minus, underscore ] };
|
||||||
|
key <AE12> { [ equal, plus ] };
|
||||||
|
key <AD11> { [ bracketleft, braceleft ] };
|
||||||
|
key <AD12> { [ bracketright, braceright ] };
|
||||||
|
key <AD01> { [ slash, question ] };
|
||||||
|
key <AC11> { [apostrophe, at, dead_circumflex, dead_caron] };
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "mac" {
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
name[Group1]= "English (UK, Macintosh)";
|
||||||
|
|
||||||
|
key <TLDE> { [ section, plusminus ] };
|
||||||
|
key <AE02> { [ 2, at, EuroSign ] };
|
||||||
|
key <AE03> { [ 3, sterling, numbersign ] };
|
||||||
|
key <LSGT> { [ grave, asciitilde ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
include "level3(enter_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "mac_intl" {
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
name[Group1]="English (UK, Macintosh, intl.)";
|
||||||
|
|
||||||
|
key <TLDE> { [ section, plusminus, notsign, notsign ] }; //dead_grave
|
||||||
|
key <AE02> { [ 2, at, EuroSign, onehalf ] };
|
||||||
|
key <AE03> { [ 3, sterling, twosuperior, onethird ] };
|
||||||
|
key <AE04> { [ 4, dollar, threesuperior, onequarter ] };
|
||||||
|
key <AE06> { [ 6, dead_circumflex, NoSymbol, onesixth ] };
|
||||||
|
key <AD09> { [ o, O, oe, OE ] };
|
||||||
|
|
||||||
|
key <AC11> { [ dead_acute, dead_diaeresis, dead_diaeresis, bar ] }; //dead_doubleacute
|
||||||
|
key <BKSL> { [ backslash, bar, numbersign, bar ] };
|
||||||
|
|
||||||
|
key <LSGT> { [ dead_grave, dead_tilde, brokenbar, bar ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "pl" {
|
||||||
|
|
||||||
|
// Polish accented letters on upper levels of corresponding base letters.
|
||||||
|
// Idea from Wawrzyniec Niewodniczański, adapted by Aleksander Kowalski.
|
||||||
|
|
||||||
|
include "gb(basic)"
|
||||||
|
|
||||||
|
name[Group1]="Polish (British keyboard)";
|
||||||
|
|
||||||
|
key <AD03> { [ e, E, eogonek, Eogonek ] };
|
||||||
|
key <AD09> { [ o, O, oacute, Oacute ] };
|
||||||
|
|
||||||
|
key <AC01> { [ a, A, aogonek, Aogonek ] };
|
||||||
|
key <AC02> { [ s, S, sacute, Sacute ] };
|
||||||
|
|
||||||
|
key <AB01> { [ z, Z, zabovedot, Zabovedot ] };
|
||||||
|
key <AB02> { [ x, X, zacute, Zacute ] };
|
||||||
|
key <AB03> { [ c, C, cacute, Cacute ] };
|
||||||
|
key <AB06> { [ n, N, nacute, Nacute ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "gla" {
|
||||||
|
|
||||||
|
// Grave-accented letters on the upper levels of the relevant vowels.
|
||||||
|
|
||||||
|
include "gb(basic)"
|
||||||
|
|
||||||
|
name[Group1]="Scottish Gaelic";
|
||||||
|
|
||||||
|
key <AD03> { [ e, E, egrave, Egrave ] };
|
||||||
|
key <AD07> { [ u, U, ugrave, Ugrave ] };
|
||||||
|
key <AD08> { [ i, I, igrave, Igrave ] };
|
||||||
|
key <AD09> { [ o, O, ograve, Ograve ] };
|
||||||
|
|
||||||
|
key <AC01> { [ a, A, agrave, Agrave ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// EXTRAS:
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "sun_type6" {
|
||||||
|
include "sun_vndr/gb(sun_type6)"
|
||||||
|
};
|
||||||
+102
@@ -0,0 +1,102 @@
|
|||||||
|
// The <KPDL> key is a mess.
|
||||||
|
// It was probably originally meant to be a decimal separator.
|
||||||
|
// Except since it was declared by USA people it didn't use the original
|
||||||
|
// SI separator "," but a "." (since then the USA managed to f-up the SI
|
||||||
|
// by making "." an accepted alternative, but standards still use "," as
|
||||||
|
// default)
|
||||||
|
// As a result users of SI-abiding countries expect either a "." or a ","
|
||||||
|
// or a "decimal_separator" which may or may not be translated in one of the
|
||||||
|
// above depending on applications.
|
||||||
|
// It's not possible to define a default per-country since user expectations
|
||||||
|
// depend on the conflicting choices of their most-used applications,
|
||||||
|
// operating system, etc. Therefore it needs to be a configuration setting
|
||||||
|
// Copyright © 2007 Nicolas Mailhot <nicolas.mailhot @ laposte.net>
|
||||||
|
|
||||||
|
|
||||||
|
// Legacy <KPDL> #1
|
||||||
|
// This assumes KP_Decimal will be translated in a dot
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "dot" {
|
||||||
|
|
||||||
|
key.type[Group1]="KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ KP_Delete, KP_Decimal ] }; // <delete> <separator>
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
// Legacy <KPDL> #2
|
||||||
|
// This assumes KP_Separator will be translated in a comma
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "comma" {
|
||||||
|
|
||||||
|
key.type[Group1]="KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ KP_Delete, KP_Separator ] }; // <delete> <separator>
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
// Period <KPDL>, usual keyboard serigraphy in most countries
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "dotoss" {
|
||||||
|
|
||||||
|
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ KP_Delete, period, comma, 0x100202F ] }; // <delete> . , ⍽ (narrow no-break space)
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
// Period <KPDL>, usual keyboard serigraphy in most countries, latin-9 restriction
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "dotoss_latin9" {
|
||||||
|
|
||||||
|
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ KP_Delete, period, comma, nobreakspace ] }; // <delete> . , ⍽ (no-break space)
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
// Comma <KPDL>, what most non anglo-saxon people consider the real separator
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "commaoss" {
|
||||||
|
|
||||||
|
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ KP_Delete, comma, period, 0x100202F ] }; // <delete> , . ⍽ (narrow no-break space)
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
// Momayyez <KPDL>: Bahrain, Iran, Iraq, Kuwait, Oman, Qatar, Saudi Arabia, Syria, UAE
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "momayyezoss" {
|
||||||
|
|
||||||
|
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ KP_Delete, 0x100066B, comma, 0x100202F ] }; // <delete> ? , ⍽ (narrow no-break space)
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
// Abstracted <KPDL>, pray everything will work out (it usually does not)
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "kposs" {
|
||||||
|
|
||||||
|
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ KP_Delete, KP_Decimal, KP_Separator, 0x100202F ] }; // <delete> ? ? ⍽ (narrow no-break space)
|
||||||
|
};
|
||||||
|
|
||||||
|
// Spreadsheets may be configured to use the dot as decimal
|
||||||
|
// punctuation, comma as a thousands separator and then semi-colon as
|
||||||
|
// the list separator. Of these, dot and semi-colon is most important
|
||||||
|
// when entering data by the keyboard; the comma can then be inferred
|
||||||
|
// and added to the presentation afterwards. Using semi-colon as a
|
||||||
|
// general separator may in fact be preferred to avoid ambiguities
|
||||||
|
// in data files. Most times a decimal separator is hard-coded, it
|
||||||
|
// seems to be period, probably since this is the syntax used in
|
||||||
|
// (most) programming languages.
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "semi" {
|
||||||
|
|
||||||
|
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ NoSymbol, NoSymbol, semicolon ] };
|
||||||
|
};
|
||||||
+255
@@ -0,0 +1,255 @@
|
|||||||
|
// Common Latin alphabet layout
|
||||||
|
|
||||||
|
default partial
|
||||||
|
xkb_symbols "basic" {
|
||||||
|
|
||||||
|
key <AE01> { [ 1, exclam, onesuperior, exclamdown ] };
|
||||||
|
key <AE02> { [ 2, at, twosuperior, oneeighth ] };
|
||||||
|
key <AE03> { [ 3, numbersign, threesuperior, sterling ] };
|
||||||
|
key <AE04> { [ 4, dollar, onequarter, dollar ] };
|
||||||
|
key <AE05> { [ 5, percent, onehalf, threeeighths ] };
|
||||||
|
key <AE06> { [ 6, asciicircum, threequarters, fiveeighths ] };
|
||||||
|
key <AE07> { [ 7, ampersand, braceleft, seveneighths ] };
|
||||||
|
key <AE08> { [ 8, asterisk, bracketleft, trademark ] };
|
||||||
|
key <AE09> { [ 9, parenleft, bracketright, plusminus ] };
|
||||||
|
key <AE10> { [ 0, parenright, braceright, degree ] };
|
||||||
|
key <AE11> { [ minus, underscore, backslash, questiondown ] };
|
||||||
|
key <AE12> { [ equal, plus, dead_cedilla, dead_ogonek ] };
|
||||||
|
|
||||||
|
key <AD01> { [ q, Q, at, Greek_OMEGA ] };
|
||||||
|
key <AD02> { [ w, W, U017F, section ] };
|
||||||
|
key <AD03> { [ e, E, e, E ] };
|
||||||
|
key <AD04> { [ r, R, paragraph, registered ] };
|
||||||
|
key <AD05> { [ t, T, tslash, Tslash ] };
|
||||||
|
key <AD06> { [ y, Y, leftarrow, yen ] };
|
||||||
|
key <AD07> { [ u, U, downarrow, uparrow ] };
|
||||||
|
key <AD08> { [ i, I, rightarrow, idotless ] };
|
||||||
|
key <AD09> { [ o, O, oslash, Oslash ] };
|
||||||
|
key <AD10> { [ p, P, thorn, THORN ] };
|
||||||
|
key <AD11> { [bracketleft, braceleft, dead_diaeresis, dead_abovering ] };
|
||||||
|
key <AD12> { [bracketright, braceright, dead_tilde, dead_macron ] };
|
||||||
|
|
||||||
|
key <AC01> { [ a, A, ae, AE ] };
|
||||||
|
key <AC02> { [ s, S, ssharp, U1E9E ] };
|
||||||
|
key <AC03> { [ d, D, eth, ETH ] };
|
||||||
|
key <AC04> { [ f, F, dstroke, ordfeminine ] };
|
||||||
|
key <AC05> { [ g, G, eng, ENG ] };
|
||||||
|
key <AC06> { [ h, H, hstroke, Hstroke ] };
|
||||||
|
key <AC07> { [ j, J, dead_hook, dead_horn ] };
|
||||||
|
key <AC08> { [ k, K, kra, ampersand ] };
|
||||||
|
key <AC09> { [ l, L, lstroke, Lstroke ] };
|
||||||
|
key <AC10> { [ semicolon, colon, dead_acute, dead_doubleacute ] };
|
||||||
|
key <AC11> { [apostrophe, quotedbl, dead_circumflex, dead_caron ] };
|
||||||
|
key <TLDE> { [ grave, asciitilde, notsign, notsign ] };
|
||||||
|
|
||||||
|
key <BKSL> { [ backslash, bar, dead_grave, dead_breve ] };
|
||||||
|
key <AB01> { [ z, Z, guillemotleft, less ] };
|
||||||
|
key <AB02> { [ x, X, guillemotright, greater ] };
|
||||||
|
key <AB03> { [ c, C, cent, copyright ] };
|
||||||
|
key <AB04> { [ v, V, doublelowquotemark, singlelowquotemark ] };
|
||||||
|
key <AB05> { [ b, B, leftdoublequotemark, leftsinglequotemark ] };
|
||||||
|
key <AB06> { [ n, N, rightdoublequotemark, rightsinglequotemark ] };
|
||||||
|
key <AB07> { [ m, M, mu, masculine ] };
|
||||||
|
key <AB08> { [ comma, less, U2022, multiply ] }; // bullet
|
||||||
|
key <AB09> { [ period, greater, periodcentered, division ] };
|
||||||
|
key <AB10> { [ slash, question, dead_belowdot, dead_abovedot ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Northern Europe ( Danish, Finnish, Norwegian, Swedish) common layout
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "type2" {
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
key <AE01> { [ 1, exclam, exclamdown, onesuperior ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, at, twosuperior ] };
|
||||||
|
key <AE03> { [ 3, numbersign, sterling, threesuperior] };
|
||||||
|
key <AE04> { [ 4, currency, dollar, onequarter ] };
|
||||||
|
key <AE05> { [ 5, percent, onehalf, cent ] };
|
||||||
|
key <AE06> { [ 6, ampersand, yen, fiveeighths ] };
|
||||||
|
key <AE07> { [ 7, slash, braceleft, division ] };
|
||||||
|
key <AE08> { [ 8, parenleft, bracketleft, guillemotleft] };
|
||||||
|
key <AE09> { [ 9, parenright, bracketright, guillemotright] };
|
||||||
|
key <AE10> { [ 0, equal, braceright, degree ] };
|
||||||
|
|
||||||
|
key <AD03> { [ e, E, EuroSign, cent ] };
|
||||||
|
key <AD04> { [ r, R, registered, registered ] };
|
||||||
|
key <AD05> { [ t, T, thorn, THORN ] };
|
||||||
|
key <AD09> { [ o, O, oe, OE ] };
|
||||||
|
key <AD11> { [ aring, Aring, dead_diaeresis, dead_abovering ] };
|
||||||
|
key <AD12> { [dead_diaeresis, dead_circumflex, dead_tilde, dead_caron ] };
|
||||||
|
|
||||||
|
key <AC01> { [ a, A, ordfeminine, masculine ] };
|
||||||
|
|
||||||
|
key <AB03> { [ c, C, copyright, copyright ] };
|
||||||
|
key <AB08> { [ comma, semicolon, dead_cedilla, dead_ogonek ] };
|
||||||
|
key <AB09> { [ period, colon, periodcentered, dead_abovedot ] };
|
||||||
|
key <AB10> { [ minus, underscore, dead_belowdot, dead_abovedot ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Slavic Latin ( Albanian, Croatian, Polish, Slovene, Yugoslav)
|
||||||
|
// common layout
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "type3" {
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
key <AD01> { [ q, Q, backslash, Greek_OMEGA ] };
|
||||||
|
key <AD02> { [ w, W, bar, section ] };
|
||||||
|
key <AD06> { [ z, Z, leftarrow, yen ] };
|
||||||
|
|
||||||
|
key <AC04> { [ f, F, bracketleft, ordfeminine ] };
|
||||||
|
key <AC05> { [ g, G, bracketright, ENG ] };
|
||||||
|
key <AC08> { [ k, K, lstroke, ampersand ] };
|
||||||
|
|
||||||
|
key <AB01> { [ y, Y, guillemotleft, less ] };
|
||||||
|
key <AB04> { [ v, V, at, grave ] };
|
||||||
|
key <AB05> { [ b, B, braceleft, apostrophe ] };
|
||||||
|
key <AB06> { [ n, N, braceright, acute ] };
|
||||||
|
key <AB07> { [ m, M, section, masculine ] };
|
||||||
|
key <AB08> { [ comma, semicolon, less, multiply ] };
|
||||||
|
key <AB09> { [ period, colon, greater, division ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Another common Latin layout
|
||||||
|
// (German, Estonian, Spanish, Icelandic, Italian, Latin American, Portuguese)
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "type4" {
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
key <AE02> { [ 2, quotedbl, at, oneeighth ] };
|
||||||
|
key <AE06> { [ 6, ampersand, notsign, fiveeighths ] };
|
||||||
|
key <AE07> { [ 7, slash, braceleft, seveneighths ] };
|
||||||
|
key <AE08> { [ 8, parenleft, bracketleft, trademark ] };
|
||||||
|
key <AE09> { [ 9, parenright, bracketright, plusminus ] };
|
||||||
|
key <AE10> { [ 0, equal, braceright, degree ] };
|
||||||
|
|
||||||
|
key <AD03> { [ e, E, EuroSign, cent ] };
|
||||||
|
|
||||||
|
key <AB08> { [ comma, semicolon, U2022, multiply ] }; // bullet
|
||||||
|
key <AB09> { [ period, colon, periodcentered, division ] };
|
||||||
|
key <AB10> { [ minus, underscore, dead_belowdot, dead_abovedot ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "nodeadkeys" {
|
||||||
|
|
||||||
|
key <AE12> { [ equal, plus, cedilla, ogonek ] };
|
||||||
|
key <AD11> { [bracketleft, braceleft, diaeresis, degree ] };
|
||||||
|
key <AD12> { [bracketright, braceright, asciitilde, macron ] };
|
||||||
|
key <AC07> { [ j, J, ezh, EZH ] };
|
||||||
|
key <AC10> { [ semicolon, colon, acute, doubleacute ] };
|
||||||
|
key <AC11> { [apostrophe, quotedbl, asciicircum, caron ] };
|
||||||
|
key <BKSL> { [ backslash, bar, grave, breve ] };
|
||||||
|
key <AB10> { [ slash, question, ellipsis, abovedot ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "type2_nodeadkeys" {
|
||||||
|
|
||||||
|
include "latin(nodeadkeys)"
|
||||||
|
|
||||||
|
key <AD11> { [ aring, Aring, diaeresis, degree ] };
|
||||||
|
key <AD12> { [ diaeresis, asciicircum, asciitilde, caron ] };
|
||||||
|
key <AB08> { [ comma, semicolon, cedilla, ogonek ] };
|
||||||
|
key <AB09> { [ period, colon, periodcentered, abovedot ] };
|
||||||
|
key <AB10> { [ minus, underscore, ellipsis, abovedot ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "type3_nodeadkeys" {
|
||||||
|
|
||||||
|
include "latin(nodeadkeys)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "type4_nodeadkeys" {
|
||||||
|
|
||||||
|
include "latin(nodeadkeys)"
|
||||||
|
|
||||||
|
key <AB10> { [ minus, underscore, ellipsis, abovedot ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Added 2008.03.05 by Marcin Woliński
|
||||||
|
// See http://marcinwolinski.pl/keyboard/ for a description.
|
||||||
|
// Used by pl(intl)
|
||||||
|
//
|
||||||
|
// ┌─────┐
|
||||||
|
// │ 2 4 │ 2 = Shift, 4 = Level3 + Shift
|
||||||
|
// │ 1 3 │ 1 = Normal, 3 = Level3
|
||||||
|
// └─────┘
|
||||||
|
// ┌─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┲━━━━━━━━━┓
|
||||||
|
// │ ~ ~ │ ! ' │ @ " │ # ˝ │ $ ¸ │ % ˇ │ ^ ^ │ & ˘ │ * ̇ │ ( ̣ │ ) ° │ _ ¯ │ + ˛ ┃ ⌫ Back- ┃
|
||||||
|
// │ ` ` │ 1 ¡ │ 2 © │ 3 • │ 4 § │ 5 € │ 6 ¢ │ 7 − │ 8 × │ 9 ÷ │ 0 ° │ - – │ = — ┃ space ┃
|
||||||
|
// ┢━━━━━┷━┱───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┺━┳━━━━━━━┫
|
||||||
|
// ┃ ┃ Q │ W │ E │ R │ T │ Y │ U │ I │ O │ P │ { « │ } » ┃ Enter ┃
|
||||||
|
// ┃Tab ↹ ┃ q │ w │ e │ r │ t │ y │ u │ i │ o │ p │ [ ‹ │ ] › ┃ ⏎ ┃
|
||||||
|
// ┣━━━━━━━┻┱────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┺┓ ┃
|
||||||
|
// ┃ ┃ A │ S │ D │ F │ G │ H │ J │ K │ L │ : “ │ " ” │ | ¶ ┃ ┃
|
||||||
|
// ┃Caps ⇬ ┃ a │ s │ d │ f │ g │ h │ j │ k │ l │ ; ‘ │ ' ’ │ \ ┃ ┃
|
||||||
|
// ┣━━━━━━━━┹────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┲┷━━━━━┻━━━━━━┫
|
||||||
|
// ┃ │ Z │ X │ C │ V │ B │ N │ M │ < „ │ > · │ ? ¿ ┃ ┃
|
||||||
|
// ┃Shift ⇧ │ z │ x │ c │ v │ b │ n │ m │ , ‚ │ . … │ / ⁄ ┃Shift ⇧ ┃
|
||||||
|
// ┣━━━━━━━┳━━━━━┷━┳━━━┷━━━┱─┴─────┴─────┴─────┴─────┴─────┴───┲━┷━━━━━╈━━━━━┻━┳━━━━━━━┳━━━┛
|
||||||
|
// ┃ ┃ ┃ ┃ ␣ ⍽ ┃ ┃ ┃ ┃
|
||||||
|
// ┃Ctrl ┃Meta ┃Alt ┃ ␣ Space ⍽ ┃AltGr ⇮┃Menu ┃Ctrl ┃
|
||||||
|
// ┗━━━━━━━┻━━━━━━━┻━━━━━━━┹───────────────────────────────────┺━━━━━━━┻━━━━━━━┻━━━━━━━┛
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "intl" {
|
||||||
|
|
||||||
|
key <TLDE> { [ grave, asciitilde, dead_grave, dead_tilde ] };
|
||||||
|
key <AE01> { [ 1, exclam, exclamdown, dead_acute ] };
|
||||||
|
key <AE02> { [ 2, at, copyright, dead_diaeresis ] };
|
||||||
|
key <AE03> { [ 3, numbersign, U2022, dead_doubleacute ] }; // U+2022 is bullet (the name bullet does not work)
|
||||||
|
key <AE04> { [ 4, dollar, section, dead_cedilla ] };
|
||||||
|
key <AE05> { [ 5, percent, EuroSign, dead_caron ] };
|
||||||
|
key <AE06> { [ 6, asciicircum, cent, dead_circumflex ] };
|
||||||
|
key <AE07> { [ 7, ampersand, U2212, dead_breve ] }; // U+2212 is MINUS SIGN
|
||||||
|
key <AE08> { [ 8, asterisk, multiply, dead_abovedot ] };
|
||||||
|
key <AE09> { [ 9, parenleft, division, dead_belowdot ] };
|
||||||
|
key <AE10> { [ 0, parenright, degree, dead_abovering ] };
|
||||||
|
key <AE11> { [ minus, underscore, endash, dead_macron ] };
|
||||||
|
key <AE12> { [ equal, plus, emdash, dead_ogonek ] };
|
||||||
|
|
||||||
|
key <AD01> { [ q, Q ] };
|
||||||
|
key <AD02> { [ w, W ] };
|
||||||
|
key <AD03> { [ e, E ] };
|
||||||
|
key <AD04> { [ r, R ] };
|
||||||
|
key <AD05> { [ t, T ] };
|
||||||
|
key <AD06> { [ y, Y ] };
|
||||||
|
key <AD07> { [ u, U ] };
|
||||||
|
key <AD08> { [ i, I ] };
|
||||||
|
key <AD09> { [ o, O ] };
|
||||||
|
key <AD10> { [ p, P ] };
|
||||||
|
key <AD11> { [bracketleft, braceleft, U2039, guillemotleft ] };
|
||||||
|
key <AD12> { [bracketright, braceright, U203A, guillemotright ] };
|
||||||
|
|
||||||
|
key <AC01> { [ a, A ] };
|
||||||
|
key <AC02> { [ s, S ] };
|
||||||
|
key <AC03> { [ d, D ] };
|
||||||
|
key <AC04> { [ f, F ] };
|
||||||
|
key <AC05> { [ g, G ] };
|
||||||
|
key <AC06> { [ h, H ] };
|
||||||
|
key <AC07> { [ j, J ] };
|
||||||
|
key <AC08> { [ k, K ] };
|
||||||
|
key <AC09> { [ l, L ] };
|
||||||
|
key <AC10> { [ semicolon, colon, leftsinglequotemark, leftdoublequotemark ] };
|
||||||
|
key <AC11> { [apostrophe, quotedbl, rightsinglequotemark, rightdoublequotemark ] };
|
||||||
|
|
||||||
|
key <BKSL> { [ backslash, bar, NoSymbol, paragraph ] };
|
||||||
|
key <AB01> { [ z, Z ] };
|
||||||
|
key <AB02> { [ x, X ] };
|
||||||
|
key <AB03> { [ c, C ] };
|
||||||
|
key <AB04> { [ v, V ] };
|
||||||
|
key <AB05> { [ b, B ] };
|
||||||
|
key <AB06> { [ n, N ] };
|
||||||
|
key <AB07> { [ m, M ] };
|
||||||
|
key <AB08> { [ comma, less, singlelowquotemark, doublelowquotemark ] };
|
||||||
|
key <AB09> { [ period, greater, ellipsis, periodcentered ] };
|
||||||
|
key <AB10> { [ slash, question, U2044, questiondown ] }; // U+2044 is FRACTION SLASH
|
||||||
|
};
|
||||||
+156
@@ -0,0 +1,156 @@
|
|||||||
|
// These variants assign ISO_Level3_Shift to various keys
|
||||||
|
// so that levels 3 and 4 can be reached.
|
||||||
|
|
||||||
|
// The default behaviour:
|
||||||
|
// the right Alt key (AltGr) chooses the third symbol engraved on a key.
|
||||||
|
default partial modifier_keys
|
||||||
|
xkb_symbols "ralt_switch" {
|
||||||
|
key <RALT> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The right Alt key never chooses the third level.
|
||||||
|
// This option attempts to undo the effect of a layout's inclusion of
|
||||||
|
// 'ralt_switch'. You may want to also select another level3 option
|
||||||
|
// to map the level3 shift to some other key.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "ralt_alt" {
|
||||||
|
key <RALT> {[ Alt_R, Meta_R ], type[group1]="TWO_LEVEL" };
|
||||||
|
modifier_map Mod1 { <RALT> };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The right Alt key (while pressed) chooses the third shift level,
|
||||||
|
// and Compose is mapped to its second level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "ralt_switch_multikey" {
|
||||||
|
key <RALT> {[ ISO_Level3_Shift, Multi_key ], type[group1]="TWO_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Either Alt key (while pressed) chooses the third shift level.
|
||||||
|
// (To be used mostly to imitate Mac OS functionality.)
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "alt_switch" {
|
||||||
|
include "level3(lalt_switch)"
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
// The left Alt key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "lalt_switch" {
|
||||||
|
key <LALT> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The right Ctrl key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "switch" {
|
||||||
|
key <RCTL> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The Menu key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "menu_switch" {
|
||||||
|
key <MENU> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Either Win key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "win_switch" {
|
||||||
|
include "level3(lwin_switch)"
|
||||||
|
include "level3(rwin_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
// The left Win key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "lwin_switch" {
|
||||||
|
key <LWIN> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The right Win key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "rwin_switch" {
|
||||||
|
key <RWIN> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The Enter key on the kepypad (while pressed) chooses the third shift level.
|
||||||
|
// (This is especially useful for Mac laptops which miss the right Alt key.)
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "enter_switch" {
|
||||||
|
key <KPEN> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The CapsLock key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "caps_switch" {
|
||||||
|
key <CAPS> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The CapsLock key (while pressed) chooses the third shift level and
|
||||||
|
// Ctrl + CapsLock has the original CapsLock function.
|
||||||
|
// The 2023 DIN standard for German keyboards recommends it as an option:
|
||||||
|
// - https://de.wikipedia.org/wiki/E1_(Tastaturbelegung)#Feststelltaste/Umschaltsperre
|
||||||
|
// - https://en.wikipedia.org/wiki/Caps_Lock#Abolition
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "caps_switch_capslock_with_ctrl" {
|
||||||
|
virtual_modifiers LevelThree;
|
||||||
|
|
||||||
|
key <CAPS> {
|
||||||
|
type[Group1] = "PC_CONTROL_LEVEL2",
|
||||||
|
symbols[Group1] = [ ISO_Level3_Shift, Caps_Lock ],
|
||||||
|
// Explicit actions are preferred over modMap None/Mod5 { Caps_Lock }
|
||||||
|
// because they have no side effect
|
||||||
|
actions[Group1] = [ SetMods(modifiers = LevelThree), LockMods(modifiers = Lock) ]
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
// The Backslash key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "bksl_switch" {
|
||||||
|
key <BKSL> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The AC11 key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "ac11_switch" {
|
||||||
|
key <AC11> {[ ISO_Level3_Shift ], type[Group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The Less/Greater key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "lsgt_switch" {
|
||||||
|
key <LSGT> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The CapsLock key (while pressed) chooses the third shift level,
|
||||||
|
// and latches when pressed together with another third-level chooser.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "caps_switch_latch" {
|
||||||
|
key <CAPS> {[ ISO_Level3_Shift, ISO_Level3_Shift, ISO_Level3_Latch ],
|
||||||
|
type[group1]="THREE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The Backslash key (while pressed) chooses the third shift level,
|
||||||
|
// and latches when pressed together with another third-level chooser.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "bksl_switch_latch" {
|
||||||
|
key <BKSL> {[ ISO_Level3_Shift, ISO_Level3_Shift, ISO_Level3_Latch ],
|
||||||
|
type[group1]="THREE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The Less/Greater key (while pressed) chooses the third shift level,
|
||||||
|
// and latches when pressed together with another third-level chooser.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "lsgt_switch_latch" {
|
||||||
|
key <LSGT> {[ ISO_Level3_Shift, ISO_Level3_Shift, ISO_Level3_Latch ],
|
||||||
|
type[group1]="THREE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Top-row digit key 4 chooses third shift level when pressed alone.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "4_switch_isolated" {
|
||||||
|
override key <AE04> {[ ISO_Level3_Shift ]};
|
||||||
|
};
|
||||||
|
|
||||||
|
// Top-row digit key 9 chooses third shift level when pressed alone.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "9_switch_isolated" {
|
||||||
|
override key <AE09> {[ ISO_Level3_Shift ]};
|
||||||
|
};
|
||||||
+2238
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,153 @@
|
|||||||
|
//! xkeyboard-config — keyboard layouts, compiled from the X11 xkeyboard-config database
|
||||||
|
//! into native Zig. It turns a physical key (a USB HID usage, as the input module delivers)
|
||||||
|
//! plus a modifier state into a **keysym** and, when the key produces one, a **character**
|
||||||
|
//! (a Unicode scalar). This is the piece that lets a `KeyEvent.keycode` become a
|
||||||
|
//! `KeyEvent.character`, without shipping an X11 runtime.
|
||||||
|
//!
|
||||||
|
//! The layout tables in `generated/layouts.zig` are produced by
|
||||||
|
//! `tools/make-xkeyboard-config.py` (see ./README.md to regenerate). Those tables are
|
||||||
|
//! deliberately pure data — each key carries its up-to-four levels and an XKB *type*. The
|
||||||
|
//! type -> level selection semantics (which modifier picks which level) live here, so the
|
||||||
|
//! data and the policy are separable.
|
||||||
|
//!
|
||||||
|
//! Scope (documented in README.md): group 1 only, no dead-key/compose composition (a dead
|
||||||
|
//! key returns its keysym with no character), and a curated set of key types. Layouts:
|
||||||
|
//! us, gb, de, fr, es, dvorak.
|
||||||
|
//!
|
||||||
|
//! Upstream xkeyboard-config and keysymdef.h are MIT/X11 licensed; see vendor/COPYING and
|
||||||
|
//! vendor/PROVENANCE.md.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const generated = @import("layouts");
|
||||||
|
|
||||||
|
pub const Level = generated.Level;
|
||||||
|
pub const KeyType = generated.KeyType;
|
||||||
|
pub const Key = generated.Key;
|
||||||
|
pub const Layout = generated.Layout;
|
||||||
|
|
||||||
|
/// The generated layouts, by name — as pointers, so they share identity with `all` and
|
||||||
|
/// `byName` (and match the `*const Layout` that `map` takes).
|
||||||
|
pub const us: *const Layout = &generated.us;
|
||||||
|
pub const gb: *const Layout = &generated.gb;
|
||||||
|
pub const de: *const Layout = &generated.de;
|
||||||
|
pub const fr: *const Layout = &generated.fr;
|
||||||
|
pub const es: *const Layout = &generated.es;
|
||||||
|
pub const dvorak: *const Layout = &generated.dvorak;
|
||||||
|
|
||||||
|
/// Every generated layout, for enumeration (e.g. a settings UI).
|
||||||
|
pub const all = generated.all;
|
||||||
|
|
||||||
|
/// The modifier state that selects a key's level. `level3` is AltGr (ISO Level3 Shift);
|
||||||
|
/// `control` is accepted for completeness but does not affect level selection here.
|
||||||
|
pub const Modifiers = struct {
|
||||||
|
shift: bool = false,
|
||||||
|
caps_lock: bool = false,
|
||||||
|
level3: bool = false,
|
||||||
|
control: bool = false,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The result of a lookup: the X11 `keysym`, and the `character` it produces (a Unicode
|
||||||
|
/// scalar) when it is a printable key — null for keys that produce none (Return, F1, a
|
||||||
|
/// bare dead key, an unmapped key).
|
||||||
|
pub const Mapping = struct {
|
||||||
|
keysym: u32,
|
||||||
|
character: ?u21,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Which level (0..3) a key of `kind` selects under `mods`. XKB's canonical semantics:
|
||||||
|
/// Shift picks the odd level, AltGr (level3) adds 2, and Caps acts like Shift for the
|
||||||
|
/// alphabetic types. See the XKB "key types" — this covers the ones the vendored layouts
|
||||||
|
/// use; anything else falls back to shift-or-not.
|
||||||
|
fn selectLevel(kind: KeyType, mods: Modifiers) usize {
|
||||||
|
const shift_or_caps = mods.shift != mods.caps_lock; // XOR: Caps behaves like Shift
|
||||||
|
const low: usize = if (mods.shift) 1 else 0;
|
||||||
|
const high: usize = if (mods.level3) 2 else 0;
|
||||||
|
return switch (kind) {
|
||||||
|
.one_level => 0,
|
||||||
|
.two_level, .keypad, .other => low,
|
||||||
|
.alphabetic => if (shift_or_caps) 1 else 0,
|
||||||
|
.four_level => low + high,
|
||||||
|
.four_level_alphabetic => (if (shift_or_caps) @as(usize, 1) else 0) + high,
|
||||||
|
// Caps affects only the base pair, not the AltGr pair.
|
||||||
|
.four_level_semialphabetic => if (mods.level3) 2 + low else (if (shift_or_caps) @as(usize, 1) else 0),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Map a physical key (`hid_usage`, a USB HID keyboard-page usage) under `mods` on
|
||||||
|
/// `layout` to its keysym and character. Falls back gracefully when the selected level is
|
||||||
|
/// undefined for the key: it drops the AltGr component, then the shift component, so a key
|
||||||
|
/// with only a base/shift pair still yields something sensible under AltGr.
|
||||||
|
pub fn map(layout: *const Layout, hid_usage: u8, mods: Modifiers) Mapping {
|
||||||
|
const key = &layout.keys[hid_usage];
|
||||||
|
var level = selectLevel(key.kind, mods);
|
||||||
|
// Fall back to a defined level: full -> without AltGr -> base.
|
||||||
|
if (key.levels[level].keysym == 0 and key.levels[level].unicode == 0) {
|
||||||
|
const candidates = [_]usize{ level & 1, 0 };
|
||||||
|
for (candidates) |candidate| {
|
||||||
|
if (key.levels[candidate].keysym != 0 or key.levels[candidate].unicode != 0) {
|
||||||
|
level = candidate;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const chosen = key.levels[level];
|
||||||
|
return .{
|
||||||
|
.keysym = chosen.keysym,
|
||||||
|
.character = if (chosen.unicode != 0) @intCast(chosen.unicode) else null,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Look up a layout by its name (`"us"`, `"gb"`, ...), or null if unknown.
|
||||||
|
pub fn byName(name: []const u8) ?*const Layout {
|
||||||
|
for (all) |layout| {
|
||||||
|
if (std.mem.eql(u8, layout.name, name)) return layout;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- tests (host-run via `zig build test`) ---------------------------------
|
||||||
|
|
||||||
|
const testing = std.testing;
|
||||||
|
|
||||||
|
// USB HID usages used in the tests (keyboard page 0x07).
|
||||||
|
const hid_a: u8 = 0x04;
|
||||||
|
const hid_1: u8 = 0x1e;
|
||||||
|
const hid_3: u8 = 0x20;
|
||||||
|
|
||||||
|
test "us: letters obey shift and caps" {
|
||||||
|
try testing.expectEqual(@as(?u21, 'a'), map(us, hid_a, .{}).character);
|
||||||
|
try testing.expectEqual(@as(?u21, 'A'), map(us, hid_a, .{ .shift = true }).character);
|
||||||
|
try testing.expectEqual(@as(?u21, 'A'), map(us, hid_a, .{ .caps_lock = true }).character);
|
||||||
|
// Shift + Caps cancels for an alphabetic key.
|
||||||
|
try testing.expectEqual(@as(?u21, 'a'), map(us, hid_a, .{ .shift = true, .caps_lock = true }).character);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "us: digits and their shifted symbols" {
|
||||||
|
try testing.expectEqual(@as(?u21, '1'), map(us, hid_1, .{}).character);
|
||||||
|
try testing.expectEqual(@as(?u21, '!'), map(us, hid_1, .{ .shift = true }).character);
|
||||||
|
try testing.expectEqual(@as(?u21, '3'), map(us, hid_3, .{}).character);
|
||||||
|
try testing.expectEqual(@as(?u21, '#'), map(us, hid_3, .{ .shift = true }).character);
|
||||||
|
// A digit is not alphabetic: Caps alone must not shift it.
|
||||||
|
try testing.expectEqual(@as(?u21, '3'), map(us, hid_3, .{ .caps_lock = true }).character);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "layouts differ: GB pound vs US hash on shift+3" {
|
||||||
|
try testing.expectEqual(@as(?u21, '#'), map(us, hid_3, .{ .shift = true }).character);
|
||||||
|
try testing.expectEqual(@as(?u21, '£'), map(gb, hid_3, .{ .shift = true }).character);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "french azerty places q where us has a" {
|
||||||
|
try testing.expectEqual(@as(?u21, 'q'), map(fr, hid_a, .{}).character);
|
||||||
|
try testing.expectEqual(@as(?u21, 'Q'), map(fr, hid_a, .{ .shift = true }).character);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "byName resolves and rejects" {
|
||||||
|
try testing.expect(byName("us") == us);
|
||||||
|
try testing.expect(byName("gb") == gb);
|
||||||
|
try testing.expect(byName("nonsense") == null);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "unmapped key yields no character" {
|
||||||
|
// HID 0x00 is not a key; every level is empty.
|
||||||
|
try testing.expectEqual(@as(?u21, null), map(us, 0x00, .{}).character);
|
||||||
|
}
|
||||||
-1086
File diff suppressed because it is too large
Load Diff
@@ -1,208 +0,0 @@
|
|||||||
//! AML (ACPI Machine Language) — the bytecode in the DSDT and SSDTs that describes
|
|
||||||
//! the parts of the machine the static tables don't.
|
|
||||||
//!
|
|
||||||
//! This module has two stages. `parser.zig` walks the entire byte stream and
|
|
||||||
//! records every named object into a namespace tree (`namespace.zig`), capturing
|
|
||||||
//! method bodies and field/region layout. `interp.zig` then *evaluates* control
|
|
||||||
//! methods on demand — running operators, control flow, and OperationRegion field
|
|
||||||
//! access — so callers can resolve device status (`_STA`), current resource
|
|
||||||
//! settings (`_CRS`), sleep states (`_Sx`), and the like against the live namespace.
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
const op = @import("opcodes.zig");
|
|
||||||
const parser = @import("parser.zig");
|
|
||||||
const namespace = @import("namespace.zig");
|
|
||||||
const interp = @import("interp.zig");
|
|
||||||
|
|
||||||
pub const Namespace = namespace.Namespace;
|
|
||||||
pub const Node = namespace.Node;
|
|
||||||
pub const NodeKind = namespace.NodeKind;
|
|
||||||
|
|
||||||
/// The AML evaluator: interprets control methods (and reads Names/Fields) far
|
|
||||||
/// enough for device discovery. See `interp.zig`.
|
|
||||||
pub const Interp = interp.Interp;
|
|
||||||
pub const Object = interp.Object;
|
|
||||||
pub const EvalHal = interp.Hal;
|
|
||||||
|
|
||||||
/// The SLP_TYP values written to PM1a/PM1b control to enter a sleep state.
|
|
||||||
pub const SleepType = struct {
|
|
||||||
slp_typ_a: u8,
|
|
||||||
slp_typ_b: u8,
|
|
||||||
};
|
|
||||||
|
|
||||||
pub const ParseResult = struct {
|
|
||||||
namespace: Namespace,
|
|
||||||
/// Bytes the parser consumed across all blocks...
|
|
||||||
consumed: usize,
|
|
||||||
/// ...out of this many. A clean full traversal has `consumed == total`.
|
|
||||||
total: usize,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Parse the given AML blocks (DSDT first, then SSDTs) into one namespace. Later
|
|
||||||
/// blocks extend the namespace built by earlier ones, exactly as ACPI intends.
|
|
||||||
pub fn parse(allocator: std.mem.Allocator, blocks: []const []const u8) !ParseResult {
|
|
||||||
var ns = try Namespace.init(allocator);
|
|
||||||
var consumed: usize = 0;
|
|
||||||
var total: usize = 0;
|
|
||||||
for (blocks) |block| {
|
|
||||||
var p = parser.Parser.init(block, &ns);
|
|
||||||
consumed += p.parseAll();
|
|
||||||
total += block.len;
|
|
||||||
}
|
|
||||||
return .{ .namespace = ns, .consumed = consumed, .total = total };
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Look up the `\_S{state}` sleep package in a parsed namespace and return its
|
|
||||||
/// first two integer elements (SLP_TYP for PM1a / PM1b), or null if absent.
|
|
||||||
pub fn sleepState(ns: *Namespace, state: u8) ?SleepType {
|
|
||||||
const seg = [4]u8{ '_', 'S', '0' + state, '_' };
|
|
||||||
const node = ns.resolve(ns.root, false, 0, &.{seg}) orelse return null;
|
|
||||||
if (node.kind != .name) return null;
|
|
||||||
return parseSleepPackage(node.value);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Decode a `Package(){ SLP_TYPa, SLP_TYPb, ... }` from the raw AML of a Name's
|
|
||||||
/// value. Returns the first two elements as bytes (missing elements default to 0).
|
|
||||||
fn parseSleepPackage(value: []const u8) ?SleepType {
|
|
||||||
if (value.len == 0 or value[0] != op.package_op) return null;
|
|
||||||
var p: usize = 1;
|
|
||||||
p += pkgLengthSize(value, p) orelse return null;
|
|
||||||
if (p >= value.len) return null;
|
|
||||||
const num_elements = value[p];
|
|
||||||
p += 1;
|
|
||||||
|
|
||||||
const a: u8 = if (num_elements >= 1) @truncate(readInteger(value, &p) orelse 0) else 0;
|
|
||||||
const b: u8 = if (num_elements >= 2) @truncate(readInteger(value, &p) orelse 0) else 0;
|
|
||||||
return .{ .slp_typ_a = a, .slp_typ_b = b };
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Bytes a PkgLength field occupies at `p` (we only need to step over it here).
|
|
||||||
fn pkgLengthSize(bytes: []const u8, p: usize) ?usize {
|
|
||||||
if (p >= bytes.len) return null;
|
|
||||||
const follow: usize = bytes[p] >> 6;
|
|
||||||
if (p + 1 + follow > bytes.len) return null;
|
|
||||||
return 1 + follow;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Read one AML integer data object at `p`, advancing `p`.
|
|
||||||
fn readInteger(bytes: []const u8, p: *usize) ?u64 {
|
|
||||||
if (p.* >= bytes.len) return null;
|
|
||||||
const opcode = bytes[p.*];
|
|
||||||
p.* += 1;
|
|
||||||
return switch (opcode) {
|
|
||||||
op.zero_op => 0,
|
|
||||||
op.one_op => 1,
|
|
||||||
op.ones_op => 0xFF,
|
|
||||||
op.byte_prefix => readLittle(bytes, p, 1),
|
|
||||||
op.word_prefix => readLittle(bytes, p, 2),
|
|
||||||
op.dword_prefix => readLittle(bytes, p, 4),
|
|
||||||
op.qword_prefix => readLittle(bytes, p, 8),
|
|
||||||
else => null,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
fn readLittle(bytes: []const u8, p: *usize, n: usize) ?u64 {
|
|
||||||
if (p.* + n > bytes.len) return null;
|
|
||||||
var v: u64 = 0;
|
|
||||||
var k: usize = 0;
|
|
||||||
while (k < n) : (k += 1) v |= @as(u64, bytes[p.* + k]) << @intCast(k * 8);
|
|
||||||
p.* += n;
|
|
||||||
return v;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- tests ------------------------------------------------------------------
|
|
||||||
|
|
||||||
test "parses a nested namespace and finds the sleep package" {
|
|
||||||
// A hand-assembled AML blob (all PkgLengths computed to be single-byte):
|
|
||||||
// Name(_S5, Package(2){0x05, 0x00})
|
|
||||||
// Scope(\_SB) { Device(PCI0) {
|
|
||||||
// Name(_HID, 0x11)
|
|
||||||
// Method(MTHD, 1) {}
|
|
||||||
// Method(CALL, 0) { MTHD(Zero) } // invocation of a 1-arg method
|
|
||||||
// } }
|
|
||||||
// OperationRegion(DBG0, SystemIO, 0x0402, 1)
|
|
||||||
// Field(DBG0, ...) { DBGB, 8 }
|
|
||||||
const blob = [_]u8{
|
|
||||||
// Name(_S5, Package(2){Byte 0x05, Byte 0x00})
|
|
||||||
0x08, 0x5F, 0x53, 0x35, 0x5F, 0x12, 0x06, 0x02, 0x0A, 0x05, 0x0A, 0x00,
|
|
||||||
// Scope(\_SB) pkglen=0x27
|
|
||||||
0x10, 0x27, 0x5C, 0x5F, 0x53, 0x42, 0x5F,
|
|
||||||
// Device(PCI0) pkglen=0x1F
|
|
||||||
0x5B, 0x82, 0x1F, 0x50, 0x43, 0x49, 0x30,
|
|
||||||
// Name(_HID, 0x11)
|
|
||||||
0x08, 0x5F, 0x48, 0x49, 0x44, 0x0A, 0x11,
|
|
||||||
// Method(MTHD, flags=1) empty, pkglen=0x06
|
|
||||||
0x14, 0x06, 0x4D, 0x54, 0x48, 0x44, 0x01,
|
|
||||||
// Method(CALL, flags=0) { MTHD(Zero) }, pkglen=0x0B
|
|
||||||
0x14, 0x0B, 0x43, 0x41, 0x4C, 0x4C, 0x00, 0x4D, 0x54, 0x48, 0x44, 0x00,
|
|
||||||
// OperationRegion(DBG0, SystemIO, Word 0x0402, Byte 1)
|
|
||||||
0x5B, 0x80, 0x44, 0x42, 0x47, 0x30, 0x01, 0x0B, 0x02, 0x04, 0x0A, 0x01,
|
|
||||||
// Field(DBG0, flags=1) { DBGB, 8 }, pkglen=0x0B
|
|
||||||
0x5B, 0x81, 0x0B, 0x44, 0x42, 0x47, 0x30, 0x01, 0x44, 0x42, 0x47, 0x42, 0x08,
|
|
||||||
};
|
|
||||||
|
|
||||||
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
|
||||||
defer arena.deinit();
|
|
||||||
var result = try parse(arena.allocator(), &.{&blob});
|
|
||||||
|
|
||||||
// Integrity: the parser consumed exactly the whole blob (no desync).
|
|
||||||
try std.testing.expectEqual(blob.len, result.consumed);
|
|
||||||
try std.testing.expectEqual(blob.len, result.total);
|
|
||||||
|
|
||||||
const ns = &result.namespace;
|
|
||||||
|
|
||||||
// Expected top-level nodes.
|
|
||||||
const sb = ns.resolve(ns.root, false, 0, &.{.{ '_', 'S', 'B', '_' }}) orelse return error.NoSB;
|
|
||||||
try std.testing.expectEqual(NodeKind.scope, sb.kind);
|
|
||||||
const pci0 = ns.resolve(sb, false, 0, &.{.{ 'P', 'C', 'I', '0' }}) orelse return error.NoPCI0;
|
|
||||||
try std.testing.expectEqual(NodeKind.device, pci0.kind);
|
|
||||||
_ = ns.resolve(pci0, false, 0, &.{.{ '_', 'H', 'I', 'D' }}) orelse return error.NoHID;
|
|
||||||
|
|
||||||
// The 1-arg method's arg count was parsed from its flags byte.
|
|
||||||
const mthd = ns.resolve(pci0, false, 0, &.{.{ 'M', 'T', 'H', 'D' }}) orelse return error.NoMTHD;
|
|
||||||
try std.testing.expectEqual(NodeKind.method, mthd.kind);
|
|
||||||
try std.testing.expectEqual(@as(u8, 1), mthd.arg_count);
|
|
||||||
|
|
||||||
// OperationRegion and the Field unit made it into the namespace.
|
|
||||||
_ = ns.resolve(ns.root, false, 0, &.{.{ 'D', 'B', 'G', '0' }}) orelse return error.NoRegion;
|
|
||||||
_ = ns.resolve(ns.root, false, 0, &.{.{ 'D', 'B', 'G', 'B' }}) orelse return error.NoField;
|
|
||||||
|
|
||||||
// The sleep package decoded.
|
|
||||||
const s5 = sleepState(ns, 5) orelse return error.NoS5;
|
|
||||||
try std.testing.expectEqual(@as(u8, 5), s5.slp_typ_a);
|
|
||||||
try std.testing.expectEqual(@as(u8, 0), s5.slp_typ_b);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn noMap(_: u64, _: u64, _: bool) void {}
|
|
||||||
fn noRead(_: u8, _: u16) u32 {
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
fn noWrite(_: u8, _: u16, _: u32) void {}
|
|
||||||
|
|
||||||
test "interpreter runs a method with args, arithmetic, and control flow" {
|
|
||||||
// Method(TST_, 1) {
|
|
||||||
// Store(Arg0, Local0); Add(Local0, 5, Local0)
|
|
||||||
// If (LGreater(Local0, 10)) { Return(One) }
|
|
||||||
// Return(Zero)
|
|
||||||
// }
|
|
||||||
const blob = [_]u8{
|
|
||||||
0x14, 0x18, 0x54, 0x53, 0x54, 0x5F, 0x01, // Method TST_, 1 arg
|
|
||||||
0x70, 0x68, 0x60, // Store(Arg0, Local0)
|
|
||||||
0x72, 0x60, 0x0A, 0x05, 0x60, // Add(Local0, 5, Local0)
|
|
||||||
0xA0, 0x07, 0x94, 0x60, 0x0A, 0x0A, 0xA4, 0x01, // If(LGreater(Local0,10)) { Return(One) }
|
|
||||||
0xA4, 0x00, // Return(Zero)
|
|
||||||
};
|
|
||||||
|
|
||||||
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
|
||||||
defer arena.deinit();
|
|
||||||
var result = try parse(arena.allocator(), &.{&blob});
|
|
||||||
const ns = &result.namespace;
|
|
||||||
const tst = ns.resolve(ns.root, false, 0, &.{.{ 'T', 'S', 'T', '_' }}) orelse return error.NoMethod;
|
|
||||||
|
|
||||||
var ev = Interp.init(ns, .{ .mapMmio = noMap, .pioRead = noRead, .pioWrite = noWrite }, arena.allocator());
|
|
||||||
|
|
||||||
const hi = try ev.evaluate(tst, &.{.{ .integer = 7 }}); // 7+5=12 > 10 -> 1
|
|
||||||
try std.testing.expectEqual(@as(u64, 1), try hi.asInt());
|
|
||||||
const lo = try ev.evaluate(tst, &.{.{ .integer = 2 }}); // 2+5=7 !> 10 -> 0
|
|
||||||
try std.testing.expectEqual(@as(u64, 0), try lo.asInt());
|
|
||||||
}
|
|
||||||
@@ -1,737 +0,0 @@
|
|||||||
//! A tree-walking AML interpreter — the evaluation stage on top of the parser's
|
|
||||||
//! structural namespace. It executes control methods (their bodies captured by
|
|
||||||
//! the parser) far enough to serve device discovery: device status (`_STA`, is a
|
|
||||||
//! device present), current resource settings (`_CRS`), and the operators, control
|
|
||||||
//! flow, locals/args, and
|
|
||||||
//! OperationRegion field access those methods reach for.
|
|
||||||
//!
|
|
||||||
//! Scope: integers, buffers, strings, packages, and references; If/Else/While/
|
|
||||||
//! Return; the arithmetic/logic operators; method invocation; Name/Local/Arg
|
|
||||||
//! access; CreateField buffer patching (the common current-resource-settings
|
|
||||||
//! (`_CRS`) idiom); and field
|
|
||||||
//! reads/writes against SystemMemory and SystemIO regions. Opcodes outside this
|
|
||||||
//! set return `error.Unsupported`, which callers treat as "couldn't evaluate" and
|
|
||||||
//! fall back — never a hard failure.
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
const op = @import("opcodes.zig");
|
|
||||||
const nsp = @import("namespace.zig");
|
|
||||||
const Node = nsp.Node;
|
|
||||||
const Namespace = nsp.Namespace;
|
|
||||||
|
|
||||||
/// Injected hardware access for OperationRegion reads/writes (the arch VMM + pio).
|
|
||||||
pub const Hal = struct {
|
|
||||||
mapMmio: *const fn (virt: u64, phys: u64, writable: bool) void,
|
|
||||||
pioRead: *const fn (width: u8, port: u16) u32,
|
|
||||||
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
|
||||||
};
|
|
||||||
|
|
||||||
pub const Error = error{ Unsupported, Truncated, DivByZero } || std.mem.Allocator.Error;
|
|
||||||
|
|
||||||
/// A runtime AML value.
|
|
||||||
pub const Object = union(enum) {
|
|
||||||
uninitialized,
|
|
||||||
integer: u64,
|
|
||||||
buffer: []u8,
|
|
||||||
string: []u8,
|
|
||||||
package: []Object,
|
|
||||||
reference: *Node,
|
|
||||||
|
|
||||||
pub fn asInt(self: Object) Error!u64 {
|
|
||||||
return switch (self) {
|
|
||||||
.integer => |v| v,
|
|
||||||
.buffer => |b| blk: {
|
|
||||||
var v: u64 = 0;
|
|
||||||
for (b, 0..) |byte, i| {
|
|
||||||
if (i >= 8) break;
|
|
||||||
v |= @as(u64, byte) << @intCast(i * 8);
|
|
||||||
}
|
|
||||||
break :blk v;
|
|
||||||
},
|
|
||||||
else => error.Unsupported,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
const max_segs = 16;
|
|
||||||
const NamePath = struct {
|
|
||||||
rooted: bool = false,
|
|
||||||
parents: u8 = 0,
|
|
||||||
segs: [max_segs][4]u8 = undefined,
|
|
||||||
count: usize = 0,
|
|
||||||
fn slice(self: *const NamePath) []const [4]u8 {
|
|
||||||
return self.segs[0..self.count];
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
const Cursor = struct {
|
|
||||||
b: []const u8,
|
|
||||||
i: usize = 0,
|
|
||||||
|
|
||||||
fn eof(self: *Cursor) bool {
|
|
||||||
return self.i >= self.b.len;
|
|
||||||
}
|
|
||||||
fn peek(self: *Cursor) ?u8 {
|
|
||||||
return if (self.eof()) null else self.b[self.i];
|
|
||||||
}
|
|
||||||
fn byte(self: *Cursor) Error!u8 {
|
|
||||||
if (self.eof()) return error.Truncated;
|
|
||||||
const v = self.b[self.i];
|
|
||||||
self.i += 1;
|
|
||||||
return v;
|
|
||||||
}
|
|
||||||
fn take(self: *Cursor, n: usize) Error![]const u8 {
|
|
||||||
if (self.i + n > self.b.len) return error.Truncated;
|
|
||||||
const s = self.b[self.i .. self.i + n];
|
|
||||||
self.i += n;
|
|
||||||
return s;
|
|
||||||
}
|
|
||||||
fn pkgLen(self: *Cursor) Error!usize {
|
|
||||||
const lead = try self.byte();
|
|
||||||
const follow: usize = lead >> 6;
|
|
||||||
if (follow == 0) return lead & 0x3F;
|
|
||||||
var value: usize = lead & 0x0F;
|
|
||||||
var k: usize = 0;
|
|
||||||
while (k < follow) : (k += 1) value |= @as(usize, try self.byte()) << @intCast(4 + k * 8);
|
|
||||||
return value;
|
|
||||||
}
|
|
||||||
fn nameString(self: *Cursor) Error!NamePath {
|
|
||||||
var np = NamePath{};
|
|
||||||
if (self.peek() == op.root_char) {
|
|
||||||
np.rooted = true;
|
|
||||||
self.i += 1;
|
|
||||||
} else {
|
|
||||||
while (self.peek() == op.parent_prefix_char) : (self.i += 1) np.parents += 1;
|
|
||||||
}
|
|
||||||
const lead = self.peek() orelse return np;
|
|
||||||
switch (lead) {
|
|
||||||
0x00 => self.i += 1,
|
|
||||||
op.dual_name_prefix => {
|
|
||||||
self.i += 1;
|
|
||||||
try self.seg(&np);
|
|
||||||
try self.seg(&np);
|
|
||||||
},
|
|
||||||
op.multi_name_prefix => {
|
|
||||||
self.i += 1;
|
|
||||||
const cnt = try self.byte();
|
|
||||||
var k: usize = 0;
|
|
||||||
while (k < cnt) : (k += 1) try self.seg(&np);
|
|
||||||
},
|
|
||||||
else => try self.seg(&np),
|
|
||||||
}
|
|
||||||
return np;
|
|
||||||
}
|
|
||||||
fn seg(self: *Cursor, np: *NamePath) Error!void {
|
|
||||||
const s = try self.take(4);
|
|
||||||
if (np.count < max_segs) {
|
|
||||||
np.segs[np.count] = s[0..4].*;
|
|
||||||
np.count += 1;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
const Frame = struct {
|
|
||||||
args: [7]Object = .{.uninitialized} ** 7,
|
|
||||||
locals: [8]Object = .{.uninitialized} ** 8,
|
|
||||||
scope: *Node,
|
|
||||||
ret: Object = .uninitialized,
|
|
||||||
returned: bool = false,
|
|
||||||
broke: bool = false,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// A CreateField binding: a name that indexes into a buffer object.
|
|
||||||
const BufField = struct { buf: *Node, byte_off: usize, bit_width: u32 };
|
|
||||||
|
|
||||||
pub const Interp = struct {
|
|
||||||
ns: *Namespace,
|
|
||||||
hal: Hal,
|
|
||||||
arena: std.mem.Allocator,
|
|
||||||
/// Runtime object overrides for Name nodes (Store targets, patched buffers).
|
|
||||||
dyn: std.AutoHashMapUnmanaged(*Node, Object) = .{},
|
|
||||||
/// CreateField bindings active for the current evaluation.
|
|
||||||
fields: std.AutoHashMapUnmanaged(*Node, BufField) = .{},
|
|
||||||
|
|
||||||
pub fn init(ns: *Namespace, hal: Hal, arena: std.mem.Allocator) Interp {
|
|
||||||
return .{ .ns = ns, .hal = hal, .arena = arena };
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Evaluate a namespace object: invoke a Method, read a Name's value, or read a
|
|
||||||
/// Field. Resets per-evaluation runtime state first.
|
|
||||||
pub fn evaluate(self: *Interp, node: *Node, args: []const Object) Error!Object {
|
|
||||||
self.dyn.clearRetainingCapacity();
|
|
||||||
self.fields.clearRetainingCapacity();
|
|
||||||
return self.invoke(node, args);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn invoke(self: *Interp, node: *Node, args: []const Object) Error!Object {
|
|
||||||
switch (node.kind) {
|
|
||||||
.method => {
|
|
||||||
var frame = Frame{ .scope = node };
|
|
||||||
for (args, 0..) |a, i| {
|
|
||||||
if (i < frame.args.len) frame.args[i] = a;
|
|
||||||
}
|
|
||||||
var cur = Cursor{ .b = node.value };
|
|
||||||
try self.execList(&cur, &frame);
|
|
||||||
return frame.ret;
|
|
||||||
},
|
|
||||||
.name => {
|
|
||||||
if (self.dyn.get(node)) |o| return o;
|
|
||||||
var cur = Cursor{ .b = node.value };
|
|
||||||
var frame = Frame{ .scope = node.parent orelse self.ns.root };
|
|
||||||
return self.term(&cur, &frame);
|
|
||||||
},
|
|
||||||
.field => return .{ .integer = try self.readField(node) },
|
|
||||||
else => return .{ .reference = node },
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Execute a TermList until it ends or the frame returns/breaks.
|
|
||||||
fn execList(self: *Interp, cur: *Cursor, frame: *Frame) Error!void {
|
|
||||||
while (!cur.eof() and !frame.returned and !frame.broke) {
|
|
||||||
_ = try self.term(cur, frame);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Evaluate/execute one term, returning its value (`.uninitialized` for pure
|
|
||||||
/// statements).
|
|
||||||
fn term(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const lead = cur.peek() orelse return error.Truncated;
|
|
||||||
if (isNameStart(lead)) return self.nameRef(cur, frame);
|
|
||||||
_ = try cur.byte();
|
|
||||||
|
|
||||||
return switch (lead) {
|
|
||||||
op.zero_op => Object{ .integer = 0 },
|
|
||||||
op.one_op => Object{ .integer = 1 },
|
|
||||||
op.ones_op => Object{ .integer = ~@as(u64, 0) },
|
|
||||||
op.byte_prefix => Object{ .integer = try self.readConst(cur, 1) },
|
|
||||||
op.word_prefix => Object{ .integer = try self.readConst(cur, 2) },
|
|
||||||
op.dword_prefix => Object{ .integer = try self.readConst(cur, 4) },
|
|
||||||
op.qword_prefix => Object{ .integer = try self.readConst(cur, 8) },
|
|
||||||
op.string_prefix => try self.readString(cur),
|
|
||||||
op.buffer_op => try self.buffer(cur, frame),
|
|
||||||
op.package_op, op.var_package_op => try self.package(cur, frame, lead == op.var_package_op),
|
|
||||||
|
|
||||||
op.local0_op...op.local7_op => frame.locals[lead - op.local0_op],
|
|
||||||
op.arg0_op...op.arg6_op => frame.args[lead - op.arg0_op],
|
|
||||||
|
|
||||||
op.return_op => blk: {
|
|
||||||
frame.ret = try self.term(cur, frame);
|
|
||||||
frame.returned = true;
|
|
||||||
break :blk .uninitialized;
|
|
||||||
},
|
|
||||||
op.break_op => blk: {
|
|
||||||
frame.broke = true;
|
|
||||||
break :blk .uninitialized;
|
|
||||||
},
|
|
||||||
op.continue_op, op.noop_op => .uninitialized,
|
|
||||||
|
|
||||||
op.if_op => try self.ifElse(cur, frame),
|
|
||||||
op.while_op => try self.whileLoop(cur, frame),
|
|
||||||
op.store_op => try self.store(cur, frame),
|
|
||||||
op.increment_op => try self.incDec(cur, frame, 1),
|
|
||||||
op.decrement_op => try self.incDec(cur, frame, -1),
|
|
||||||
|
|
||||||
op.add_op => try self.binary(cur, frame, .add),
|
|
||||||
op.subtract_op => try self.binary(cur, frame, .sub),
|
|
||||||
op.multiply_op => try self.binary(cur, frame, .mul),
|
|
||||||
op.mod_op => try self.binary(cur, frame, .mod),
|
|
||||||
op.and_op => try self.binary(cur, frame, .band),
|
|
||||||
op.or_op => try self.binary(cur, frame, .bor),
|
|
||||||
op.xor_op => try self.binary(cur, frame, .bxor),
|
|
||||||
op.nand_op => try self.binary(cur, frame, .nand),
|
|
||||||
op.nor_op => try self.binary(cur, frame, .nor),
|
|
||||||
op.shift_left_op => try self.binary(cur, frame, .shl),
|
|
||||||
op.shift_right_op => try self.binary(cur, frame, .shr),
|
|
||||||
op.divide_op => try self.divide(cur, frame),
|
|
||||||
|
|
||||||
op.land_op => try self.logic2(cur, frame, .land),
|
|
||||||
op.lor_op => try self.logic2(cur, frame, .lor),
|
|
||||||
op.lequal_op => try self.logic2(cur, frame, .eq),
|
|
||||||
op.lgreater_op => try self.logic2(cur, frame, .gt),
|
|
||||||
op.lless_op => try self.logic2(cur, frame, .lt),
|
|
||||||
op.lnot_op => try self.lnot(cur, frame),
|
|
||||||
|
|
||||||
op.not_op => blk: {
|
|
||||||
const v = try self.evalInt(cur, frame);
|
|
||||||
const r = ~v;
|
|
||||||
try self.storeTarget(cur, frame, .{ .integer = r });
|
|
||||||
break :blk .{ .integer = r };
|
|
||||||
},
|
|
||||||
|
|
||||||
op.size_of_op => try self.sizeOf(cur, frame),
|
|
||||||
op.index_op => try self.index(cur, frame),
|
|
||||||
op.deref_of_op => try self.derefOf(cur, frame),
|
|
||||||
op.to_integer_op => blk: {
|
|
||||||
const v = try self.evalInt(cur, frame);
|
|
||||||
try self.storeTarget(cur, frame, .{ .integer = v });
|
|
||||||
break :blk .{ .integer = v };
|
|
||||||
},
|
|
||||||
op.to_buffer_op => try self.passThroughUnary(cur, frame),
|
|
||||||
|
|
||||||
op.ext_op_prefix => try self.ext(cur, frame),
|
|
||||||
|
|
||||||
// CreateXField: source, index, name (bit widths differ by op)
|
|
||||||
op.create_bit_field_op => try self.createField(cur, frame, 1),
|
|
||||||
op.create_byte_field_op => try self.createField(cur, frame, 8),
|
|
||||||
op.create_word_field_op => try self.createField(cur, frame, 16),
|
|
||||||
op.create_dword_field_op => try self.createField(cur, frame, 32),
|
|
||||||
op.create_qword_field_op => try self.createField(cur, frame, 64),
|
|
||||||
|
|
||||||
else => error.Unsupported,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- name references ----------------------------------------------------
|
|
||||||
|
|
||||||
fn nameRef(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const np = try cur.nameString();
|
|
||||||
const node = self.ns.resolve(frame.scope, np.rooted, np.parents, np.slice()) orelse
|
|
||||||
return .uninitialized; // unknown name -> treat as uninitialised
|
|
||||||
switch (node.kind) {
|
|
||||||
.method => {
|
|
||||||
var argbuf: [7]Object = undefined;
|
|
||||||
var i: usize = 0;
|
|
||||||
while (i < node.arg_count and i < argbuf.len) : (i += 1) argbuf[i] = try self.term(cur, frame);
|
|
||||||
return self.invoke(node, argbuf[0..@min(node.arg_count, argbuf.len)]);
|
|
||||||
},
|
|
||||||
.field => return .{ .integer = try self.readField(node) },
|
|
||||||
.name => return self.invoke(node, &.{}),
|
|
||||||
else => return .{ .reference = node },
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- data objects -------------------------------------------------------
|
|
||||||
|
|
||||||
fn readConst(self: *Interp, cur: *Cursor, n: usize) Error!u64 {
|
|
||||||
_ = self;
|
|
||||||
const bytes = try cur.take(n);
|
|
||||||
var v: u64 = 0;
|
|
||||||
for (bytes, 0..) |b, i| v |= @as(u64, b) << @intCast(i * 8);
|
|
||||||
return v;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn readString(self: *Interp, cur: *Cursor) Error!Object {
|
|
||||||
const start = cur.i;
|
|
||||||
while (cur.peek()) |c| {
|
|
||||||
cur.i += 1;
|
|
||||||
if (c == 0) break;
|
|
||||||
}
|
|
||||||
const raw = cur.b[start .. cur.i - 1];
|
|
||||||
const s = try self.arena.dupe(u8, raw);
|
|
||||||
return .{ .string = s };
|
|
||||||
}
|
|
||||||
|
|
||||||
fn buffer(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const start = cur.i;
|
|
||||||
const len = try cur.pkgLen();
|
|
||||||
const end = @min(start + len, cur.b.len);
|
|
||||||
const size = try self.evalInt(cur, frame);
|
|
||||||
const data = cur.b[@min(cur.i, end)..end];
|
|
||||||
const buf = try self.arena.alloc(u8, @intCast(size));
|
|
||||||
@memset(buf, 0);
|
|
||||||
@memcpy(buf[0..@min(buf.len, data.len)], data[0..@min(buf.len, data.len)]);
|
|
||||||
cur.i = end;
|
|
||||||
return .{ .buffer = buf };
|
|
||||||
}
|
|
||||||
|
|
||||||
fn package(self: *Interp, cur: *Cursor, frame: *Frame, variable: bool) Error!Object {
|
|
||||||
const start = cur.i;
|
|
||||||
const len = try cur.pkgLen();
|
|
||||||
const end = @min(start + len, cur.b.len);
|
|
||||||
const count: usize = if (variable) @intCast(try self.evalInt(cur, frame)) else try cur.byte();
|
|
||||||
const elems = try self.arena.alloc(Object, count);
|
|
||||||
var i: usize = 0;
|
|
||||||
while (i < count and cur.i < end) : (i += 1) elems[i] = try self.term(cur, frame);
|
|
||||||
while (i < count) : (i += 1) elems[i] = .uninitialized;
|
|
||||||
cur.i = end;
|
|
||||||
return .{ .package = elems };
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- operators ----------------------------------------------------------
|
|
||||||
|
|
||||||
const BinOp = enum { add, sub, mul, mod, band, bor, bxor, nand, nor, shl, shr };
|
|
||||||
|
|
||||||
fn binary(self: *Interp, cur: *Cursor, frame: *Frame, kind: BinOp) Error!Object {
|
|
||||||
const a = try self.evalInt(cur, frame);
|
|
||||||
const b = try self.evalInt(cur, frame);
|
|
||||||
const r: u64 = switch (kind) {
|
|
||||||
.add => a +% b,
|
|
||||||
.sub => a -% b,
|
|
||||||
.mul => a *% b,
|
|
||||||
.mod => if (b == 0) return error.DivByZero else a % b,
|
|
||||||
.band => a & b,
|
|
||||||
.bor => a | b,
|
|
||||||
.bxor => a ^ b,
|
|
||||||
.nand => ~(a & b),
|
|
||||||
.nor => ~(a | b),
|
|
||||||
.shl => if (b >= 64) 0 else a << @intCast(b),
|
|
||||||
.shr => if (b >= 64) 0 else a >> @intCast(b),
|
|
||||||
};
|
|
||||||
try self.storeTarget(cur, frame, .{ .integer = r });
|
|
||||||
return .{ .integer = r };
|
|
||||||
}
|
|
||||||
|
|
||||||
fn divide(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const a = try self.evalInt(cur, frame);
|
|
||||||
const b = try self.evalInt(cur, frame);
|
|
||||||
if (b == 0) return error.DivByZero;
|
|
||||||
try self.storeTarget(cur, frame, .{ .integer = a % b }); // remainder target
|
|
||||||
try self.storeTarget(cur, frame, .{ .integer = a / b }); // quotient target
|
|
||||||
return .{ .integer = a / b };
|
|
||||||
}
|
|
||||||
|
|
||||||
const LogicOp = enum { land, lor, eq, gt, lt };
|
|
||||||
|
|
||||||
fn logic2(self: *Interp, cur: *Cursor, frame: *Frame, kind: LogicOp) Error!Object {
|
|
||||||
const a = try self.evalInt(cur, frame);
|
|
||||||
const b = try self.evalInt(cur, frame);
|
|
||||||
const r = switch (kind) {
|
|
||||||
.land => a != 0 and b != 0,
|
|
||||||
.lor => a != 0 or b != 0,
|
|
||||||
.eq => a == b,
|
|
||||||
.gt => a > b,
|
|
||||||
.lt => a < b,
|
|
||||||
};
|
|
||||||
return .{ .integer = if (r) ~@as(u64, 0) else 0 };
|
|
||||||
}
|
|
||||||
|
|
||||||
fn lnot(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
// 0x92 0x93/94/95 are the compound comparisons.
|
|
||||||
const b = cur.peek() orelse return error.Truncated;
|
|
||||||
switch (b) {
|
|
||||||
op.lnot.not_equal => {
|
|
||||||
cur.i += 1;
|
|
||||||
const x = try self.evalInt(cur, frame);
|
|
||||||
const y = try self.evalInt(cur, frame);
|
|
||||||
return .{ .integer = if (x != y) ~@as(u64, 0) else 0 };
|
|
||||||
},
|
|
||||||
op.lnot.less_equal => {
|
|
||||||
cur.i += 1;
|
|
||||||
const x = try self.evalInt(cur, frame);
|
|
||||||
const y = try self.evalInt(cur, frame);
|
|
||||||
return .{ .integer = if (x <= y) ~@as(u64, 0) else 0 };
|
|
||||||
},
|
|
||||||
op.lnot.greater_equal => {
|
|
||||||
cur.i += 1;
|
|
||||||
const x = try self.evalInt(cur, frame);
|
|
||||||
const y = try self.evalInt(cur, frame);
|
|
||||||
return .{ .integer = if (x >= y) ~@as(u64, 0) else 0 };
|
|
||||||
},
|
|
||||||
else => {
|
|
||||||
const x = try self.evalInt(cur, frame);
|
|
||||||
return .{ .integer = if (x == 0) ~@as(u64, 0) else 0 };
|
|
||||||
},
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn incDec(self: *Interp, cur: *Cursor, frame: *Frame, delta: i64) Error!Object {
|
|
||||||
// Operand is a SuperName that is both read and written.
|
|
||||||
const save = cur.i;
|
|
||||||
const cur_val = try self.term(cur, frame);
|
|
||||||
const v = try cur_val.asInt();
|
|
||||||
const r = if (delta > 0) v +% 1 else v -% 1;
|
|
||||||
var tcur = Cursor{ .b = cur.b, .i = save };
|
|
||||||
try self.storeInto(&tcur, frame, .{ .integer = r });
|
|
||||||
return .{ .integer = r };
|
|
||||||
}
|
|
||||||
|
|
||||||
fn sizeOf(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const o = try self.term(cur, frame);
|
|
||||||
return .{ .integer = switch (o) {
|
|
||||||
.buffer => |b| b.len,
|
|
||||||
.string => |s| s.len,
|
|
||||||
.package => |p| p.len,
|
|
||||||
else => 0,
|
|
||||||
} };
|
|
||||||
}
|
|
||||||
|
|
||||||
fn passThroughUnary(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const o = try self.term(cur, frame);
|
|
||||||
try self.storeTarget(cur, frame, o);
|
|
||||||
return o;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn index(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const src = try self.term(cur, frame);
|
|
||||||
const idx: usize = @intCast(try self.evalInt(cur, frame));
|
|
||||||
// Optional target (a reference); we don't materialise references, so store
|
|
||||||
// the indexed value if a target is present.
|
|
||||||
const val: Object = switch (src) {
|
|
||||||
.buffer => |b| .{ .integer = if (idx < b.len) b[idx] else 0 },
|
|
||||||
.package => |p| if (idx < p.len) p[idx] else .uninitialized,
|
|
||||||
.string => |s| .{ .integer = if (idx < s.len) s[idx] else 0 },
|
|
||||||
else => .uninitialized,
|
|
||||||
};
|
|
||||||
try self.storeTarget(cur, frame, val);
|
|
||||||
return val;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn derefOf(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const o = try self.term(cur, frame);
|
|
||||||
return switch (o) {
|
|
||||||
.reference => |n| self.invoke(n, &.{}),
|
|
||||||
else => o,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- control flow -------------------------------------------------------
|
|
||||||
|
|
||||||
fn ifElse(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const start = cur.i;
|
|
||||||
const end = @min(start + try cur.pkgLen(), cur.b.len);
|
|
||||||
const cond = try self.evalInt(cur, frame);
|
|
||||||
if (cond != 0) {
|
|
||||||
var body = Cursor{ .b = cur.b[0..end], .i = cur.i };
|
|
||||||
try self.execList(&body, frame);
|
|
||||||
cur.i = end;
|
|
||||||
// Skip a trailing Else.
|
|
||||||
if (cur.peek() == op.else_op) {
|
|
||||||
cur.i += 1;
|
|
||||||
const es = cur.i;
|
|
||||||
const ee = @min(es + try cur.pkgLen(), cur.b.len);
|
|
||||||
cur.i = ee;
|
|
||||||
}
|
|
||||||
} else {
|
|
||||||
cur.i = end;
|
|
||||||
if (cur.peek() == op.else_op) {
|
|
||||||
cur.i += 1;
|
|
||||||
const es = cur.i;
|
|
||||||
const ee = @min(es + try cur.pkgLen(), cur.b.len);
|
|
||||||
var body = Cursor{ .b = cur.b[0..ee], .i = cur.i };
|
|
||||||
try self.execList(&body, frame);
|
|
||||||
cur.i = ee;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return .uninitialized;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn whileLoop(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const start = cur.i;
|
|
||||||
const end = @min(start + try cur.pkgLen(), cur.b.len);
|
|
||||||
const pred_at = cur.i;
|
|
||||||
var guard: usize = 0;
|
|
||||||
while (guard < 100_000) : (guard += 1) {
|
|
||||||
var pc = Cursor{ .b = cur.b[0..end], .i = pred_at };
|
|
||||||
const cond = try self.evalInt(&pc, frame);
|
|
||||||
if (cond == 0) break;
|
|
||||||
var body = Cursor{ .b = cur.b[0..end], .i = pc.i };
|
|
||||||
try self.execList(&body, frame);
|
|
||||||
if (frame.returned) break;
|
|
||||||
if (frame.broke) {
|
|
||||||
frame.broke = false;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
cur.i = end;
|
|
||||||
return .uninitialized;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- store --------------------------------------------------------------
|
|
||||||
|
|
||||||
fn store(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const value = try self.term(cur, frame);
|
|
||||||
try self.storeInto(cur, frame, value);
|
|
||||||
return value;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// A Store *target* that may be NullName (no store).
|
|
||||||
fn storeTarget(self: *Interp, cur: *Cursor, frame: *Frame, value: Object) Error!void {
|
|
||||||
if (cur.peek() == 0x00) {
|
|
||||||
cur.i += 1; // NullName
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
try self.storeInto(cur, frame, value);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn storeInto(self: *Interp, cur: *Cursor, frame: *Frame, value: Object) Error!void {
|
|
||||||
const lead = cur.peek() orelse return error.Truncated;
|
|
||||||
if (isNameStart(lead)) {
|
|
||||||
const np = try cur.nameString();
|
|
||||||
const node = self.ns.resolve(frame.scope, np.rooted, np.parents, np.slice()) orelse return;
|
|
||||||
if (self.fields.get(node)) |bf| {
|
|
||||||
try self.writeBufField(bf, try value.asInt());
|
|
||||||
} else if (node.kind == .field) {
|
|
||||||
try self.writeField(node, try value.asInt());
|
|
||||||
} else {
|
|
||||||
try self.dyn.put(self.arena, node, value);
|
|
||||||
}
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
_ = try cur.byte();
|
|
||||||
switch (lead) {
|
|
||||||
0x00 => {}, // NullName
|
|
||||||
op.local0_op...op.local7_op => frame.locals[lead - op.local0_op] = value,
|
|
||||||
op.arg0_op...op.arg6_op => frame.args[lead - op.arg0_op] = value,
|
|
||||||
op.index_op => {
|
|
||||||
const src = try self.term(cur, frame);
|
|
||||||
const idx: usize = @intCast(try self.evalInt(cur, frame));
|
|
||||||
switch (src) {
|
|
||||||
.buffer => |b| if (idx < b.len) {
|
|
||||||
b[idx] = @truncate(try value.asInt());
|
|
||||||
},
|
|
||||||
.package => |p| if (idx < p.len) {
|
|
||||||
p[idx] = value;
|
|
||||||
},
|
|
||||||
else => {},
|
|
||||||
}
|
|
||||||
},
|
|
||||||
else => return error.Unsupported,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- CreateField (buffer patching) --------------------------------------
|
|
||||||
|
|
||||||
fn createField(self: *Interp, cur: *Cursor, frame: *Frame, bit_width: u32) Error!Object {
|
|
||||||
const src = try self.term(cur, frame); // source buffer (as a reference or value)
|
|
||||||
const bit_index = try self.evalInt(cur, frame);
|
|
||||||
const np = try cur.nameString();
|
|
||||||
const node = self.ns.resolve(frame.scope, np.rooted, np.parents, np.slice()) orelse return .uninitialized;
|
|
||||||
|
|
||||||
// Bind the new name to the source buffer's node so stores land in it.
|
|
||||||
const buf_node: *Node = switch (src) {
|
|
||||||
.reference => |n| n,
|
|
||||||
else => return .uninitialized,
|
|
||||||
};
|
|
||||||
// Materialise the buffer into `dyn` so patches persist and are returned.
|
|
||||||
if (self.dyn.get(buf_node) == null) {
|
|
||||||
const val = try self.invoke(buf_node, &.{});
|
|
||||||
try self.dyn.put(self.arena, buf_node, val);
|
|
||||||
}
|
|
||||||
const byte_off: usize = @intCast(bit_index / 8);
|
|
||||||
try self.fields.put(self.arena, node, .{ .buf = buf_node, .byte_off = byte_off, .bit_width = bit_width });
|
|
||||||
return .uninitialized;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn writeBufField(self: *Interp, bf: BufField, value: u64) Error!void {
|
|
||||||
const obj = self.dyn.get(bf.buf) orelse return;
|
|
||||||
const buf = switch (obj) {
|
|
||||||
.buffer => |b| b,
|
|
||||||
else => return,
|
|
||||||
};
|
|
||||||
const nbytes = (bf.bit_width + 7) / 8;
|
|
||||||
var k: usize = 0;
|
|
||||||
while (k < nbytes and bf.byte_off + k < buf.len) : (k += 1) {
|
|
||||||
buf[bf.byte_off + k] = @truncate(value >> @intCast(k * 8));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- OperationRegion field access ---------------------------------------
|
|
||||||
|
|
||||||
fn readField(self: *Interp, field: *Node) Error!u64 {
|
|
||||||
const region = field.region orelse return error.Unsupported;
|
|
||||||
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
|
||||||
const base = try self.regionBase(region);
|
|
||||||
const start_byte = base + field.bit_offset / 8;
|
|
||||||
const shift: u7 = @intCast(field.bit_offset % 8);
|
|
||||||
const total = @as(usize, shift) + field.bit_width;
|
|
||||||
const nbytes = (total + 7) / 8;
|
|
||||||
var raw: u128 = 0;
|
|
||||||
var k: usize = 0;
|
|
||||||
while (k < nbytes) : (k += 1) {
|
|
||||||
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
|
||||||
}
|
|
||||||
const masked = (raw >> shift) & bitMask(field.bit_width);
|
|
||||||
return @truncate(masked);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn writeField(self: *Interp, field: *Node, value: u64) Error!void {
|
|
||||||
const region = field.region orelse return error.Unsupported;
|
|
||||||
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
|
||||||
const base = try self.regionBase(region);
|
|
||||||
const start_byte = base + field.bit_offset / 8;
|
|
||||||
const shift: u7 = @intCast(field.bit_offset % 8);
|
|
||||||
const total = @as(usize, shift) + field.bit_width;
|
|
||||||
const nbytes = (total + 7) / 8;
|
|
||||||
// Read-modify-write byte by byte.
|
|
||||||
var raw: u128 = 0;
|
|
||||||
var k: usize = 0;
|
|
||||||
while (k < nbytes) : (k += 1) {
|
|
||||||
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
|
||||||
}
|
|
||||||
const mask = bitMask(field.bit_width) << shift;
|
|
||||||
raw = (raw & ~mask) | ((@as(u128, value) << shift) & mask);
|
|
||||||
k = 0;
|
|
||||||
while (k < nbytes) : (k += 1) {
|
|
||||||
try self.writeRegionByte(region.region_space, start_byte + k, @truncate(raw >> @intCast(k * 8)));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn regionBase(self: *Interp, region: *Node) Error!u64 {
|
|
||||||
var cur = Cursor{ .b = region.region_offset_aml };
|
|
||||||
var frame = Frame{ .scope = region.parent orelse self.ns.root };
|
|
||||||
return (try self.term(&cur, &frame)).asInt();
|
|
||||||
}
|
|
||||||
|
|
||||||
fn readRegionByte(self: *Interp, space: u8, addr: u64) Error!u8 {
|
|
||||||
switch (space) {
|
|
||||||
0 => { // SystemMemory
|
|
||||||
self.hal.mapMmio(addr & ~@as(u64, 0xFFF), addr & ~@as(u64, 0xFFF), true);
|
|
||||||
const p: *align(1) const volatile u8 = @ptrFromInt(addr);
|
|
||||||
return p.*;
|
|
||||||
},
|
|
||||||
1 => return @truncate(self.hal.pioRead(1, @intCast(addr & 0xFFFF))), // SystemIO
|
|
||||||
else => return error.Unsupported,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn writeRegionByte(self: *Interp, space: u8, addr: u64, value: u8) Error!void {
|
|
||||||
switch (space) {
|
|
||||||
0 => {
|
|
||||||
self.hal.mapMmio(addr & ~@as(u64, 0xFFF), addr & ~@as(u64, 0xFFF), true);
|
|
||||||
const p: *align(1) volatile u8 = @ptrFromInt(addr);
|
|
||||||
p.* = value;
|
|
||||||
},
|
|
||||||
1 => self.hal.pioWrite(1, @intCast(addr & 0xFFFF), value),
|
|
||||||
else => return error.Unsupported,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- extended opcodes ---------------------------------------------------
|
|
||||||
|
|
||||||
fn ext(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const e = try cur.byte();
|
|
||||||
switch (e) {
|
|
||||||
op.ext.debug => return .uninitialized,
|
|
||||||
op.ext.revision => return .{ .integer = 2 },
|
|
||||||
op.ext.timer => return .{ .integer = 0 },
|
|
||||||
// Mutex/Event ops are no-ops in this single-threaded evaluator.
|
|
||||||
op.ext.acquire => {
|
|
||||||
_ = try self.term(cur, frame); // mutex SuperName
|
|
||||||
_ = try cur.take(2); // timeout
|
|
||||||
return .{ .integer = 0 }; // acquired
|
|
||||||
},
|
|
||||||
op.ext.release, op.ext.reset, op.ext.signal => {
|
|
||||||
_ = try self.term(cur, frame);
|
|
||||||
return .uninitialized;
|
|
||||||
},
|
|
||||||
op.ext.wait => {
|
|
||||||
_ = try self.term(cur, frame);
|
|
||||||
_ = try self.term(cur, frame);
|
|
||||||
return .{ .integer = 0 };
|
|
||||||
},
|
|
||||||
op.ext.sleep, op.ext.stall => {
|
|
||||||
_ = try self.term(cur, frame);
|
|
||||||
return .uninitialized;
|
|
||||||
},
|
|
||||||
else => return error.Unsupported,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn evalInt(self: *Interp, cur: *Cursor, frame: *Frame) Error!u64 {
|
|
||||||
return (try self.term(cur, frame)).asInt();
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
fn bitMask(width: u32) u128 {
|
|
||||||
if (width >= 128) return ~@as(u128, 0);
|
|
||||||
return (@as(u128, 1) << @intCast(width)) - 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn isNameStart(b: u8) bool {
|
|
||||||
return (b >= op.name_char_start and b <= op.name_char_end) or
|
|
||||||
b == op.name_char_underscore or
|
|
||||||
b == op.root_char or
|
|
||||||
b == op.parent_prefix_char or
|
|
||||||
b == op.dual_name_prefix or
|
|
||||||
b == op.multi_name_prefix;
|
|
||||||
}
|
|
||||||
@@ -1,137 +0,0 @@
|
|||||||
//! AML opcode constants — the full ACPI Machine Language opcode table.
|
|
||||||
//!
|
|
||||||
//! Single-byte opcodes are plain values. Extended opcodes are a two-byte sequence
|
|
||||||
//! `ext_prefix` (0x5B) followed by a byte listed under `ext`. A few comparison
|
|
||||||
//! opcodes are `lnot_op` (0x92) followed by a second byte (see `lnot`).
|
|
||||||
|
|
||||||
// --- name / path characters -------------------------------------------------
|
|
||||||
pub const zero_op = 0x00;
|
|
||||||
pub const one_op = 0x01;
|
|
||||||
pub const alias_op = 0x06;
|
|
||||||
pub const name_op = 0x08;
|
|
||||||
pub const byte_prefix = 0x0A;
|
|
||||||
pub const word_prefix = 0x0B;
|
|
||||||
pub const dword_prefix = 0x0C;
|
|
||||||
pub const string_prefix = 0x0D;
|
|
||||||
pub const qword_prefix = 0x0E;
|
|
||||||
pub const scope_op = 0x10;
|
|
||||||
pub const buffer_op = 0x11;
|
|
||||||
pub const package_op = 0x12;
|
|
||||||
pub const var_package_op = 0x13;
|
|
||||||
pub const method_op = 0x14;
|
|
||||||
pub const external_op = 0x15;
|
|
||||||
|
|
||||||
pub const dual_name_prefix = 0x2E;
|
|
||||||
pub const multi_name_prefix = 0x2F;
|
|
||||||
pub const ext_op_prefix = 0x5B;
|
|
||||||
pub const root_char = 0x5C;
|
|
||||||
pub const parent_prefix_char = 0x5E;
|
|
||||||
pub const name_char_underscore = 0x5F;
|
|
||||||
|
|
||||||
pub const digit_char_start = 0x30;
|
|
||||||
pub const digit_char_end = 0x39;
|
|
||||||
pub const name_char_start = 0x41; // 'A'
|
|
||||||
pub const name_char_end = 0x5A; // 'Z'
|
|
||||||
|
|
||||||
// --- locals / args ----------------------------------------------------------
|
|
||||||
pub const local0_op = 0x60;
|
|
||||||
pub const local7_op = 0x67;
|
|
||||||
pub const arg0_op = 0x68;
|
|
||||||
pub const arg6_op = 0x6E;
|
|
||||||
|
|
||||||
// --- store / references / arithmetic ---------------------------------------
|
|
||||||
pub const store_op = 0x70;
|
|
||||||
pub const ref_of_op = 0x71;
|
|
||||||
pub const add_op = 0x72;
|
|
||||||
pub const concat_op = 0x73;
|
|
||||||
pub const subtract_op = 0x74;
|
|
||||||
pub const increment_op = 0x75;
|
|
||||||
pub const decrement_op = 0x76;
|
|
||||||
pub const multiply_op = 0x77;
|
|
||||||
pub const divide_op = 0x78;
|
|
||||||
pub const shift_left_op = 0x79;
|
|
||||||
pub const shift_right_op = 0x7A;
|
|
||||||
pub const and_op = 0x7B;
|
|
||||||
pub const nand_op = 0x7C;
|
|
||||||
pub const or_op = 0x7D;
|
|
||||||
pub const nor_op = 0x7E;
|
|
||||||
pub const xor_op = 0x7F;
|
|
||||||
pub const not_op = 0x80;
|
|
||||||
pub const find_set_left_bit_op = 0x81;
|
|
||||||
pub const find_set_right_bit_op = 0x82;
|
|
||||||
pub const deref_of_op = 0x83;
|
|
||||||
pub const concat_res_op = 0x84;
|
|
||||||
pub const mod_op = 0x85;
|
|
||||||
pub const notify_op = 0x86;
|
|
||||||
pub const size_of_op = 0x87;
|
|
||||||
pub const index_op = 0x88;
|
|
||||||
pub const match_op = 0x89;
|
|
||||||
pub const create_dword_field_op = 0x8A;
|
|
||||||
pub const create_word_field_op = 0x8B;
|
|
||||||
pub const create_byte_field_op = 0x8C;
|
|
||||||
pub const create_bit_field_op = 0x8D;
|
|
||||||
pub const object_type_op = 0x8E;
|
|
||||||
pub const create_qword_field_op = 0x8F;
|
|
||||||
|
|
||||||
pub const land_op = 0x90;
|
|
||||||
pub const lor_op = 0x91;
|
|
||||||
pub const lnot_op = 0x92; // may be followed by a second byte (see `lnot`)
|
|
||||||
pub const lequal_op = 0x93;
|
|
||||||
pub const lgreater_op = 0x94;
|
|
||||||
pub const lless_op = 0x95;
|
|
||||||
pub const to_buffer_op = 0x96;
|
|
||||||
pub const to_decimal_string_op = 0x97;
|
|
||||||
pub const to_hex_string_op = 0x98;
|
|
||||||
pub const to_integer_op = 0x99;
|
|
||||||
pub const to_string_op = 0x9C;
|
|
||||||
pub const copy_object_op = 0x9D;
|
|
||||||
pub const mid_op = 0x9E;
|
|
||||||
pub const continue_op = 0x9F;
|
|
||||||
pub const if_op = 0xA0;
|
|
||||||
pub const else_op = 0xA1;
|
|
||||||
pub const while_op = 0xA2;
|
|
||||||
pub const noop_op = 0xA3;
|
|
||||||
pub const return_op = 0xA4;
|
|
||||||
pub const break_op = 0xA5;
|
|
||||||
pub const break_point_op = 0xCC;
|
|
||||||
pub const ones_op = 0xFF;
|
|
||||||
|
|
||||||
/// Second bytes of the `lnot_op` (0x92) compound comparison opcodes.
|
|
||||||
pub const lnot = struct {
|
|
||||||
pub const not_equal = 0x93; // LNotEqualOp: 0x92 0x93
|
|
||||||
pub const less_equal = 0x94; // LLessEqualOp: 0x92 0x94
|
|
||||||
pub const greater_equal = 0x95; // LGreaterEqualOp: 0x92 0x95
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Second bytes of extended opcodes (prefixed by `ext_op_prefix`, 0x5B).
|
|
||||||
pub const ext = struct {
|
|
||||||
pub const mutex = 0x01;
|
|
||||||
pub const event = 0x02;
|
|
||||||
pub const cond_ref_of = 0x12;
|
|
||||||
pub const create_field = 0x13;
|
|
||||||
pub const load_table = 0x1F;
|
|
||||||
pub const load = 0x20;
|
|
||||||
pub const stall = 0x21;
|
|
||||||
pub const sleep = 0x22;
|
|
||||||
pub const acquire = 0x23;
|
|
||||||
pub const signal = 0x24;
|
|
||||||
pub const wait = 0x25;
|
|
||||||
pub const reset = 0x26;
|
|
||||||
pub const release = 0x27;
|
|
||||||
pub const from_bcd = 0x28;
|
|
||||||
pub const to_bcd = 0x29;
|
|
||||||
pub const unload = 0x2A;
|
|
||||||
pub const revision = 0x30;
|
|
||||||
pub const debug = 0x31;
|
|
||||||
pub const fatal = 0x32;
|
|
||||||
pub const timer = 0x33;
|
|
||||||
pub const op_region = 0x80;
|
|
||||||
pub const field = 0x81;
|
|
||||||
pub const device = 0x82;
|
|
||||||
pub const processor = 0x83;
|
|
||||||
pub const power_res = 0x84;
|
|
||||||
pub const thermal_zone = 0x85;
|
|
||||||
pub const index_field = 0x86;
|
|
||||||
pub const bank_field = 0x87;
|
|
||||||
pub const data_region = 0x88;
|
|
||||||
};
|
|
||||||
@@ -1,517 +0,0 @@
|
|||||||
//! Recursive-descent AML parser. Walks the entire byte stream — including method
|
|
||||||
//! bodies — building the ACPI namespace as it goes. It does not *evaluate*
|
|
||||||
//! anything (no OperationRegion reads, no arithmetic); it parses structure so the
|
|
||||||
//! cursor stays aligned and every named object is recorded.
|
|
||||||
//!
|
|
||||||
//! The one genuine ambiguity in AML is method invocation: a bare NameString in an
|
|
||||||
//! operand position is a call whose argument count is only known from the method's
|
|
||||||
//! (earlier) declaration. Because we build the namespace in the same in-order pass,
|
|
||||||
//! `resolve` finds that declaration and tells us how many operands to consume.
|
|
||||||
//!
|
|
||||||
//! Safety net: every object delimited by a PkgLength (Scope/Device/Method/If/While/
|
|
||||||
//! Field/Buffer/Package/…) is parsed within its known extent, and the cursor is
|
|
||||||
//! snapped to that extent afterwards. So a mis-resolved invocation can only desync
|
|
||||||
//! *within* one such object; the enclosing walk realigns at the boundary.
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
const op = @import("opcodes.zig");
|
|
||||||
const ns = @import("namespace.zig");
|
|
||||||
const Namespace = ns.Namespace;
|
|
||||||
const Node = ns.Node;
|
|
||||||
|
|
||||||
pub const Error = error{ Truncated, Malformed } || std.mem.Allocator.Error;
|
|
||||||
|
|
||||||
const max_segs = 64;
|
|
||||||
|
|
||||||
/// A parsed NameString: an optional root anchor or some parent hops, then a list
|
|
||||||
/// of 4-byte segments.
|
|
||||||
const NamePath = struct {
|
|
||||||
rooted: bool = false,
|
|
||||||
parents: u8 = 0,
|
|
||||||
segs: [max_segs][4]u8 = undefined,
|
|
||||||
count: usize = 0,
|
|
||||||
|
|
||||||
fn slice(self: *const NamePath) []const [4]u8 {
|
|
||||||
return self.segs[0..self.count];
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
pub const Parser = struct {
|
|
||||||
aml: []const u8,
|
|
||||||
pos: usize = 0,
|
|
||||||
namespace: *Namespace,
|
|
||||||
|
|
||||||
pub fn init(aml: []const u8, namespace: *Namespace) Parser {
|
|
||||||
return .{ .aml = aml, .namespace = namespace };
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Parse the whole block as a TermList under the namespace root. Returns the
|
|
||||||
/// number of bytes consumed — equal to `aml.len` for a clean full traversal.
|
|
||||||
pub fn parseAll(self: *Parser) usize {
|
|
||||||
self.termList(self.aml.len, self.namespace.root);
|
|
||||||
return self.pos;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- cursor primitives --------------------------------------------------
|
|
||||||
|
|
||||||
fn eof(self: *Parser) bool {
|
|
||||||
return self.pos >= self.aml.len;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn peek(self: *Parser) ?u8 {
|
|
||||||
return if (self.eof()) null else self.aml[self.pos];
|
|
||||||
}
|
|
||||||
|
|
||||||
fn readByte(self: *Parser) Error!u8 {
|
|
||||||
if (self.eof()) return error.Truncated;
|
|
||||||
const b = self.aml[self.pos];
|
|
||||||
self.pos += 1;
|
|
||||||
return b;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn skip(self: *Parser, n: usize) Error!void {
|
|
||||||
if (self.pos + n > self.aml.len) return error.Truncated;
|
|
||||||
self.pos += n;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn skipCString(self: *Parser) Error!void {
|
|
||||||
while (true) {
|
|
||||||
const b = try self.readByte();
|
|
||||||
if (b == 0) return;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// AML PkgLength: the lead byte's top two bits give how many extra bytes
|
|
||||||
/// follow; the value counts from the start of the PkgLength field.
|
|
||||||
fn readPkgLength(self: *Parser) Error!usize {
|
|
||||||
const lead = try self.readByte();
|
|
||||||
const follow: usize = lead >> 6;
|
|
||||||
if (follow == 0) return lead & 0x3F;
|
|
||||||
var value: usize = lead & 0x0F;
|
|
||||||
var i: usize = 0;
|
|
||||||
while (i < follow) : (i += 1) {
|
|
||||||
const b = try self.readByte();
|
|
||||||
value |= @as(usize, b) << @intCast(4 + i * 8);
|
|
||||||
}
|
|
||||||
return value;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn readNameSeg(self: *Parser) Error![4]u8 {
|
|
||||||
if (self.pos + 4 > self.aml.len) return error.Truncated;
|
|
||||||
const seg = self.aml[self.pos..][0..4].*;
|
|
||||||
self.pos += 4;
|
|
||||||
return seg;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn readNameString(self: *Parser) Error!NamePath {
|
|
||||||
var np = NamePath{};
|
|
||||||
// A NameString is either root-anchored or parent-relative, not both.
|
|
||||||
if (self.peek() == op.root_char) {
|
|
||||||
np.rooted = true;
|
|
||||||
self.pos += 1;
|
|
||||||
} else {
|
|
||||||
while (self.peek() == op.parent_prefix_char) : (self.pos += 1) np.parents += 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
const lead = self.peek() orelse return np;
|
|
||||||
switch (lead) {
|
|
||||||
0x00 => self.pos += 1, // NullName
|
|
||||||
op.dual_name_prefix => {
|
|
||||||
self.pos += 1;
|
|
||||||
try self.appendSeg(&np);
|
|
||||||
try self.appendSeg(&np);
|
|
||||||
},
|
|
||||||
op.multi_name_prefix => {
|
|
||||||
self.pos += 1;
|
|
||||||
const cnt = try self.readByte();
|
|
||||||
var i: usize = 0;
|
|
||||||
while (i < cnt) : (i += 1) try self.appendSeg(&np);
|
|
||||||
},
|
|
||||||
else => {
|
|
||||||
if (isNameStart(lead)) try self.appendSeg(&np);
|
|
||||||
},
|
|
||||||
}
|
|
||||||
return np;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn appendSeg(self: *Parser, np: *NamePath) Error!void {
|
|
||||||
const seg = try self.readNameSeg();
|
|
||||||
if (np.count < max_segs) {
|
|
||||||
np.segs[np.count] = seg;
|
|
||||||
np.count += 1;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- term list / object -------------------------------------------------
|
|
||||||
|
|
||||||
/// Parse objects until `end`, then snap to `end`. Any parse error resyncs to
|
|
||||||
/// the boundary rather than propagating — containment for the rare desync.
|
|
||||||
fn termList(self: *Parser, end: usize, scope: *Node) void {
|
|
||||||
while (self.pos < end) {
|
|
||||||
self.object(scope) catch break;
|
|
||||||
}
|
|
||||||
self.pos = end;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Parse exactly one object/term at the cursor. Used for both TermObjs and
|
|
||||||
/// operands (TermArg / SuperName / Target all reduce to "one object" for the
|
|
||||||
/// purpose of advancing the cursor).
|
|
||||||
fn object(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const lead = self.peek() orelse return error.Truncated;
|
|
||||||
if (isNameStart(lead)) return self.nameInvocation(scope);
|
|
||||||
|
|
||||||
_ = try self.readByte();
|
|
||||||
switch (lead) {
|
|
||||||
// constants and no-operand statements
|
|
||||||
op.zero_op, op.one_op, op.ones_op => {},
|
|
||||||
op.noop_op, op.continue_op, op.break_op, op.break_point_op => {},
|
|
||||||
op.local0_op...op.local7_op => {},
|
|
||||||
op.arg0_op...op.arg6_op => {},
|
|
||||||
|
|
||||||
// literal data
|
|
||||||
op.byte_prefix => try self.skip(1),
|
|
||||||
op.word_prefix => try self.skip(2),
|
|
||||||
op.dword_prefix => try self.skip(4),
|
|
||||||
op.qword_prefix => try self.skip(8),
|
|
||||||
op.string_prefix => try self.skipCString(),
|
|
||||||
|
|
||||||
// data containers (contents skipped via their PkgLength)
|
|
||||||
op.buffer_op, op.package_op, op.var_package_op => try self.skipPkg(),
|
|
||||||
|
|
||||||
// namespace modifiers / named objects
|
|
||||||
op.name_op => try self.opName(scope),
|
|
||||||
op.alias_op => try self.opAlias(scope),
|
|
||||||
op.scope_op => try self.opScopeLike(scope, .scope),
|
|
||||||
op.method_op => try self.opMethod(scope),
|
|
||||||
op.external_op => try self.opExternal(scope),
|
|
||||||
op.ext_op_prefix => try self.opExt(scope),
|
|
||||||
|
|
||||||
// control flow
|
|
||||||
op.if_op => try self.opIf(scope),
|
|
||||||
op.else_op => try self.opElse(scope),
|
|
||||||
op.while_op => try self.opWhile(scope),
|
|
||||||
op.return_op => try self.object(scope),
|
|
||||||
op.notify_op => try self.args(scope, 2),
|
|
||||||
|
|
||||||
// stores / references / unary+target
|
|
||||||
op.store_op => try self.args(scope, 2),
|
|
||||||
op.ref_of_op, op.deref_of_op, op.size_of_op, op.object_type_op => try self.args(scope, 1),
|
|
||||||
op.increment_op, op.decrement_op => try self.args(scope, 1),
|
|
||||||
op.not_op, op.find_set_left_bit_op, op.find_set_right_bit_op => try self.args(scope, 2),
|
|
||||||
op.to_buffer_op, op.to_decimal_string_op, op.to_hex_string_op, op.to_integer_op => try self.args(scope, 2),
|
|
||||||
op.copy_object_op => try self.args(scope, 2),
|
|
||||||
|
|
||||||
// binary + target
|
|
||||||
op.add_op, op.subtract_op, op.multiply_op, op.mod_op => try self.args(scope, 3),
|
|
||||||
op.and_op, op.nand_op, op.or_op, op.nor_op, op.xor_op => try self.args(scope, 3),
|
|
||||||
op.shift_left_op, op.shift_right_op, op.concat_op, op.concat_res_op, op.index_op => try self.args(scope, 3),
|
|
||||||
op.divide_op => try self.args(scope, 4),
|
|
||||||
op.to_string_op => try self.args(scope, 3),
|
|
||||||
op.mid_op => try self.args(scope, 4),
|
|
||||||
|
|
||||||
// logical
|
|
||||||
op.land_op, op.lor_op => try self.args(scope, 2),
|
|
||||||
op.lequal_op, op.lgreater_op, op.lless_op => try self.args(scope, 2),
|
|
||||||
op.lnot_op => try self.opLnot(scope),
|
|
||||||
|
|
||||||
op.match_op => try self.opMatch(scope),
|
|
||||||
|
|
||||||
// CreateXField: <source> <index> NameString
|
|
||||||
op.create_dword_field_op,
|
|
||||||
op.create_word_field_op,
|
|
||||||
op.create_byte_field_op,
|
|
||||||
op.create_bit_field_op,
|
|
||||||
op.create_qword_field_op,
|
|
||||||
=> try self.opCreateField(scope, 2),
|
|
||||||
|
|
||||||
else => return error.Malformed,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Parse `n` operands.
|
|
||||||
fn args(self: *Parser, scope: *Node, n: usize) Error!void {
|
|
||||||
var i: usize = 0;
|
|
||||||
while (i < n) : (i += 1) try self.object(scope);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// A NameString in operand/statement position: a method invocation (consuming
|
|
||||||
/// the callee's declared argument count) or a plain name reference.
|
|
||||||
fn nameInvocation(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
if (self.namespace.resolve(scope, np.rooted, np.parents, np.slice())) |node| {
|
|
||||||
if ((node.kind == .method or node.kind == .external) and node.arg_count > 0) {
|
|
||||||
try self.args(scope, node.arg_count);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Skip a PkgLength-delimited body wholesale (Buffer / Package / VarPackage):
|
|
||||||
/// the contents are pure data, never namespace declarations.
|
|
||||||
fn skipPkg(self: *Parser) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const len = try self.readPkgLength();
|
|
||||||
const end = start + len;
|
|
||||||
if (end > self.aml.len) return error.Truncated;
|
|
||||||
self.pos = end;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- namespace objects --------------------------------------------------
|
|
||||||
|
|
||||||
fn opName(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
const val_start = self.pos;
|
|
||||||
try self.object(scope); // the DataRefObject value
|
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .name);
|
|
||||||
node.value = self.aml[val_start..self.pos];
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opAlias(self: *Parser, scope: *Node) Error!void {
|
|
||||||
_ = try self.readNameString(); // source
|
|
||||||
const np = try self.readNameString(); // the alias name
|
|
||||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .alias);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opMethod(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
const np = try self.readNameString();
|
|
||||||
const flags = try self.readByte();
|
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .method);
|
|
||||||
node.arg_count = flags & 0x7;
|
|
||||||
// Capture the body for on-demand evaluation and skip it — objects declared
|
|
||||||
// inside a method are created at *runtime*, not at load, so they must not
|
|
||||||
// become permanent namespace nodes.
|
|
||||||
node.value = self.aml[self.pos..@min(end, self.aml.len)];
|
|
||||||
self.pos = end;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opExternal(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
_ = try self.readByte(); // object type
|
|
||||||
const arg_count = try self.readByte();
|
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .external);
|
|
||||||
node.arg_count = arg_count;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Scope / Device / ThermalZone: PkgLength, NameString, then a nested TermList.
|
|
||||||
fn opScopeLike(self: *Parser, scope: *Node, kind: ns.NodeKind) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
const np = try self.readNameString();
|
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), kind);
|
|
||||||
self.termList(end, node);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opProcessor(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
const np = try self.readNameString();
|
|
||||||
try self.skip(6); // ProcID(byte) + PblkAddr(dword) + PblkLen(byte)
|
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .processor);
|
|
||||||
self.termList(end, node);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opPowerRes(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
const np = try self.readNameString();
|
|
||||||
try self.skip(3); // SystemLevel(byte) + ResourceOrder(word)
|
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .power_res);
|
|
||||||
self.termList(end, node);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// OperationRegion: NameString, RegionSpace(byte), Offset(TermArg), Len(TermArg).
|
|
||||||
/// The offset/length expressions are kept as AML for lazy evaluation.
|
|
||||||
fn opRegion(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
const space = try self.readByte();
|
|
||||||
const off_start = self.pos;
|
|
||||||
try self.object(scope);
|
|
||||||
const off_end = self.pos;
|
|
||||||
try self.object(scope);
|
|
||||||
const len_end = self.pos;
|
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .region);
|
|
||||||
node.region_space = space;
|
|
||||||
node.region_offset_aml = self.aml[off_start..off_end];
|
|
||||||
node.region_len_aml = self.aml[off_end..len_end];
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opDataRegion(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
try self.args(scope, 3); // signature, oem id, oem table id (TermArgs)
|
|
||||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .region);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opMutex(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
try self.skip(1); // sync flags
|
|
||||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .mutex);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opEvent(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .event);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// CreateXField: `count` TermArgs then the new field's NameString.
|
|
||||||
fn opCreateField(self: *Parser, scope: *Node, count: usize) Error!void {
|
|
||||||
try self.args(scope, count);
|
|
||||||
const np = try self.readNameString();
|
|
||||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .name);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Field / IndexField / BankField: a region/bank reference, flags, then a
|
|
||||||
/// FieldList whose NamedFields become nodes in the current scope. For a plain
|
|
||||||
/// Field, the first NameString is the backing region — captured so field units
|
|
||||||
/// carry a region + bit position the evaluator can read/write.
|
|
||||||
fn opField(self: *Parser, scope: *Node, name_strings: u8, bank: bool) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
var region: ?*Node = null;
|
|
||||||
var i: u8 = 0;
|
|
||||||
while (i < name_strings) : (i += 1) {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
// Only a plain Field's single NameString denotes an OperationRegion.
|
|
||||||
if (name_strings == 1) region = self.namespace.resolve(scope, np.rooted, np.parents, np.slice());
|
|
||||||
}
|
|
||||||
if (bank) try self.object(scope); // bank value TermArg
|
|
||||||
const flags = try self.readByte();
|
|
||||||
self.fieldList(end, scope, region, flags & 0x0F);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn fieldList(self: *Parser, end: usize, scope: *Node, region: ?*Node, initial_access: u8) void {
|
|
||||||
var bit_offset: u32 = 0;
|
|
||||||
var access = initial_access;
|
|
||||||
while (self.pos < end) {
|
|
||||||
const lead = self.peek() orelse break;
|
|
||||||
switch (lead) {
|
|
||||||
0x00 => { // ReservedField: advances the bit position
|
|
||||||
self.pos += 1;
|
|
||||||
const width = self.readPkgLength() catch break;
|
|
||||||
bit_offset += @intCast(width);
|
|
||||||
},
|
|
||||||
0x01 => { // AccessField: AccessType (low nibble) + AccessAttrib
|
|
||||||
self.pos += 1;
|
|
||||||
const at = self.readByte() catch break;
|
|
||||||
self.skip(1) catch break;
|
|
||||||
access = at & 0x0F;
|
|
||||||
},
|
|
||||||
0x02 => { // ConnectField: NameString | BufferData
|
|
||||||
self.pos += 1;
|
|
||||||
self.object(scope) catch break;
|
|
||||||
},
|
|
||||||
0x03 => { // ExtendedAccessField: type + attrib + length
|
|
||||||
self.pos += 1;
|
|
||||||
self.skip(3) catch break;
|
|
||||||
},
|
|
||||||
else => { // NamedField: NameSeg + PkgLength (bit width)
|
|
||||||
const seg = self.readNameSeg() catch break;
|
|
||||||
const width = self.readPkgLength() catch break;
|
|
||||||
const unit = self.namespace.newFieldUnit(scope, seg) catch break;
|
|
||||||
unit.region = region;
|
|
||||||
unit.bit_offset = bit_offset;
|
|
||||||
unit.bit_width = @intCast(width);
|
|
||||||
unit.access_type = access;
|
|
||||||
bit_offset += @intCast(width);
|
|
||||||
},
|
|
||||||
}
|
|
||||||
}
|
|
||||||
self.pos = end;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- control flow -------------------------------------------------------
|
|
||||||
|
|
||||||
fn opIf(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
try self.object(scope); // predicate
|
|
||||||
self.termList(end, scope);
|
|
||||||
if (self.peek() == op.else_op) {
|
|
||||||
self.pos += 1;
|
|
||||||
try self.opElse(scope);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opElse(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
self.termList(end, scope);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opWhile(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
try self.object(scope); // predicate
|
|
||||||
self.termList(end, scope);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opLnot(self: *Parser, scope: *Node) Error!void {
|
|
||||||
// 0x92 followed by 0x93/94/95 is a compound comparison (two operands);
|
|
||||||
// otherwise it is a plain LNot of one operand.
|
|
||||||
const b = self.peek() orelse return error.Truncated;
|
|
||||||
switch (b) {
|
|
||||||
op.lnot.not_equal, op.lnot.less_equal, op.lnot.greater_equal => {
|
|
||||||
self.pos += 1;
|
|
||||||
try self.args(scope, 2);
|
|
||||||
},
|
|
||||||
else => try self.object(scope),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opMatch(self: *Parser, scope: *Node) Error!void {
|
|
||||||
try self.object(scope); // search package
|
|
||||||
try self.skip(1); // match opcode 1
|
|
||||||
try self.object(scope); // operand 1
|
|
||||||
try self.skip(1); // match opcode 2
|
|
||||||
try self.object(scope); // operand 2
|
|
||||||
try self.object(scope); // start index
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- extended opcodes (0x5B xx) -----------------------------------------
|
|
||||||
|
|
||||||
fn opExt(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const e = try self.readByte();
|
|
||||||
switch (e) {
|
|
||||||
op.ext.mutex => try self.opMutex(scope),
|
|
||||||
op.ext.event => try self.opEvent(scope),
|
|
||||||
op.ext.op_region => try self.opRegion(scope),
|
|
||||||
op.ext.data_region => try self.opDataRegion(scope),
|
|
||||||
op.ext.field => try self.opField(scope, 1, false),
|
|
||||||
op.ext.index_field => try self.opField(scope, 2, false),
|
|
||||||
op.ext.bank_field => try self.opField(scope, 2, true),
|
|
||||||
op.ext.device => try self.opScopeLike(scope, .device),
|
|
||||||
op.ext.thermal_zone => try self.opScopeLike(scope, .thermal_zone),
|
|
||||||
op.ext.processor => try self.opProcessor(scope),
|
|
||||||
op.ext.power_res => try self.opPowerRes(scope),
|
|
||||||
|
|
||||||
op.ext.cond_ref_of => try self.args(scope, 2), // SuperName, Target
|
|
||||||
op.ext.create_field => try self.opCreateField(scope, 3),
|
|
||||||
op.ext.load_table => try self.args(scope, 6),
|
|
||||||
op.ext.load => try self.args(scope, 2), // NameString, Target
|
|
||||||
op.ext.stall, op.ext.sleep => try self.args(scope, 1),
|
|
||||||
op.ext.acquire => {
|
|
||||||
try self.object(scope); // mutex SuperName
|
|
||||||
try self.skip(2); // timeout WordData
|
|
||||||
},
|
|
||||||
op.ext.signal, op.ext.reset, op.ext.release, op.ext.unload => try self.args(scope, 1),
|
|
||||||
op.ext.wait => try self.args(scope, 2),
|
|
||||||
op.ext.from_bcd, op.ext.to_bcd => try self.args(scope, 2),
|
|
||||||
op.ext.fatal => {
|
|
||||||
try self.skip(5); // Type(byte) + Code(dword)
|
|
||||||
try self.object(scope); // Arg TermArg
|
|
||||||
},
|
|
||||||
op.ext.revision, op.ext.debug, op.ext.timer => {},
|
|
||||||
|
|
||||||
else => return error.Malformed,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
fn isNameStart(b: u8) bool {
|
|
||||||
return (b >= op.name_char_start and b <= op.name_char_end) or
|
|
||||||
b == op.name_char_underscore or
|
|
||||||
b == op.root_char or
|
|
||||||
b == op.parent_prefix_char or
|
|
||||||
b == op.dual_name_prefix or
|
|
||||||
b == op.multi_name_prefix;
|
|
||||||
}
|
|
||||||
@@ -1,76 +0,0 @@
|
|||||||
//! The firmware-agnostic discovery facade.
|
|
||||||
//!
|
|
||||||
//! The kernel calls `platform.discover()` and gets back a generic `DeviceTree`
|
|
||||||
//! without ever naming ACPI or device-tree — the same way it imports `arch`
|
|
||||||
//! without naming x86_64. Which backend runs is decided *at runtime* from what
|
|
||||||
//! the bootloader handed us (an ACPI RSDP today, a device-tree blob later),
|
|
||||||
//! because a single image — a future ARM kernel especially — may boot under
|
|
||||||
//! either firmware. That's a deliberate divergence from `arch`, which is a
|
|
||||||
//! compile-time choice.
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
const danos = @import("danos");
|
|
||||||
const device = @import("device.zig");
|
|
||||||
const acpi = @import("acpi.zig");
|
|
||||||
const power = @import("power.zig");
|
|
||||||
const devicetree = @import("devicetree.zig");
|
|
||||||
|
|
||||||
pub const DeviceTree = device.DeviceTree;
|
|
||||||
pub const Device = device.Device;
|
|
||||||
pub const DeviceClass = device.DeviceClass;
|
|
||||||
pub const Hal = device.Hal;
|
|
||||||
pub const PowerInfo = acpi.PowerInfo;
|
|
||||||
pub const AmlStats = acpi.AmlStats;
|
|
||||||
pub const PlatformInfo = acpi.PlatformInfo;
|
|
||||||
pub const RegAccess = acpi.RegAccess;
|
|
||||||
pub const IsoEntry = acpi.IsoEntry;
|
|
||||||
|
|
||||||
/// The register map + sleep types discovery extracted, for logging/diagnostics.
|
|
||||||
pub fn powerInfo() PowerInfo {
|
|
||||||
return acpi.power_info;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The scalar firmware facts the arch layer needs to avoid legacy assumptions
|
|
||||||
/// (8259 presence, LAPIC base, PM timer, SPCR UART, IRQ overrides).
|
|
||||||
pub fn platformInfo() PlatformInfo {
|
|
||||||
return acpi.platform_info;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// AML parse integrity/diagnostics (namespace node count, bytes consumed).
|
|
||||||
pub fn amlStats() AmlStats {
|
|
||||||
return acpi.aml_stats;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Enumerate hardware into a fresh device tree. `hal` supplies the hardware
|
|
||||||
/// primitives the backend needs (MMIO mapping for PCIe config space, port I/O for
|
|
||||||
/// ACPI registers); pass the arch implementation. Errors leave nothing to clean up
|
|
||||||
/// beyond the tree's own allocations.
|
|
||||||
pub fn discover(
|
|
||||||
boot_info: *const danos.BootInfo,
|
|
||||||
allocator: std.mem.Allocator,
|
|
||||||
hal: Hal,
|
|
||||||
) !DeviceTree {
|
|
||||||
var dt = try DeviceTree.init(allocator);
|
|
||||||
|
|
||||||
if (boot_info.acpi_rsdp != 0) {
|
|
||||||
try acpi.discover(boot_info.acpi_rsdp, &dt, hal);
|
|
||||||
} else {
|
|
||||||
// No ACPI RSDP. A device-tree boot would parse its blob here; today that
|
|
||||||
// path is a stub, so this reports the machine described itself no way we
|
|
||||||
// understand yet.
|
|
||||||
try devicetree.discover(&dt);
|
|
||||||
}
|
|
||||||
|
|
||||||
return dt;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Restart the machine. Never returns on success; returns only if no reset method
|
|
||||||
/// worked (extremely unlikely). Backend-agnostic entry the kernel calls.
|
|
||||||
pub fn reboot(hal: Hal) void {
|
|
||||||
power.reboot(hal);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Power the machine off (ACPI S5). Never returns on success.
|
|
||||||
pub fn shutdown(hal: Hal) void {
|
|
||||||
power.shutdown(hal);
|
|
||||||
}
|
|
||||||
@@ -1,358 +0,0 @@
|
|||||||
//! Local APIC and its timer — the source of device interrupts.
|
|
||||||
//!
|
|
||||||
//! Modern x86 routes interrupts through the per-CPU Local APIC (the legacy 8259
|
|
||||||
//! PIC is remapped out of the way and masked). The LAPIC also has a built-in
|
|
||||||
//! timer, which is the simplest device interrupt to bring up: it needs no
|
|
||||||
//! external routing, just a vector and a count. We use it as danos's heartbeat.
|
|
||||||
//!
|
|
||||||
//! The LAPIC is memory-mapped (default physical 0xFEE00000, inside our identity
|
|
||||||
//! map). Every interrupt must be acknowledged with an end-of-interrupt write, or
|
|
||||||
//! the LAPIC won't deliver the next one.
|
|
||||||
|
|
||||||
const io = @import("io.zig");
|
|
||||||
const paging = @import("paging.zig");
|
|
||||||
|
|
||||||
/// The ACPI PM timer, as a calibration reference: an I/O port or MMIO counter.
|
|
||||||
pub const PmTimer = struct { mmio: bool, address: u64, is_32bit: bool };
|
|
||||||
|
|
||||||
// Platform facts from discovery (set by `configure` before bring-up). Defaults are
|
|
||||||
// the legacy-safe assumptions so the code still works if discovery never ran.
|
|
||||||
var cfg_pic_present: bool = true;
|
|
||||||
var cfg_hpet_base: u64 = 0; // 0 = no HPET discovered
|
|
||||||
var cfg_pm_timer: ?PmTimer = null;
|
|
||||||
/// Which reference the last calibration used, for logging.
|
|
||||||
var cal_source: []const u8 = "none";
|
|
||||||
|
|
||||||
/// Hand the LAPIC bring-up the discovered platform facts. Call before `init`.
|
|
||||||
pub fn configure(pic_present: bool, hpet_base: u64, pm_timer: ?PmTimer) void {
|
|
||||||
cfg_pic_present = pic_present;
|
|
||||||
cfg_hpet_base = hpet_base;
|
|
||||||
cfg_pm_timer = pm_timer;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The calibration reference the timer was measured against ("cpuid"/"hpet"/…).
|
|
||||||
pub fn calibrationSource() []const u8 {
|
|
||||||
return cal_source;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// IDT vector the timer fires on (in the device range, >= 32).
|
|
||||||
pub const timer_vector = 32;
|
|
||||||
/// Spurious-interrupt vector. Low nibble 0xF by convention; also in our gate
|
|
||||||
/// range so a stray spurious interrupt lands on a valid (no-op) handler.
|
|
||||||
const spurious_vector = 47;
|
|
||||||
|
|
||||||
// LAPIC register offsets.
|
|
||||||
const reg_spurious = 0x0F0;
|
|
||||||
const reg_eoi = 0x0B0;
|
|
||||||
const reg_lvt_timer = 0x320;
|
|
||||||
const reg_timer_initial = 0x380;
|
|
||||||
const reg_timer_current = 0x390;
|
|
||||||
const reg_timer_divide = 0x3E0;
|
|
||||||
|
|
||||||
const lvt_masked = 1 << 16;
|
|
||||||
const lvt_periodic = 1 << 17;
|
|
||||||
const timer_divide_16 = 0x3;
|
|
||||||
|
|
||||||
const ia32_apic_base_msr = 0x1B;
|
|
||||||
|
|
||||||
/// LAPIC MMIO base. A runtime var (not a constant) both because we read it from
|
|
||||||
/// the MSR and so register writes compile to normal stores rather than a
|
|
||||||
/// `mov moffs`, which the self-hosted backend can't encode.
|
|
||||||
var base: usize = 0xFEE00000;
|
|
||||||
|
|
||||||
var tick_count: u64 = 0;
|
|
||||||
|
|
||||||
/// LAPIC timer counts per millisecond, measured against the PIT (see calibrate).
|
|
||||||
/// At divide-by-16, this is the effective counting rate.
|
|
||||||
var ticks_per_ms: u32 = 0;
|
|
||||||
/// The periodic-interrupt frequency the timer is armed at, once initTimer runs.
|
|
||||||
var timer_hz: u32 = 0;
|
|
||||||
|
|
||||||
/// TSC (Time Stamp Counter) calibration: cycles per second, and the count at boot.
|
|
||||||
/// The TSC is a per-core cycle counter, giving a ~nanosecond high-resolution
|
|
||||||
/// monotonic clock — far finer than the millisecond timer tick.
|
|
||||||
var tsc_hz: u64 = 0;
|
|
||||||
var tsc_base: u64 = 0;
|
|
||||||
|
|
||||||
/// Read the 64-bit Time Stamp Counter.
|
|
||||||
fn rdtsc() u64 {
|
|
||||||
var low: u32 = undefined;
|
|
||||||
var high: u32 = undefined;
|
|
||||||
asm volatile ("rdtsc"
|
|
||||||
: [low] "={eax}" (low),
|
|
||||||
[high] "={edx}" (high),
|
|
||||||
);
|
|
||||||
return (@as(u64, high) << 32) | low;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn read(reg: u32) u32 {
|
|
||||||
return @as(*volatile u32, @ptrFromInt(base + reg)).*;
|
|
||||||
}
|
|
||||||
fn write(reg: u32, value: u32) void {
|
|
||||||
@as(*volatile u32, @ptrFromInt(base + reg)).* = value;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Move the legacy 8259 PIC's vectors to 0x20-0x2F (clear of the CPU exception
|
|
||||||
/// vectors) and mask every line, so it can't deliver interrupts behind the APIC.
|
|
||||||
fn remapAndMaskPic() void {
|
|
||||||
io.outb(0x20, 0x11); // start init (cascade mode)
|
|
||||||
io.outb(0xA0, 0x11);
|
|
||||||
io.outb(0x21, 0x20); // master offset 0x20
|
|
||||||
io.outb(0xA1, 0x28); // slave offset 0x28
|
|
||||||
io.outb(0x21, 0x04); // tell master about slave on IRQ2
|
|
||||||
io.outb(0xA1, 0x02);
|
|
||||||
io.outb(0x21, 0x01); // 8086 mode
|
|
||||||
io.outb(0xA1, 0x01);
|
|
||||||
io.outb(0x21, 0xFF); // mask all
|
|
||||||
io.outb(0xA1, 0xFF);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Enable the Local APIC: mask the PIC (only if one is present — a legacy-free
|
|
||||||
/// UEFI Class 3 machine may have none), set the global-enable MSR bit, and
|
|
||||||
/// software-enable the APIC via its spurious-vector register.
|
|
||||||
pub fn init() void {
|
|
||||||
if (cfg_pic_present) remapAndMaskPic();
|
|
||||||
|
|
||||||
const msr = io.rdmsr(ia32_apic_base_msr);
|
|
||||||
base = @intCast(msr & 0xFFFFF000); // physical base is bits 12+
|
|
||||||
io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable
|
|
||||||
|
|
||||||
write(reg_spurious, 0x100 | spurious_vector); // bit 8 = software enable
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The calibration window: we time everything against a 10 ms reference interval.
|
|
||||||
const calib_ms = 10;
|
|
||||||
|
|
||||||
/// Measure the LAPIC timer's and the TSC's rates. The PIT (legacy 8254) can be
|
|
||||||
/// absent on UEFI Class 3 firmware — and polling it would hang — so we pick a
|
|
||||||
/// reference clock in order of preference: the CPU's own TSC frequency (CPUID leaf
|
|
||||||
/// 0x15, no external timer needed), then the discovered HPET, then the ACPI PM
|
|
||||||
/// timer, and only the PIT as a last resort. Each path yields the same two rates.
|
|
||||||
pub fn calibrate() void {
|
|
||||||
var done = false;
|
|
||||||
|
|
||||||
// 1. CPUID leaf 0x15 gives the TSC frequency directly — measure the LAPIC
|
|
||||||
// against the TSC itself, needing no external timer at all.
|
|
||||||
if (cpuidTscHz()) |hz| {
|
|
||||||
measure(hz, ~@as(u64, 0), rdtsc);
|
|
||||||
tsc_hz = hz; // keep the exact enumerated value
|
|
||||||
cal_source = "cpuid";
|
|
||||||
done = true;
|
|
||||||
}
|
|
||||||
|
|
||||||
// 2. The discovered HPET.
|
|
||||||
if (!done and cfg_hpet_base != 0) {
|
|
||||||
if (hpetHz()) |hpet_hz| {
|
|
||||||
measure(hpet_hz, hpetMask(), readHpet);
|
|
||||||
cal_source = "hpet";
|
|
||||||
done = true;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// 3. The ACPI PM timer (fixed 3.579545 MHz).
|
|
||||||
if (!done) {
|
|
||||||
if (cfg_pm_timer) |pt| {
|
|
||||||
measure(3_579_545, if (pt.is_32bit) 0xFFFF_FFFF else 0xFF_FFFF, readPmTimer);
|
|
||||||
cal_source = "pm-timer";
|
|
||||||
done = true;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// 4. The legacy PIT, last resort.
|
|
||||||
if (!done) {
|
|
||||||
calibratePit();
|
|
||||||
cal_source = "pit";
|
|
||||||
}
|
|
||||||
|
|
||||||
// A bad measurement (no reference actually ticked) leaves nonsense; fall back.
|
|
||||||
if (ticks_per_ms == 0 or tsc_hz == 0) {
|
|
||||||
calibratePit();
|
|
||||||
cal_source = "pit";
|
|
||||||
}
|
|
||||||
|
|
||||||
tsc_base = rdtsc(); // the clock's zero point (boot)
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Run the LAPIC timer one-shot from its max count while a monotonic reference
|
|
||||||
/// clock (frequency `ref_hz`, counter width `ref_mask`) counts out `calib_ms`, and
|
|
||||||
/// snapshot the TSC across the same window. Yields `ticks_per_ms` and `tsc_hz`.
|
|
||||||
fn measure(ref_hz: u64, ref_mask: u64, refNow: *const fn () u64) void {
|
|
||||||
const calib_ticks = ref_hz / (1000 / calib_ms); // reference ticks in calib_ms
|
|
||||||
|
|
||||||
write(reg_timer_divide, timer_divide_16);
|
|
||||||
write(reg_lvt_timer, lvt_masked);
|
|
||||||
write(reg_timer_initial, 0xFFFFFFFF);
|
|
||||||
|
|
||||||
const ref0 = refNow();
|
|
||||||
const tsc0 = rdtsc();
|
|
||||||
while (((refNow() -% ref0) & ref_mask) < calib_ticks) {}
|
|
||||||
const tsc1 = rdtsc();
|
|
||||||
|
|
||||||
const elapsed = 0xFFFFFFFF - read(reg_timer_current);
|
|
||||||
write(reg_timer_initial, 0);
|
|
||||||
|
|
||||||
ticks_per_ms = elapsed / calib_ms;
|
|
||||||
tsc_hz = (tsc1 -% tsc0) * (1000 / calib_ms);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The PIT fallback (legacy 8254 channel 2, polled). Only reached when no better
|
|
||||||
/// reference exists — on a legacy-free machine this path isn't taken.
|
|
||||||
fn calibratePit() void {
|
|
||||||
const pit_hz = 1_193_182;
|
|
||||||
const pit_count: u16 = @intCast(pit_hz / 1000 * calib_ms);
|
|
||||||
|
|
||||||
write(reg_timer_divide, timer_divide_16);
|
|
||||||
write(reg_lvt_timer, lvt_masked);
|
|
||||||
write(reg_timer_initial, 0xFFFFFFFF);
|
|
||||||
|
|
||||||
io.outb(0x61, io.inb(0x61) & 0xFC); // speaker off, gate low
|
|
||||||
io.outb(0x43, 0xB0); // channel 2, lo/hi byte, mode 0
|
|
||||||
io.outb(0x42, @truncate(pit_count));
|
|
||||||
io.outb(0x42, @truncate(pit_count >> 8));
|
|
||||||
|
|
||||||
const tsc_start = rdtsc();
|
|
||||||
io.outb(0x61, (io.inb(0x61) & 0xFC) | 0x01); // gate high -> start
|
|
||||||
var guard: u64 = 0;
|
|
||||||
while (io.inb(0x61) & 0x20 == 0 and guard < 100_000_000) : (guard += 1) {} // bounded
|
|
||||||
const tsc_end = rdtsc();
|
|
||||||
|
|
||||||
const elapsed = 0xFFFFFFFF - read(reg_timer_current);
|
|
||||||
write(reg_timer_initial, 0);
|
|
||||||
|
|
||||||
ticks_per_ms = elapsed / calib_ms;
|
|
||||||
tsc_hz = (tsc_end -% tsc_start) * (1000 / calib_ms);
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- reference clocks ------------------------------------------------------
|
|
||||||
|
|
||||||
/// TSC frequency from CPUID leaf 0x15 (crystal_hz * numerator / denominator), or
|
|
||||||
/// null if the CPU doesn't enumerate it (common under QEMU).
|
|
||||||
fn cpuidTscHz() ?u64 {
|
|
||||||
if (cpuid(0).eax < 0x15) return null;
|
|
||||||
const r = cpuid(0x15);
|
|
||||||
if (r.eax == 0 or r.ebx == 0 or r.ecx == 0) return null; // ratio/crystal not given
|
|
||||||
return @as(u64, r.ecx) * r.ebx / r.eax;
|
|
||||||
}
|
|
||||||
|
|
||||||
const CpuidRegs = struct { eax: u32, ebx: u32, ecx: u32, edx: u32 };
|
|
||||||
|
|
||||||
fn cpuid(leaf: u32) CpuidRegs {
|
|
||||||
var a: u32 = undefined;
|
|
||||||
var b: u32 = undefined;
|
|
||||||
var c: u32 = undefined;
|
|
||||||
var d: u32 = undefined;
|
|
||||||
asm volatile ("cpuid"
|
|
||||||
: [a] "={eax}" (a),
|
|
||||||
[b] "={ebx}" (b),
|
|
||||||
[c] "={ecx}" (c),
|
|
||||||
[d] "={edx}" (d),
|
|
||||||
: [leaf] "{eax}" (leaf),
|
|
||||||
[sub] "{ecx}" (@as(u32, 0)),
|
|
||||||
);
|
|
||||||
return .{ .eax = a, .ebx = b, .ecx = c, .edx = d };
|
|
||||||
}
|
|
||||||
|
|
||||||
// HPET registers: capabilities at +0x00 (period in the high dword, in fs; bit 13 =
|
|
||||||
// 64-bit-counter capable), general config at +0x10, main counter at +0xF0.
|
|
||||||
fn hpetRead64(off: usize) u64 {
|
|
||||||
return @as(*volatile u64, @ptrFromInt(cfg_hpet_base + off)).*;
|
|
||||||
}
|
|
||||||
fn hpetWrite64(off: usize, value: u64) void {
|
|
||||||
@as(*volatile u64, @ptrFromInt(cfg_hpet_base + off)).* = value;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Map + enable the HPET and return its tick frequency, or null if unusable.
|
|
||||||
fn hpetHz() ?u64 {
|
|
||||||
paging.map(cfg_hpet_base & ~@as(u64, 0xFFF), cfg_hpet_base & ~@as(u64, 0xFFF), true);
|
|
||||||
const caps = hpetRead64(0x00);
|
|
||||||
const period_fs = caps >> 32; // femtoseconds per tick
|
|
||||||
if (period_fs == 0) return null;
|
|
||||||
hpetWrite64(0x10, hpetRead64(0x10) | 1); // ENABLE_CNF: start the main counter
|
|
||||||
return 1_000_000_000_000_000 / period_fs; // 1e15 fs/s ÷ fs/tick
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The HPET counter width mask (64- or 32-bit, per caps bit 13).
|
|
||||||
fn hpetMask() u64 {
|
|
||||||
return if (hpetRead64(0x00) & (1 << 13) != 0) ~@as(u64, 0) else 0xFFFF_FFFF;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn readHpet() u64 {
|
|
||||||
return hpetRead64(0xF0);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn readPmTimer() u64 {
|
|
||||||
const pt = cfg_pm_timer.?;
|
|
||||||
if (pt.mmio) return @as(*volatile u32, @ptrFromInt(pt.address)).*;
|
|
||||||
return io.inl(@intCast(pt.address));
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Arm the LAPIC timer to fire on `timer_vector` at `hz` (periodic). Requires
|
|
||||||
/// calibrate() to have run.
|
|
||||||
pub fn initTimer(hz: u32) void {
|
|
||||||
timer_hz = hz;
|
|
||||||
const count = @as(u64, ticks_per_ms) * 1000 / hz; // counts per (1/hz) second
|
|
||||||
write(reg_timer_divide, timer_divide_16);
|
|
||||||
write(reg_lvt_timer, timer_vector | lvt_periodic);
|
|
||||||
write(reg_timer_initial, @intCast(count));
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Configured periodic-interrupt frequency (Hz).
|
|
||||||
pub fn frequencyHz() u32 {
|
|
||||||
return timer_hz;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Measured LAPIC timer frequency (Hz), for reporting/sanity checks.
|
|
||||||
pub fn lapicHz() u64 {
|
|
||||||
return @as(u64, ticks_per_ms) * 1000;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Measured TSC frequency (Hz).
|
|
||||||
pub fn tscHz() u64 {
|
|
||||||
return tsc_hz;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Monotonic high-resolution clock, from the TSC. A function per resolution, each
|
|
||||||
// scaling the cycle delta directly at its unit (the 128-bit intermediate avoids
|
|
||||||
// overflow across a long uptime). nanos() resolves to a few ns; millis() is what
|
|
||||||
// the scheduler uses for sleep deadlines.
|
|
||||||
|
|
||||||
pub fn nanos() u64 {
|
|
||||||
if (tsc_hz == 0) return 0;
|
|
||||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000_000 / tsc_hz);
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn micros() u64 {
|
|
||||||
if (tsc_hz == 0) return 0;
|
|
||||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000_000 / tsc_hz);
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn millis() u64 {
|
|
||||||
if (tsc_hz == 0) return 0;
|
|
||||||
return @intCast(@as(u128, rdtsc() -% tsc_base) * 1_000 / tsc_hz);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Acknowledge the current interrupt so the LAPIC will deliver the next one.
|
|
||||||
pub fn eoi() void {
|
|
||||||
write(reg_eoi, 0);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Optional callback run each tick (the scheduler registers it for preemption).
|
|
||||||
var on_tick: ?*const fn () void = null;
|
|
||||||
|
|
||||||
pub fn setTickHook(hook: *const fn () void) void {
|
|
||||||
on_tick = hook;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The timer interrupt handler: advance the monotonic tick count, then run the
|
|
||||||
/// tick hook (which may switch tasks). The interrupt is already acknowledged by
|
|
||||||
/// the dispatcher before we get here, so a task switch here doesn't stall it.
|
|
||||||
pub fn timerTick() void {
|
|
||||||
tick_count +%= 1;
|
|
||||||
if (on_tick) |hook| hook();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Number of timer ticks so far. Volatile load: the count is bumped
|
|
||||||
/// asynchronously by the interrupt handler, so callers must re-read memory.
|
|
||||||
pub fn ticks() u64 {
|
|
||||||
return @as(*const volatile u64, &tick_count).*;
|
|
||||||
}
|
|
||||||
@@ -1,284 +0,0 @@
|
|||||||
//! x86_64 CPU operations. This is the "arch" module: the generic kernel imports
|
|
||||||
//! it as `@import("arch")` and never names x86_64 directly, so a second
|
|
||||||
//! architecture is added by pointing that module at a different directory in
|
|
||||||
//! build.zig — no change to the generic code. Keep everything CPU-specific here
|
|
||||||
//! (halt, the descriptor tables, later paging), and nothing generic.
|
|
||||||
|
|
||||||
const danos = @import("danos");
|
|
||||||
const gdt = @import("gdt.zig");
|
|
||||||
const tss = @import("tss.zig");
|
|
||||||
const idt = @import("idt.zig");
|
|
||||||
const paging = @import("paging.zig");
|
|
||||||
const serial = @import("serial.zig");
|
|
||||||
const apic = @import("apic.zig");
|
|
||||||
const ioapic = @import("ioapic.zig");
|
|
||||||
const io = @import("io.zig");
|
|
||||||
|
|
||||||
/// The saved register/trap frame passed to a fault handler.
|
|
||||||
pub const CpuState = idt.CpuState;
|
|
||||||
|
|
||||||
/// Bring up the serial port (the kernel's machine-readable log). No dependencies,
|
|
||||||
/// so it can be the very first thing called.
|
|
||||||
pub fn serialInit() void {
|
|
||||||
serial.init();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Write bytes to the serial port.
|
|
||||||
pub fn serialWrite(bytes: []const u8) void {
|
|
||||||
serial.write(bytes);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Emit a one-byte checkpoint to the POST diagnostic port (0x80). A POST card or
|
|
||||||
/// BMC displays it; it's the last-resort progress signal when there's no text
|
|
||||||
/// output at all. Writing 0x80 is universally safe (it's the legacy I/O-delay port).
|
|
||||||
pub fn postCode(code: u8) void {
|
|
||||||
io.outb(0x80, code);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Whether a Bochs/QEMU-style debug console is on port 0xE9 (it returns 0xE9 when
|
|
||||||
/// read). On real hardware the port reads back 0xFF, so this stays false — a safe
|
|
||||||
/// probe before we write to it.
|
|
||||||
pub fn debugconPresent() bool {
|
|
||||||
return io.inb(0xE9) == 0xE9;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Output sink: write bytes to the 0xE9 debug console (see `debugconPresent`).
|
|
||||||
pub fn debugconWrite(bytes: []const u8) void {
|
|
||||||
for (bytes) |b| io.outb(0xE9, b);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Set up the CPU's descriptor tables: our own GDT, the TSS (with an interrupt
|
|
||||||
/// stack for double faults), then the IDT with exception handlers. After this a
|
|
||||||
/// CPU fault is reported instead of triple-faulting. Install the fault handler
|
|
||||||
/// (setFaultHandler) first so early faults are caught.
|
|
||||||
pub fn init() void {
|
|
||||||
gdt.init();
|
|
||||||
tss.init();
|
|
||||||
idt.init();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Build the kernel's own page tables (with real permissions) and switch onto
|
|
||||||
/// them. Needs the frame allocator and the boot info (for the memory map and the
|
|
||||||
/// kernel's segment layout). Call once the frame allocator is up.
|
|
||||||
pub fn enablePaging(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void {
|
|
||||||
paging.init(allocFrame, boot_info);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Map a page into the kernel address space (non-executable). For the heap, etc.
|
|
||||||
pub fn mapPage(virt: u64, phys: u64, writable: bool) void {
|
|
||||||
paging.map(virt, phys, writable);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Remove a kernel mapping.
|
|
||||||
pub fn unmapPage(virt: u64) void {
|
|
||||||
paging.unmap(virt);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// CR3 holds the physical address of the active top-level page table.
|
|
||||||
pub fn readCr3() u64 {
|
|
||||||
return asm volatile ("mov %%cr3, %[out]"
|
|
||||||
: [out] "=r" (-> u64),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Kernel tick rate: 1000 Hz (1 ms), the scheduler's time quantum.
|
|
||||||
pub const timer_hz = 1000;
|
|
||||||
|
|
||||||
/// The ACPI PM timer, as a calibration reference (re-exported for the config).
|
|
||||||
pub const PmTimer = apic.PmTimer;
|
|
||||||
/// A MADT interrupt-source override (re-exported for the config).
|
|
||||||
pub const IsoEntry = ioapic.IsoEntry;
|
|
||||||
|
|
||||||
/// Discovered platform facts the arch layer needs so it makes no legacy
|
|
||||||
/// assumptions — sourced from the device tree + ACPI, passed in by the kernel.
|
|
||||||
pub const PlatformConfig = struct {
|
|
||||||
/// Whether the legacy 8259 PIC is present (skip programming it if not).
|
|
||||||
pic_present: bool = true,
|
|
||||||
/// HPET MMIO base (0 = none) — a calibration reference for the timer.
|
|
||||||
hpet_base: u64 = 0,
|
|
||||||
/// The ACPI PM timer, another calibration reference.
|
|
||||||
pm_timer: ?PmTimer = null,
|
|
||||||
/// I/O APIC MMIO base + its first global system interrupt (0 = none).
|
|
||||||
ioapic_base: u64 = 0,
|
|
||||||
ioapic_gsi_base: u32 = 0,
|
|
||||||
/// MADT ISA-IRQ overrides, for I/O APIC routing.
|
|
||||||
overrides: []const IsoEntry = &.{},
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Apply the discovered platform config. Must run before `startTimer` (the timer
|
|
||||||
/// calibration reads `hpet_base`/`pm_timer`) and before any interrupt routing.
|
|
||||||
/// Maps + masks the I/O APIC immediately.
|
|
||||||
pub fn configurePlatform(cfg: PlatformConfig) void {
|
|
||||||
apic.configure(cfg.pic_present, cfg.hpet_base, cfg.pm_timer);
|
|
||||||
ioapic.configure(cfg.ioapic_base, cfg.ioapic_gsi_base, cfg.overrides);
|
|
||||||
ioapic.init();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Point the serial console at the UART ACPI's SPCR table named (MMIO or I/O port).
|
|
||||||
pub fn serialReconfigure(is_mmio: bool, addr: u64) void {
|
|
||||||
serial.reconfigure(is_mmio, addr);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The reference clock the timer was calibrated against ("cpuid"/"hpet"/…).
|
|
||||||
pub fn timerCalibrationSource() []const u8 {
|
|
||||||
return apic.calibrationSource();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// I/O APIC diagnostics (for boot logging / verification).
|
|
||||||
pub fn ioapicEntryCount() u32 {
|
|
||||||
return ioapic.entryCount();
|
|
||||||
}
|
|
||||||
pub fn ioapicEntryLow(n: u32) u32 {
|
|
||||||
return ioapic.entryLow(n);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Enable the Local APIC, calibrate its timer against the best available reference
|
|
||||||
/// (see apic.calibrate — no longer the PIT by default), and start it firing at
|
|
||||||
/// `timer_hz` — the kernel's real-time heartbeat. Interrupts still have to be
|
|
||||||
/// unmasked with enableInterrupts() to be delivered. Run `configurePlatform` first.
|
|
||||||
pub fn startTimer() void {
|
|
||||||
apic.init();
|
|
||||||
apic.calibrate();
|
|
||||||
idt.setHandler(apic.timer_vector, apic.timerTick);
|
|
||||||
apic.initTimer(timer_hz);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Number of timer ticks since startTimer().
|
|
||||||
pub fn ticks() u64 {
|
|
||||||
return apic.ticks();
|
|
||||||
}
|
|
||||||
|
|
||||||
// Monotonic high-resolution clock (from the TSC), one function per resolution.
|
|
||||||
pub fn nanos() u64 {
|
|
||||||
return apic.nanos();
|
|
||||||
}
|
|
||||||
pub fn micros() u64 {
|
|
||||||
return apic.micros();
|
|
||||||
}
|
|
||||||
pub fn millis() u64 {
|
|
||||||
return apic.millis();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Measured LAPIC timer / TSC frequencies in Hz (from calibration).
|
|
||||||
pub fn lapicHz() u64 {
|
|
||||||
return apic.lapicHz();
|
|
||||||
}
|
|
||||||
pub fn tscHz() u64 {
|
|
||||||
return apic.tscHz();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Unmask maskable interrupts (`sti`) so device interrupts get delivered.
|
|
||||||
pub fn enableInterrupts() void {
|
|
||||||
asm volatile ("sti");
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Mask maskable interrupts (`cli`).
|
|
||||||
pub fn disableInterrupts() void {
|
|
||||||
asm volatile ("cli");
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Disable interrupts and return the previous flags, so a nested critical section
|
|
||||||
/// can restore the caller's state rather than blindly re-enabling. Pairs with
|
|
||||||
/// restoreInterrupts.
|
|
||||||
pub fn saveInterrupts() u64 {
|
|
||||||
var flags: u64 = undefined;
|
|
||||||
asm volatile (
|
|
||||||
\\pushfq
|
|
||||||
\\pop %[f]
|
|
||||||
\\cli
|
|
||||||
: [f] "=r" (flags),
|
|
||||||
:
|
|
||||||
: .{ .memory = true }
|
|
||||||
);
|
|
||||||
return flags;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Re-enable interrupts only if they were enabled when `flags` was captured.
|
|
||||||
pub fn restoreInterrupts(flags: u64) void {
|
|
||||||
if (flags & 0x200 != 0) asm volatile ("sti" ::: .{ .memory = true }); // bit 9 = IF
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Register a callback the timer interrupt invokes each tick (e.g. the scheduler).
|
|
||||||
pub fn setTickHook(hook: *const fn () void) void {
|
|
||||||
apic.setTickHook(hook);
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- context switching (for the scheduler) -------------------------------
|
|
||||||
|
|
||||||
/// Save the current task's registers/stack and resume `new_rsp`; the old stack
|
|
||||||
/// pointer is written to `old_rsp`. Defined in isr.s.
|
|
||||||
extern fn switch_context(old_rsp: *usize, new_rsp: usize) callconv(.c) void;
|
|
||||||
|
|
||||||
pub fn switchContext(old_rsp: *usize, new_rsp: usize) void {
|
|
||||||
switch_context(old_rsp, new_rsp);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Build the initial stack for a new task so that switching to it lands in
|
|
||||||
/// `task_trampoline`, which then calls `entry`. Returns the saved stack pointer.
|
|
||||||
/// The layout must match switch_context's push order (callee-saved, then the
|
|
||||||
/// return address on top); `entry` is smuggled in via the r15 slot.
|
|
||||||
pub fn initTaskStack(stack_top: usize, entry: usize) usize {
|
|
||||||
const trampoline = @extern(*const anyopaque, .{ .name = "task_trampoline" });
|
|
||||||
var sp = stack_top;
|
|
||||||
const push = struct {
|
|
||||||
fn f(p: *usize, value: usize) void {
|
|
||||||
p.* -= @sizeOf(usize);
|
|
||||||
@as(*usize, @ptrFromInt(p.*)).* = value;
|
|
||||||
}
|
|
||||||
}.f;
|
|
||||||
push(&sp, @intFromPtr(trampoline)); // return address for switch_context's `ret`
|
|
||||||
push(&sp, 0); // rbx
|
|
||||||
push(&sp, 0); // rbp
|
|
||||||
push(&sp, 0); // r12
|
|
||||||
push(&sp, 0); // r13
|
|
||||||
push(&sp, 0); // r14
|
|
||||||
push(&sp, entry); // r15 -> task entry, read by task_trampoline
|
|
||||||
return sp;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Route CPU exceptions to `handler`, which receives the trap frame and does not
|
|
||||||
/// return. Until set, faults just halt the core.
|
|
||||||
pub fn setFaultHandler(handler: *const fn (*const CpuState) noreturn) void {
|
|
||||||
idt.on_fault = handler;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// A human-readable name for a CPU exception vector.
|
|
||||||
pub fn vectorName(vector: u64) []const u8 {
|
|
||||||
return idt.vectorName(vector);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Read `width` bytes (1/2/4) from an I/O port. The generic device layer drives
|
|
||||||
/// ACPI registers through this rather than naming x86 port instructions; on an
|
|
||||||
/// MMIO-only architecture this would be implemented differently.
|
|
||||||
pub fn pioRead(width: u8, port: u16) u32 {
|
|
||||||
return switch (width) {
|
|
||||||
1 => io.inb(port),
|
|
||||||
2 => io.inw(port),
|
|
||||||
4 => io.inl(port),
|
|
||||||
else => 0,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Write `width` bytes (1/2/4) to an I/O port.
|
|
||||||
pub fn pioWrite(width: u8, port: u16, value: u32) void {
|
|
||||||
switch (width) {
|
|
||||||
1 => io.outb(port, @truncate(value)),
|
|
||||||
2 => io.outw(port, @truncate(value)),
|
|
||||||
4 => io.outl(port, value),
|
|
||||||
else => {},
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// CR2 holds the faulting linear address after a page fault (#PF, vector 14).
|
|
||||||
pub fn readCr2() u64 {
|
|
||||||
return asm volatile ("mov %%cr2, %[out]"
|
|
||||||
: [out] "=r" (-> u64),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Park the core forever. `hlt` drops it into a low-power idle until the next
|
|
||||||
/// interrupt; the loop re-halts on every wake so the stop is permanent. See
|
|
||||||
/// docs/halting.md for the full reasoning.
|
|
||||||
pub fn halt() noreturn {
|
|
||||||
while (true) asm volatile ("hlt");
|
|
||||||
}
|
|
||||||
@@ -1,54 +0,0 @@
|
|||||||
//! Global Descriptor Table. In long mode segmentation is mostly vestigial, but
|
|
||||||
//! the CPU still needs valid code/data segment descriptors, and the IDT's gates
|
|
||||||
//! reference a code selector — so we install our own flat GDT with known
|
|
||||||
//! selectors (0x08 kernel code, 0x10 kernel data) rather than trusting whatever
|
|
||||||
//! the firmware left in place.
|
|
||||||
|
|
||||||
/// Selectors into the table below (index * 8).
|
|
||||||
pub const kernel_code = 0x08;
|
|
||||||
pub const kernel_data = 0x10;
|
|
||||||
pub const tss_selector = 0x18;
|
|
||||||
|
|
||||||
/// Flat 64-bit descriptors. Base/limit are ignored in long mode; what matters is
|
|
||||||
/// the access byte and, for code, the long-mode (L) flag.
|
|
||||||
/// code: present, ring 0, executable, readable, L=1 -> 0x00AF9A00_0000FFFF
|
|
||||||
/// data: present, ring 0, writable -> 0x00CF9200_0000FFFF
|
|
||||||
/// The last two slots hold one 16-byte TSS descriptor, filled in by setTss.
|
|
||||||
var table = [_]u64{
|
|
||||||
0, // null descriptor (required)
|
|
||||||
0x00AF9A000000FFFF, // kernel code (0x08)
|
|
||||||
0x00CF92000000FFFF, // kernel data (0x10)
|
|
||||||
0, // TSS descriptor low (0x18)
|
|
||||||
0, // TSS descriptor high
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Fill the 64-bit TSS system descriptor (two GDT slots) so the task register can
|
|
||||||
/// point at our TSS. Type 0x89 = present, ring 0, available 64-bit TSS.
|
|
||||||
pub fn setTss(base: u64, limit: u64) void {
|
|
||||||
table[3] = (limit & 0xFFFF) |
|
|
||||||
((base & 0xFFFF) << 16) |
|
|
||||||
(((base >> 16) & 0xFF) << 32) |
|
|
||||||
(@as(u64, 0x89) << 40) |
|
|
||||||
(((limit >> 16) & 0xF) << 48) |
|
|
||||||
(((base >> 24) & 0xFF) << 56);
|
|
||||||
table[4] = (base >> 32) & 0xFFFFFFFF;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The operand `lgdt` wants: table byte-length minus one, then its address.
|
|
||||||
const Descriptor = packed struct {
|
|
||||||
limit: u16,
|
|
||||||
base: u64,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Loads the GDT and reloads the segment registers (including CS). Defined in
|
|
||||||
/// isr.s — it uses the selectors 0x08 (code) and 0x10 (data) that match `table`.
|
|
||||||
extern fn gdt_flush(descriptor: *const Descriptor) callconv(.c) void;
|
|
||||||
|
|
||||||
/// Install our GDT and switch onto its segments.
|
|
||||||
pub fn init() void {
|
|
||||||
const descriptor = Descriptor{
|
|
||||||
.limit = @sizeOf(@TypeOf(table)) - 1,
|
|
||||||
.base = @intFromPtr(&table),
|
|
||||||
};
|
|
||||||
gdt_flush(&descriptor);
|
|
||||||
}
|
|
||||||
@@ -1,97 +0,0 @@
|
|||||||
//! I/O APIC — routes external device interrupts (a device's line) to a LAPIC
|
|
||||||
//! vector on a chosen CPU. Its address and the ISA-IRQ-to-GSI remappings come from
|
|
||||||
//! ACPI's MADT (via discovery), never assumed.
|
|
||||||
//!
|
|
||||||
//! Status: groundwork. The only interrupt danos handles today is the LAPIC's own
|
|
||||||
//! timer, which needs no I/O APIC — so nothing calls `routeIrq` yet. What runs now
|
|
||||||
//! is `init`, which maps the I/O APIC and **masks every input**, the correct
|
|
||||||
//! quiescent state on a legacy-free machine. `routeIrq` is ready for the first real
|
|
||||||
//! device driver (a keyboard, say).
|
|
||||||
|
|
||||||
const paging = @import("paging.zig");
|
|
||||||
|
|
||||||
/// A MADT Interrupt Source Override: an ISA IRQ that appears at a different global
|
|
||||||
/// system interrupt, with its own polarity/trigger (MPS INTI `flags`).
|
|
||||||
pub const IsoEntry = struct { source: u8, gsi: u32, flags: u16 };
|
|
||||||
|
|
||||||
var base: u64 = 0; // 0 = no I/O APIC discovered
|
|
||||||
var gsi_base: u32 = 0;
|
|
||||||
var max_entries: u32 = 0;
|
|
||||||
var overrides: [16]IsoEntry = undefined;
|
|
||||||
var override_count: usize = 0;
|
|
||||||
|
|
||||||
// The I/O APIC exposes an index register (IOREGSEL) and a data window (IOWIN).
|
|
||||||
const reg_ioregsel = 0x00;
|
|
||||||
const reg_iowin = 0x10;
|
|
||||||
const reg_version = 0x01;
|
|
||||||
const redir_base = 0x10; // redirection table: two 32-bit regs per entry
|
|
||||||
const redir_mask = 1 << 16; // mask bit in the low dword
|
|
||||||
|
|
||||||
/// Supply the discovered I/O APIC location + the MADT IRQ overrides. Call before `init`.
|
|
||||||
pub fn configure(ioapic_base: u64, ioapic_gsi_base: u32, isos: []const IsoEntry) void {
|
|
||||||
base = ioapic_base;
|
|
||||||
gsi_base = ioapic_gsi_base;
|
|
||||||
override_count = @min(isos.len, overrides.len);
|
|
||||||
for (isos[0..override_count], 0..) |iso, i| overrides[i] = iso;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn regRead(index: u32) u32 {
|
|
||||||
@as(*volatile u32, @ptrFromInt(base + reg_ioregsel)).* = index;
|
|
||||||
return @as(*volatile u32, @ptrFromInt(base + reg_iowin)).*;
|
|
||||||
}
|
|
||||||
fn regWrite(index: u32, value: u32) void {
|
|
||||||
@as(*volatile u32, @ptrFromInt(base + reg_ioregsel)).* = index;
|
|
||||||
@as(*volatile u32, @ptrFromInt(base + reg_iowin)).* = value;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn writeEntry(n: u32, low: u32, high: u32) void {
|
|
||||||
regWrite(redir_base + 2 * n, low);
|
|
||||||
regWrite(redir_base + 2 * n + 1, high);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Map the I/O APIC and mask every redirection entry — the safe quiescent state.
|
|
||||||
pub fn init() void {
|
|
||||||
if (base == 0) return;
|
|
||||||
paging.map(base & ~@as(u64, 0xFFF), base & ~@as(u64, 0xFFF), true);
|
|
||||||
max_entries = ((regRead(reg_version) >> 16) & 0xFF) + 1;
|
|
||||||
var n: u32 = 0;
|
|
||||||
while (n < max_entries) : (n += 1) writeEntry(n, redir_mask, 0);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Route ISA `irq` to `vector` on the LAPIC `apic_id`, honouring a MADT override
|
|
||||||
/// for its GSI/polarity/trigger, and unmask it. No caller yet — groundwork for the
|
|
||||||
/// first device driver.
|
|
||||||
pub fn routeIrq(irq: u8, vector: u8, apic_id: u8) void {
|
|
||||||
if (base == 0) return;
|
|
||||||
|
|
||||||
var gsi: u32 = irq;
|
|
||||||
var flags: u16 = 0;
|
|
||||||
for (overrides[0..override_count]) |o| {
|
|
||||||
if (o.source == irq) {
|
|
||||||
gsi = o.gsi;
|
|
||||||
flags = o.flags;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (gsi < gsi_base) return;
|
|
||||||
const n = gsi - gsi_base;
|
|
||||||
if (n >= max_entries) return;
|
|
||||||
|
|
||||||
// Low dword: vector + delivery mode fixed(0) + physical dest(0), unmasked.
|
|
||||||
// MPS INTI flags: bits [1:0] polarity (3 = active low), [3:2] trigger (3 = level).
|
|
||||||
var low: u32 = vector;
|
|
||||||
if (flags & 0x3 == 3) low |= (1 << 13);
|
|
||||||
if ((flags >> 2) & 0x3 == 3) low |= (1 << 15);
|
|
||||||
const high: u32 = @as(u32, apic_id) << 24; // destination APIC ID
|
|
||||||
writeEntry(n, low, high);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Number of redirection entries the I/O APIC advertises (0 until `init`).
|
|
||||||
pub fn entryCount() u32 {
|
|
||||||
return max_entries;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The low dword of redirection entry `n` — for diagnostics/read-back.
|
|
||||||
pub fn entryLow(n: u32) u32 {
|
|
||||||
if (base == 0) return 0;
|
|
||||||
return regRead(redir_base + 2 * n);
|
|
||||||
}
|
|
||||||
@@ -1,180 +0,0 @@
|
|||||||
# x86_64 low-level entry code: the CPU-exception stubs, plus the GDT/IDT load
|
|
||||||
# helpers. Kept in a dedicated assembly file rather than inline asm because these
|
|
||||||
# need real labels and cross-symbol jumps/calls (isr_common, exceptionHandler),
|
|
||||||
# and because `lgdt`/`lidt` memory operands aren't expressible in Zig inline asm.
|
|
||||||
#
|
|
||||||
# Each exception vector normalises the stack to a uniform trap frame — a dummy
|
|
||||||
# error code where the CPU pushes none, then the vector number — and jumps to the
|
|
||||||
# shared tail, which saves the general registers and calls the Zig handler with a
|
|
||||||
# pointer to the frame (matching src/arch/x86_64/idt.zig's CpuState).
|
|
||||||
|
|
||||||
.text
|
|
||||||
|
|
||||||
# gdt_flush(rdi = *GDT descriptor): load the GDT, reload the data segment
|
|
||||||
# registers to the data selector, and reload CS to the code selector. CS can't be
|
|
||||||
# set with mov, so we far-return through the caller's own return address.
|
|
||||||
.global gdt_flush
|
|
||||||
gdt_flush:
|
|
||||||
lgdt (%rdi)
|
|
||||||
mov $0x10, %ax # kernel data selector
|
|
||||||
mov %ax, %ds
|
|
||||||
mov %ax, %es
|
|
||||||
mov %ax, %ss
|
|
||||||
mov %ax, %fs
|
|
||||||
mov %ax, %gs
|
|
||||||
pop %rax # caller's return address
|
|
||||||
push $0x08 # kernel code selector (new CS)
|
|
||||||
push %rax # return address (new RIP)
|
|
||||||
lretq
|
|
||||||
|
|
||||||
# idt_flush(rdi = *IDT descriptor): load the IDT.
|
|
||||||
.global idt_flush
|
|
||||||
idt_flush:
|
|
||||||
lidt (%rdi)
|
|
||||||
ret
|
|
||||||
|
|
||||||
# load_tr(di = TSS selector): load the task register.
|
|
||||||
.global load_tr
|
|
||||||
load_tr:
|
|
||||||
ltr %di
|
|
||||||
ret
|
|
||||||
|
|
||||||
# switch_context(rdi = &old_task.rsp, rsi = new_task.rsp)
|
|
||||||
# Cooperative context switch: save the callee-saved registers on the current
|
|
||||||
# stack, stash the stack pointer in the old task, load the new task's stack
|
|
||||||
# pointer, restore its callee-saved registers, and return into it. Caller-saved
|
|
||||||
# registers are the compiler's responsibility (this looks like a normal call).
|
|
||||||
.global switch_context
|
|
||||||
switch_context:
|
|
||||||
push %rbx
|
|
||||||
push %rbp
|
|
||||||
push %r12
|
|
||||||
push %r13
|
|
||||||
push %r14
|
|
||||||
push %r15
|
|
||||||
mov %rsp, (%rdi) # save old stack pointer into old_task.rsp
|
|
||||||
mov %rsi, %rsp # switch to the new task's stack
|
|
||||||
pop %r15
|
|
||||||
pop %r14
|
|
||||||
pop %r13
|
|
||||||
pop %r12
|
|
||||||
pop %rbp
|
|
||||||
pop %rbx
|
|
||||||
ret # return into the new task's saved instruction pointer
|
|
||||||
|
|
||||||
# task_trampoline: the first thing a freshly-spawned task runs. init_task_stack
|
|
||||||
# leaves its entry function in r15. New tasks start with interrupts enabled.
|
|
||||||
.global task_trampoline
|
|
||||||
task_trampoline:
|
|
||||||
sti
|
|
||||||
call *%r15 # call the task entry (fn() void)
|
|
||||||
1: hlt # if the entry returns, idle (still preemptible)
|
|
||||||
jmp 1b
|
|
||||||
|
|
||||||
# Stub for a vector the CPU does NOT push an error code for: push a dummy 0.
|
|
||||||
.macro STUB_NOERR vec
|
|
||||||
.global isr\vec
|
|
||||||
isr\vec:
|
|
||||||
pushq $0
|
|
||||||
pushq $\vec
|
|
||||||
jmp isr_common
|
|
||||||
.endm
|
|
||||||
|
|
||||||
# Stub for a vector the CPU DOES push an error code for: leave it in place.
|
|
||||||
.macro STUB_ERR vec
|
|
||||||
.global isr\vec
|
|
||||||
isr\vec:
|
|
||||||
pushq $\vec
|
|
||||||
jmp isr_common
|
|
||||||
.endm
|
|
||||||
|
|
||||||
STUB_NOERR 0
|
|
||||||
STUB_NOERR 1
|
|
||||||
STUB_NOERR 2
|
|
||||||
STUB_NOERR 3
|
|
||||||
STUB_NOERR 4
|
|
||||||
STUB_NOERR 5
|
|
||||||
STUB_NOERR 6
|
|
||||||
STUB_NOERR 7
|
|
||||||
STUB_ERR 8
|
|
||||||
STUB_NOERR 9
|
|
||||||
STUB_ERR 10
|
|
||||||
STUB_ERR 11
|
|
||||||
STUB_ERR 12
|
|
||||||
STUB_ERR 13
|
|
||||||
STUB_ERR 14
|
|
||||||
STUB_NOERR 15
|
|
||||||
STUB_NOERR 16
|
|
||||||
STUB_ERR 17
|
|
||||||
STUB_NOERR 18
|
|
||||||
STUB_NOERR 19
|
|
||||||
STUB_NOERR 20
|
|
||||||
STUB_ERR 21
|
|
||||||
STUB_NOERR 22
|
|
||||||
STUB_NOERR 23
|
|
||||||
STUB_NOERR 24
|
|
||||||
STUB_NOERR 25
|
|
||||||
STUB_NOERR 26
|
|
||||||
STUB_NOERR 27
|
|
||||||
STUB_NOERR 28
|
|
||||||
STUB_NOERR 29
|
|
||||||
STUB_NOERR 30
|
|
||||||
STUB_NOERR 31
|
|
||||||
|
|
||||||
# Device-interrupt vectors (timer, spurious, room for more). None push an error
|
|
||||||
# code, so they all use the dummy-zero form.
|
|
||||||
STUB_NOERR 32
|
|
||||||
STUB_NOERR 33
|
|
||||||
STUB_NOERR 34
|
|
||||||
STUB_NOERR 35
|
|
||||||
STUB_NOERR 36
|
|
||||||
STUB_NOERR 37
|
|
||||||
STUB_NOERR 38
|
|
||||||
STUB_NOERR 39
|
|
||||||
STUB_NOERR 40
|
|
||||||
STUB_NOERR 41
|
|
||||||
STUB_NOERR 42
|
|
||||||
STUB_NOERR 43
|
|
||||||
STUB_NOERR 44
|
|
||||||
STUB_NOERR 45
|
|
||||||
STUB_NOERR 46
|
|
||||||
STUB_NOERR 47
|
|
||||||
|
|
||||||
.extern interruptDispatch
|
|
||||||
|
|
||||||
# Shared tail. Register push order here defines the CpuState field order.
|
|
||||||
isr_common:
|
|
||||||
push %rax
|
|
||||||
push %rbx
|
|
||||||
push %rcx
|
|
||||||
push %rdx
|
|
||||||
push %rsi
|
|
||||||
push %rdi
|
|
||||||
push %rbp
|
|
||||||
push %r8
|
|
||||||
push %r9
|
|
||||||
push %r10
|
|
||||||
push %r11
|
|
||||||
push %r12
|
|
||||||
push %r13
|
|
||||||
push %r14
|
|
||||||
push %r15
|
|
||||||
mov %rsp, %rdi # first argument: pointer to the trap frame
|
|
||||||
call interruptDispatch
|
|
||||||
pop %r15
|
|
||||||
pop %r14
|
|
||||||
pop %r13
|
|
||||||
pop %r12
|
|
||||||
pop %r11
|
|
||||||
pop %r10
|
|
||||||
pop %r9
|
|
||||||
pop %r8
|
|
||||||
pop %rbp
|
|
||||||
pop %rdi
|
|
||||||
pop %rsi
|
|
||||||
pop %rdx
|
|
||||||
pop %rcx
|
|
||||||
pop %rbx
|
|
||||||
pop %rax
|
|
||||||
add $16, %rsp # drop the vector and error code
|
|
||||||
iretq
|
|
||||||
@@ -1,47 +0,0 @@
|
|||||||
/* Kernel link layout.
|
|
||||||
*
|
|
||||||
* The kernel is linked at a fixed low physical address (set by `image_base` in
|
|
||||||
* build.zig). UEFI runs with memory identity-mapped, so the bootloader can load
|
|
||||||
* each PT_LOAD segment to the physical address matching its virtual address and
|
|
||||||
* jump straight to _start — no page tables to build yet. (Moving to a
|
|
||||||
* higher-half virtual base is a later step, once the bootloader sets up paging.)
|
|
||||||
*/
|
|
||||||
|
|
||||||
ENTRY(_start)
|
|
||||||
|
|
||||||
/* One loadable segment per permission set, so the loader can map .text as R+X,
|
|
||||||
* .rodata as R, and .data/.bss as R+W. FLAGS bits: 1=X, 2=W, 4=R. */
|
|
||||||
PHDRS {
|
|
||||||
text PT_LOAD FLAGS(5); /* R + X */
|
|
||||||
rodata PT_LOAD FLAGS(4); /* R */
|
|
||||||
data PT_LOAD FLAGS(6); /* R + W */
|
|
||||||
}
|
|
||||||
|
|
||||||
SECTIONS {
|
|
||||||
.text ALIGN(4K) : {
|
|
||||||
*(.text .text.*)
|
|
||||||
} :text
|
|
||||||
|
|
||||||
.rodata ALIGN(4K) : {
|
|
||||||
*(.rodata .rodata.*)
|
|
||||||
} :rodata
|
|
||||||
|
|
||||||
.data ALIGN(4K) : {
|
|
||||||
*(.data .data.*)
|
|
||||||
} :data
|
|
||||||
|
|
||||||
/* .bss occupies memory but not file space. The loader zeroes it via the
|
|
||||||
* gap between each PT_LOAD segment's file size and memory size, so no
|
|
||||||
* boundary symbols are needed here. (Zig's self-hosted linker also does not
|
|
||||||
* yet honour linker-script symbol assignments.) */
|
|
||||||
.bss ALIGN(4K) : {
|
|
||||||
*(.bss .bss.*)
|
|
||||||
*(COMMON)
|
|
||||||
} :data
|
|
||||||
|
|
||||||
/DISCARD/ : {
|
|
||||||
*(.comment)
|
|
||||||
*(.note .note.*)
|
|
||||||
*(.eh_frame .eh_frame_hdr)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,153 +0,0 @@
|
|||||||
//! The kernel's page tables and virtual memory manager.
|
|
||||||
//!
|
|
||||||
//! Builds our own 4-level page tables and switches CR3 onto them, replacing the
|
|
||||||
//! firmware's. Unlike the earlier bootstrap this maps with real permissions:
|
|
||||||
//! RAM is identity-mapped read-write + no-execute, the kernel's own segments get
|
|
||||||
//! their ELF permissions (code R+X, rodata R, data R+W+NX), and page 0 is left
|
|
||||||
//! unmapped as a null guard. It also exposes map/unmap for on-demand mapping,
|
|
||||||
//! which the kernel heap will build on.
|
|
||||||
//!
|
|
||||||
//! Everything is 4 KiB pages — precise and simple; the extra table memory is
|
|
||||||
//! negligible against available RAM.
|
|
||||||
|
|
||||||
const danos = @import("danos");
|
|
||||||
const io = @import("io.zig");
|
|
||||||
|
|
||||||
const page_size = danos.page_size;
|
|
||||||
|
|
||||||
// Page-table entry bits.
|
|
||||||
const present: u64 = 1 << 0;
|
|
||||||
const writable: u64 = 1 << 1;
|
|
||||||
const no_execute: u64 = 1 << 63;
|
|
||||||
const addr_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
|
||||||
|
|
||||||
// ELF segment flags (p_flags).
|
|
||||||
const pf_x: u32 = 1;
|
|
||||||
const pf_w: u32 = 2;
|
|
||||||
|
|
||||||
// State kept after init so map()/unmap() can serve later callers (e.g. the heap).
|
|
||||||
var kernel_pml4: u64 = 0;
|
|
||||||
var alloc_frame: *const fn () ?u64 = undefined;
|
|
||||||
|
|
||||||
fn tableAt(phys: u64) *[512]u64 {
|
|
||||||
return @ptrFromInt(phys);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn allocTable() u64 {
|
|
||||||
const frame = alloc_frame() orelse @panic("paging: out of memory building page tables");
|
|
||||||
@memset(tableAt(frame)[0..], 0);
|
|
||||||
return frame;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Return the table an entry points at, creating it if empty. Intermediate
|
|
||||||
/// entries are writable and executable so the leaf's bits govern (a page is
|
|
||||||
/// writable only if every level is; non-executable if any level is).
|
|
||||||
fn descend(entry: *u64) u64 {
|
|
||||||
if (entry.* & present != 0) return entry.* & addr_mask;
|
|
||||||
const frame = allocTable();
|
|
||||||
entry.* = frame | present | writable;
|
|
||||||
return frame;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Map one 4 KiB page `virt` -> `phys` with `flags` (present is added).
|
|
||||||
fn mapPage(pml4: u64, virt: u64, phys: u64, flags: u64) void {
|
|
||||||
const pml4e = &tableAt(pml4)[(virt >> 39) & 0x1FF];
|
|
||||||
const pdpt = descend(pml4e);
|
|
||||||
const pdpte = &tableAt(pdpt)[(virt >> 30) & 0x1FF];
|
|
||||||
const pd = descend(pdpte);
|
|
||||||
const pde = &tableAt(pd)[(virt >> 21) & 0x1FF];
|
|
||||||
const pt = descend(pde);
|
|
||||||
tableAt(pt)[(virt >> 12) & 0x1FF] = (phys & addr_mask) | flags | present;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Identity-map [base, base+len) with `flags`, rounded out to whole pages.
|
|
||||||
fn mapRangeIdentity(pml4: u64, base: u64, len: u64, flags: u64) void {
|
|
||||||
var addr = base & ~@as(u64, page_size - 1);
|
|
||||||
const end = base + len;
|
|
||||||
while (addr < end) : (addr += page_size) {
|
|
||||||
if (addr == 0) continue; // leave page 0 unmapped: the null guard
|
|
||||||
mapPage(pml4, addr, addr, flags);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn regions(mm: danos.MemoryMap) []const danos.MemoryRegion {
|
|
||||||
return @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Enable the NX bit in the page-table format (EFER.NXE). Must happen before we
|
|
||||||
/// load a CR3 whose entries set the NX bit, or those bits are reserved and fault.
|
|
||||||
fn enableNx() void {
|
|
||||||
const efer_msr = 0xC0000080;
|
|
||||||
io.wrmsr(efer_msr, io.rdmsr(efer_msr) | (1 << 11));
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Build the address space and switch onto it.
|
|
||||||
pub fn init(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void {
|
|
||||||
alloc_frame = allocFrame;
|
|
||||||
enableNx();
|
|
||||||
const pml4 = allocTable();
|
|
||||||
|
|
||||||
// 1. All RAM identity-mapped RW + NX. Non-RAM (MMIO) is skipped and stays
|
|
||||||
// unmapped unless mapped explicitly below.
|
|
||||||
for (regions(boot_info.memory_map)) |r| {
|
|
||||||
if (r.kind == .mmio) continue;
|
|
||||||
mapRangeIdentity(pml4, r.base, r.pages * page_size, present | writable | no_execute);
|
|
||||||
}
|
|
||||||
|
|
||||||
// 2. The framebuffer and the Local APIC (device memory we need), RW + NX.
|
|
||||||
const fb = boot_info.framebuffer;
|
|
||||||
mapRangeIdentity(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute);
|
|
||||||
mapPage(pml4, 0xFEE00000, 0xFEE00000, present | writable | no_execute);
|
|
||||||
|
|
||||||
// 3. Overlay the kernel's own segments with their real ELF permissions,
|
|
||||||
// replacing the blanket RW+NX from step 1: code becomes R+X, rodata R,
|
|
||||||
// data R+W+NX. This is the W^X guarantee.
|
|
||||||
for (boot_info.kernel_segments[0..boot_info.kernel_segment_count]) |seg| {
|
|
||||||
var flags: u64 = present;
|
|
||||||
if (seg.flags & pf_w != 0) flags |= writable;
|
|
||||||
if (seg.flags & pf_x == 0) flags |= no_execute;
|
|
||||||
var addr = seg.virt;
|
|
||||||
const end = seg.virt + seg.pages * page_size;
|
|
||||||
while (addr < end) : (addr += page_size) mapPage(pml4, addr, addr, flags);
|
|
||||||
}
|
|
||||||
|
|
||||||
kernel_pml4 = pml4;
|
|
||||||
asm volatile ("mov %[pml4], %%cr3"
|
|
||||||
:
|
|
||||||
: [pml4] "r" (pml4),
|
|
||||||
: .{ .memory = true }
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Map a page into the kernel address space on demand (for the heap, etc.).
|
|
||||||
/// `writable_page` controls W; pages are always mapped non-executable.
|
|
||||||
pub fn map(virt: u64, phys: u64, writable_page: bool) void {
|
|
||||||
var flags: u64 = present | no_execute;
|
|
||||||
if (writable_page) flags |= writable;
|
|
||||||
mapPage(kernel_pml4, virt, phys, flags);
|
|
||||||
invalidate(virt);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Remove a mapping and flush it from the TLB.
|
|
||||||
pub fn unmap(virt: u64) void {
|
|
||||||
const pml4e = tableAt(kernel_pml4)[(virt >> 39) & 0x1FF];
|
|
||||||
if (pml4e & present == 0) return;
|
|
||||||
const pdpte = tableAt(pml4e & addr_mask)[(virt >> 30) & 0x1FF];
|
|
||||||
if (pdpte & present == 0) return;
|
|
||||||
const pde = tableAt(pdpte & addr_mask)[(virt >> 21) & 0x1FF];
|
|
||||||
if (pde & present == 0) return;
|
|
||||||
tableAt(pde & addr_mask)[(virt >> 12) & 0x1FF] = 0;
|
|
||||||
invalidate(virt);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn invalidate(virt: u64) void {
|
|
||||||
// invlpg needs its operand via a register-indirect memory reference that Zig
|
|
||||||
// inline asm won't form directly, so stage the address in a register first.
|
|
||||||
asm volatile (
|
|
||||||
\\mov %[v], %%rax
|
|
||||||
\\invlpg (%%rax)
|
|
||||||
:
|
|
||||||
: [v] "r" (virt),
|
|
||||||
: .{ .rax = true, .memory = true }
|
|
||||||
);
|
|
||||||
}
|
|
||||||
@@ -1,48 +0,0 @@
|
|||||||
//! Task State Segment and its interrupt stack. In long mode the TSS's main job
|
|
||||||
//! is the Interrupt Stack Table: an IDT gate can name an IST entry, and the CPU
|
|
||||||
//! switches to that stack when the exception fires — no matter how broken the
|
|
||||||
//! interrupted stack was. We use IST1 for the double-fault handler, so a fault
|
|
||||||
//! that happens *because* the current stack is unusable still lands on solid
|
|
||||||
//! ground instead of triple-faulting.
|
|
||||||
|
|
||||||
const gdt = @import("gdt.zig");
|
|
||||||
|
|
||||||
/// x86_64 TSS. `packed` because several 64-bit fields sit at 4-byte-unaligned
|
|
||||||
/// offsets (rsp0 at byte 4), which a normal struct would pad away.
|
|
||||||
const Tss = packed struct {
|
|
||||||
reserved0: u32 = 0,
|
|
||||||
rsp0: u64 = 0,
|
|
||||||
rsp1: u64 = 0,
|
|
||||||
rsp2: u64 = 0,
|
|
||||||
reserved1: u64 = 0,
|
|
||||||
ist1: u64 = 0,
|
|
||||||
ist2: u64 = 0,
|
|
||||||
ist3: u64 = 0,
|
|
||||||
ist4: u64 = 0,
|
|
||||||
ist5: u64 = 0,
|
|
||||||
ist6: u64 = 0,
|
|
||||||
ist7: u64 = 0,
|
|
||||||
reserved2: u64 = 0,
|
|
||||||
reserved3: u16 = 0,
|
|
||||||
iomap_base: u16 = 0,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// The IST slot (1-based, as the IDT gate encodes it) used for critical faults.
|
|
||||||
pub const double_fault_ist = 1;
|
|
||||||
|
|
||||||
var tss: Tss align(16) = .{};
|
|
||||||
|
|
||||||
/// Dedicated stack for IST1. Static so it needs no allocator and is always valid.
|
|
||||||
var ist1_stack: [16 * 1024]u8 align(16) = undefined;
|
|
||||||
|
|
||||||
/// Loads the task register with the TSS selector. Defined in isr.s.
|
|
||||||
extern fn load_tr(selector: u16) callconv(.c) void;
|
|
||||||
|
|
||||||
/// Point IST1 at its stack, publish the TSS through the GDT, and load it into the
|
|
||||||
/// task register. Requires the GDT to already be loaded (gdt.init first).
|
|
||||||
pub fn init() void {
|
|
||||||
tss.ist1 = @intFromPtr(&ist1_stack) + ist1_stack.len; // stacks grow down
|
|
||||||
tss.iomap_base = @sizeOf(Tss); // == limit: no I/O permission bitmap
|
|
||||||
gdt.setTss(@intFromPtr(&tss), @sizeOf(Tss) - 1);
|
|
||||||
load_tr(gdt.tss_selector);
|
|
||||||
}
|
|
||||||
@@ -1,290 +0,0 @@
|
|||||||
const std = @import("std");
|
|
||||||
const danos = @import("danos");
|
|
||||||
const arch = @import("arch");
|
|
||||||
const console = @import("console.zig");
|
|
||||||
const log = @import("log.zig");
|
|
||||||
const pmm = @import("pmm.zig");
|
|
||||||
const heap = @import("heap.zig");
|
|
||||||
const scheduler = @import("scheduler.zig");
|
|
||||||
const platform = @import("platform");
|
|
||||||
const tests = @import("tests.zig");
|
|
||||||
const build_options = @import("build_options");
|
|
||||||
const BootInfo = danos.BootInfo;
|
|
||||||
|
|
||||||
/// The calling convention used to enter the kernel. Pinned to SysV explicitly:
|
|
||||||
/// the bootloader is built for the UEFI target, whose C convention is Microsoft
|
|
||||||
/// x64 (first argument in RCX), while the kernel is SysV (first argument in
|
|
||||||
/// RDI). Both sides reference this so the `boot_info` pointer lands in the
|
|
||||||
/// register the other expects. `danos.kernel_abi` re-exports it to the loader.
|
|
||||||
pub const kernel_abi = danos.kernel_abi;
|
|
||||||
|
|
||||||
// POST/checkpoint codes emitted to I/O port 0x80 at boot milestones — the
|
|
||||||
// last-resort progress signal on a machine with no text output at all.
|
|
||||||
const cp_entry = 0x10;
|
|
||||||
const cp_paging = 0x20;
|
|
||||||
const cp_heap = 0x30;
|
|
||||||
const cp_discovery = 0x40;
|
|
||||||
const cp_scheduler = 0x50;
|
|
||||||
const cp_timer = 0x60;
|
|
||||||
const cp_running = 0x70;
|
|
||||||
const cp_exception = 0xE0;
|
|
||||||
const cp_panic = 0xEE;
|
|
||||||
|
|
||||||
/// Kernel entry point. The bootloader jumps here after `ExitBootServices` with a
|
|
||||||
/// pointer to the handoff data. There is no runtime, no stack unwinding, and no
|
|
||||||
/// caller to return to, so this never returns.
|
|
||||||
export fn _start(boot_info: *const BootInfo) callconv(kernel_abi) noreturn {
|
|
||||||
kmain(boot_info);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn kmain(boot_info: *const BootInfo) noreturn {
|
|
||||||
// The **log** is the machine-readable diagnostic stream: it fans out to every
|
|
||||||
// *diagnostic* channel that exists (serial, the 0xE9 debug console, and later a
|
|
||||||
// file on a ramdisk/USB/SSD), so a message survives as long as any is present.
|
|
||||||
// A headless, serial-less machine still boots correctly — it just goes quiet,
|
|
||||||
// with port-0x80 checkpoints as the only progress signal.
|
|
||||||
arch.serialInit();
|
|
||||||
log.addSink(arch.serialWrite);
|
|
||||||
if (arch.debugconPresent()) log.addSink(arch.debugconWrite);
|
|
||||||
|
|
||||||
// The **framebuffer** is deliberately *not* a log sink. It's a separate output
|
|
||||||
// surface — a bootstrap text console today, a graphics device driver later — so
|
|
||||||
// we never assume the OS is text-based. Only a few user-facing status lines
|
|
||||||
// (via `status`) and panics are mirrored to it; the verbose log stays out.
|
|
||||||
const fb = boot_info.framebuffer;
|
|
||||||
console.init(fb);
|
|
||||||
|
|
||||||
log.checkpoint(cp_entry);
|
|
||||||
|
|
||||||
// Catch CPU exceptions before doing anything that might fault: install our
|
|
||||||
// reporter, then bring up the GDT + IDT.
|
|
||||||
arch.setFaultHandler(onException);
|
|
||||||
arch.init();
|
|
||||||
|
|
||||||
status("danos: initialising kernel...\n");
|
|
||||||
log.write(if (console.present())
|
|
||||||
"danos: framebuffer console online (bootstrap; graphics driver later)\n"
|
|
||||||
else
|
|
||||||
"danos: no framebuffer (headless) -> logging to serial/debugcon only\n");
|
|
||||||
log.write("danos: cpu tables online (GDT, IDT, TSS)\n");
|
|
||||||
log.print(" resolution : {d}x{d}\n", .{ fb.width, fb.height });
|
|
||||||
log.print(" pitch : {d} bytes\n", .{fb.pitch});
|
|
||||||
log.print(" format : {s}\n", .{@tagName(fb.format)});
|
|
||||||
log.print(" framebuffer: 0x{x:0>16}\n", .{fb.base});
|
|
||||||
log.print (" footprint : {d} MiB\n", .{(fb.pitch * fb.height) / (1024 * 1024)});
|
|
||||||
|
|
||||||
// Summarise the physical memory the loader handed us. The array is danos's
|
|
||||||
// own MemoryRegion, so this is a plain slice — no firmware layout in sight.
|
|
||||||
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(boot_info.memory_map.regions))[0..boot_info.memory_map.len];
|
|
||||||
var usable_pages: u64 = 0;
|
|
||||||
var reserved_pages: u64 = 0; // reserved RAM only — MMIO is device space, not RAM
|
|
||||||
for (regions) |r| {
|
|
||||||
switch (r.kind) {
|
|
||||||
.usable => usable_pages += r.pages,
|
|
||||||
.reserved, .acpi_tables, .acpi_nvs => reserved_pages += r.pages,
|
|
||||||
.mmio => {},
|
|
||||||
}
|
|
||||||
}
|
|
||||||
const total_pages = usable_pages + reserved_pages;
|
|
||||||
const total_bytes = total_pages * danos.page_size;
|
|
||||||
const gib = 1 << 30;
|
|
||||||
|
|
||||||
log.write("\ndanos: physical memory\n");
|
|
||||||
log.print(" total RAM : {d}.{d:0>2} GiB ({d} MiB) - RAM the firmware reported\n", .{ total_bytes / gib, (total_bytes % gib) * 100 / gib, mib(total_pages) });
|
|
||||||
log.print(" usable : {d} MiB - free RAM (incl. reclaimed boot-services memory)\n", .{mib(usable_pages)});
|
|
||||||
log.print(" reserved : {d} MiB - kernel image, boot stack, ACPI, runtime services\n", .{mib(reserved_pages)});
|
|
||||||
log.print(" regions : {d} - entries in the firmware memory map\n", .{regions.len});
|
|
||||||
|
|
||||||
// Bring up the physical frame allocator over that map, and prove it works:
|
|
||||||
// allocate three frames, then hand them back.
|
|
||||||
pmm.init(boot_info.memory_map);
|
|
||||||
const s1 = pmm.stats();
|
|
||||||
log.print("\ndanos: frame allocator online\n", .{});
|
|
||||||
log.print(" free frames: {d} ({d} MiB)\n", .{ s1.free_frames, mib(s1.free_frames) });
|
|
||||||
const f0 = pmm.alloc();
|
|
||||||
const f1 = pmm.alloc();
|
|
||||||
const f2 = pmm.alloc();
|
|
||||||
log.print(" alloc x3 : 0x{x} 0x{x} 0x{x}\n", .{ f0 orelse 0, f1 orelse 0, f2 orelse 0 });
|
|
||||||
if (f0) |p| pmm.free(p);
|
|
||||||
if (f1) |p| pmm.free(p);
|
|
||||||
if (f2) |p| pmm.free(p);
|
|
||||||
log.print(" after free : {d} frames free\n", .{pmm.stats().free_frames});
|
|
||||||
|
|
||||||
// Switch off the firmware's page tables onto our own (with real permissions).
|
|
||||||
arch.enablePaging(pmm.alloc, boot_info);
|
|
||||||
log.checkpoint(cp_paging);
|
|
||||||
log.print("\ndanos: paging enabled\n", .{});
|
|
||||||
log.print(" page tables: CR3 = 0x{x:0>16}\n", .{arch.readCr3()});
|
|
||||||
log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_info.kernel_segment_count});
|
|
||||||
|
|
||||||
// Bring up the kernel heap (dynamic allocation), built on the VMM.
|
|
||||||
heap.init();
|
|
||||||
log.checkpoint(cp_heap);
|
|
||||||
log.write("\ndanos: kernel heap online\n");
|
|
||||||
// Measure the amount of resources the kernel is actually using
|
|
||||||
const s2 = pmm.stats();
|
|
||||||
log.print(" Kernel footprint: {d} KiB\n", .{kib(s1.free_frames - s2.free_frames)});
|
|
||||||
|
|
||||||
// Enumerate hardware from the firmware tables (ACPI here) into a generic
|
|
||||||
// device tree, then list it. Discovery walks ACPI memory directly (identity-
|
|
||||||
// mapped) and maps PCIe config space on demand via the VMM. A failure here is
|
|
||||||
// not fatal yet — log it and carry on.
|
|
||||||
const hal = platform.Hal{
|
|
||||||
.mapMmio = arch.mapPage,
|
|
||||||
.pioRead = arch.pioRead,
|
|
||||||
.pioWrite = arch.pioWrite,
|
|
||||||
};
|
|
||||||
if (platform.discover(boot_info, heap.allocator(), hal)) |devtree| {
|
|
||||||
var dt = devtree;
|
|
||||||
log.write("\ndanos: device discovery online\n");
|
|
||||||
dt.dump(log.write);
|
|
||||||
|
|
||||||
// Power register map extracted from the FADT + AML, for confidence it parsed.
|
|
||||||
const pw = platform.powerInfo();
|
|
||||||
log.write("danos: power\n");
|
|
||||||
log.print(" pm1a_cnt : {s} 0x{x} (width {d})\n", .{ if (pw.pm1a_cnt.mmio) "mmio" else "io", pw.pm1a_cnt.address, pw.pm1a_cnt.width });
|
|
||||||
if (pw.s5) |s| {
|
|
||||||
log.print(" S5 slp_typ : a={d} b={d}\n", .{ s.slp_typ_a, s.slp_typ_b });
|
|
||||||
} else {
|
|
||||||
log.write(" S5 slp_typ : (not found)\n");
|
|
||||||
}
|
|
||||||
log.print(" reset : supported={} {s} 0x{x} val 0x{x}\n", .{ pw.reset_supported, if (pw.reset.mmio) "mmio" else "io", pw.reset.address, pw.reset_value });
|
|
||||||
|
|
||||||
// AML namespace parse integrity: consumed should equal total.
|
|
||||||
const am = platform.amlStats();
|
|
||||||
log.print(" aml : {d} namespace nodes, parsed {d}/{d} bytes\n", .{ am.nodes, am.consumed, am.total });
|
|
||||||
|
|
||||||
// Feed the arch layer the discovered addresses/facts so it makes no legacy
|
|
||||||
// assumptions — the point of all this on UEFI Class 3 firmware. MMIO bases
|
|
||||||
// (HPET, I/O APIC) come from the device tree; scalar facts from ACPI.
|
|
||||||
const pinfo = platform.platformInfo();
|
|
||||||
const hpet_base: u64 = if (dt.firstOfClass(.timer)) |t|
|
|
||||||
(if (t.firstResource(.memory)) |r| r.start else 0)
|
|
||||||
else
|
|
||||||
0;
|
|
||||||
var ioapic_base: u64 = 0;
|
|
||||||
var ioapic_gsi: u32 = 0;
|
|
||||||
if (dt.firstOfClass(.interrupt_controller)) |ic| {
|
|
||||||
if (ic.firstResource(.memory)) |r| ioapic_base = r.start;
|
|
||||||
if (ic.firstResource(.irq)) |r| ioapic_gsi = @intCast(r.start);
|
|
||||||
}
|
|
||||||
var isos: [16]arch.IsoEntry = undefined;
|
|
||||||
const iso_n = @min(pinfo.override_count, isos.len);
|
|
||||||
for (0..iso_n) |i| isos[i] = .{
|
|
||||||
.source = pinfo.overrides[i].source,
|
|
||||||
.gsi = pinfo.overrides[i].gsi,
|
|
||||||
.flags = pinfo.overrides[i].flags,
|
|
||||||
};
|
|
||||||
const pm_timer: ?arch.PmTimer = if (pinfo.pm_timer.present())
|
|
||||||
.{ .mmio = pinfo.pm_timer.mmio, .address = pinfo.pm_timer.address, .is_32bit = pinfo.pm_timer_32bit }
|
|
||||||
else
|
|
||||||
null;
|
|
||||||
arch.configurePlatform(.{
|
|
||||||
.pic_present = pinfo.pic_present,
|
|
||||||
.hpet_base = hpet_base,
|
|
||||||
.pm_timer = pm_timer,
|
|
||||||
.ioapic_base = ioapic_base,
|
|
||||||
.ioapic_gsi_base = ioapic_gsi,
|
|
||||||
.overrides = isos[0..iso_n],
|
|
||||||
});
|
|
||||||
if (pinfo.spcr_uart) |u| arch.serialReconfigure(u.mmio, u.address);
|
|
||||||
|
|
||||||
log.write("danos: platform\n");
|
|
||||||
log.print(" 8259 PIC : {s}\n", .{if (pinfo.pic_present) "present" else "absent"});
|
|
||||||
log.print(" lapic base : 0x{x}\n", .{pinfo.lapic_base});
|
|
||||||
log.print(" hpet base : 0x{x}\n", .{hpet_base});
|
|
||||||
log.print(" pm timer : {s} 0x{x} ({s})\n", .{ if (pinfo.pm_timer.mmio) "mmio" else "io", pinfo.pm_timer.address, if (pinfo.pm_timer_32bit) "32-bit" else "24-bit" });
|
|
||||||
if (pinfo.spcr_uart) |u| {
|
|
||||||
log.print(" console UART: {s} 0x{x} (SPCR type {d})\n", .{ if (u.mmio) "mmio" else "io", u.address, pinfo.spcr_kind });
|
|
||||||
} else {
|
|
||||||
log.write(" console UART: none in SPCR -> legacy COM1\n");
|
|
||||||
}
|
|
||||||
log.print(" ioapic : base 0x{x}, {d} inputs (masked); entry0 low 0x{x}\n", .{ ioapic_base, arch.ioapicEntryCount(), arch.ioapicEntryLow(0) });
|
|
||||||
} else |err| {
|
|
||||||
log.print("\ndanos: device discovery failed: {s}\n", .{@errorName(err)});
|
|
||||||
}
|
|
||||||
log.checkpoint(cp_discovery);
|
|
||||||
|
|
||||||
// Register the current context as the first task before enabling preemption.
|
|
||||||
scheduler.init(4);
|
|
||||||
log.checkpoint(cp_scheduler);
|
|
||||||
log.write("\ndanos: scheduler online\n");
|
|
||||||
|
|
||||||
// Start the timer and unmask interrupts — the kernel now has a heartbeat, and
|
|
||||||
// the timer preempts among tasks.
|
|
||||||
arch.startTimer();
|
|
||||||
arch.enableInterrupts();
|
|
||||||
log.checkpoint(cp_timer);
|
|
||||||
log.print("danos: timer online ({d} Hz tick; LAPIC {d} MHz, TSC {d} MHz; calibrated via {s})\n", .{ arch.timer_hz, arch.lapicHz() / 1_000_000, arch.tscHz() / 1_000_000, arch.timerCalibrationSource() });
|
|
||||||
|
|
||||||
// In a test build (`zig build -Dtest-case=<name>`), run that case and stop.
|
|
||||||
// Normal builds fall through to the idle halt.
|
|
||||||
if (build_options.test_case) |case| {
|
|
||||||
tests.run(case, boot_info);
|
|
||||||
arch.halt();
|
|
||||||
}
|
|
||||||
|
|
||||||
log.checkpoint(cp_running);
|
|
||||||
status("kernel initialised.\n");
|
|
||||||
|
|
||||||
// TODO: init process
|
|
||||||
|
|
||||||
status("\nnothing left to do; halting CPU.\n");
|
|
||||||
|
|
||||||
arch.halt();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// A user-facing status line: to the diagnostic `log` *and* the on-screen console
|
|
||||||
/// (if a framebuffer is present). The verbose log uses `log.*` directly and never
|
|
||||||
/// touches the framebuffer.
|
|
||||||
fn status(msg: []const u8) void {
|
|
||||||
log.write(msg);
|
|
||||||
console.write(msg);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn statusPrint(comptime fmt: []const u8, args: anytype) void {
|
|
||||||
var buf: [256]u8 = undefined;
|
|
||||||
status(std.fmt.bufPrint(&buf, fmt, args) catch return);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Frames (4 KiB pages) to whole MiB.
|
|
||||||
fn mib(pages: u64) u64 {
|
|
||||||
return pages * danos.page_size / (1024 * 1024);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn kib(frames: u64) u64 {
|
|
||||||
return frames * danos.page_size / (1024);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Report a CPU exception and halt. There's no fault recovery yet, so any
|
|
||||||
/// exception is terminal — but it reports what and where (to every output sink,
|
|
||||||
/// plus a POST code and a persistent breadcrumb) instead of silently resetting.
|
|
||||||
fn onException(state: *const arch.CpuState) noreturn {
|
|
||||||
log.checkpoint(cp_exception);
|
|
||||||
// A fault is user-facing enough to paint on screen too (via statusPrint), on
|
|
||||||
// top of the diagnostic log.
|
|
||||||
statusPrint("\nCPU EXCEPTION: {s} (vector {d})\n", .{ arch.vectorName(state.vector), state.vector });
|
|
||||||
statusPrint(" error code : 0x{x}\n", .{state.error_code});
|
|
||||||
statusPrint(" RIP : 0x{x:0>16}\n", .{state.rip});
|
|
||||||
statusPrint(" RSP : 0x{x:0>16}\n", .{state.rsp});
|
|
||||||
if (state.vector == 14) statusPrint(" CR2 (addr) : 0x{x:0>16}\n", .{arch.readCr2()});
|
|
||||||
|
|
||||||
var buf: [128]u8 = undefined;
|
|
||||||
log.recordPanic(std.fmt.bufPrint(&buf, "CPU exception {s} (vector {d}) at RIP 0x{x}", .{ arch.vectorName(state.vector), state.vector, state.rip }) catch "cpu exception");
|
|
||||||
arch.halt();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Freestanding has no OS to receive a panic. Emit it to every output sink, drop a
|
|
||||||
/// POST code + a persistent breadcrumb (so a post-mortem can recover it even with
|
|
||||||
/// no live console), then halt. Assumes no console — the sinks self-guard.
|
|
||||||
pub const panic = std.debug.FullPanic(struct {
|
|
||||||
fn panic(msg: []const u8, first_trace_addr: ?usize) noreturn {
|
|
||||||
_ = first_trace_addr;
|
|
||||||
log.checkpoint(cp_panic);
|
|
||||||
log.recordPanic(msg);
|
|
||||||
status("\nKERNEL PANIC: ");
|
|
||||||
status(msg);
|
|
||||||
status("\n");
|
|
||||||
arch.halt();
|
|
||||||
}
|
|
||||||
}.panic);
|
|
||||||
@@ -1,251 +0,0 @@
|
|||||||
//! The scheduler: fixed-priority preemptive multitasking.
|
|
||||||
//!
|
|
||||||
//! Tasks are kernel threads (ring 0, each with its own stack). The **highest-
|
|
||||||
//! priority ready task always runs**; within a priority level, tasks round-robin.
|
|
||||||
//! Selection is O(1) — a bitmap of non-empty priority levels plus a FIFO queue per
|
|
||||||
//! level — which keeps scheduling deterministic, as a real-time kernel needs (see
|
|
||||||
//! docs/vision.md).
|
|
||||||
//!
|
|
||||||
//! Switching happens both cooperatively (`yield`) and preemptively (the timer
|
|
||||||
//! calls `tick`). See docs/scheduling.md for the interrupt-flag discipline that
|
|
||||||
//! makes those two paths coexist.
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
const arch = @import("arch");
|
|
||||||
const heap = @import("heap.zig");
|
|
||||||
|
|
||||||
/// Priority level: 0 (lowest) .. 7 (highest). 8 levels total.
|
|
||||||
pub const Priority = u3;
|
|
||||||
const num_priorities = 8;
|
|
||||||
|
|
||||||
const stack_size = 16 * 1024; // each task's kernel stack is 16 KiB
|
|
||||||
const max_tasks = 16; // the maximum number of tasks alive at once is 16 in a static sized pool
|
|
||||||
|
|
||||||
const State = enum { free, ready, running, blocked };
|
|
||||||
|
|
||||||
const Task = struct {
|
|
||||||
id: u32 = 0,
|
|
||||||
state: State = .free,
|
|
||||||
priority: Priority = 0,
|
|
||||||
rsp: usize = 0, // saved stack pointer, valid while not running
|
|
||||||
stack: []u8 = &.{},
|
|
||||||
wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping
|
|
||||||
next: ?*Task = null, // ready-queue link
|
|
||||||
};
|
|
||||||
|
|
||||||
var tasks = [_]Task{.{}} ** max_tasks;
|
|
||||||
var current: *Task = undefined;
|
|
||||||
var next_id: u32 = 1;
|
|
||||||
|
|
||||||
// Per-priority FIFO ready queues, and a bitmap of which levels are non-empty.
|
|
||||||
var ready_head: [num_priorities]?*Task = .{null} ** num_priorities;
|
|
||||||
var ready_tail: [num_priorities]?*Task = .{null} ** num_priorities;
|
|
||||||
var ready_bitmap: u8 = 0;
|
|
||||||
|
|
||||||
var preemption_enabled = true;
|
|
||||||
|
|
||||||
/// Register the currently-running kernel context as the first task, spawn the
|
|
||||||
/// idle task, and hook the timer for preemption.
|
|
||||||
pub fn init(boot_priority: Priority) void {
|
|
||||||
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
|
|
||||||
current = &tasks[0];
|
|
||||||
spawn(idle, 0); // lowest priority, always runnable — runs when nothing else is
|
|
||||||
arch.setTickHook(tick);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The idle task: run when every other task is blocked or sleeping. `hlt` waits
|
|
||||||
/// for the next interrupt at near-zero power (see docs/halting.md).
|
|
||||||
fn idle() void {
|
|
||||||
while (true) asm volatile ("hlt");
|
|
||||||
}
|
|
||||||
|
|
||||||
fn enqueue(t: *Task) void {
|
|
||||||
t.next = null;
|
|
||||||
const p: usize = t.priority;
|
|
||||||
if (ready_tail[p]) |tail| tail.next = t else ready_head[p] = t;
|
|
||||||
ready_tail[p] = t;
|
|
||||||
ready_bitmap |= levelBit(t.priority);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn dequeueHighest() ?*Task {
|
|
||||||
if (ready_bitmap == 0) return null;
|
|
||||||
const level: Priority = @intCast(num_priorities - 1 - @clz(ready_bitmap));
|
|
||||||
const t = ready_head[level].?;
|
|
||||||
ready_head[level] = t.next;
|
|
||||||
if (ready_head[level] == null) {
|
|
||||||
ready_tail[level] = null;
|
|
||||||
ready_bitmap &= ~levelBit(level);
|
|
||||||
}
|
|
||||||
t.next = null;
|
|
||||||
return t;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn levelBit(p: Priority) u8 {
|
|
||||||
return @as(u8, 1) << p;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Create a task that runs `entry` at `priority`. It becomes ready immediately.
|
|
||||||
pub fn spawn(entry: *const fn () void, priority: Priority) void {
|
|
||||||
const t = freeSlot() orelse @panic("sched: task table full");
|
|
||||||
const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack");
|
|
||||||
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack };
|
|
||||||
next_id += 1;
|
|
||||||
const top = @intFromPtr(stack.ptr) + stack.len;
|
|
||||||
t.rsp = arch.initTaskStack(top, @intFromPtr(entry));
|
|
||||||
enqueue(t);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn freeSlot() ?*Task {
|
|
||||||
for (&tasks) |*t| {
|
|
||||||
if (t.state == .free) return t;
|
|
||||||
}
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Pick the highest-priority ready task and switch to it. Interrupts must be
|
|
||||||
/// disabled by the caller.
|
|
||||||
fn schedule() void {
|
|
||||||
const prev = current;
|
|
||||||
if (prev.state == .running) {
|
|
||||||
prev.state = .ready;
|
|
||||||
enqueue(prev); // back of its level's queue (round-robin)
|
|
||||||
}
|
|
||||||
const next = dequeueHighest() orelse {
|
|
||||||
prev.state = .running; // nothing else ready — keep running
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
next.state = .running;
|
|
||||||
current = next;
|
|
||||||
if (next != prev) arch.switchContext(&prev.rsp, next.rsp);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Voluntarily give up the CPU to the next ready task.
|
|
||||||
pub fn yield() void {
|
|
||||||
const flags = arch.saveInterrupts();
|
|
||||||
schedule();
|
|
||||||
arch.restoreInterrupts(flags);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Block the current task for `ms` milliseconds, then let it become runnable
|
|
||||||
/// again. The idle task (or other work) runs in the meantime.
|
|
||||||
pub fn sleep(ms: u64) void {
|
|
||||||
const flags = arch.saveInterrupts();
|
|
||||||
current.wake_at = arch.millis() + ms;
|
|
||||||
current.state = .blocked;
|
|
||||||
schedule(); // current is blocked, so schedule() won't re-enqueue it
|
|
||||||
arch.restoreInterrupts(flags);
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- event-based blocking -------------------------------------------------
|
|
||||||
//
|
|
||||||
// A WaitQueue is a set of tasks blocked waiting for something (a resource, a
|
|
||||||
// message). Tasks link into it through the same `next` field the ready queues
|
|
||||||
// use — a task is in exactly one queue at a time. These are the primitive locks,
|
|
||||||
// semaphores and IPC channels are built on.
|
|
||||||
|
|
||||||
pub const WaitQueue = struct {
|
|
||||||
head: ?*Task = null,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Block the current task on `wq` and switch away. Precondition: interrupts are
|
|
||||||
/// disabled (the caller holds them, so a condition can be checked and the block
|
|
||||||
/// committed atomically). On return — when woken — interrupts are still disabled.
|
|
||||||
pub fn waitLocked(wq: *WaitQueue) void {
|
|
||||||
current.state = .blocked;
|
|
||||||
current.next = wq.head;
|
|
||||||
wq.head = current;
|
|
||||||
schedule();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Move the highest-priority waiter on `wq` (if any) to the ready queue.
|
|
||||||
/// Precondition: interrupts disabled. Does not preempt — the caller decides.
|
|
||||||
pub fn wakeLocked(wq: *WaitQueue) void {
|
|
||||||
// Find the highest-priority waiter (bounded scan) and unlink it.
|
|
||||||
var best_prev: ?*Task = null;
|
|
||||||
var best: ?*Task = null;
|
|
||||||
var prev: ?*Task = null;
|
|
||||||
var cur = wq.head;
|
|
||||||
while (cur) |t| : ({
|
|
||||||
prev = t;
|
|
||||||
cur = t.next;
|
|
||||||
}) {
|
|
||||||
if (best == null or t.priority > best.?.priority) {
|
|
||||||
best = t;
|
|
||||||
best_prev = prev;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
const t = best orelse return;
|
|
||||||
if (best_prev) |p| p.next = t.next else wq.head = t.next;
|
|
||||||
t.state = .ready;
|
|
||||||
enqueue(t);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Block on `wq` (a self-contained critical section).
|
|
||||||
pub fn wait(wq: *WaitQueue) void {
|
|
||||||
const flags = arch.saveInterrupts();
|
|
||||||
waitLocked(wq);
|
|
||||||
arch.restoreInterrupts(flags);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Wake the highest-priority waiter on `wq`, preempting if it outranks us.
|
|
||||||
pub fn wake(wq: *WaitQueue) void {
|
|
||||||
const flags = arch.saveInterrupts();
|
|
||||||
wakeLocked(wq);
|
|
||||||
// If a higher-priority task is now ready, run it immediately.
|
|
||||||
if (highestReadyPriority()) |p| {
|
|
||||||
if (p > current.priority) schedule();
|
|
||||||
}
|
|
||||||
arch.restoreInterrupts(flags);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn highestReadyPriority() ?Priority {
|
|
||||||
if (ready_bitmap == 0) return null;
|
|
||||||
return @intCast(num_priorities - 1 - @clz(ready_bitmap));
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Wake any sleeping task whose deadline has passed. Bounded by the task count,
|
|
||||||
/// so it stays deterministic. Called from the timer tick (interrupts disabled).
|
|
||||||
fn wakeExpired() void {
|
|
||||||
const now = arch.millis();
|
|
||||||
for (&tasks) |*t| {
|
|
||||||
if (t.state == .blocked and t.wake_at != 0 and now >= t.wake_at) {
|
|
||||||
t.wake_at = 0;
|
|
||||||
t.state = .ready;
|
|
||||||
enqueue(t);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Called from the timer interrupt (interrupts already disabled): wake due
|
|
||||||
/// sleepers, then preempt.
|
|
||||||
pub fn tick() void {
|
|
||||||
wakeExpired();
|
|
||||||
if (preemption_enabled) schedule();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Enable or disable timer-driven preemption (cooperative-only when off).
|
|
||||||
pub fn setPreemption(enabled: bool) void {
|
|
||||||
preemption_enabled = enabled;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// End the current task and switch away for good; never returns. The task's stack
|
|
||||||
/// is leaked for now (no reaper yet).
|
|
||||||
pub fn exit() noreturn {
|
|
||||||
arch.disableInterrupts();
|
|
||||||
current.state = .free;
|
|
||||||
const next = dequeueHighest() orelse @panic("sched: no task left to run");
|
|
||||||
next.state = .running;
|
|
||||||
current = next;
|
|
||||||
var discard: usize = 0;
|
|
||||||
arch.switchContext(&discard, next.rsp);
|
|
||||||
unreachable;
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn currentId() u32 {
|
|
||||||
return current.id;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Change the running task's priority (takes effect next time it's enqueued).
|
|
||||||
pub fn setPriority(p: Priority) void {
|
|
||||||
current.priority = p;
|
|
||||||
}
|
|
||||||
@@ -1,491 +0,0 @@
|
|||||||
//! In-kernel test cases, run at the end of bring-up when the kernel is built with
|
|
||||||
//! `-Dtest-case=<name>`. Each case writes structured markers to the serial port
|
|
||||||
//! that the QEMU harness (test/qemu_test.py) asserts on:
|
|
||||||
//!
|
|
||||||
//! [PASS]/[FAIL] <check> per assertion
|
|
||||||
//! DANOS-TEST-RESULT: PASS|FAIL overall, for non-faulting cases
|
|
||||||
//!
|
|
||||||
//! Faulting cases (fault-ud, fault-pf, fault-df) deliberately don't return a
|
|
||||||
//! result line — they trigger a CPU exception, and the harness asserts on the
|
|
||||||
//! exception report the handler prints (which also reaches serial).
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
const danos = @import("danos");
|
|
||||||
const arch = @import("arch");
|
|
||||||
const platform = @import("platform");
|
|
||||||
const pmm = @import("pmm.zig");
|
|
||||||
const heap = @import("heap.zig");
|
|
||||||
const sched = @import("scheduler.zig");
|
|
||||||
const ipc = @import("ipc.zig");
|
|
||||||
|
|
||||||
/// Formatted write straight to serial, independent of the framebuffer console.
|
|
||||||
fn log(comptime fmt: []const u8, args: anytype) void {
|
|
||||||
var buf: [128]u8 = undefined;
|
|
||||||
arch.serialWrite(std.fmt.bufPrint(&buf, fmt, args) catch return);
|
|
||||||
}
|
|
||||||
|
|
||||||
var passed: u32 = 0;
|
|
||||||
var failed: u32 = 0;
|
|
||||||
|
|
||||||
fn check(name: []const u8, ok: bool) void {
|
|
||||||
if (ok) {
|
|
||||||
passed += 1;
|
|
||||||
log("[PASS] {s}\n", .{name});
|
|
||||||
} else {
|
|
||||||
failed += 1;
|
|
||||||
log("[FAIL] {s}\n", .{name});
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Emit the overall result line the harness matches, then the done sentinel.
|
|
||||||
fn result() void {
|
|
||||||
log("DANOS-TEST-RESULT: {s} ({d} passed, {d} failed)\n", .{
|
|
||||||
if (failed == 0) "PASS" else "FAIL",
|
|
||||||
passed,
|
|
||||||
failed,
|
|
||||||
});
|
|
||||||
log("DANOS-TEST-DONE\n", .{});
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn run(case: []const u8, boot_info: *const BootInfo) void {
|
|
||||||
if (eql(case, "smoke")) {
|
|
||||||
smoke(boot_info);
|
|
||||||
} else if (eql(case, "timer")) {
|
|
||||||
timer();
|
|
||||||
} else if (eql(case, "clock")) {
|
|
||||||
clock();
|
|
||||||
} else if (eql(case, "vmm")) {
|
|
||||||
vmm();
|
|
||||||
} else if (eql(case, "heap")) {
|
|
||||||
heapTest();
|
|
||||||
} else if (eql(case, "sched")) {
|
|
||||||
schedTest();
|
|
||||||
} else if (eql(case, "priority")) {
|
|
||||||
priorityTest();
|
|
||||||
} else if (eql(case, "sleep")) {
|
|
||||||
sleepTest();
|
|
||||||
} else if (eql(case, "event")) {
|
|
||||||
eventTest();
|
|
||||||
} else if (eql(case, "ipc")) {
|
|
||||||
ipcTest();
|
|
||||||
} else if (eql(case, "fault-ud")) {
|
|
||||||
faultInvalidOpcode();
|
|
||||||
} else if (eql(case, "fault-pf")) {
|
|
||||||
faultPageFault();
|
|
||||||
} else if (eql(case, "fault-df")) {
|
|
||||||
faultDoubleFault();
|
|
||||||
} else if (eql(case, "fault-nx")) {
|
|
||||||
faultNoExecute();
|
|
||||||
} else if (eql(case, "fault-null")) {
|
|
||||||
faultNull();
|
|
||||||
} else if (eql(case, "poweroff")) {
|
|
||||||
powerTest(.off);
|
|
||||||
} else if (eql(case, "reboot")) {
|
|
||||||
powerTest(.reboot);
|
|
||||||
} else {
|
|
||||||
log("DANOS-TEST-RESULT: FAIL (unknown case '{s}')\n", .{case});
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn platformHal() platform.Hal {
|
|
||||||
return .{
|
|
||||||
.mapMmio = arch.mapPage,
|
|
||||||
.pioRead = arch.pioRead,
|
|
||||||
.pioWrite = arch.pioWrite,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Drive an ACPI power transition. On success the machine powers off or resets,
|
|
||||||
/// so QEMU exits — the harness observes the process exit. If control returns, the
|
|
||||||
/// transition failed and we emit a FAIL result.
|
|
||||||
fn powerTest(comptime action: enum { off, reboot }) void {
|
|
||||||
const name = if (action == .off) "poweroff" else "reboot";
|
|
||||||
log("DANOS-TEST-BEGIN: {s}\n", .{name});
|
|
||||||
const hal = platformHal();
|
|
||||||
log("DANOS-POWER: attempting {s}\n", .{name});
|
|
||||||
switch (action) {
|
|
||||||
.off => platform.shutdown(hal),
|
|
||||||
.reboot => platform.reboot(hal),
|
|
||||||
}
|
|
||||||
check("power transition took effect", false);
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
const BootInfo = danos.BootInfo;
|
|
||||||
|
|
||||||
fn eql(a: []const u8, b: []const u8) bool {
|
|
||||||
return std.mem.eql(u8, a, b);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Non-destructive checks of the memory map and frame allocator.
|
|
||||||
fn smoke(boot_info: *const BootInfo) void {
|
|
||||||
log("DANOS-TEST-BEGIN: smoke\n", .{});
|
|
||||||
|
|
||||||
// The memory map has some usable RAM.
|
|
||||||
const mm = boot_info.memory_map;
|
|
||||||
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
|
||||||
var usable: u64 = 0;
|
|
||||||
for (regions) |r| {
|
|
||||||
if (r.kind == .usable) usable += r.pages;
|
|
||||||
}
|
|
||||||
check("memory map reports usable RAM", usable > 0);
|
|
||||||
|
|
||||||
// The frame allocator hands out distinct, page-aligned frames.
|
|
||||||
const a = pmm.alloc();
|
|
||||||
const b = pmm.alloc();
|
|
||||||
check("alloc returns a frame", a != null);
|
|
||||||
check("alloc returns distinct frames", a != null and b != null and a.? != b.?);
|
|
||||||
check("frames are page-aligned", (a orelse 1) % danos.page_size == 0);
|
|
||||||
|
|
||||||
// Freeing restores the count.
|
|
||||||
const before = pmm.stats().free_frames;
|
|
||||||
if (a) |p| pmm.free(p);
|
|
||||||
if (b) |p| pmm.free(p);
|
|
||||||
check("free returns frames to the pool", pmm.stats().free_frames == before + 2);
|
|
||||||
|
|
||||||
// Paging is active on our own tables (CR3 is non-zero and page-aligned).
|
|
||||||
const cr3 = arch.readCr3();
|
|
||||||
check("paging active (CR3 set)", cr3 != 0 and cr3 % danos.page_size == 0);
|
|
||||||
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Verify device interrupts fire and return: the timer tick counter must advance
|
|
||||||
/// on its own. Interrupts are already enabled by kmain before tests run.
|
|
||||||
fn timer() void {
|
|
||||||
log("DANOS-TEST-BEGIN: timer\n", .{});
|
|
||||||
const start = arch.ticks();
|
|
||||||
// Busy-wait for the counter to advance. arch.ticks() is a volatile load, so
|
|
||||||
// the compiler re-reads it each iteration and sees the interrupt's update.
|
|
||||||
// The cap is only a safety net; the harness timeout is the real backstop.
|
|
||||||
var spins: u64 = 0;
|
|
||||||
while (arch.ticks() == start and spins < 5_000_000_000) spins +%= 1;
|
|
||||||
check("timer interrupts advance the tick count", arch.ticks() > start);
|
|
||||||
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Verify the on-demand VMM: map a fresh frame at an unused virtual address, and
|
|
||||||
/// check it's writable and reads back.
|
|
||||||
fn vmm() void {
|
|
||||||
log("DANOS-TEST-BEGIN: vmm\n", .{});
|
|
||||||
const frame = pmm.alloc();
|
|
||||||
check("frame available to map", frame != null);
|
|
||||||
if (frame) |phys| {
|
|
||||||
var virt: u64 = 0x0000_4000_0000_0000; // canonical, well clear of everything mapped
|
|
||||||
arch.mapPage(virt, phys, true);
|
|
||||||
const p: *volatile u64 = @ptrFromInt(virt);
|
|
||||||
p.* = 0xdead_c0de_cafe_babe;
|
|
||||||
check("mapped page is writable and reads back", p.* == 0xdead_c0de_cafe_babe);
|
|
||||||
arch.unmapPage(virt);
|
|
||||||
pmm.free(phys);
|
|
||||||
virt += 0;
|
|
||||||
}
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Exercise the kernel heap: basic alloc/write/free, reuse, growth beyond the
|
|
||||||
/// initial region, and a std container backed by it.
|
|
||||||
fn heapTest() void {
|
|
||||||
log("DANOS-TEST-BEGIN: heap\n", .{});
|
|
||||||
const a = heap.allocator();
|
|
||||||
|
|
||||||
// Allocate, write a pattern, read it back, free.
|
|
||||||
const buf = a.alloc(u8, 4096) catch null;
|
|
||||||
check("alloc 4096 bytes", buf != null);
|
|
||||||
if (buf) |b| {
|
|
||||||
@memset(b, 0xAB);
|
|
||||||
check("heap memory is writable and reads back", b[0] == 0xAB and b[4095] == 0xAB);
|
|
||||||
a.free(b);
|
|
||||||
}
|
|
||||||
|
|
||||||
// Freeing then re-allocating the same size should reuse the block.
|
|
||||||
const p1 = a.alloc(u64, 8) catch null;
|
|
||||||
const addr1 = if (p1) |p| @intFromPtr(p.ptr) else 0;
|
|
||||||
if (p1) |p| a.free(p);
|
|
||||||
const p2 = a.alloc(u64, 8) catch null;
|
|
||||||
const addr2 = if (p2) |p| @intFromPtr(p.ptr) else 0;
|
|
||||||
check("freed block is reused", addr1 != 0 and addr1 == addr2);
|
|
||||||
if (p2) |p| a.free(p);
|
|
||||||
|
|
||||||
// Force growth past the initial page and check every block is usable.
|
|
||||||
var blocks: [64]?[]u8 = .{null} ** 64;
|
|
||||||
var ok = true;
|
|
||||||
for (&blocks, 0..) |*slot, i| {
|
|
||||||
const b = a.alloc(u8, 4096) catch null;
|
|
||||||
slot.* = b;
|
|
||||||
if (b) |bb| @memset(bb, @intCast(i & 0xff)) else {
|
|
||||||
ok = false;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for (blocks, 0..) |slot, i| {
|
|
||||||
if (slot) |bb| {
|
|
||||||
if (bb[0] != @as(u8, @intCast(i & 0xff)) or bb[4095] != @as(u8, @intCast(i & 0xff))) ok = false;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
check("many allocations (heap growth) stay valid", ok);
|
|
||||||
for (blocks) |slot| {
|
|
||||||
if (slot) |bb| a.free(bb);
|
|
||||||
}
|
|
||||||
|
|
||||||
// A std container backed by the kernel heap.
|
|
||||||
var list: std.ArrayList(u32) = .empty;
|
|
||||||
var sum: u64 = 0;
|
|
||||||
var expected: u64 = 0;
|
|
||||||
var i: u32 = 0;
|
|
||||||
var list_ok = true;
|
|
||||||
while (i < 1000) : (i += 1) {
|
|
||||||
list.append(a, i) catch {
|
|
||||||
list_ok = false;
|
|
||||||
};
|
|
||||||
expected += i;
|
|
||||||
}
|
|
||||||
for (list.items) |v| sum += v;
|
|
||||||
list.deinit(a);
|
|
||||||
check("std.ArrayList on the kernel heap", list_ok and sum == expected);
|
|
||||||
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Verify the calibrated clocks: sane measured frequencies, monotonic uptime that
|
|
||||||
/// advances with real ticks, and — the point of the TSC clock — nanosecond
|
|
||||||
/// resolution far finer than the 1 ms tick, with the unit functions consistent.
|
|
||||||
fn clock() void {
|
|
||||||
log("DANOS-TEST-BEGIN: clock\n", .{});
|
|
||||||
|
|
||||||
const lapic = arch.lapicHz();
|
|
||||||
check("LAPIC frequency measured", lapic > 1_000_000 and lapic < 100_000_000_000);
|
|
||||||
const tsc = arch.tscHz();
|
|
||||||
check("TSC frequency measured", tsc > 100_000_000 and tsc < 100_000_000_000);
|
|
||||||
|
|
||||||
// Uptime advances over ~5 real ticks (1000 Hz => 1 tick == 1 ms).
|
|
||||||
const start_ticks = arch.ticks();
|
|
||||||
const start_ms = arch.millis();
|
|
||||||
var spins: u64 = 0;
|
|
||||||
while (arch.ticks() < start_ticks + 5 and spins < 5_000_000_000) spins +%= 1;
|
|
||||||
const elapsed_ms = arch.millis() - start_ms;
|
|
||||||
check("uptime advances with ticks", elapsed_ms >= 5 and elapsed_ms < 100);
|
|
||||||
|
|
||||||
// Sub-millisecond resolution: spin until nanos() first advances, then confirm
|
|
||||||
// that first step happened within a millisecond — so nanos() resolves finer
|
|
||||||
// than the 1 ms tick (a tick clock's smallest step *is* 1 ms). Spinning to the
|
|
||||||
// first change is robust to QEMU's coarse TSC update granularity.
|
|
||||||
const n1 = arch.nanos();
|
|
||||||
var s2: u64 = 0;
|
|
||||||
while (arch.nanos() == n1 and s2 < 10_000_000) s2 +%= 1;
|
|
||||||
const n2 = arch.nanos();
|
|
||||||
check("nanos() has sub-millisecond resolution", n2 > n1 and (n2 - n1) < 1_000_000);
|
|
||||||
|
|
||||||
// The unit functions agree (within rounding).
|
|
||||||
const ns = arch.nanos();
|
|
||||||
check("nanos/micros/millis are consistent", diffWithin(arch.micros(), ns / 1000, 1000) and diffWithin(arch.millis(), ns / 1_000_000, 2));
|
|
||||||
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
fn diffWithin(a: u64, b: u64, tol: u64) bool {
|
|
||||||
return if (a > b) a - b <= tol else b - a <= tol;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- scheduler tests ------------------------------------------------------
|
|
||||||
|
|
||||||
var counters = [_]u64{0} ** 3;
|
|
||||||
|
|
||||||
fn spin0() void {
|
|
||||||
const p: *volatile u64 = &counters[0];
|
|
||||||
while (true) p.* = p.* +% 1;
|
|
||||||
}
|
|
||||||
fn spin1() void {
|
|
||||||
const p: *volatile u64 = &counters[1];
|
|
||||||
while (true) p.* = p.* +% 1;
|
|
||||||
}
|
|
||||||
fn spin2() void {
|
|
||||||
const p: *volatile u64 = &counters[2];
|
|
||||||
while (true) p.* = p.* +% 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Preemption: spawn three tasks that busy-loop *without* yielding. If they all
|
|
||||||
/// make progress, the timer must be preempting between them (and the context
|
|
||||||
/// switch works) — because nothing yields voluntarily.
|
|
||||||
fn schedTest() void {
|
|
||||||
log("DANOS-TEST-BEGIN: sched\n", .{});
|
|
||||||
counters = .{ 0, 0, 0 };
|
|
||||||
sched.spawn(spin0, 4);
|
|
||||||
sched.spawn(spin1, 4);
|
|
||||||
sched.spawn(spin2, 4);
|
|
||||||
|
|
||||||
const c0: *volatile u64 = &counters[0];
|
|
||||||
const c1: *volatile u64 = &counters[1];
|
|
||||||
const c2: *volatile u64 = &counters[2];
|
|
||||||
var spins: u64 = 0;
|
|
||||||
while ((c0.* == 0 or c1.* == 0 or c2.* == 0) and spins < 5_000_000_000) spins +%= 1;
|
|
||||||
|
|
||||||
check("all three non-yielding tasks made progress (preemption)", c0.* > 0 and c1.* > 0 and c2.* > 0);
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
var run_order = [_]u8{0} ** 4;
|
|
||||||
var run_n: usize = 0;
|
|
||||||
|
|
||||||
fn recordExit(priority: u8) void {
|
|
||||||
run_order[run_n] = priority;
|
|
||||||
run_n += 1;
|
|
||||||
sched.exit();
|
|
||||||
}
|
|
||||||
fn taskHigh() void {
|
|
||||||
recordExit(6);
|
|
||||||
}
|
|
||||||
fn taskMid() void {
|
|
||||||
recordExit(4);
|
|
||||||
}
|
|
||||||
fn taskLow() void {
|
|
||||||
recordExit(2);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Fixed priority: with preemption off (deterministic), spawn tasks at three
|
|
||||||
/// priorities and let them run cooperatively. They must run highest-first.
|
|
||||||
fn priorityTest() void {
|
|
||||||
log("DANOS-TEST-BEGIN: priority\n", .{});
|
|
||||||
sched.setPreemption(false);
|
|
||||||
sched.setPriority(1); // above the idle task (0), below the workers — runs last
|
|
||||||
run_n = 0;
|
|
||||||
|
|
||||||
sched.spawn(taskLow, 2);
|
|
||||||
sched.spawn(taskMid, 4);
|
|
||||||
sched.spawn(taskHigh, 6);
|
|
||||||
|
|
||||||
while (run_n < 3) sched.yield(); // regain control only once the workers are done
|
|
||||||
|
|
||||||
check("tasks ran highest-priority first", run_order[0] == 6 and run_order[1] == 4 and run_order[2] == 2);
|
|
||||||
|
|
||||||
sched.setPriority(4);
|
|
||||||
sched.setPreemption(true);
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
var event_wq: sched.WaitQueue = .{};
|
|
||||||
var event_stage: u32 = 0;
|
|
||||||
|
|
||||||
fn eventWaiter() void {
|
|
||||||
event_stage = 1; // reached the wait
|
|
||||||
sched.wait(&event_wq); // block until woken
|
|
||||||
event_stage = 3; // woken and resumed
|
|
||||||
sched.exit();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Event-based blocking: a task blocks on a wait queue and is woken. The waiter is
|
|
||||||
/// higher priority, so waking it preempts us and it runs to completion at once.
|
|
||||||
fn eventTest() void {
|
|
||||||
log("DANOS-TEST-BEGIN: event\n", .{});
|
|
||||||
event_stage = 0;
|
|
||||||
sched.spawn(eventWaiter, 6); // higher priority than this task (4)
|
|
||||||
|
|
||||||
var spins: u64 = 0;
|
|
||||||
while (event_stage != 1 and spins < 1_000_000_000) : (spins += 1) sched.yield();
|
|
||||||
check("waiter reached the wait and blocked", event_stage == 1);
|
|
||||||
|
|
||||||
sched.wake(&event_wq);
|
|
||||||
check("wake resumed the blocked waiter (preempting)", event_stage == 3);
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
var channel: ipc.Channel(u64, 4) = .{};
|
|
||||||
var recv_sum: u64 = 0;
|
|
||||||
var recv_count: u64 = 0;
|
|
||||||
|
|
||||||
fn producer() void {
|
|
||||||
var i: u64 = 1;
|
|
||||||
while (i <= 100) : (i += 1) channel.send(i);
|
|
||||||
sched.exit();
|
|
||||||
}
|
|
||||||
fn consumer() void {
|
|
||||||
var n: u64 = 0;
|
|
||||||
while (n < 100) : (n += 1) {
|
|
||||||
recv_sum += channel.recv();
|
|
||||||
recv_count += 1;
|
|
||||||
}
|
|
||||||
sched.exit();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// IPC: a producer and consumer pass 100 messages through a 4-slot channel. The
|
|
||||||
/// small buffer forces the channel full and empty repeatedly, exercising both the
|
|
||||||
/// blocking-send and blocking-recv paths. The messages must arrive intact.
|
|
||||||
fn ipcTest() void {
|
|
||||||
log("DANOS-TEST-BEGIN: ipc\n", .{});
|
|
||||||
channel = .{};
|
|
||||||
recv_sum = 0;
|
|
||||||
recv_count = 0;
|
|
||||||
sched.spawn(consumer, 5); // above this task (4) so they run and we observe after
|
|
||||||
sched.spawn(producer, 5);
|
|
||||||
|
|
||||||
var spins: u64 = 0;
|
|
||||||
while (recv_count < 100 and spins < 2_000_000_000) : (spins += 1) sched.yield();
|
|
||||||
|
|
||||||
check("all 100 messages received", recv_count == 100);
|
|
||||||
check("messages arrived intact (sum 1..100 == 5050)", recv_sum == 5050);
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Blocking: sleep(50) should block this task for about 50 ms (measured on the
|
|
||||||
/// calibrated clock) — not busy-wait — while the idle task runs.
|
|
||||||
fn sleepTest() void {
|
|
||||||
log("DANOS-TEST-BEGIN: sleep\n", .{});
|
|
||||||
const t0 = arch.millis();
|
|
||||||
sched.sleep(50);
|
|
||||||
const elapsed = arch.millis() - t0;
|
|
||||||
check("sleep(50) blocked for ~50 ms", elapsed >= 50 and elapsed <= 70);
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
fn faultInvalidOpcode() void {
|
|
||||||
log("DANOS-TEST-BEGIN: fault-ud\n", .{});
|
|
||||||
asm volatile ("ud2");
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Verify NX: fetching an instruction from a data page (mapped no-execute) faults.
|
|
||||||
fn faultNoExecute() void {
|
|
||||||
log("DANOS-TEST-BEGIN: fault-nx\n", .{});
|
|
||||||
var scratch: u64 = 0xC3; // a lone `ret` — harmless if NX somehow let it run
|
|
||||||
const f: *const fn () void = @ptrFromInt(@intFromPtr(&scratch));
|
|
||||||
f(); // instruction fetch from an NX page -> #PF before it executes
|
|
||||||
log("DANOS-TEST-RESULT: FAIL (NX not enforced)\n", .{});
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Verify the null guard: dereferencing address 0 (page 0 left unmapped) faults.
|
|
||||||
fn faultNull() void {
|
|
||||||
log("DANOS-TEST-BEGIN: fault-null\n", .{});
|
|
||||||
// Launder the address through empty asm so the compiler no longer knows it's
|
|
||||||
// 0 (otherwise it folds a null-pointer safety panic instead of doing the real
|
|
||||||
// access). `allowzero` skips the same null check on the cast. The write then
|
|
||||||
// hits the unmapped page 0 and takes a real hardware #PF.
|
|
||||||
var addr: u64 = 0;
|
|
||||||
addr = asm ("" : [ret] "=r" (-> u64) : [in] "0" (addr));
|
|
||||||
const p: *allowzero volatile u64 = @ptrFromInt(addr);
|
|
||||||
p.* = 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn faultPageFault() void {
|
|
||||||
log("DANOS-TEST-BEGIN: fault-pf\n", .{});
|
|
||||||
// Runtime address so the backend emits a register store (not a `mov moffs`,
|
|
||||||
// which the self-hosted x86_64 backend can't encode).
|
|
||||||
var addr: u64 = 0xdeadbeef000; // well above all mapped RAM
|
|
||||||
const p: *volatile u64 = @ptrFromInt(addr);
|
|
||||||
p.* = 1;
|
|
||||||
addr += 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn faultDoubleFault() void {
|
|
||||||
log("DANOS-TEST-BEGIN: fault-df\n", .{});
|
|
||||||
arch.disableInterrupts(); // so only the ud2 delivery (not a timer tick) triggers the #DF
|
|
||||||
// Point RSP at unmapped memory, then fault: the CPU can't push the fault
|
|
||||||
// frame, which escalates to #DF — survivable only because #DF runs on IST1.
|
|
||||||
var bad_sp: u64 = 0x5000000000;
|
|
||||||
asm volatile (
|
|
||||||
\\mov %[sp], %%rsp
|
|
||||||
\\ud2
|
|
||||||
:
|
|
||||||
: [sp] "r" (bad_sp),
|
|
||||||
: .{ .memory = true }
|
|
||||||
);
|
|
||||||
bad_sp += 0;
|
|
||||||
}
|
|
||||||
-112
@@ -1,112 +0,0 @@
|
|||||||
//! Shared definitions that form the contract between a bootloader
|
|
||||||
//! (src/boot/, e.g. efi.zig built as BOOTX64.efi) and the kernel (src/kernel/main.zig).
|
|
||||||
//!
|
|
||||||
//! Both binaries import this as the "danos" module, so the handoff layout is
|
|
||||||
//! defined in exactly one place.
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
|
|
||||||
/// Calling convention for the bootloader→kernel jump. Pinned to SysV so it does
|
|
||||||
/// not depend on each binary's target default: the UEFI bootloader's C
|
|
||||||
/// convention is Microsoft x64 (first arg in RCX), the freestanding kernel's is
|
|
||||||
/// SysV (first arg in RDI). Both reference this to agree on where `*BootInfo`
|
|
||||||
/// is passed.
|
|
||||||
pub const kernel_abi: std.builtin.CallingConvention = .{ .x86_64_sysv = .{} };
|
|
||||||
|
|
||||||
/// Pixel byte order of the linear framebuffer the firmware handed us.
|
|
||||||
pub const PixelFormat = enum(u32) {
|
|
||||||
/// Byte 0 = Red, 1 = Green, 2 = Blue, 3 = reserved.
|
|
||||||
rgbx,
|
|
||||||
/// Byte 0 = Blue, 1 = Green, 2 = Red, 3 = reserved.
|
|
||||||
bgrx,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// A linear framebuffer: `width`x`height` pixels, each a 32-bit value, with
|
|
||||||
/// `pitch` bytes between the start of one row and the next (which may be larger
|
|
||||||
/// than `width * 4` due to hardware padding).
|
|
||||||
///
|
|
||||||
/// A `base` of 0 means **no framebuffer** — the firmware exposed no Graphics
|
|
||||||
/// Output Protocol (a headless server, say). The kernel must treat on-screen
|
|
||||||
/// output as optional and never assume a framebuffer exists.
|
|
||||||
pub const Framebuffer = extern struct {
|
|
||||||
base: usize, // the memory address where pixel data starts (0 = none)
|
|
||||||
width: u32, // visible pixels per row (e.g. 1920)
|
|
||||||
height: u32, // visible rows (e.g. 1080)
|
|
||||||
pitch: u32, // bytes from the start of one row to the start of the next
|
|
||||||
format: PixelFormat,
|
|
||||||
|
|
||||||
/// Whether a usable framebuffer was handed over.
|
|
||||||
pub fn present(self: Framebuffer) bool {
|
|
||||||
return self.base != 0 and self.width != 0 and self.height != 0;
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Page size the memory map is measured in. 4 KiB on every architecture danos
|
|
||||||
/// targets so far.
|
|
||||||
pub const page_size = 4096;
|
|
||||||
|
|
||||||
/// danos's own classification of a span of physical memory — deliberately not
|
|
||||||
/// UEFI's vocabulary. Each boot path (UEFI now, device tree later) translates its
|
|
||||||
/// native memory description into these kinds, so the kernel never learns what
|
|
||||||
/// booted it. [[arch]] keeps the same discipline for CPU code.
|
|
||||||
pub const MemoryKind = enum(u32) {
|
|
||||||
/// Free RAM the kernel may allocate. Each boot path folds its own transient
|
|
||||||
/// memory into this once it's genuinely free (e.g. the UEFI loader classifies
|
|
||||||
/// boot-services memory as usable after ExitBootServices), so the kernel never
|
|
||||||
/// has to know about boot-protocol-specific "reclaimable" states.
|
|
||||||
usable,
|
|
||||||
/// Firmware, MMIO, the kernel image, our own boot buffers, the boot stack —
|
|
||||||
/// never hand out.
|
|
||||||
reserved,
|
|
||||||
/// ACPI tables: parse, then reclaim.
|
|
||||||
acpi_tables,
|
|
||||||
/// ACPI non-volatile storage: preserve across sleep, do not allocate.
|
|
||||||
acpi_nvs,
|
|
||||||
/// Not backed by RAM: memory-mapped device registers or a reserved
|
|
||||||
/// address-space window (e.g. PCIe config space). Kept distinct from
|
|
||||||
/// `reserved` so RAM accounting doesn't count device address space.
|
|
||||||
mmio,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// One contiguous span of physical memory. Because danos defines this layout
|
|
||||||
/// itself (unlike the UEFI descriptor it's built from), `@sizeOf` is
|
|
||||||
/// authoritative — the kernel walks a plain `[]MemoryRegion`, with none of the
|
|
||||||
/// firmware's variable descriptor-stride to worry about.
|
|
||||||
pub const MemoryRegion = extern struct {
|
|
||||||
base: u64, // physical start address
|
|
||||||
pages: u64, // length in `page_size` units
|
|
||||||
kind: MemoryKind,
|
|
||||||
_pad: u32 = 0,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// The physical memory layout handed to the kernel: a pointer to an array of
|
|
||||||
/// `len` `MemoryRegion`s, in a buffer that outlives the loader.
|
|
||||||
pub const MemoryMap = extern struct {
|
|
||||||
regions: usize, // address of a `[len]MemoryRegion`
|
|
||||||
len: usize,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// One PT_LOAD segment of the kernel image, so the kernel can re-map itself with
|
|
||||||
/// correct permissions (code R+X, rodata R, data R+W+NX). `flags` are raw ELF
|
|
||||||
/// segment flags: PF_X=1, PF_W=2, PF_R=4.
|
|
||||||
pub const KernelSegment = extern struct {
|
|
||||||
virt: u64,
|
|
||||||
pages: u64,
|
|
||||||
flags: u32,
|
|
||||||
_pad: u32 = 0,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Handoff structure the bootloader fills in and passes to the kernel's
|
|
||||||
/// `_start` in RDI (the first argument under the SysV AMD64 C ABI).
|
|
||||||
pub const BootInfo = extern struct {
|
|
||||||
framebuffer: Framebuffer,
|
|
||||||
memory_map: MemoryMap,
|
|
||||||
/// The kernel's own PT_LOAD segments (it has three: text, rodata, data).
|
|
||||||
kernel_segments: [8]KernelSegment,
|
|
||||||
kernel_segment_count: u32,
|
|
||||||
/// Physical address of the ACPI RSDP the firmware exposed, or 0 if none. The
|
|
||||||
/// kernel's device layer parses the ACPI tables from here to discover hardware.
|
|
||||||
/// A device-tree boot path leaves this 0 and (later) fills a `device_tree_blob`
|
|
||||||
/// field instead, so the kernel discovers devices without knowing what booted it.
|
|
||||||
acpi_rsdp: u64 = 0,
|
|
||||||
};
|
|
||||||
+197
@@ -0,0 +1,197 @@
|
|||||||
|
//! The **private kernel ↔ runtime** ABI: the raw system_call contract — the call
|
||||||
|
//! numbers, `mmap` protection flags, the page size those calls work in, and the IPC
|
||||||
|
//! name-registry ids and notification bit. Shared by the kernel dispatcher
|
||||||
|
//! (system/kernel/process.zig) and the user-space runtime library (library/runtime/),
|
||||||
|
//! so the two can never drift.
|
||||||
|
//!
|
||||||
|
//! **Application code does not speak this.** danos programs call the `runtime` library —
|
||||||
|
//! the stable, danos-native ABI — and the runtime is the one thing that issues the
|
||||||
|
//! actual system calls (POSIX code layers over the runtime, never on this directly). It
|
||||||
|
//! is the same split as libSystem on macOS or win32 over the NT syscalls: the numbers
|
||||||
|
//! here are an implementation detail the runtime hides and may renumber, not a public
|
||||||
|
//! interface. See docs/coding-standards.md and library/runtime/.
|
||||||
|
//!
|
||||||
|
//! This is the *core* contract; the device half — `DeviceDescriptor` and friends, which
|
||||||
|
//! also cross this boundary — lives with the device sub-project as [[device-abi]]
|
||||||
|
//! (system/devices/device-abi.zig). The loader↔kernel handoff is [[boot-handoff]].
|
||||||
|
|
||||||
|
/// Page size every `mmap`/`munmap` grant and the boot memory map are measured in.
|
||||||
|
/// 4 KiB on every architecture danos targets so far. Part of the ABI because the
|
||||||
|
/// runtime aligns to it (grants are page-granular) and the kernel guarantees it.
|
||||||
|
pub const page_size = 4096;
|
||||||
|
|
||||||
|
/// The kernel system_call numbers — the single source of truth shared by the kernel
|
||||||
|
/// dispatcher (system/kernel/process.zig) and the user runtime library, so the two
|
||||||
|
/// can never drift. The set is deliberately microkernel-minimal: file/device I/O
|
||||||
|
/// is not here — it lives in user-space servers reached through the IPC calls.
|
||||||
|
/// The table grows one milestone at a time; see docs/syscall.md.
|
||||||
|
pub const SystemCall = enum(u64) {
|
||||||
|
exit = 0, // exit(code): end the calling process
|
||||||
|
yield = 1, // yield(): give up the rest of this quantum
|
||||||
|
debug_write = 2, // debug_write(ptr, len): raw bytes to the kernel log (bring-up only)
|
||||||
|
sleep = 3, // sleep(ms): block the caller for ms milliseconds
|
||||||
|
mmap = 4, // mmap(len, prot) -> base: grant zeroed, page-aligned user pages
|
||||||
|
munmap = 5, // munmap(base, len): release pages from a prior mmap
|
||||||
|
create_ipc_endpoint = 6, // create_ipc_endpoint() -> handle: a new IPC endpoint
|
||||||
|
ipc_register = 7, // ipc_register(service_id, handle): publish an endpoint by well-known id
|
||||||
|
ipc_lookup = 8, // ipc_lookup(service_id) -> handle: find a published endpoint
|
||||||
|
ipc_call = 9, // ipc_call(h, message, len, reply, cap) -> reply_len: send + block for reply
|
||||||
|
ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, receive, cap) -> receive_len (+badge in rdx)
|
||||||
|
device_enumerate = 11, // device_enumerate(buffer, maximum) -> count: snapshot the device table
|
||||||
|
device_claim = 12, // device_claim(id) -> ok: take exclusive ownership of a device
|
||||||
|
mmio_map = 13, // mmio_map(id, resource_index) -> vaddr: map a claimed device's MMIO into this AS
|
||||||
|
irq_bind = 14, // irq_bind(id, resource_index, endpoint): deliver a device IRQ as an IPC notification
|
||||||
|
irq_ack = 15, // irq_ack(id, resource_index): re-arm a bound IRQ after servicing it
|
||||||
|
device_register = 16, // device_register(parent_id, descriptor) -> id: publish a child of a device you claimed
|
||||||
|
system_spawn = 17, // system_spawn(name_ptr, name_len, arguments_ptr, arguments_len, exit_endpoint) -> child process id: start a named initial-ramdisk binary as a new ring-3 process
|
||||||
|
dma_alloc = 18, // dma_alloc(len, flags) -> vaddr (rax), paddr (rdx): contiguous, pinned, uncacheable DMA memory
|
||||||
|
dma_free = 19, // dma_free(vaddr, len) -> 0: release a prior dma_alloc
|
||||||
|
msi_bind = 20, // msi_bind(device_id, endpoint) -> address (rax), data (rdx): a per-device MSI vector for a claimed device
|
||||||
|
io_read = 21, // io_read(device_id, resource_index, offset, width) -> value: read a port in a claimed device's io_port resource
|
||||||
|
io_write = 22, // io_write(device_id, resource_index, offset, width, value) -> 0: write a port in a claimed device's io_port resource
|
||||||
|
clock = 23, // clock() -> nanoseconds since boot: a monotonic time source (for timeouts/delays)
|
||||||
|
process_enumerate = 24, // process_enumerate(buffer, maximum) -> total: snapshot the task table
|
||||||
|
process_kill = 25, // process_kill(id) -> 0/-errno: end a process this process spawned
|
||||||
|
ipc_send = 26, // ipc_send(handle, message_ptr, message_len) -> 0/-errno: post a payload to an endpoint's async queue without blocking
|
||||||
|
process_exit_reason = 27, // process_exit_reason(id) -> ExitReason/-errno: how a dead child ended (its supervisor only)
|
||||||
|
process_subscribe = 28, // process_subscribe(endpoint) -> 0/-errno: subscribe to published exit events — every death posts a notification
|
||||||
|
signal_bind = 29, // signal_bind(endpoint) -> 0/-errno: nominate the endpoint this process's signals arrive on
|
||||||
|
process_signal = 30, // process_signal(id, signal) -> 0/-errno: post a signal to a child (or to yourself)
|
||||||
|
timer_bind = 31, // timer_bind(endpoint, ms) -> 0/-errno: one-shot timer — posts a notification when ms elapse
|
||||||
|
klog_read = 32, // klog_read(offset, ptr, len) -> bytes copied: copy the kernel RAM log buffer out to a user buffer (for persisting the boot log to disk)
|
||||||
|
wall_clock = 33, // wall_clock() -> Unix epoch seconds (UTC): the RTC wall-clock time, for filesystem timestamps (mtime). Monotonic time is `clock`.
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// How a process ended — recorded by the kernel at death, queried by the
|
||||||
|
/// supervisor with `process_exit_reason`, and the input to its restart decision
|
||||||
|
/// (docs/process-lifecycle.md): a clean exit meant to stop, a fault wants a
|
||||||
|
/// restart with backoff, killed means the supervisor did it itself. The faults
|
||||||
|
/// mirror the CPU exceptions a ring-3 process can die of; they are exit reasons,
|
||||||
|
/// never delivered to the faulting process (recovery is restart, not a handler).
|
||||||
|
pub const ExitReason = enum(u8) {
|
||||||
|
exited = 0, // returned from main / called exit
|
||||||
|
aborted = 1, // deliberate self-termination (reserved: no abort path yet)
|
||||||
|
segmentation_fault = 2, // page fault
|
||||||
|
illegal_instruction = 3, // invalid opcode
|
||||||
|
arithmetic_fault = 4, // divide error, x87 or SIMD fault
|
||||||
|
protection_fault = 5, // general protection fault
|
||||||
|
fault = 6, // any other CPU exception
|
||||||
|
killed = 7, // process_kill
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The x86 MSI message address base (`0xFEE0_0000`): a device raises an MSI by writing
|
||||||
|
/// `data` to this address, which the Local APIC turns into an interrupt at the vector
|
||||||
|
/// in `data`. The kernel returns the concrete (address, data) from `msi_bind`; this is
|
||||||
|
/// the fixed prefix, exposed so a driver's config-space programming reads clearly.
|
||||||
|
pub const msi_address_base: u64 = 0xFEE0_0000;
|
||||||
|
|
||||||
|
/// `dma_alloc` flags. `coherent` (uncacheable) is the portable default; the others are
|
||||||
|
/// opt-in for specific hardware. `write_combining` needs PAT programming (not yet — it
|
||||||
|
/// currently falls back to coherent); see docs/driver-model.md (M14).
|
||||||
|
pub const dma_coherent: u64 = 1; // strong-uncacheable — the default, the only portable one
|
||||||
|
pub const dma_write_combining: u64 = 2; // write-combining (framebuffers); needs PAT
|
||||||
|
pub const dma_below_4g: u64 = 4; // physical address must fit 32 bits (legacy DMA engines)
|
||||||
|
|
||||||
|
/// Set in the badge returned by `ipc_reply_wait` when what arrived is an
|
||||||
|
/// **asynchronous notification** (a device interrupt bound with `irq_bind`, or a
|
||||||
|
/// child-exit notice — see `notify_exit_bit`) rather than a message from a client.
|
||||||
|
/// There is no payload and no reply owed; the low bits carry the source. Shared so
|
||||||
|
/// the kernel's ISR and the driver's event loop can't disagree about which bit
|
||||||
|
/// means "the hardware spoke".
|
||||||
|
pub const notify_badge_bit: u64 = 1 << 63;
|
||||||
|
|
||||||
|
/// Set (alongside `notify_badge_bit`) in the badge of a **child-exit notification**:
|
||||||
|
/// posted to the endpoint a supervisor passed to `system_spawn` when that child ends
|
||||||
|
/// — by clean exit, by a fault, or by `process_kill`. The low bits carry the child's
|
||||||
|
/// process id, so one endpoint can supervise many children (and even share with IRQ
|
||||||
|
/// notifications, which never set this bit). The microkernel's SIGCHLD.
|
||||||
|
pub const notify_exit_bit: u64 = 1 << 62;
|
||||||
|
|
||||||
|
/// Set (alongside `notify_badge_bit`) in the badge of a **buffered message** — a payload
|
||||||
|
/// posted to an endpoint's async queue by `ipc_send`, delivered through `ipc_reply_wait`
|
||||||
|
/// like a notification (no reply owed) but carrying bytes in the receive buffer, not just
|
||||||
|
/// a badge. This is what distinguishes a payload-bearing async message from a bare IRQ /
|
||||||
|
/// child-exit notification (which sets neither this nor `notify_exit_bit`). The low bits
|
||||||
|
/// carry the sender's task id. The async counterpart of the synchronous `ipc_call`, for
|
||||||
|
/// broadcasts where a rendezvous is the wrong shape (the input service is the first user).
|
||||||
|
pub const notify_message_bit: u64 = 1 << 61;
|
||||||
|
|
||||||
|
/// Set (alongside `notify_badge_bit`) in the badge of a **signal notification** —
|
||||||
|
/// the process-lifecycle vocabulary of docs/process-lifecycle.md, delivered to the
|
||||||
|
/// endpoint the process nominated with `signal_bind`. The low bits carry the
|
||||||
|
/// coalesced pending mask (bit positions = `Signal` values): signals are
|
||||||
|
/// statements, not questions, and two pending terminates are one terminate.
|
||||||
|
pub const notify_signal_bit: u64 = 1 << 60;
|
||||||
|
|
||||||
|
/// Set (alongside `notify_badge_bit`) in the badge of a **timer notification** —
|
||||||
|
/// a one-shot `timer_bind` deadline landing. No payload bits: what to do when the
|
||||||
|
/// deadline fires is whatever the receiver armed it for (a stop-sequence
|
||||||
|
/// escalation, a restart backoff, an alarm).
|
||||||
|
pub const notify_timer_bit: u64 = 1 << 59;
|
||||||
|
|
||||||
|
/// The signal vocabulary (docs/process-lifecycle.md): POSIX's concepts, danos's
|
||||||
|
/// names, message delivery. The value is the bit position in the pending mask — a
|
||||||
|
/// private kernel/runtime detail, free to change while they ship together. Kill
|
||||||
|
/// is not here (it is `process_kill`, unhandleable by definition); faults are not
|
||||||
|
/// here (they are `ExitReason`s — recovery is restart, not a handler); liveness is
|
||||||
|
/// not here (a question, asked as the zero-length ping call, not a statement).
|
||||||
|
pub const Signal = enum(u5) {
|
||||||
|
terminate = 0, // finish up and exit (the polite half of the stop sequence)
|
||||||
|
reload = 1, // re-read configuration / re-scan
|
||||||
|
interrupt = 2, // interactive interrupt (no sender until a console exists)
|
||||||
|
quit = 3, // as interrupt, by convention more final
|
||||||
|
alarm = 4, // a timer the process armed for itself (unbuilt: no consumer yet)
|
||||||
|
user_1 = 5, // service-defined
|
||||||
|
user_2 = 6, // service-defined
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Capacity of `ProcessDescriptor.name` — matches the longest name `system_spawn`
|
||||||
|
/// accepts, so a process's recorded name (its argv[0]) is never truncated.
|
||||||
|
pub const maximum_process_name = 64;
|
||||||
|
|
||||||
|
/// What a process is doing right now, as reported by `process_enumerate`. Crosses
|
||||||
|
/// the system_call boundary as `ProcessDescriptor.state`.
|
||||||
|
pub const ProcessState = enum(u32) {
|
||||||
|
ready = 0, // runnable, waiting for a core
|
||||||
|
running = 1, // executing on a core right now
|
||||||
|
blocked = 2, // waiting (sleeping, or blocked in IPC)
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One `process_enumerate` entry — the kernel's view of a live task, kernel tasks
|
||||||
|
/// included (they carry an empty name and id 0 is the boot task). Fixed layout
|
||||||
|
/// (extern) because it crosses the kernel↔user boundary by memory copy, like
|
||||||
|
/// `DeviceDescriptor` in the device ABI.
|
||||||
|
pub const ProcessDescriptor = extern struct {
|
||||||
|
id: u32, // kernel-assigned process id; never reused (monotonic)
|
||||||
|
supervisor: u32, // id of the process that spawned it (0 = the kernel)
|
||||||
|
state: u32, // a ProcessState value
|
||||||
|
priority: u32,
|
||||||
|
name_length: u32,
|
||||||
|
name: [maximum_process_name]u8, // argv[0] at spawn; empty for kernel tasks
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Well-known IPC service ids for the bootstrap name registry (create_ipc_endpoint +
|
||||||
|
/// ipc_register/ipc_lookup). Small integers, so no string interning is needed
|
||||||
|
/// during bring-up. The VFS server registers under `vfs`; clients look it up.
|
||||||
|
pub const ServiceId = enum(u32) {
|
||||||
|
vfs = 1,
|
||||||
|
input = 2,
|
||||||
|
ps2_bus = 3, // the 8042 owner; child device drivers attach here for raw bytes
|
||||||
|
device_manager = 4, // the tree, the matcher, the supervisor (docs/device-manager.md)
|
||||||
|
power = 5, // system power: events (button, lid, battery) + shutdown (docs/power.md; domain-named per docs/discovery.md — the acpi service registers it on x86, a PSCI service will on ARM)
|
||||||
|
usb_bus = 6, // the xHCI host-controller driver's transfer endpoint; USB class drivers look it up and `callCap`-open their device to get a private per-device transfer channel (docs/driver-model.md)
|
||||||
|
block = 7, // a block-device driver (USB mass storage today): read/write of fixed-size blocks, the storage a filesystem sits on
|
||||||
|
fat = 8, // the FAT filesystem server; the VFS mounts it and forwards paths under its mount point (/mnt/usb) to it
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Protection flags for `mmap` (matching the usual C bit values).
|
||||||
|
pub const prot_read: u64 = 1;
|
||||||
|
pub const prot_write: u64 = 2;
|
||||||
|
pub const prot_exec: u64 = 4;
|
||||||
|
|
||||||
|
/// `send_cap` / `received_cap` sentinel meaning "no capability" on the `ipc_call` /
|
||||||
|
/// `ipc_reply_wait` cap-passing path (M13). `~0`, like `no_parent` — a real handle is
|
||||||
|
/// a small index, so it can never collide.
|
||||||
|
pub const no_cap: u64 = ~@as(u64, 0);
|
||||||
@@ -0,0 +1,156 @@
|
|||||||
|
//! The **loader ↔ kernel** contract: everything a bootloader (boot/, e.g. efi.zig
|
||||||
|
//! built as BOOTX64.efi) and the kernel (system/kernel/kernel.zig) must agree on to
|
||||||
|
//! hand control over — the handoff structures the loader fills in, plus the kernel's
|
||||||
|
//! virtual-memory layout and the physical↔virtual addressing both sides use.
|
||||||
|
//!
|
||||||
|
//! Both binaries import this as the `boot-handoff` module, so the layout is defined
|
||||||
|
//! in exactly one place. **User space never sees this** — the kernel↔user contract is
|
||||||
|
//! [[abi]] (system/abi.zig); device types are [[device-abi]] (system/devices/device-abi.zig).
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
/// Calling convention for the bootloader→kernel jump. Pinned to SystemV so it does
|
||||||
|
/// not depend on each binary's target default: the UEFI bootloader's C
|
||||||
|
/// convention is Microsoft x64 (first arg in RCX), the freestanding kernel's is
|
||||||
|
/// SystemV (first arg in RDI). Both reference this to agree on where `*BootInformation`
|
||||||
|
/// is passed.
|
||||||
|
pub const kernel_abi: std.builtin.CallingConvention = .{ .x86_64_sysv = .{} };
|
||||||
|
|
||||||
|
/// Pixel byte order of the linear framebuffer the firmware handed us.
|
||||||
|
pub const PixelFormat = enum(u32) {
|
||||||
|
/// Byte 0 = Red, 1 = Green, 2 = Blue, 3 = reserved.
|
||||||
|
rgbx,
|
||||||
|
/// Byte 0 = Blue, 1 = Green, 2 = Red, 3 = reserved.
|
||||||
|
bgrx,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A linear framebuffer: `width`x`height` pixels, each a 32-bit value, with
|
||||||
|
/// `pitch` bytes between the start of one row and the next (which may be larger
|
||||||
|
/// than `width * 4` due to hardware padding).
|
||||||
|
///
|
||||||
|
/// A `base` of 0 means **no framebuffer** — the firmware exposed no Graphics
|
||||||
|
/// Output Protocol (a headless server, say). The kernel must treat on-screen
|
||||||
|
/// output as optional and never assume a framebuffer exists.
|
||||||
|
pub const Framebuffer = extern struct {
|
||||||
|
base: usize, // the memory address where pixel data starts (0 = none)
|
||||||
|
width: u32, // visible pixels per row (e.g. 1920)
|
||||||
|
height: u32, // visible rows (e.g. 1080)
|
||||||
|
pitch: u32, // bytes from the start of one row to the start of the next
|
||||||
|
format: PixelFormat,
|
||||||
|
|
||||||
|
/// Whether a usable framebuffer was handed over.
|
||||||
|
pub fn present(self: Framebuffer) bool {
|
||||||
|
return self.base != 0 and self.width != 0 and self.height != 0;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The kernel's virtual-memory layout (higher-half). The kernel is linked at
|
||||||
|
/// `kernel_virt_base` but loaded at a low physical address; all of RAM (and the
|
||||||
|
/// device MMIO windows) is also mapped at `physmap_base + physical`, so the kernel
|
||||||
|
/// can reach any physical address by adding a constant. The low half is left
|
||||||
|
/// entirely to user space.
|
||||||
|
///
|
||||||
|
/// user image + stack : 0x0000_7000_0000_0000 (PML4[224], low half)
|
||||||
|
/// kernel heap : 0xFFFF_8000_0000_0000 (PML4[256])
|
||||||
|
/// physmap : 0xFFFF_8800_0000_0000 (PML4[272]) + physical
|
||||||
|
/// kernel image : 0xFFFF_FFFF_8000_0000 (PML4[511])
|
||||||
|
pub const physmap_base: u64 = 0xFFFF_8800_0000_0000;
|
||||||
|
pub const kernel_virt_base: u64 = 0xFFFF_FFFF_8000_0000;
|
||||||
|
|
||||||
|
/// Physical address -> its virtual address in the physmap. The single way the
|
||||||
|
/// kernel dereferences a physical address once paging is up.
|
||||||
|
///
|
||||||
|
/// **Hazard:** valid only once the (bootstrap or final) page tables are live.
|
||||||
|
/// The bootloader may use the *constant* `physmap_base` to build those tables,
|
||||||
|
/// but must not call this to dereference memory before its own CR3 is loaded —
|
||||||
|
/// it runs under the firmware's identity map, where these addresses are unmapped.
|
||||||
|
pub inline fn physicalToVirtual(physical: u64) u64 {
|
||||||
|
return physical + physmap_base;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Physmap virtual address -> physical. Inverse of `physicalToVirtual`; for producing
|
||||||
|
/// the physical address of something the kernel holds a physmap pointer to
|
||||||
|
/// (e.g. a page-table frame for CR3, a post-mortem breadcrumb's RAM location).
|
||||||
|
pub inline fn virtualToPhysical(virtual: u64) u64 {
|
||||||
|
return virtual - physmap_base;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// danos's own classification of a span of physical memory — deliberately not
|
||||||
|
/// UEFI's vocabulary. Each boot path (UEFI now, device tree later) translates its
|
||||||
|
/// native memory description into these kinds, so the kernel never learns what
|
||||||
|
/// booted it. [[architecture]] keeps the same discipline for CPU code.
|
||||||
|
pub const MemoryKind = enum(u32) {
|
||||||
|
/// Free RAM the kernel may allocate. Each boot path folds its own transient
|
||||||
|
/// memory into this once it's genuinely free (e.g. the UEFI loader classifies
|
||||||
|
/// boot-services memory as usable after ExitBootServices), so the kernel never
|
||||||
|
/// has to know about boot-protocol-specific "reclaimable" states.
|
||||||
|
usable,
|
||||||
|
/// Firmware, MMIO, the kernel image, our own boot buffers, the boot stack —
|
||||||
|
/// never hand out.
|
||||||
|
reserved,
|
||||||
|
/// ACPI tables: parse, then reclaim.
|
||||||
|
acpi_tables,
|
||||||
|
/// ACPI non-volatile storage: preserve across sleep, do not allocate.
|
||||||
|
acpi_nvs,
|
||||||
|
/// Not backed by RAM: memory-mapped device registers or a reserved
|
||||||
|
/// address-space window (e.g. PCIe configuration space). Kept distinct from
|
||||||
|
/// `reserved` so RAM accounting doesn't count device address space.
|
||||||
|
mmio,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One contiguous span of physical memory. Because danos defines this layout
|
||||||
|
/// itself (unlike the UEFI descriptor it's built from), `@sizeOf` is
|
||||||
|
/// authoritative — the kernel walks a plain `[]MemoryRegion`, with none of the
|
||||||
|
/// firmware's variable descriptor-stride to worry about.
|
||||||
|
pub const MemoryRegion = extern struct {
|
||||||
|
base: u64, // physical start address
|
||||||
|
pages: u64, // length in 4 KiB pages (the [[abi]] `page_size` unit)
|
||||||
|
kind: MemoryKind,
|
||||||
|
_pad: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The physical memory layout handed to the kernel: a pointer to an array of
|
||||||
|
/// `len` `MemoryRegion`s, in a buffer that outlives the loader.
|
||||||
|
pub const MemoryMap = extern struct {
|
||||||
|
regions: usize, // address of a `[len]MemoryRegion`
|
||||||
|
len: usize,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One PT_LOAD segment of the kernel image, so the kernel can re-map itself with
|
||||||
|
/// correct permissions (code R+X, rodata R, data R+W+NX). `flags` are raw ELF
|
||||||
|
/// segment flags: PF_X=1, PF_W=2, PF_R=4. `virtual` is the higher-half link address;
|
||||||
|
/// `physical` is where the loader actually placed the segment (they differ once the
|
||||||
|
/// kernel links high — the loader records the real load address here).
|
||||||
|
pub const KernelSegment = extern struct {
|
||||||
|
virtual: u64,
|
||||||
|
physical: u64,
|
||||||
|
pages: u64,
|
||||||
|
flags: u32,
|
||||||
|
_pad: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Handoff structure the bootloader fills in and passes to the kernel's
|
||||||
|
/// `_start` in RDI (the first argument under the SystemV AMD64 C ABI).
|
||||||
|
pub const BootInformation = extern struct {
|
||||||
|
framebuffer: Framebuffer,
|
||||||
|
memory_map: MemoryMap,
|
||||||
|
/// The kernel's own PT_LOAD segments (it has three: text, rodata, data).
|
||||||
|
kernel_segments: [8]KernelSegment,
|
||||||
|
kernel_segment_count: u32,
|
||||||
|
/// Physical address of the ACPI RSDP the firmware exposed, or 0 if none. The
|
||||||
|
/// kernel's device layer parses the ACPI tables from here to discover hardware.
|
||||||
|
/// A device-tree boot path leaves this 0 and (later) fills a `device_tree_blob`
|
||||||
|
/// field instead, so the kernel discovers devices without knowing what booted it.
|
||||||
|
acpi_rsdp: u64 = 0,
|
||||||
|
/// The raw `/system/services/init` ELF image, read off the boot volume by the loader
|
||||||
|
/// into memory that survives the handoff (classified reserved, so the kernel
|
||||||
|
/// identity-maps it and never allocates over it). 0/0 = no init found — the
|
||||||
|
/// kernel boots without user space. Grows into a full initial_ramdisk handoff later.
|
||||||
|
init_base: u64 = 0,
|
||||||
|
init_len: u64 = 0,
|
||||||
|
/// The initial_ramdisk image (a bundle of extra user binaries — the VFS server and
|
||||||
|
/// device drivers), read off the boot volume into memory that survives the
|
||||||
|
/// handoff, same as `init` above. 0/0 = no initial_ramdisk. See system/initial-ramdisk.zig.
|
||||||
|
initial_ramdisk_base: u64 = 0,
|
||||||
|
initial_ramdisk_len: u64 = 0,
|
||||||
|
};
|
||||||
@@ -0,0 +1,138 @@
|
|||||||
|
//! ACPI / PnP hardware-ID (`_HID`) names: the flat analog of pci-class.zig for
|
||||||
|
//! `acpi_device` nodes. Unlike PCI, ACPI has no class/subclass/prog-IF taxonomy — a
|
||||||
|
//! device's identity *is* its `_HID` string (`PNP0303` simply means "PS/2 keyboard"),
|
||||||
|
//! so this is a plain id <-> name registry rather than a hierarchical decoder.
|
||||||
|
//! The well-known PnP/ACPI IDs; vendor-specific ids (e.g. `QEMU0002`, `INTC1234`) have
|
||||||
|
//! no standard name and decode to nothing. Pure reference data, so it is shared by
|
||||||
|
//! kernel discovery (the device-tree dump) and any user-space driver or tool.
|
||||||
|
//!
|
||||||
|
//! Code that means a specific device names the `HardwareId` variant instead of its
|
||||||
|
//! `_HID` string — `HardwareId.ps2_keyboard.hid()` reads without a registry lookup,
|
||||||
|
//! where a bare `"PNP0303"` does not.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
/// The common standard PnP/ACPI hardware IDs, as named values. Prefix ranges hint at
|
||||||
|
/// the grouping (PNP03xx keyboards, PNP0Fxx pointing devices, PNP0Cxx ACPI
|
||||||
|
/// power/thermal, PNP0Axx buses), but there is no formal hierarchy — hence a flat
|
||||||
|
/// enum over a flat registry.
|
||||||
|
pub const HardwareId = enum {
|
||||||
|
programmable_interrupt_controller,
|
||||||
|
system_timer,
|
||||||
|
high_precision_event_timer,
|
||||||
|
dma_controller,
|
||||||
|
ps2_keyboard,
|
||||||
|
parallel_port,
|
||||||
|
ecp_parallel_port,
|
||||||
|
serial_port,
|
||||||
|
floppy_disk_controller,
|
||||||
|
system_speaker,
|
||||||
|
pci_bus,
|
||||||
|
generic_container,
|
||||||
|
/// The second id the ACPI spec assigns the same "Generic Container Device" name.
|
||||||
|
generic_container_extended,
|
||||||
|
pci_express_root_bridge,
|
||||||
|
real_time_clock,
|
||||||
|
system_board,
|
||||||
|
motherboard_reserved_resources,
|
||||||
|
math_coprocessor,
|
||||||
|
acpi_system_board,
|
||||||
|
embedded_controller,
|
||||||
|
control_method_battery,
|
||||||
|
fan,
|
||||||
|
power_button,
|
||||||
|
lid,
|
||||||
|
sleep_button,
|
||||||
|
pci_interrupt_link,
|
||||||
|
microsoft_ps2_mouse,
|
||||||
|
ps2_mouse,
|
||||||
|
ac_adapter,
|
||||||
|
processor_device,
|
||||||
|
processor_aggregator,
|
||||||
|
processor_container,
|
||||||
|
|
||||||
|
const Entry = struct { hid: []const u8, name: []const u8 };
|
||||||
|
|
||||||
|
/// The registry row for this id: its `_HID` string and human-readable name.
|
||||||
|
fn entry(self: HardwareId) Entry {
|
||||||
|
return switch (self) {
|
||||||
|
.programmable_interrupt_controller => .{ .hid = "PNP0000", .name = "Programmable Interrupt Controller (PIC)" },
|
||||||
|
.system_timer => .{ .hid = "PNP0100", .name = "System Timer (PIT)" },
|
||||||
|
.high_precision_event_timer => .{ .hid = "PNP0103", .name = "High Precision Event Timer (HPET)" },
|
||||||
|
.dma_controller => .{ .hid = "PNP0200", .name = "DMA Controller" },
|
||||||
|
.ps2_keyboard => .{ .hid = "PNP0303", .name = "PS/2 Keyboard" },
|
||||||
|
.parallel_port => .{ .hid = "PNP0400", .name = "Standard LPT Parallel Port" },
|
||||||
|
.ecp_parallel_port => .{ .hid = "PNP0401", .name = "ECP Parallel Port" },
|
||||||
|
.serial_port => .{ .hid = "PNP0501", .name = "16550A-compatible Serial Port" },
|
||||||
|
.floppy_disk_controller => .{ .hid = "PNP0700", .name = "PC Floppy Disk Controller" },
|
||||||
|
.system_speaker => .{ .hid = "PNP0800", .name = "System Speaker" },
|
||||||
|
.pci_bus => .{ .hid = "PNP0A03", .name = "PCI Bus" },
|
||||||
|
.generic_container => .{ .hid = "PNP0A05", .name = "Generic Container Device" },
|
||||||
|
.generic_container_extended => .{ .hid = "PNP0A06", .name = "Generic Container Device" },
|
||||||
|
.pci_express_root_bridge => .{ .hid = "PNP0A08", .name = "PCI Express Root Bridge" },
|
||||||
|
.real_time_clock => .{ .hid = "PNP0B00", .name = "Real-Time Clock (RTC)" },
|
||||||
|
.system_board => .{ .hid = "PNP0C01", .name = "System Board" },
|
||||||
|
.motherboard_reserved_resources => .{ .hid = "PNP0C02", .name = "Motherboard Reserved Resources" },
|
||||||
|
.math_coprocessor => .{ .hid = "PNP0C04", .name = "Math Coprocessor" },
|
||||||
|
.acpi_system_board => .{ .hid = "PNP0C08", .name = "ACPI System Board" },
|
||||||
|
.embedded_controller => .{ .hid = "PNP0C09", .name = "ACPI Embedded Controller" },
|
||||||
|
.control_method_battery => .{ .hid = "PNP0C0A", .name = "ACPI Control Method Battery" },
|
||||||
|
.fan => .{ .hid = "PNP0C0B", .name = "ACPI Fan" },
|
||||||
|
.power_button => .{ .hid = "PNP0C0C", .name = "ACPI Power Button" },
|
||||||
|
.lid => .{ .hid = "PNP0C0D", .name = "ACPI Lid" },
|
||||||
|
.sleep_button => .{ .hid = "PNP0C0E", .name = "ACPI Sleep Button" },
|
||||||
|
.pci_interrupt_link => .{ .hid = "PNP0C0F", .name = "PCI Interrupt Link Device" },
|
||||||
|
.microsoft_ps2_mouse => .{ .hid = "PNP0F03", .name = "Microsoft PS/2 Mouse" },
|
||||||
|
.ps2_mouse => .{ .hid = "PNP0F13", .name = "PS/2 Mouse" },
|
||||||
|
.ac_adapter => .{ .hid = "ACPI0003", .name = "AC Adapter" },
|
||||||
|
.processor_device => .{ .hid = "ACPI0007", .name = "Processor Device" },
|
||||||
|
.processor_aggregator => .{ .hid = "ACPI000C", .name = "Processor Aggregator" },
|
||||||
|
.processor_container => .{ .hid = "ACPI0010", .name = "Processor Container" },
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// This id's `_HID` string (e.g. `.ps2_keyboard` -> "PNP0303").
|
||||||
|
pub fn hid(self: HardwareId) []const u8 {
|
||||||
|
return self.entry().hid;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// This id's human-readable name (e.g. `.ps2_keyboard` -> "PS/2 Keyboard").
|
||||||
|
pub fn description(self: HardwareId) []const u8 {
|
||||||
|
return self.entry().name;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The named value for a `_HID` string, or null if it is not a known standard
|
||||||
|
/// id (vendor-specific ids are not in the registry).
|
||||||
|
pub fn fromHid(hid_string: []const u8) ?HardwareId {
|
||||||
|
for (std.enums.values(HardwareId)) |id| {
|
||||||
|
if (std.mem.eql(u8, id.hid(), hid_string)) return id;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The human-readable name for a `_HID` string, or "" if it is not a known standard
|
||||||
|
/// id (vendor-specific ids have no registry name — callers just print the raw HID).
|
||||||
|
pub fn description(hid: []const u8) []const u8 {
|
||||||
|
return (HardwareId.fromHid(hid) orelse return "").description();
|
||||||
|
}
|
||||||
|
|
||||||
|
test "decodes standard PnP/ACPI ids and leaves the rest alone" {
|
||||||
|
const eq = std.testing.expectEqualStrings;
|
||||||
|
try eq("PS/2 Keyboard", description("PNP0303"));
|
||||||
|
try eq("PS/2 Mouse", description("PNP0F13"));
|
||||||
|
try eq("PCI Express Root Bridge", description("PNP0A08"));
|
||||||
|
try eq("Real-Time Clock (RTC)", description("PNP0B00"));
|
||||||
|
try eq("", description("QEMU0002")); // vendor-specific: no standard name
|
||||||
|
try eq("", description("")); // no HID at all
|
||||||
|
}
|
||||||
|
|
||||||
|
test "named values round-trip through their _HID strings" {
|
||||||
|
const testing = std.testing;
|
||||||
|
try testing.expectEqualStrings("PNP0303", HardwareId.ps2_keyboard.hid());
|
||||||
|
try testing.expectEqual(@as(?HardwareId, .ps2_mouse), HardwareId.fromHid("PNP0F13"));
|
||||||
|
try testing.expectEqual(@as(?HardwareId, null), HardwareId.fromHid("QEMU0002"));
|
||||||
|
for (std.enums.values(HardwareId)) |id| {
|
||||||
|
try testing.expectEqual(@as(?HardwareId, id), HardwareId.fromHid(id.hid()));
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,875 @@
|
|||||||
|
//! ACPI discovery backend.
|
||||||
|
//!
|
||||||
|
//! Walks the ACPI tables the firmware left in memory (starting from the RSDP the
|
||||||
|
//! bootloader handed us) and translates the static tables into the generic
|
||||||
|
//! `device` model, so the kernel enumerates hardware without knowing ACPI is the
|
||||||
|
//! source. This is deliberately the *static-table* path: MADT (CPUs / interrupt
|
||||||
|
//! controllers), MCFG (PCIe ECAM -> PCI enumeration), HPET (timer), and FADT
|
||||||
|
//! (power register map). The DSDT/SSDT bytecode is handed to the `aml` submodule
|
||||||
|
//! only to extract the sleep-state (`_Sx`) values for power management; full AML namespace
|
||||||
|
//! interpretation is a separate, larger subproject.
|
||||||
|
//!
|
||||||
|
//! ACPI tables live in `.acpi_tables` / `.acpi_nvs` memory, which the kernel
|
||||||
|
//! identity-maps, so table addresses are dereferenced directly. PCIe ECAM is MMIO
|
||||||
|
//! and is *not* mapped up front, so configuration-space pages are mapped on demand via
|
||||||
|
//! the `Hal.mapMmio` callback the caller supplies (the architecture VMM's map primitive).
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const boot_handoff = @import("boot-handoff");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const parameters = @import("parameters");
|
||||||
|
const device_model = @import("device-model.zig");
|
||||||
|
const aml = @import("aml/aml.zig");
|
||||||
|
const DeviceTree = device_model.DeviceTree;
|
||||||
|
const Hal = device_model.Hal;
|
||||||
|
|
||||||
|
/// A hardware register located either in MMIO or I/O-port space, as ACPI's
|
||||||
|
/// Generic Address Structure describes. `address == 0` means "not present".
|
||||||
|
pub const RegisterAccess = struct {
|
||||||
|
/// true = system memory (MMIO), false = system I/O port space.
|
||||||
|
mmio: bool = false,
|
||||||
|
address: u64 = 0,
|
||||||
|
/// Access width in bytes.
|
||||||
|
width: u8 = 0,
|
||||||
|
|
||||||
|
pub fn present(self: RegisterAccess) bool {
|
||||||
|
return self.address != 0;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Everything the power subsystem needs, extracted from the FADT and the AML
|
||||||
|
/// sleep packages during discovery. Populated by `discover`, read by `power`.
|
||||||
|
pub const PowerInformation = struct {
|
||||||
|
/// The System Control Interrupt's GSI (FADT SCI_INT) — the line ACPI events
|
||||||
|
/// (power button, GPEs) arrive on. Published to the acpi service for M21.
|
||||||
|
sci_interrupt: u16 = 0,
|
||||||
|
/// The SMM command port and the value that switches the platform into ACPI mode.
|
||||||
|
smi_cmd: u16 = 0,
|
||||||
|
acpi_enable: u8 = 0,
|
||||||
|
acpi_disable: u8 = 0,
|
||||||
|
/// PM1 control registers — writing SLP_TYP|SLP_EN here enters a sleep state.
|
||||||
|
pm1a_cnt: RegisterAccess = .{},
|
||||||
|
pm1b_cnt: RegisterAccess = .{},
|
||||||
|
/// The FADT reset register and the value to write to it.
|
||||||
|
reset: RegisterAccess = .{},
|
||||||
|
reset_value: u8 = 0,
|
||||||
|
reset_supported: bool = false,
|
||||||
|
/// SLP_TYP values for S5 (soft off) and S3 (suspend), from the AML sleep-state (`_Sx`)
|
||||||
|
/// packages.
|
||||||
|
s5: ?aml.SleepType = null,
|
||||||
|
s3: ?aml.SleepType = null,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Filled in by `discover`; the power service reads it to reboot/shutdown.
|
||||||
|
pub var power_information: PowerInformation = .{};
|
||||||
|
|
||||||
|
/// A legacy ISA IRQ remapped to a different global system interrupt (GSI), from a
|
||||||
|
/// MADT Interrupt Source Override. `flags` are the MPS INTI polarity/trigger bits.
|
||||||
|
pub const IsoEntry = struct {
|
||||||
|
source: u8,
|
||||||
|
gsi: u32,
|
||||||
|
flags: u16,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Firmware facts the architecture layer needs to avoid legacy assumptions (so danos boots
|
||||||
|
/// on legacy-free UEFI Class 3 machines). MMIO device *addresses* (HPET, IOAPIC)
|
||||||
|
/// come from the device tree instead; this holds the scalar facts that have no
|
||||||
|
/// natural device node.
|
||||||
|
pub const PlatformInformation = struct {
|
||||||
|
/// Whether the legacy 8259 PIC is present (MADT flags bit 0, PCAT_COMPAT). When
|
||||||
|
/// false, the PIC must not be programmed (it may not exist).
|
||||||
|
pic_present: bool = false,
|
||||||
|
/// Local APIC MMIO base (MADT, honouring a type-5 address override).
|
||||||
|
lapic_base: u64 = 0xFEE00000,
|
||||||
|
/// The ACPI power-management timer — a fixed 3.579545 MHz counter usable as a
|
||||||
|
/// calibration reference when no HPET is present.
|
||||||
|
pm_timer: RegisterAccess = .{},
|
||||||
|
/// true = 32-bit PM timer counter, false = 24-bit (FADT flag TMR_VALUE_EXT).
|
||||||
|
pm_timer_32bit: bool = false,
|
||||||
|
/// The console UART the firmware points at (SPCR), if any — MMIO or I/O port.
|
||||||
|
spcr_uart: ?RegisterAccess = null,
|
||||||
|
/// SPCR interface type (0/1 = 16550/16450, …).
|
||||||
|
spcr_kind: u8 = 0,
|
||||||
|
/// ISA-IRQ-to-GSI remappings from the MADT (for future IOAPIC routing).
|
||||||
|
overrides: [16]IsoEntry = undefined,
|
||||||
|
override_count: usize = 0,
|
||||||
|
/// Whether an IOMMU (VT-d DMA-remapping unit) was found in the ACPI DMAR table.
|
||||||
|
/// When false, `device_claim` on a DMA-capable device is equivalent to granting
|
||||||
|
/// ring 0 — a device can DMA to any physical address (docs/driver-model.md M16).
|
||||||
|
/// Detection is the first step; per-device domain enforcement lands with the first
|
||||||
|
/// DMA driver.
|
||||||
|
iommu_present: bool = false,
|
||||||
|
/// MMIO base of the first DMA-remapping hardware unit (DMAR DRHD), when present.
|
||||||
|
iommu_base: u64 = 0,
|
||||||
|
/// The unit's Version register (offset 0x00) — its low byte is major.minor;
|
||||||
|
/// reading it back nonzero confirms a real, mappable VT-d unit.
|
||||||
|
iommu_version: u32 = 0,
|
||||||
|
/// The unit's Capability register (offset 0x08): supported address widths, number
|
||||||
|
/// of domains, etc. Recorded now; consumed when enforcement is built.
|
||||||
|
iommu_capabilities: u64 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Filled in by `discover`; the architecture layer reads it during bring-up.
|
||||||
|
pub var platform_information: PlatformInformation = .{};
|
||||||
|
|
||||||
|
/// One usable logical processor, from a MADT type-0 (Local APIC) record. The
|
||||||
|
/// `apic_id` is the Local APIC ID that SMP bring-up targets to wake this core
|
||||||
|
/// (INIT–SIPI–SIPI); `processor_id` is the ACPI namespace handle. Only processors
|
||||||
|
/// the firmware marks *enabled* are recorded — a disabled one can't be started.
|
||||||
|
pub const Cpu = struct {
|
||||||
|
processor_id: u8,
|
||||||
|
apic_id: u8,
|
||||||
|
/// MADT flags bit 1: usable but firmware-started offline (hot-plug / deferred
|
||||||
|
/// bring-up), as opposed to already available. Informational for now.
|
||||||
|
online_capable: bool,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The set of usable logical processors the MADT listed — the hardware's degree of
|
||||||
|
/// parallelism. Includes the bootstrap processor danos already runs on; the rest
|
||||||
|
/// are the application processors SMP bring-up would start (see docs/smp.md).
|
||||||
|
pub const CpuInformation = struct {
|
||||||
|
/// A static pool sized well above any danos target (a desktop, two 4-core Pis).
|
||||||
|
/// If the MADT ever lists more, the surplus is dropped and counted in `dropped`
|
||||||
|
/// so the truncation is never silent.
|
||||||
|
cpus: [maximum_cpus]Cpu = undefined,
|
||||||
|
count: usize = 0,
|
||||||
|
dropped: usize = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
const maximum_cpus = parameters.maximum_cpus;
|
||||||
|
|
||||||
|
/// Filled in by `discover` (from the MADT); SMP bring-up reads it to wake the APs.
|
||||||
|
pub var cpu_information: CpuInformation = .{};
|
||||||
|
|
||||||
|
/// Integrity/diagnostics for the AML parse. `consumed == total` means the parser
|
||||||
|
/// walked every byte of the DSDT/SSDTs without desyncing.
|
||||||
|
pub const AmlStats = struct {
|
||||||
|
nodes: usize = 0,
|
||||||
|
consumed: usize = 0,
|
||||||
|
total: usize = 0,
|
||||||
|
};
|
||||||
|
pub var aml_stats: AmlStats = .{};
|
||||||
|
|
||||||
|
/// The ACPI namespace built from the DSDT/SSDTs, kept for sleep-state (`_Sx`) lookup now and
|
||||||
|
/// device enumeration later. Null until `discover` runs successfully.
|
||||||
|
pub var namespace: ?aml.Namespace = null;
|
||||||
|
|
||||||
|
/// Physical address of the DSDT the FADT points at, or 0.
|
||||||
|
pub var dsdt_physical: u64 = 0;
|
||||||
|
|
||||||
|
/// The FADT itself (physical + length), published on the acpi-tables node so
|
||||||
|
/// the ring-3 acpi service can read the PM1 event and GPE blocks it needs for
|
||||||
|
/// the event side (docs/acpi.md — ACPI events). Distinguished from the AML
|
||||||
|
/// blob resources by its intact "FACP" header — the blobs are header-stripped.
|
||||||
|
var fadt_physical: u64 = 0;
|
||||||
|
var fadt_length: u64 = 0;
|
||||||
|
|
||||||
|
// AML blocks (DSDT + any SSDTs) collected during the table walk, as physical
|
||||||
|
// address + length of each table's post-header bytecode. Scanned after the walk
|
||||||
|
// for the sleep-state (`_Sx`) packages.
|
||||||
|
var aml_block_physical: [32]u64 = undefined;
|
||||||
|
var aml_block_len: [32]usize = undefined;
|
||||||
|
var aml_block_count: usize = 0;
|
||||||
|
|
||||||
|
fn addAmlBlock(sdt_physical: u64) void {
|
||||||
|
if (aml_block_count >= aml_block_physical.len or sdt_physical == 0) return;
|
||||||
|
const h: *const SystemDescriptorTableHeader = @ptrFromInt(boot_handoff.physicalToVirtual(sdt_physical));
|
||||||
|
if (h.length <= @sizeOf(SystemDescriptorTableHeader)) return;
|
||||||
|
aml_block_physical[aml_block_count] = sdt_physical + @sizeOf(SystemDescriptorTableHeader);
|
||||||
|
aml_block_len[aml_block_count] = h.length - @sizeOf(SystemDescriptorTableHeader);
|
||||||
|
aml_block_count += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// RSDP structure for revision 0 (version 1.0)
|
||||||
|
const RootSystemDescriptionPointer = extern struct {
|
||||||
|
/// An 8 byte magic number used for locating the RSDP, containing RSD PTR.
|
||||||
|
signature: [8]u8,
|
||||||
|
/// A byte used to verify the first 20 bytes of the RSDP
|
||||||
|
checksum: u8,
|
||||||
|
/// An OEM-supplied string that identified the OEM.
|
||||||
|
oem_id: [6]u8,
|
||||||
|
/// The RSDP revision, used for determining which fields are available.
|
||||||
|
revision: u8,
|
||||||
|
/// A 32-bit physical address pointing to the RSDT.
|
||||||
|
root_system_description_table_address: u32 align(1),
|
||||||
|
};
|
||||||
|
|
||||||
|
/// XSDP structure for revision 2 (version 2.0+)
|
||||||
|
const ExtendedSystemDescriptorPointer = extern struct {
|
||||||
|
/// An 8 byte magic number used for locating the RSDP, containing RSD PTR.
|
||||||
|
signature: [8]u8,
|
||||||
|
/// A byte used to verify the first 20 bytes of the RSDP
|
||||||
|
checksum: u8,
|
||||||
|
/// An OEM-supplied string that identified the OEM.
|
||||||
|
oem_id: [6]u8,
|
||||||
|
/// The RSDP revision, used for determining which fields are available.
|
||||||
|
revision: u8,
|
||||||
|
/// deprecated since version 2.0. A 32-bit physical address pointing to the RSDT.
|
||||||
|
root_system_description_table_address: u32 align(1),
|
||||||
|
/// The size of the RSDP.
|
||||||
|
length: u32 align(1),
|
||||||
|
/// A 64-bit physical address pointing to the XSDT. If the revision is at least 2, the XSDT
|
||||||
|
/// should be used regardless of architecture, as the RSDT was deprecated.
|
||||||
|
extended_system_descriptor_table_address: u64 align(1),
|
||||||
|
/// A checksum used for the entire table.
|
||||||
|
extended_checksum: u8,
|
||||||
|
reserved: [3]u8,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Multiple APIC Description Table (MADT)
|
||||||
|
const APIC: [4]u8 = "APIC".*;
|
||||||
|
/// Boot Error Record Table (BERT)
|
||||||
|
const BERT: [4]u8 = "BERT".*;
|
||||||
|
/// Corrected Platform Error Polling Table (CPEP)
|
||||||
|
const CPEP: [4]u8 = "CPEP".*;
|
||||||
|
/// Differentiated System Description Table (DSDT)
|
||||||
|
const DSDT: [4]u8 = "DSDT".*;
|
||||||
|
/// Embedded Controller Boot Resources Table (ECDT)
|
||||||
|
const ECDT: [4]u8 = "ECDT".*;
|
||||||
|
/// Error Injection Table (EINJ)
|
||||||
|
const EINJ: [4]u8 = "EINJ".*;
|
||||||
|
/// Error Record Serialization Table (ERST)
|
||||||
|
const ERST: [4]u8 = "ERST".*;
|
||||||
|
/// Fixed ACPI Description Table (FADT)
|
||||||
|
const FACP: [4]u8 = "FACP".*;
|
||||||
|
/// Firmware ACPI Control Structure (FACS)
|
||||||
|
const FACS: [4]u8 = "FACS".*;
|
||||||
|
/// Hardware Error Source Table (HEST)
|
||||||
|
const HEST: [4]u8 = "HEST".*;
|
||||||
|
/// High Precision Event Timer table (HPET)
|
||||||
|
const HPET: [4]u8 = "HPET".*;
|
||||||
|
/// PCI Express memory-mapped configuration space table (MCFG)
|
||||||
|
const MCFG: [4]u8 = "MCFG".*;
|
||||||
|
/// Maximum System Characteristics Table (MSCT)
|
||||||
|
const MSCT: [4]u8 = "MSCT".*;
|
||||||
|
/// Memory Power State Table (MPST)
|
||||||
|
const MPST: [4]u8 = "MPST".*;
|
||||||
|
// Platform Memory Topology Table (PMTT)
|
||||||
|
const PMTT: [4]u8 = "PMTT".*;
|
||||||
|
/// Persistent System Description Table (PSDT)
|
||||||
|
const PSDT: [4]u8 = "PSDT".*;
|
||||||
|
/// ACPI RAS Feature Table (RASF)
|
||||||
|
const RASF: [4]u8 = "RASF".*;
|
||||||
|
/// Root System Description Table
|
||||||
|
const RSDT: [4]u8 = "RSDT".*;
|
||||||
|
/// Smart Battery Specification Table (SBST)
|
||||||
|
const SBST: [4]u8 = "SBST".*;
|
||||||
|
/// System Locality System Information Table (SLIT)
|
||||||
|
const SLIT: [4]u8 = "SLIT".*;
|
||||||
|
/// System Resource Affinity Table (SRAT)
|
||||||
|
const SRAT: [4]u8 = "SRAT".*;
|
||||||
|
/// Secondary System Description Table (SSDT)
|
||||||
|
const DMAR: [4]u8 = "DMAR".*;
|
||||||
|
const SSDT: [4]u8 = "SSDT".*;
|
||||||
|
/// Serial Port Console Redirection table (SPCR) — the firmware's console UART.
|
||||||
|
const SPCR: [4]u8 = "SPCR".*;
|
||||||
|
/// Extended System Description Table (XSDT; 64-bit version of the RSDT)
|
||||||
|
const XSDT: [4]u8 = "XSDT".*;
|
||||||
|
|
||||||
|
/// The header every system descriptor table (RSDT/XSDT and each SDT) begins with.
|
||||||
|
const SystemDescriptorTableHeader = extern struct {
|
||||||
|
/// A 4 byte signature used for identification (e.g. "RSDT", "APIC").
|
||||||
|
signature: [4]u8,
|
||||||
|
/// The length of the entire table, including the header.
|
||||||
|
length: u32 align(1),
|
||||||
|
/// The revision of the ACPI spec this table conforms to.
|
||||||
|
revision: u8,
|
||||||
|
/// An 8-bit checksum field for the whole table, inclusive of the header.
|
||||||
|
checksum: u8,
|
||||||
|
/// An OEM-supplied string that identified the OEM.
|
||||||
|
oem_id: [6]u8,
|
||||||
|
oem_table_id: [8]u8,
|
||||||
|
oem_revision: u32 align(1),
|
||||||
|
creator_id: u32 align(1),
|
||||||
|
creator_revision: u32 align(1),
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- MADT: Multiple APIC Description Table (signature "APIC") ---------------
|
||||||
|
|
||||||
|
const Madt = extern struct {
|
||||||
|
header: SystemDescriptorTableHeader,
|
||||||
|
local_apic_address: u32 align(1),
|
||||||
|
flags: u32 align(1),
|
||||||
|
// Followed by a variable-length run of interrupt-controller records, each a
|
||||||
|
// MadtRecordHeader plus a type-specific body.
|
||||||
|
};
|
||||||
|
|
||||||
|
const MadtRecordHeader = extern struct {
|
||||||
|
type: u8,
|
||||||
|
length: u8,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// MADT record type 0: a processor's Local APIC.
|
||||||
|
const MadtLocalApic = extern struct {
|
||||||
|
record: MadtRecordHeader,
|
||||||
|
processor_id: u8,
|
||||||
|
apic_id: u8,
|
||||||
|
/// bit 0 = enabled, bit 1 = online-capable.
|
||||||
|
flags: u32 align(1),
|
||||||
|
};
|
||||||
|
|
||||||
|
/// MADT record type 1: an I/O APIC.
|
||||||
|
const MadtIoApic = extern struct {
|
||||||
|
record: MadtRecordHeader,
|
||||||
|
io_apic_id: u8,
|
||||||
|
reserved: u8,
|
||||||
|
address: u32 align(1),
|
||||||
|
/// First global system interrupt this I/O APIC handles.
|
||||||
|
gsi_base: u32 align(1),
|
||||||
|
};
|
||||||
|
|
||||||
|
/// MADT record type 2: an Interrupt Source Override (ISA IRQ -> GSI remap).
|
||||||
|
const MadtIso = extern struct {
|
||||||
|
record: MadtRecordHeader,
|
||||||
|
bus: u8,
|
||||||
|
source: u8,
|
||||||
|
gsi: u32 align(1),
|
||||||
|
flags: u16 align(1),
|
||||||
|
};
|
||||||
|
|
||||||
|
/// MADT record type 5: Local APIC Address Override (64-bit MMIO base).
|
||||||
|
const MadtLapicOverride = extern struct {
|
||||||
|
record: MadtRecordHeader,
|
||||||
|
reserved: u16 align(1),
|
||||||
|
address: u64 align(1),
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- MCFG: PCIe ECAM configuration space (signature "MCFG") -----------------
|
||||||
|
|
||||||
|
const Mcfg = extern struct {
|
||||||
|
header: SystemDescriptorTableHeader,
|
||||||
|
reserved: u64 align(1),
|
||||||
|
// Followed by one or more McfgAllocation entries.
|
||||||
|
};
|
||||||
|
|
||||||
|
const McfgAllocation = extern struct {
|
||||||
|
/// Physical base of this segment group's ECAM window.
|
||||||
|
base_address: u64 align(1),
|
||||||
|
segment_group: u16 align(1),
|
||||||
|
start_bus: u8,
|
||||||
|
end_bus: u8,
|
||||||
|
reserved: u32 align(1),
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- HPET (signature "HPET") ------------------------------------------------
|
||||||
|
|
||||||
|
const Hpet = extern struct {
|
||||||
|
header: SystemDescriptorTableHeader,
|
||||||
|
hardware_rev_id: u8,
|
||||||
|
flags: u8,
|
||||||
|
pci_vendor_id: u16 align(1),
|
||||||
|
// Generic Address Structure describing the register block.
|
||||||
|
address_space_id: u8,
|
||||||
|
register_bit_width: u8,
|
||||||
|
register_bit_offset: u8,
|
||||||
|
gas_reserved: u8,
|
||||||
|
address: u64 align(1),
|
||||||
|
hpet_number: u8,
|
||||||
|
minimum_tick: u16 align(1),
|
||||||
|
page_protection: u8,
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- Entry point ------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Discover hardware from the ACPI tables rooted at `rsdp_physical` and populate
|
||||||
|
/// `device_tree`. `hal` provides MMIO mapping (for PCIe ECAM) and port I/O. Also parses the
|
||||||
|
/// FADT and the AML sleep-state (`_Sx`) packages into `power_information` for the power service.
|
||||||
|
pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryRegion, device_tree: *DeviceTree, hal: Hal) !void {
|
||||||
|
if (rsdp_physical == 0) return error.NoRsdp;
|
||||||
|
boot_memory_regions = memory_regions;
|
||||||
|
|
||||||
|
// Start clean so a re-run doesn't accumulate stale state.
|
||||||
|
power_information = .{};
|
||||||
|
fadt_physical = 0;
|
||||||
|
fadt_length = 0;
|
||||||
|
platform_information = .{};
|
||||||
|
aml_stats = .{};
|
||||||
|
namespace = null;
|
||||||
|
dsdt_physical = 0;
|
||||||
|
aml_block_count = 0;
|
||||||
|
|
||||||
|
const rsdp: *const RootSystemDescriptionPointer = @ptrFromInt(boot_handoff.physicalToVirtual(rsdp_physical));
|
||||||
|
if (!std.mem.eql(u8, &rsdp.signature, "RSD PTR ")) return error.BadRsdpSignature;
|
||||||
|
// Revision 0 checksums only the first 20 bytes (the v1.0 RSDP).
|
||||||
|
if (!checksumOk(@ptrFromInt(boot_handoff.physicalToVirtual(rsdp_physical)), 20)) return error.BadRsdpChecksum;
|
||||||
|
|
||||||
|
if (rsdp.revision >= 2) {
|
||||||
|
const xsdp: *const ExtendedSystemDescriptorPointer = @ptrFromInt(boot_handoff.physicalToVirtual(rsdp_physical));
|
||||||
|
if (!checksumOk(@ptrFromInt(boot_handoff.physicalToVirtual(rsdp_physical)), xsdp.length)) return error.BadXsdpChecksum;
|
||||||
|
try walkRoot(u64, xsdp.extended_system_descriptor_table_address, device_tree, hal);
|
||||||
|
} else {
|
||||||
|
try walkRoot(u32, rsdp.root_system_description_table_address, device_tree, hal);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Now that the DSDT and any SSDTs are collected, build the AML namespace and
|
||||||
|
// read the sleep types from it.
|
||||||
|
var blocks: [aml_block_physical.len][]const u8 = undefined;
|
||||||
|
for (0..aml_block_count) |i| {
|
||||||
|
blocks[i] = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(aml_block_physical[i])))[0..aml_block_len[i]];
|
||||||
|
}
|
||||||
|
const active = blocks[0..aml_block_count];
|
||||||
|
if (aml.parse(device_tree.allocator, active)) |pr| {
|
||||||
|
namespace = pr.namespace;
|
||||||
|
aml_stats = .{ .nodes = namespace.?.nodeCount(), .consumed = pr.consumed, .total = pr.total };
|
||||||
|
power_information.s5 = aml.sleepState(&namespace.?, 5);
|
||||||
|
power_information.s3 = aml.sleepState(&namespace.?, 3);
|
||||||
|
// The namespace's Device objects are no longer folded into the kernel
|
||||||
|
// tree (M20.3): the ring-3 acpi service claims the acpi-tables node
|
||||||
|
// (published below), re-parses the same blobs, and registers + reports
|
||||||
|
// the _HID devices itself. The kernel keeps the namespace only for the
|
||||||
|
// \_S5 sleep type above.
|
||||||
|
} else |_| {
|
||||||
|
// AML parse failed (e.g. out of memory); power stays best-effort with
|
||||||
|
// whatever the FADT alone provided.
|
||||||
|
}
|
||||||
|
|
||||||
|
// Publish the acpi-tables node (docs/discovery.md): the AML blobs as
|
||||||
|
// memory resources for the acpi service to map and parse in ring 3, a broad
|
||||||
|
// io_port grant for the OperationRegion access its interpreter needs, and
|
||||||
|
// the SCI for the events track (M21). Exactly one node, one trusted
|
||||||
|
// claimant. Kept even when the kernel-side device building (above) retires
|
||||||
|
// in M20.3 — the kernel still owns the *static* tables and \_S5.
|
||||||
|
publishAcpiTablesNode(device_tree) catch {};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Build the acpi-tables node (see the call site in discover). Best-effort: a
|
||||||
|
/// failure here leaves the kernel-seeded tree working, only the ring-3 service
|
||||||
|
/// finds nothing to claim.
|
||||||
|
fn publishAcpiTablesNode(device_tree: *DeviceTree) !void {
|
||||||
|
const node = try device_tree.addChild(device_tree.root, .acpi_tables, "acpi-tables");
|
||||||
|
// One memory resource per AML block — page-aligned base down, length padded
|
||||||
|
// up to cover the bytecode, so mmio_map hands the service a pointer into it.
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < aml_block_count and i < device_model.maximum_resources - 2) : (i += 1) {
|
||||||
|
// mmio_map preserves the sub-page offset, so the service maps this and
|
||||||
|
// gets a pointer straight to the bytecode.
|
||||||
|
_ = node.addResource(.memory, aml_block_physical[i], aml_block_len[i]);
|
||||||
|
}
|
||||||
|
// The broad I/O grant: OperationRegions name whatever ports the firmware
|
||||||
|
// chose (EC, PM1, GPE, SMBus); which ports cannot be known before the AML
|
||||||
|
// that names them is parsed, so the grant is the whole space — the honest
|
||||||
|
// trust boundary of docs/discovery.md (the acpi service's one trusted node).
|
||||||
|
_ = node.addResource(.io_port, 0, 1 << 16);
|
||||||
|
// A broad interrupt window: ACPI _CRS names legacy ISA IRQs (the PS/2 lines
|
||||||
|
// 1 and 12, the RTC, …), and the service registers those devices under this
|
||||||
|
// node, so it must own a superset. The range [0, 256) covers every GSI; the
|
||||||
|
// SCI (recorded first, len 1) stays distinct so M21 can pick it out.
|
||||||
|
if (power_information.sci_interrupt != 0) _ = node.addResource(.irq, power_information.sci_interrupt, 1);
|
||||||
|
_ = node.addResource(.irq, 0, 256);
|
||||||
|
// The FADT rides along (M21): the service reads the PM1 event / GPE blocks
|
||||||
|
// from its own copy, telling it apart from the AML blobs by signature.
|
||||||
|
if (fadt_physical != 0) _ = node.addResource(.memory, fadt_physical, fadt_length);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The number of Device objects in the namespace built during discovery, or 0.
|
||||||
|
pub fn amlDeviceCount() usize {
|
||||||
|
if (namespace) |*ns| return aml.deviceCount(ns);
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Walk the RSDT (Entry = u32) or XSDT (Entry = u64): validate it, then dispatch
|
||||||
|
/// each SDT it points at. A bad individual table is skipped, not fatal.
|
||||||
|
fn walkRoot(comptime Entry: type, root_physical: u64, device_tree: *DeviceTree, hal: Hal) !void {
|
||||||
|
const header: *const SystemDescriptorTableHeader = @ptrFromInt(boot_handoff.physicalToVirtual(root_physical));
|
||||||
|
if (!checksumOk(@ptrFromInt(boot_handoff.physicalToVirtual(root_physical)), header.length)) return error.BadRootChecksum;
|
||||||
|
|
||||||
|
const count = (header.length - @sizeOf(SystemDescriptorTableHeader)) / @sizeOf(Entry);
|
||||||
|
const base: [*]const u8 = @ptrFromInt(boot_handoff.physicalToVirtual(root_physical));
|
||||||
|
const entries: [*]align(1) const Entry = @ptrCast(base + @sizeOf(SystemDescriptorTableHeader));
|
||||||
|
|
||||||
|
for (entries[0..count]) |ent| {
|
||||||
|
const sdt_physical: u64 = ent; // u32 entries widen; u64 pass through
|
||||||
|
handleTable(device_tree, hal, sdt_physical) catch continue;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Dispatch a single SDT on its signature.
|
||||||
|
fn handleTable(device_tree: *DeviceTree, hal: Hal, sdt_physical: u64) !void {
|
||||||
|
const header: *const SystemDescriptorTableHeader = @ptrFromInt(boot_handoff.physicalToVirtual(sdt_physical));
|
||||||
|
const sig = header.signature;
|
||||||
|
if (std.mem.eql(u8, &sig, &APIC)) {
|
||||||
|
try parseMadt(device_tree, header);
|
||||||
|
} else if (std.mem.eql(u8, &sig, &MCFG)) {
|
||||||
|
try parseMcfg(device_tree, header);
|
||||||
|
} else if (std.mem.eql(u8, &sig, &HPET)) {
|
||||||
|
try parseHpet(device_tree, hal, header);
|
||||||
|
} else if (std.mem.eql(u8, &sig, &FACP)) {
|
||||||
|
fadt_physical = sdt_physical;
|
||||||
|
fadt_length = header.length;
|
||||||
|
parseFadt(header);
|
||||||
|
} else if (std.mem.eql(u8, &sig, &SPCR)) {
|
||||||
|
parseSpcr(header);
|
||||||
|
} else if (std.mem.eql(u8, &sig, &DMAR)) {
|
||||||
|
parseDmar(hal, header);
|
||||||
|
} else if (std.mem.eql(u8, &sig, &SSDT)) {
|
||||||
|
// Secondary namespace bytecode — collect for the sleep-state (`_Sx`) scan.
|
||||||
|
addAmlBlock(sdt_physical);
|
||||||
|
}
|
||||||
|
// Any other signature is recognised but left opaque for now.
|
||||||
|
}
|
||||||
|
|
||||||
|
/// MADT -> one processor node per Local APIC, one interrupt_controller per I/O APIC.
|
||||||
|
fn parseMadt(device_tree: *DeviceTree, header: *const SystemDescriptorTableHeader) !void {
|
||||||
|
const madt: *const Madt = @ptrCast(header);
|
||||||
|
const total: usize = header.length;
|
||||||
|
const base: [*]const u8 = @ptrCast(header);
|
||||||
|
var ioapic_index: usize = 0;
|
||||||
|
|
||||||
|
// MADT header: local APIC base + flags (bit 0 = 8259 PIC present).
|
||||||
|
platform_information.lapic_base = madt.local_apic_address;
|
||||||
|
platform_information.pic_present = madt.flags & 1 != 0;
|
||||||
|
|
||||||
|
var off: usize = @sizeOf(Madt);
|
||||||
|
while (off + @sizeOf(MadtRecordHeader) <= total) {
|
||||||
|
const rec: *const MadtRecordHeader = @ptrCast(base + off);
|
||||||
|
if (rec.length < @sizeOf(MadtRecordHeader)) break; // malformed; avoid a spin
|
||||||
|
switch (rec.type) {
|
||||||
|
0 => {
|
||||||
|
const la: *const MadtLocalApic = @ptrCast(base + off);
|
||||||
|
// bit 0 = enabled: skip processors the firmware marks unusable.
|
||||||
|
if (la.flags & 1 != 0) {
|
||||||
|
var nb: [24]u8 = undefined;
|
||||||
|
const nm = std.fmt.bufPrint(&nb, "cpu{d}", .{la.processor_id}) catch "cpu";
|
||||||
|
_ = try device_tree.addChild(device_tree.root, .processor, nm);
|
||||||
|
// Also record it as a schedulable core (with the APIC ID an AP
|
||||||
|
// wake needs, which the device node name doesn't preserve).
|
||||||
|
if (cpu_information.count < cpu_information.cpus.len) {
|
||||||
|
cpu_information.cpus[cpu_information.count] = .{
|
||||||
|
.processor_id = la.processor_id,
|
||||||
|
.apic_id = la.apic_id,
|
||||||
|
.online_capable = la.flags & 2 != 0,
|
||||||
|
};
|
||||||
|
cpu_information.count += 1;
|
||||||
|
} else {
|
||||||
|
cpu_information.dropped += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
1 => {
|
||||||
|
const io: *const MadtIoApic = @ptrCast(base + off);
|
||||||
|
var nb: [24]u8 = undefined;
|
||||||
|
const nm = std.fmt.bufPrint(&nb, "ioapic{d}", .{ioapic_index}) catch "ioapic";
|
||||||
|
ioapic_index += 1;
|
||||||
|
const d = try device_tree.addChild(device_tree.root, .interrupt_controller, nm);
|
||||||
|
_ = d.addResource(.memory, io.address, 0x20);
|
||||||
|
// The GSI range this I/O APIC handles, starting at gsi_base.
|
||||||
|
_ = d.addResource(.irq, io.gsi_base, 0);
|
||||||
|
},
|
||||||
|
2 => {
|
||||||
|
const iso: *const MadtIso = @ptrCast(base + off);
|
||||||
|
if (platform_information.override_count < platform_information.overrides.len) {
|
||||||
|
platform_information.overrides[platform_information.override_count] = .{
|
||||||
|
.source = iso.source,
|
||||||
|
.gsi = iso.gsi,
|
||||||
|
.flags = iso.flags,
|
||||||
|
};
|
||||||
|
platform_information.override_count += 1;
|
||||||
|
}
|
||||||
|
},
|
||||||
|
5 => {
|
||||||
|
const ovr: *const MadtLapicOverride = @ptrCast(base + off);
|
||||||
|
platform_information.lapic_base = ovr.address;
|
||||||
|
},
|
||||||
|
else => {},
|
||||||
|
}
|
||||||
|
off += rec.length;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// MCFG -> a pci_host_bridge per ECAM segment, then a PCI enumeration underneath.
|
||||||
|
fn parseMcfg(device_tree: *DeviceTree, header: *const SystemDescriptorTableHeader) !void {
|
||||||
|
const total: usize = header.length;
|
||||||
|
const base: [*]const u8 = @ptrCast(header);
|
||||||
|
|
||||||
|
var off: usize = @sizeOf(Mcfg);
|
||||||
|
while (off + @sizeOf(McfgAllocation) <= total) : (off += @sizeOf(McfgAllocation)) {
|
||||||
|
const alloc: *const McfgAllocation = @ptrCast(base + off);
|
||||||
|
const bus_count: u64 = @as(u64, alloc.end_bus - alloc.start_bus) + 1;
|
||||||
|
|
||||||
|
var nb: [24]u8 = undefined;
|
||||||
|
const nm = std.fmt.bufPrint(&nb, "pci{d}", .{alloc.segment_group}) catch "pci";
|
||||||
|
const bridge = try device_tree.addChild(device_tree.root, .pci_host_bridge, nm);
|
||||||
|
// ECAM window: 1 MiB of configuration space per bus.
|
||||||
|
_ = bridge.addResource(.memory, alloc.base_address, bus_count << 20);
|
||||||
|
_ = bridge.addResource(.bus_range, alloc.start_bus, bus_count);
|
||||||
|
addBridgeApertures(bridge);
|
||||||
|
// The bridge decodes the whole 16-bit I/O space toward its bus — the
|
||||||
|
// window functions' I/O BARs must register-contain within (M19.2).
|
||||||
|
_ = bridge.addResource(.io_port, 0, 1 << 16);
|
||||||
|
|
||||||
|
// The function walk itself retired to ring 3 (M19.3): the pci-bus
|
||||||
|
// driver claims this bridge, repeats the scan through its ECAM grant,
|
||||||
|
// and device_registers what it finds — the kernel seeds only the
|
||||||
|
// bridge. The scan's equivalence was proven before the hand-off
|
||||||
|
// (pci-scan), and the walk's history is in git if archaeology calls.
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The boot memory map, stored at discover() entry for the aperture derivation
|
||||||
|
/// below (and, in M20, for the acpi-tables node's containment windows).
|
||||||
|
var boot_memory_regions: []const boot_handoff.MemoryRegion = &.{};
|
||||||
|
|
||||||
|
/// The bridge's MMIO apertures, derived from the boot memory map's holes
|
||||||
|
/// (docs/discovery.md — apertures from the memory map): registered PCI functions carry BAR
|
||||||
|
/// resources, and `device_register` containment demands the bridge own windows
|
||||||
|
/// that cover them. Everything the firmware described is "not hole"; the low
|
||||||
|
/// aperture runs from the end of the described space below 4 GiB up to the
|
||||||
|
/// I/O-APIC region, the high one from 4 GiB (or the end of RAM above it) to
|
||||||
|
/// the 46-bit line. Coarse, mechanical, and AML-free — available at boot no
|
||||||
|
/// matter what later moved to user space.
|
||||||
|
fn addBridgeApertures(bridge: *device_model.Device) void {
|
||||||
|
// Below 4 GiB the described regions are sparse (RAM low, firmware flash
|
||||||
|
// and tables high), so the holes are the *gaps between* them — a single
|
||||||
|
// "after the last region" rule dies on OVMF's flash at the very top.
|
||||||
|
// Sort-merge the described ranges, then keep the three largest gaps
|
||||||
|
// (resource slots are bounded at 8 per device; ECAM + bus range + 3 + the
|
||||||
|
// high aperture fits). Above 4 GiB one aperture runs from the end of the
|
||||||
|
// described space to the 46-bit line.
|
||||||
|
const Range = struct { base: u64, end: u64 };
|
||||||
|
var below: [64]Range = undefined;
|
||||||
|
var below_count: usize = 0;
|
||||||
|
var high_end: u64 = 1 << 32;
|
||||||
|
for (boot_memory_regions) |region| {
|
||||||
|
const end = region.base + region.pages * 4096;
|
||||||
|
// Above 4 GiB only *usable RAM* blocks the aperture: OVMF describes
|
||||||
|
// its own 64-bit PCI window as a reserved region and then programs
|
||||||
|
// BARs inside it — honoring reserved there would exclude the very
|
||||||
|
// space BARs live in. Below 4 GiB every described region blocks (the
|
||||||
|
// kernel image, the tables, the ramdisk all live there). Bring-up
|
||||||
|
// trust: only the bridge's claimant can register into the aperture.
|
||||||
|
if (region.kind == .usable and end > high_end) high_end = end;
|
||||||
|
if (region.base >= (1 << 32) or below_count == below.len) continue;
|
||||||
|
below[below_count] = .{ .base = region.base, .end = @min(end, 1 << 32) };
|
||||||
|
below_count += 1;
|
||||||
|
}
|
||||||
|
// Insertion sort by base (the map is small and this runs once at boot).
|
||||||
|
for (1..below_count) |i| {
|
||||||
|
const key = below[i];
|
||||||
|
var j = i;
|
||||||
|
while (j > 0 and below[j - 1].base > key.base) : (j -= 1) below[j] = below[j - 1];
|
||||||
|
below[j] = key;
|
||||||
|
}
|
||||||
|
// Walk the sorted ranges, collecting inter-region gaps of at least 1 MiB.
|
||||||
|
var gaps: [3]Range = .{Range{ .base = 0, .end = 0 }} ** 3;
|
||||||
|
var cursor: u64 = 0;
|
||||||
|
var index: usize = 0;
|
||||||
|
while (index <= below_count) : (index += 1) {
|
||||||
|
const gap_end = if (index == below_count) (1 << 32) else below[index].base;
|
||||||
|
if (gap_end > cursor and gap_end - cursor >= (1 << 20)) {
|
||||||
|
// Keep the three largest, replacing the smallest kept so far.
|
||||||
|
var smallest: usize = 0;
|
||||||
|
for (gaps, 0..) |gap, gi| {
|
||||||
|
if (gap.end - gap.base < gaps[smallest].end - gaps[smallest].base) smallest = gi;
|
||||||
|
}
|
||||||
|
if (gap_end - cursor > gaps[smallest].end - gaps[smallest].base) {
|
||||||
|
gaps[smallest] = .{ .base = cursor, .end = gap_end };
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (index < below_count and below[index].end > cursor) cursor = below[index].end;
|
||||||
|
}
|
||||||
|
for (gaps) |gap| {
|
||||||
|
if (gap.end > gap.base) _ = bridge.addResource(.memory, gap.base, gap.end - gap.base);
|
||||||
|
}
|
||||||
|
_ = bridge.addResource(.memory, high_end, (@as(u64, 1) << 46) - high_end);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// HPET -> a timer node with its register block as an MMIO resource, plus the GSI
|
||||||
|
/// its comparators can raise.
|
||||||
|
///
|
||||||
|
/// Unlike a PCI device or an ACPI `_CRS` node, the HPET table carries **no interrupt
|
||||||
|
/// number**: which I/O APIC inputs a comparator may drive is advertised at runtime,
|
||||||
|
/// as a bitmask in `Tn_INT_ROUTE_CAP` (bits 63:32 of the Timer 0 configuration register).
|
||||||
|
/// So discovery maps the register block, reads the mask, and records one concrete
|
||||||
|
/// `irq` resource — the GSI a driver is entitled to bind. The driver commits to it
|
||||||
|
/// by writing `Tn_INT_ROUTE_CNF`; the kernel checks the binding against this
|
||||||
|
/// resource (see process.ownedGsi), which is what keeps `irq_bind` a capability
|
||||||
|
/// rather than a request for an arbitrary interrupt line.
|
||||||
|
fn parseHpet(device_tree: *DeviceTree, hal: Hal, header: *const SystemDescriptorTableHeader) !void {
|
||||||
|
const hpet: *const Hpet = @ptrCast(header);
|
||||||
|
const d = try device_tree.addChild(device_tree.root, .timer, "hpet");
|
||||||
|
|
||||||
|
// The GAS tag must say System Memory (0) before we treat `address` as a physical
|
||||||
|
// address. The HPET spec mandates it, but firmware is not a thing to trust: a
|
||||||
|
// System I/O (1) tag here would have us map an arbitrary page and read a bogus
|
||||||
|
// route-capability mask out of it.
|
||||||
|
if (hpet.address_space_id != gas_system_memory) return;
|
||||||
|
|
||||||
|
_ = d.addResource(.memory, hpet.address, 0x400);
|
||||||
|
|
||||||
|
const regs = hal.mapMmio(hpet.address, 0x400, true);
|
||||||
|
const t0_configuration: *const volatile u64 = @ptrFromInt(regs + 0x100);
|
||||||
|
const route_cap: u32 = @truncate(t0_configuration.* >> 32);
|
||||||
|
if (hpetGsi(route_cap)) |gsi| _ = d.addResource(.irq, gsi, 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// ACPI Generic Address Structure address-space ids we care about.
|
||||||
|
const gas_system_memory: u8 = 0;
|
||||||
|
|
||||||
|
/// Pick a GSI for the HPET out of its route-capability mask. Prefer an input at or
|
||||||
|
/// above 16: the low ones overlap the legacy ISA lines (2 = cascaded PIT, 8 = RTC),
|
||||||
|
/// which the MADT may separately override, whereas 16+ are the free upper inputs on
|
||||||
|
/// every I/O APIC we care about. Falls back to the lowest bit set if there are none.
|
||||||
|
fn hpetGsi(route_cap: u32) ?u32 {
|
||||||
|
if (route_cap == 0) return null;
|
||||||
|
var gsi: u32 = 16;
|
||||||
|
while (gsi < 32) : (gsi += 1) {
|
||||||
|
if (route_cap & (@as(u32, 1) << @intCast(gsi)) != 0) return gsi;
|
||||||
|
}
|
||||||
|
return @ctz(route_cap);
|
||||||
|
}
|
||||||
|
|
||||||
|
// FADT field offsets (bytes from the table start). The FADT grew across ACPI
|
||||||
|
// revisions, so every field is read through `fadt()` with a length guard rather
|
||||||
|
// than a fixed struct — an older/shorter FADT simply lacks the later (X_) fields.
|
||||||
|
const fadt_dsdt = 40; // u32
|
||||||
|
const fadt_smi_cmd = 48; // u32 (an I/O port)
|
||||||
|
const fadt_acpi_enable = 52; // u8
|
||||||
|
const fadt_acpi_disable = 53; // u8
|
||||||
|
const fadt_pm1a_cnt_blk = 64; // u32 (I/O port)
|
||||||
|
const fadt_pm1b_cnt_blk = 68; // u32 (I/O port)
|
||||||
|
const fadt_pm_tmr_blk = 76; // u32 (I/O port) — the PM timer counter
|
||||||
|
const fadt_pm1_cnt_len = 89; // u8 (bytes)
|
||||||
|
const fadt_sci_int = 46; // u16 (the SCI's GSI)
|
||||||
|
const fadt_flags = 112; // u32
|
||||||
|
const fadt_reset_register = 116; // GAS (12 bytes)
|
||||||
|
const fadt_reset_value = 128; // u8
|
||||||
|
const fadt_x_dsdt = 140; // u64
|
||||||
|
const fadt_x_pm1a_cnt_blk = 172; // GAS
|
||||||
|
const fadt_x_pm1b_cnt_blk = 184; // GAS
|
||||||
|
const fadt_x_pm_tmr_blk = 208; // GAS
|
||||||
|
const flag_reset_register_supported = 1 << 10;
|
||||||
|
const flag_tmr_value_ext = 1 << 8; // PM timer counter is 32-bit (else 24-bit)
|
||||||
|
|
||||||
|
/// FADT -> the power register map (into `power_information`) and the DSDT address, which
|
||||||
|
/// is queued for the AML sleep-state (`_Sx`) scan. No AML interpretation happens here.
|
||||||
|
fn parseFadt(header: *const SystemDescriptorTableHeader) void {
|
||||||
|
const base: [*]align(1) const u8 = @ptrCast(header);
|
||||||
|
const len: usize = header.length;
|
||||||
|
const pi = &power_information;
|
||||||
|
|
||||||
|
pi.sci_interrupt = @truncate(fadt(u16, base, len, fadt_sci_int) orelse 0);
|
||||||
|
pi.smi_cmd = @truncate(fadt(u32, base, len, fadt_smi_cmd) orelse 0);
|
||||||
|
pi.acpi_enable = fadt(u8, base, len, fadt_acpi_enable) orelse 0;
|
||||||
|
pi.acpi_disable = fadt(u8, base, len, fadt_acpi_disable) orelse 0;
|
||||||
|
|
||||||
|
const cnt_width = fadt(u8, base, len, fadt_pm1_cnt_len) orelse 2;
|
||||||
|
pi.pm1a_cnt = readCntRegister(base, len, fadt_x_pm1a_cnt_blk, fadt_pm1a_cnt_blk, cnt_width);
|
||||||
|
pi.pm1b_cnt = readCntRegister(base, len, fadt_x_pm1b_cnt_blk, fadt_pm1b_cnt_blk, cnt_width);
|
||||||
|
|
||||||
|
const flags = fadt(u32, base, len, fadt_flags) orelse 0;
|
||||||
|
pi.reset_supported = flags & flag_reset_register_supported != 0;
|
||||||
|
pi.reset = readGas(base, len, fadt_reset_register) orelse .{};
|
||||||
|
pi.reset_value = fadt(u8, base, len, fadt_reset_value) orelse 0;
|
||||||
|
|
||||||
|
// The PM timer — a fixed-rate counter used as a calibration reference when no
|
||||||
|
// HPET is present. Prefer the 64-bit-capable X_ GAS, fall back to the port.
|
||||||
|
platform_information.pm_timer = readCntRegister(base, len, fadt_x_pm_tmr_blk, fadt_pm_tmr_blk, 4);
|
||||||
|
platform_information.pm_timer_32bit = flags & flag_tmr_value_ext != 0;
|
||||||
|
|
||||||
|
var dsdt: u64 = fadt(u32, base, len, fadt_dsdt) orelse 0;
|
||||||
|
if (fadt(u64, base, len, fadt_x_dsdt)) |x| {
|
||||||
|
if (x != 0) dsdt = x;
|
||||||
|
}
|
||||||
|
dsdt_physical = dsdt;
|
||||||
|
addAmlBlock(dsdt);
|
||||||
|
}
|
||||||
|
|
||||||
|
// SPCR field offsets (bytes from the table start).
|
||||||
|
const spcr_interface_type = 36; // u8
|
||||||
|
const spcr_base_address = 40; // GAS (12 bytes)
|
||||||
|
|
||||||
|
/// SPCR -> the console UART's address + interface type, so serial can target the
|
||||||
|
/// firmware's actual debug port instead of assuming legacy COM1.
|
||||||
|
fn parseSpcr(header: *const SystemDescriptorTableHeader) void {
|
||||||
|
const base: [*]align(1) const u8 = @ptrCast(header);
|
||||||
|
const len: usize = header.length;
|
||||||
|
const gas = readGas(base, len, spcr_base_address) orelse return;
|
||||||
|
if (gas.address == 0) return;
|
||||||
|
platform_information.spcr_uart = gas;
|
||||||
|
platform_information.spcr_kind = fadt(u8, base, len, spcr_interface_type) orelse 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
// DMAR remapping-structure layout (Intel VT-d spec §8): the DMAR-specific header is 12
|
||||||
|
// bytes (host-address-width, flags, 10 reserved), then a list of {type u16, length u16}
|
||||||
|
// structures. Type 0 is a DRHD (DMA Remapping Hardware Unit Definition), whose 64-bit
|
||||||
|
// register base sits at offset 8 within it.
|
||||||
|
const dmar_structures_offset = 48; // 36-byte ACPI header + 12-byte DMAR header
|
||||||
|
const dmar_type_drhd: u16 = 0;
|
||||||
|
const drhd_register_base_offset = 8;
|
||||||
|
|
||||||
|
/// DMAR -> detect the IOMMU. Find the first DMA-remapping hardware unit, map its
|
||||||
|
/// register block, and record its version and capabilities. This is *detection only*:
|
||||||
|
/// it tells the system an IOMMU exists (so `device_claim` on a DMA device could one day
|
||||||
|
/// be gated by a per-device translation domain), but no domains are programmed yet —
|
||||||
|
/// enforcement is built with the first DMA driver, which is what there is to protect and
|
||||||
|
/// test against. See docs/driver-model.md (M16), the honest caveat.
|
||||||
|
fn parseDmar(hal: Hal, header: *const SystemDescriptorTableHeader) void {
|
||||||
|
const base: [*]align(1) const u8 = @ptrCast(header);
|
||||||
|
const total: usize = header.length;
|
||||||
|
|
||||||
|
var off: usize = dmar_structures_offset;
|
||||||
|
while (off + 4 <= total) {
|
||||||
|
const kind = fadt(u16, base, total, off) orelse break;
|
||||||
|
const length = fadt(u16, base, total, off + 2) orelse break;
|
||||||
|
if (length < 4 or off + length > total) break; // malformed; stop rather than loop
|
||||||
|
if (kind == dmar_type_drhd) {
|
||||||
|
const register_base = fadt(u64, base, total, off + drhd_register_base_offset) orelse 0;
|
||||||
|
if (register_base != 0) {
|
||||||
|
const regs = hal.mapMmio(register_base, abi.page_size, true);
|
||||||
|
platform_information.iommu_present = true;
|
||||||
|
platform_information.iommu_base = register_base;
|
||||||
|
platform_information.iommu_version = @as(*const volatile u32, @ptrFromInt(regs + 0x00)).*;
|
||||||
|
platform_information.iommu_capabilities = @as(*const volatile u64, @ptrFromInt(regs + 0x08)).*;
|
||||||
|
return; // first unit is enough for detection; multi-unit is future
|
||||||
|
}
|
||||||
|
}
|
||||||
|
off += length;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- helpers ----------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Sum `len` bytes; an ACPI table/pointer is valid when the low 8 bits are zero.
|
||||||
|
fn checksumOk(bytes: [*]const u8, len: usize) bool {
|
||||||
|
var sum: u8 = 0;
|
||||||
|
for (0..len) |i| sum +%= bytes[i];
|
||||||
|
return sum == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read a FADT field of type `T` at `off`, or null if the table is too short to
|
||||||
|
/// contain it (a legal state for older FADT revisions).
|
||||||
|
fn fadt(comptime T: type, base: [*]align(1) const u8, len: usize, off: usize) ?T {
|
||||||
|
if (off + @sizeOf(T) > len) return null;
|
||||||
|
return rd(T, base, off);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode a Generic Address Structure at `off` into a `RegisterAccess`. GAS layout:
|
||||||
|
/// address_space(u8), bit_width(u8), bit_offset(u8), access_size(u8), address(u64).
|
||||||
|
fn readGas(base: [*]align(1) const u8, len: usize, off: usize) ?RegisterAccess {
|
||||||
|
if (off + 12 > len) return null;
|
||||||
|
const address_space = rd(u8, base, off);
|
||||||
|
const bit_width = rd(u8, base, off + 1);
|
||||||
|
const address = rd(u64, base, off + 4);
|
||||||
|
return .{
|
||||||
|
.mmio = address_space == 0, // 0 = system memory, 1 = system I/O
|
||||||
|
.address = address,
|
||||||
|
.width = bit_width / 8,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A PM1 control register: prefer the 64-bit-capable X_ GAS form; fall back to the
|
||||||
|
/// legacy 32-bit I/O-port field. Width comes from PM1_CNT_LEN either way.
|
||||||
|
fn readCntRegister(base: [*]align(1) const u8, len: usize, xoff: usize, legacy_off: usize, width: u8) RegisterAccess {
|
||||||
|
if (readGas(base, len, xoff)) |g| {
|
||||||
|
if (g.address != 0) return .{ .mmio = g.mmio, .address = g.address, .width = width };
|
||||||
|
}
|
||||||
|
const port = fadt(u32, base, len, legacy_off) orelse 0;
|
||||||
|
return .{ .mmio = false, .address = port, .width = width };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read a little-endian integer at `off` from a (possibly unaligned) byte pointer.
|
||||||
|
/// x86 is little-endian and native, so an unaligned load suffices.
|
||||||
|
fn rd(comptime T: type, bytes: [*]align(1) const u8, off: usize) T {
|
||||||
|
const p: *align(1) const T = @ptrCast(bytes + off);
|
||||||
|
return p.*;
|
||||||
|
}
|
||||||
@@ -0,0 +1,260 @@
|
|||||||
|
//! AML (ACPI Machine Language) — the bytecode in the DSDT and SSDTs that describes
|
||||||
|
//! the parts of the machine the static tables don't.
|
||||||
|
//!
|
||||||
|
//! This module has two stages. `parser.zig` walks the entire byte stream and
|
||||||
|
//! records every named object into a namespace tree (`namespace.zig`), capturing
|
||||||
|
//! method bodies and field/region layout. `interpreter.zig` then *evaluates* control
|
||||||
|
//! methods on demand — running operators, control flow, and OperationRegion field
|
||||||
|
//! access — so callers can resolve device status (`_STA`), current resource
|
||||||
|
//! settings (`_CRS`), sleep states (`_Sx`), and the like against the live namespace.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const opcode = @import("opcodes.zig");
|
||||||
|
const parser = @import("parser.zig");
|
||||||
|
|
||||||
|
/// The named AML opcode/prefix bytes (`zero_opcode`, `byte_prefix`, …). Re-exported so
|
||||||
|
/// callers that decode raw AML bytes — e.g. the acpi service reading a `_HID` integer —
|
||||||
|
/// name the opcodes instead of writing bare 0x0A/0x0B/… literals (docs/coding-standards.md).
|
||||||
|
pub const opcodes = @import("opcodes.zig");
|
||||||
|
|
||||||
|
pub const Namespace = @import("namespace.zig").Namespace;
|
||||||
|
pub const Node = @import("namespace.zig").Node;
|
||||||
|
pub const NodeKind = @import("namespace.zig").NodeKind;
|
||||||
|
|
||||||
|
/// The AML evaluator: interprets control methods (and reads Names/Fields) far
|
||||||
|
/// enough for device discovery. See `interpreter.zig`.
|
||||||
|
pub const Interpreter = @import("interpreter.zig").Interpreter;
|
||||||
|
pub const Object = @import("interpreter.zig").Object;
|
||||||
|
pub const EvaluateHal = @import("interpreter.zig").Hal;
|
||||||
|
|
||||||
|
/// The SLP_TYP values written to PM1a/PM1b control to enter a sleep state.
|
||||||
|
pub const SleepType = struct {
|
||||||
|
slp_typ_a: u8,
|
||||||
|
slp_typ_b: u8,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const ParseResult = struct {
|
||||||
|
namespace: Namespace,
|
||||||
|
/// Bytes the parser consumed across all blocks...
|
||||||
|
consumed: usize,
|
||||||
|
/// ...out of this many. A clean full traversal has `consumed == total`.
|
||||||
|
total: usize,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Parse the given AML blocks (DSDT first, then SSDTs) into one namespace. Later
|
||||||
|
/// blocks extend the namespace built by earlier ones, exactly as ACPI intends.
|
||||||
|
pub fn parse(allocator: std.mem.Allocator, blocks: []const []const u8) !ParseResult {
|
||||||
|
var namespace = try Namespace.init(allocator);
|
||||||
|
var consumed: usize = 0;
|
||||||
|
var total: usize = 0;
|
||||||
|
for (blocks) |block| {
|
||||||
|
var p = parser.Parser.init(block, &namespace);
|
||||||
|
consumed += p.parseAll();
|
||||||
|
total += block.len;
|
||||||
|
}
|
||||||
|
return .{ .namespace = namespace, .consumed = consumed, .total = total };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Count the Device objects in a parsed namespace — what the acpi service
|
||||||
|
/// (docs/discovery.md) reports, and what the kernel's own parse counts
|
||||||
|
/// so the two can be checked equal across the ring-3 move.
|
||||||
|
pub fn deviceCount(namespace: *const Namespace) usize {
|
||||||
|
return countKind(namespace.root, .device);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn countKind(node: *const Node, kind: NodeKind) usize {
|
||||||
|
var n: usize = if (node.kind == kind) 1 else 0;
|
||||||
|
var c = node.first_child;
|
||||||
|
while (c) |child| : (c = child.next_sibling) n += countKind(child, kind);
|
||||||
|
return n;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Look up the `\_S{state}` sleep package in a parsed namespace and return its
|
||||||
|
/// first two integer elements (SLP_TYP for PM1a / PM1b), or null if absent.
|
||||||
|
pub fn sleepState(namespace: *Namespace, state: u8) ?SleepType {
|
||||||
|
const segment = [4]u8{ '_', 'S', '0' + state, '_' };
|
||||||
|
const node = namespace.resolve(namespace.root, false, 0, &.{segment}) orelse return null;
|
||||||
|
if (node.kind != .name) return null;
|
||||||
|
return parseSleepPackage(node.value);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode a `Package(){ SLP_TYPa, SLP_TYPb, ... }` from the raw AML of a Name's
|
||||||
|
/// value. Returns the first two elements as bytes (missing elements default to 0).
|
||||||
|
fn parseSleepPackage(value: []const u8) ?SleepType {
|
||||||
|
if (value.len == 0 or value[0] != opcode.package_opcode) return null;
|
||||||
|
var p: usize = 1;
|
||||||
|
p += packageLengthSize(value, p) orelse return null;
|
||||||
|
if (p >= value.len) return null;
|
||||||
|
const number_elements = value[p];
|
||||||
|
p += 1;
|
||||||
|
|
||||||
|
const a: u8 = if (number_elements >= 1) @truncate(readInteger(value, &p) orelse 0) else 0;
|
||||||
|
const b: u8 = if (number_elements >= 2) @truncate(readInteger(value, &p) orelse 0) else 0;
|
||||||
|
return .{ .slp_typ_a = a, .slp_typ_b = b };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Bytes a PkgLength field occupies at `p` (we only need to step over it here).
|
||||||
|
fn packageLengthSize(bytes: []const u8, p: usize) ?usize {
|
||||||
|
if (p >= bytes.len) return null;
|
||||||
|
const follow: usize = bytes[p] >> 6;
|
||||||
|
if (p + 1 + follow > bytes.len) return null;
|
||||||
|
return 1 + follow;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read one AML integer data object at `p`, advancing `p`.
|
||||||
|
fn readInteger(bytes: []const u8, p: *usize) ?u64 {
|
||||||
|
if (p.* >= bytes.len) return null;
|
||||||
|
const opcode_byte = bytes[p.*];
|
||||||
|
p.* += 1;
|
||||||
|
return switch (opcode_byte) {
|
||||||
|
opcode.zero_opcode => 0,
|
||||||
|
opcode.one_opcode => 1,
|
||||||
|
opcode.ones_opcode => 0xFF,
|
||||||
|
opcode.byte_prefix => readLittle(bytes, p, 1),
|
||||||
|
opcode.word_prefix => readLittle(bytes, p, 2),
|
||||||
|
opcode.dword_prefix => readLittle(bytes, p, 4),
|
||||||
|
opcode.qword_prefix => readLittle(bytes, p, 8),
|
||||||
|
else => null,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
fn readLittle(bytes: []const u8, p: *usize, n: usize) ?u64 {
|
||||||
|
if (p.* + n > bytes.len) return null;
|
||||||
|
var v: u64 = 0;
|
||||||
|
var k: usize = 0;
|
||||||
|
while (k < n) : (k += 1) v |= @as(u64, bytes[p.* + k]) << @intCast(k * 8);
|
||||||
|
p.* += n;
|
||||||
|
return v;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- tests ------------------------------------------------------------------
|
||||||
|
|
||||||
|
test "parses a nested namespace and finds the sleep package" {
|
||||||
|
// A hand-assembled AML blob (all PkgLengths computed to be single-byte):
|
||||||
|
// Name(_S5, Package(2){0x05, 0x00})
|
||||||
|
// Scope(\_SB) { Device(PCI0) {
|
||||||
|
// Name(_HID, 0x11)
|
||||||
|
// Method(MTHD, 1) {}
|
||||||
|
// Method(CALL, 0) { MTHD(Zero) } // invocation of a 1-arg method
|
||||||
|
// } }
|
||||||
|
// OperationRegion(DBG0, SystemIO, 0x0402, 1)
|
||||||
|
// Field(DBG0, ...) { DBGB, 8 }
|
||||||
|
const blob = [_]u8{
|
||||||
|
// Name(_S5, Package(2){Byte 0x05, Byte 0x00})
|
||||||
|
0x08, 0x5F, 0x53, 0x35, 0x5F, 0x12, 0x06, 0x02, 0x0A, 0x05, 0x0A, 0x00,
|
||||||
|
// Scope(\_SB) packagelen=0x27
|
||||||
|
0x10, 0x27, 0x5C, 0x5F, 0x53, 0x42, 0x5F,
|
||||||
|
// Device(PCI0) packagelen=0x1F
|
||||||
|
0x5B, 0x82, 0x1F, 0x50, 0x43,
|
||||||
|
0x49, 0x30,
|
||||||
|
// Name(_HID, 0x11)
|
||||||
|
0x08, 0x5F, 0x48, 0x49, 0x44, 0x0A, 0x11,
|
||||||
|
// Method(MTHD, flags=1) empty, packagelen=0x06
|
||||||
|
0x14, 0x06, 0x4D,
|
||||||
|
0x54, 0x48, 0x44, 0x01,
|
||||||
|
// Method(CALL, flags=0) { MTHD(Zero) }, packagelen=0x0B
|
||||||
|
0x14, 0x0B, 0x43, 0x41, 0x4C, 0x4C, 0x00, 0x4D,
|
||||||
|
0x54, 0x48, 0x44, 0x00,
|
||||||
|
// OperationRegion(DBG0, SystemIO, Word 0x0402, Byte 1)
|
||||||
|
0x5B, 0x80, 0x44, 0x42, 0x47, 0x30, 0x01, 0x0B,
|
||||||
|
0x02, 0x04, 0x0A, 0x01,
|
||||||
|
// Field(DBG0, flags=1) { DBGB, 8 }, packagelen=0x0B
|
||||||
|
0x5B, 0x81, 0x0B, 0x44, 0x42, 0x47, 0x30, 0x01,
|
||||||
|
0x44, 0x42, 0x47, 0x42, 0x08,
|
||||||
|
};
|
||||||
|
|
||||||
|
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
||||||
|
defer arena.deinit();
|
||||||
|
var result = try parse(arena.allocator(), &.{&blob});
|
||||||
|
|
||||||
|
// Integrity: the parser consumed exactly the whole blob (no desync).
|
||||||
|
try std.testing.expectEqual(blob.len, result.consumed);
|
||||||
|
try std.testing.expectEqual(blob.len, result.total);
|
||||||
|
|
||||||
|
const namespace = &result.namespace;
|
||||||
|
|
||||||
|
// Expected top-level nodes.
|
||||||
|
const sb = namespace.resolve(namespace.root, false, 0, &.{.{ '_', 'S', 'B', '_' }}) orelse return error.NoSB;
|
||||||
|
try std.testing.expectEqual(NodeKind.scope, sb.kind);
|
||||||
|
const pci0 = namespace.resolve(sb, false, 0, &.{.{ 'P', 'C', 'I', '0' }}) orelse return error.NoPCI0;
|
||||||
|
try std.testing.expectEqual(NodeKind.device, pci0.kind);
|
||||||
|
_ = namespace.resolve(pci0, false, 0, &.{.{ '_', 'H', 'I', 'D' }}) orelse return error.NoHID;
|
||||||
|
|
||||||
|
// The 1-arg method's arg count was parsed from its flags byte.
|
||||||
|
const mthd = namespace.resolve(pci0, false, 0, &.{.{ 'M', 'T', 'H', 'D' }}) orelse return error.NoMTHD;
|
||||||
|
try std.testing.expectEqual(NodeKind.method, mthd.kind);
|
||||||
|
try std.testing.expectEqual(@as(u8, 1), mthd.arg_count);
|
||||||
|
|
||||||
|
// OperationRegion and the Field unit made it into the namespace.
|
||||||
|
_ = namespace.resolve(namespace.root, false, 0, &.{.{ 'D', 'B', 'G', '0' }}) orelse return error.NoRegion;
|
||||||
|
_ = namespace.resolve(namespace.root, false, 0, &.{.{ 'D', 'B', 'G', 'B' }}) orelse return error.NoField;
|
||||||
|
|
||||||
|
// The sleep package decoded.
|
||||||
|
const s5 = sleepState(namespace, 5) orelse return error.NoS5;
|
||||||
|
try std.testing.expectEqual(@as(u8, 5), s5.slp_typ_a);
|
||||||
|
try std.testing.expectEqual(@as(u8, 0), s5.slp_typ_b);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn noMap(physical: u64, _: u64, _: bool) u64 {
|
||||||
|
return physical;
|
||||||
|
}
|
||||||
|
fn noRead(_: u8, _: u16) u32 {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
fn noWrite(_: u8, _: u16, _: u32) void {}
|
||||||
|
|
||||||
|
test "interpreter runs a method with args, arithmetic, and control flow" {
|
||||||
|
// Method(TST_, 1) {
|
||||||
|
// Store(Arg0, Local0); Add(Local0, 5, Local0)
|
||||||
|
// If (LGreater(Local0, 10)) { Return(One) }
|
||||||
|
// Return(Zero)
|
||||||
|
// }
|
||||||
|
const blob = [_]u8{
|
||||||
|
0x14, 0x18, 0x54, 0x53, 0x54, 0x5F, 0x01, // Method TST_, 1 arg
|
||||||
|
0x70, 0x68, 0x60, // Store(Arg0, Local0)
|
||||||
|
0x72, 0x60, 0x0A, 0x05, 0x60, // Add(Local0, 5, Local0)
|
||||||
|
0xA0, 0x07, 0x94, 0x60, 0x0A, 0x0A, 0xA4, 0x01, // If(LGreater(Local0,10)) { Return(One) }
|
||||||
|
0xA4, 0x00, // Return(Zero)
|
||||||
|
};
|
||||||
|
|
||||||
|
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
||||||
|
defer arena.deinit();
|
||||||
|
var result = try parse(arena.allocator(), &.{&blob});
|
||||||
|
const namespace = &result.namespace;
|
||||||
|
const tst = namespace.resolve(namespace.root, false, 0, &.{.{ 'T', 'S', 'T', '_' }}) orelse return error.NoMethod;
|
||||||
|
|
||||||
|
var interpreter = Interpreter.init(namespace, .{ .mapMmio = noMap, .pioRead = noRead, .pioWrite = noWrite }, arena.allocator());
|
||||||
|
|
||||||
|
const hi = try interpreter.evaluate(tst, &.{.{ .integer = 7 }}); // 7+5=12 > 10 -> 1
|
||||||
|
try std.testing.expectEqual(@as(u64, 1), try hi.asInteger());
|
||||||
|
const lo = try interpreter.evaluate(tst, &.{.{ .integer = 2 }}); // 2+5=7 !> 10 -> 0
|
||||||
|
try std.testing.expectEqual(@as(u64, 0), try lo.asInteger());
|
||||||
|
}
|
||||||
|
|
||||||
|
test "interpreter records Notify(device, code)" {
|
||||||
|
// Device(DEV_) { Name(_HID, 0x030AD041) } // PNP0A03-ish placeholder
|
||||||
|
// Method(TST_, 0) { Notify(DEV_, 0x80); Return(Zero) }
|
||||||
|
// Encoded: a Device holding a Name, then a Method issuing Notify on it.
|
||||||
|
const blob = [_]u8{
|
||||||
|
0x5B, 0x82, 0x0F, 0x44, 0x45, 0x56, 0x5F, // Device(DEV_) len=0x0F (pkglen + DEV_ + Name)
|
||||||
|
0x08, 0x5F, 0x48, 0x49, 0x44, 0x0C, 0x41, 0xD0, 0x0A, 0x03, // Name(_HID, DWord 0x030AD041)
|
||||||
|
0x14, 0x0F, 0x54, 0x53, 0x54, 0x5F, 0x00, // Method(TST_, 0) len=0x0F (pkglen + TST_ + flags + body)
|
||||||
|
0x86, 0x44, 0x45, 0x56, 0x5F, 0x0A, 0x80, // Notify(DEV_, 0x80)
|
||||||
|
0xA4, 0x00, // Return(Zero)
|
||||||
|
};
|
||||||
|
|
||||||
|
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
||||||
|
defer arena.deinit();
|
||||||
|
var result = try parse(arena.allocator(), &.{&blob});
|
||||||
|
const namespace = &result.namespace;
|
||||||
|
const tst = namespace.resolve(namespace.root, false, 0, &.{.{ 'T', 'S', 'T', '_' }}) orelse return error.NoMethod;
|
||||||
|
const dev = namespace.resolve(namespace.root, false, 0, &.{.{ 'D', 'E', 'V', '_' }}) orelse return error.NoDevice;
|
||||||
|
|
||||||
|
var interpreter = Interpreter.init(namespace, .{ .mapMmio = noMap, .pioRead = noRead, .pioWrite = noWrite }, arena.allocator());
|
||||||
|
_ = try interpreter.evaluate(tst, &.{});
|
||||||
|
|
||||||
|
const events = interpreter.takeNotifications();
|
||||||
|
try std.testing.expectEqual(@as(usize, 1), events.len);
|
||||||
|
try std.testing.expectEqual(dev, events[0].node);
|
||||||
|
try std.testing.expectEqual(@as(u64, 0x80), events[0].code);
|
||||||
|
}
|
||||||
@@ -0,0 +1,777 @@
|
|||||||
|
//! A tree-walking AML interpreter — the evaluation stage on top of the parser's
|
||||||
|
//! structural namespace. It executes control methods (their bodies captured by
|
||||||
|
//! the parser) far enough to serve device discovery: device status (`_STA`, is a
|
||||||
|
//! device present), current resource settings (`_CRS`), and the operators, control
|
||||||
|
//! flow, locals/args, and
|
||||||
|
//! OperationRegion field access those methods reach for.
|
||||||
|
//!
|
||||||
|
//! Scope: integers, buffers, strings, packages, and references; If/Else/While/
|
||||||
|
//! Return; the arithmetic/logic operators; method invocation; Name/Local/Arg
|
||||||
|
//! access; CreateField buffer patching (the common current-resource-settings
|
||||||
|
//! (`_CRS`) idiom); and field
|
||||||
|
//! reads/writes against SystemMemory and SystemIO regions. Opcodes outside this
|
||||||
|
//! set return `error.Unsupported`, which callers treat as "couldn't evaluate" and
|
||||||
|
//! fall back — never a hard failure.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const opcode = @import("opcodes.zig");
|
||||||
|
const Node = @import("namespace.zig").Node;
|
||||||
|
const Namespace = @import("namespace.zig").Namespace;
|
||||||
|
|
||||||
|
/// Injected hardware access for OperationRegion reads/writes (the architecture VMM + pio).
|
||||||
|
pub const Hal = struct {
|
||||||
|
mapMmio: *const fn (physical: u64, len: u64, writable: bool) u64,
|
||||||
|
pioRead: *const fn (width: u8, port: u16) u32,
|
||||||
|
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const Error = error{ Unsupported, Truncated, DivByZero } || std.mem.Allocator.Error;
|
||||||
|
|
||||||
|
/// A runtime AML value.
|
||||||
|
pub const Object = union(enum) {
|
||||||
|
uninitialized,
|
||||||
|
integer: u64,
|
||||||
|
buffer: []u8,
|
||||||
|
string: []u8,
|
||||||
|
package: []Object,
|
||||||
|
reference: *Node,
|
||||||
|
|
||||||
|
pub fn asInteger(self: Object) Error!u64 {
|
||||||
|
return switch (self) {
|
||||||
|
.integer => |v| v,
|
||||||
|
.buffer => |b| blk: {
|
||||||
|
var v: u64 = 0;
|
||||||
|
for (b, 0..) |byte, i| {
|
||||||
|
if (i >= 8) break;
|
||||||
|
v |= @as(u64, byte) << @intCast(i * 8);
|
||||||
|
}
|
||||||
|
break :blk v;
|
||||||
|
},
|
||||||
|
else => error.Unsupported,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
const maximum_segments = 16;
|
||||||
|
const NamePath = struct {
|
||||||
|
rooted: bool = false,
|
||||||
|
parents: u8 = 0,
|
||||||
|
segments: [maximum_segments][4]u8 = undefined,
|
||||||
|
count: usize = 0,
|
||||||
|
fn slice(self: *const NamePath) []const [4]u8 {
|
||||||
|
return self.segments[0..self.count];
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
const Cursor = struct {
|
||||||
|
b: []const u8,
|
||||||
|
i: usize = 0,
|
||||||
|
|
||||||
|
fn eof(self: *Cursor) bool {
|
||||||
|
return self.i >= self.b.len;
|
||||||
|
}
|
||||||
|
fn peek(self: *Cursor) ?u8 {
|
||||||
|
return if (self.eof()) null else self.b[self.i];
|
||||||
|
}
|
||||||
|
fn byte(self: *Cursor) Error!u8 {
|
||||||
|
if (self.eof()) return error.Truncated;
|
||||||
|
const v = self.b[self.i];
|
||||||
|
self.i += 1;
|
||||||
|
return v;
|
||||||
|
}
|
||||||
|
fn take(self: *Cursor, n: usize) Error![]const u8 {
|
||||||
|
if (self.i + n > self.b.len) return error.Truncated;
|
||||||
|
const s = self.b[self.i .. self.i + n];
|
||||||
|
self.i += n;
|
||||||
|
return s;
|
||||||
|
}
|
||||||
|
fn packageLength(self: *Cursor) Error!usize {
|
||||||
|
const lead = try self.byte();
|
||||||
|
const follow: usize = lead >> 6;
|
||||||
|
if (follow == 0) return lead & 0x3F;
|
||||||
|
var value: usize = lead & 0x0F;
|
||||||
|
var k: usize = 0;
|
||||||
|
while (k < follow) : (k += 1) value |= @as(usize, try self.byte()) << @intCast(4 + k * 8);
|
||||||
|
return value;
|
||||||
|
}
|
||||||
|
fn nameString(self: *Cursor) Error!NamePath {
|
||||||
|
var name_path = NamePath{};
|
||||||
|
if (self.peek() == opcode.root_char) {
|
||||||
|
name_path.rooted = true;
|
||||||
|
self.i += 1;
|
||||||
|
} else {
|
||||||
|
while (self.peek() == opcode.parent_prefix_char) : (self.i += 1) name_path.parents += 1;
|
||||||
|
}
|
||||||
|
const lead = self.peek() orelse return name_path;
|
||||||
|
switch (lead) {
|
||||||
|
0x00 => self.i += 1,
|
||||||
|
opcode.dual_name_prefix => {
|
||||||
|
self.i += 1;
|
||||||
|
try self.segment(&name_path);
|
||||||
|
try self.segment(&name_path);
|
||||||
|
},
|
||||||
|
opcode.multi_name_prefix => {
|
||||||
|
self.i += 1;
|
||||||
|
const count = try self.byte();
|
||||||
|
var k: usize = 0;
|
||||||
|
while (k < count) : (k += 1) try self.segment(&name_path);
|
||||||
|
},
|
||||||
|
else => try self.segment(&name_path),
|
||||||
|
}
|
||||||
|
return name_path;
|
||||||
|
}
|
||||||
|
fn segment(self: *Cursor, name_path: *NamePath) Error!void {
|
||||||
|
const s = try self.take(4);
|
||||||
|
if (name_path.count < maximum_segments) {
|
||||||
|
name_path.segments[name_path.count] = s[0..4].*;
|
||||||
|
name_path.count += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
const Frame = struct {
|
||||||
|
args: [7]Object = .{.uninitialized} ** 7,
|
||||||
|
locals: [8]Object = .{.uninitialized} ** 8,
|
||||||
|
scope: *Node,
|
||||||
|
ret: Object = .uninitialized,
|
||||||
|
returned: bool = false,
|
||||||
|
broke: bool = false,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A CreateField binding: a name that indexes into a buffer object.
|
||||||
|
const BufferField = struct { buffer: *Node, byte_off: usize, bit_width: u32 };
|
||||||
|
|
||||||
|
/// One Notify(device, code) the interpreter executed.
|
||||||
|
pub const NotifyEvent = struct { node: *Node, code: u64 };
|
||||||
|
|
||||||
|
pub const Interpreter = struct {
|
||||||
|
namespace: *Namespace,
|
||||||
|
hal: Hal,
|
||||||
|
arena: std.mem.Allocator,
|
||||||
|
/// Runtime object overrides for Name nodes (Store targets, patched buffers).
|
||||||
|
dynamic_overrides: std.AutoHashMapUnmanaged(*Node, Object) = .{},
|
||||||
|
/// CreateField bindings active for the current evaluation.
|
||||||
|
fields: std.AutoHashMapUnmanaged(*Node, BufferField) = .{},
|
||||||
|
/// Notify(device, code) operations the last evaluation executed — a GPE or
|
||||||
|
/// EC handler tells the OS "look at this device" this way. Bounded; the
|
||||||
|
/// caller drains it with `takeNotifications` after `evaluate` (M21).
|
||||||
|
notify_queue: [16]NotifyEvent = undefined,
|
||||||
|
notify_count: usize = 0,
|
||||||
|
|
||||||
|
pub fn init(namespace: *Namespace, hal: Hal, arena: std.mem.Allocator) Interpreter {
|
||||||
|
return .{ .namespace = namespace, .hal = hal, .arena = arena };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Evaluate a namespace object: invoke a Method, read a Name's value, or read a
|
||||||
|
/// Field. Resets per-evaluation runtime state first.
|
||||||
|
pub fn evaluate(self: *Interpreter, node: *Node, args: []const Object) Error!Object {
|
||||||
|
self.notify_count = 0;
|
||||||
|
self.dynamic_overrides.clearRetainingCapacity();
|
||||||
|
self.fields.clearRetainingCapacity();
|
||||||
|
return self.invoke(node, args);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn invoke(self: *Interpreter, node: *Node, args: []const Object) Error!Object {
|
||||||
|
switch (node.kind) {
|
||||||
|
.method => {
|
||||||
|
var frame = Frame{ .scope = node };
|
||||||
|
for (args, 0..) |a, i| {
|
||||||
|
if (i < frame.args.len) frame.args[i] = a;
|
||||||
|
}
|
||||||
|
var current = Cursor{ .b = node.value };
|
||||||
|
try self.executeList(¤t, &frame);
|
||||||
|
return frame.ret;
|
||||||
|
},
|
||||||
|
.name => {
|
||||||
|
if (self.dynamic_overrides.get(node)) |o| return o;
|
||||||
|
var current = Cursor{ .b = node.value };
|
||||||
|
var frame = Frame{ .scope = node.parent orelse self.namespace.root };
|
||||||
|
return self.term(¤t, &frame);
|
||||||
|
},
|
||||||
|
.field => return .{ .integer = try self.readField(node) },
|
||||||
|
else => return .{ .reference = node },
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Execute a TermList until it ends or the frame returns/breaks.
|
||||||
|
fn executeList(self: *Interpreter, current: *Cursor, frame: *Frame) Error!void {
|
||||||
|
while (!current.eof() and !frame.returned and !frame.broke) {
|
||||||
|
_ = try self.term(current, frame);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Evaluate/execute one term, returning its value (`.uninitialized` for pure
|
||||||
|
/// statements).
|
||||||
|
fn term(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const lead = current.peek() orelse return error.Truncated;
|
||||||
|
if (isNameStart(lead)) return self.nameReference(current, frame);
|
||||||
|
_ = try current.byte();
|
||||||
|
|
||||||
|
return switch (lead) {
|
||||||
|
opcode.zero_opcode => Object{ .integer = 0 },
|
||||||
|
opcode.one_opcode => Object{ .integer = 1 },
|
||||||
|
opcode.ones_opcode => Object{ .integer = ~@as(u64, 0) },
|
||||||
|
opcode.byte_prefix => Object{ .integer = try self.readConstant(current, 1) },
|
||||||
|
opcode.word_prefix => Object{ .integer = try self.readConstant(current, 2) },
|
||||||
|
opcode.dword_prefix => Object{ .integer = try self.readConstant(current, 4) },
|
||||||
|
opcode.qword_prefix => Object{ .integer = try self.readConstant(current, 8) },
|
||||||
|
opcode.string_prefix => try self.readString(current),
|
||||||
|
opcode.buffer_opcode => try self.buffer(current, frame),
|
||||||
|
opcode.package_opcode, opcode.var_package_opcode => try self.package(current, frame, lead == opcode.var_package_opcode),
|
||||||
|
|
||||||
|
opcode.local0_opcode...opcode.local7_opcode => frame.locals[lead - opcode.local0_opcode],
|
||||||
|
opcode.arg0_opcode...opcode.arg6_opcode => frame.args[lead - opcode.arg0_opcode],
|
||||||
|
|
||||||
|
opcode.return_opcode => blk: {
|
||||||
|
frame.ret = try self.term(current, frame);
|
||||||
|
frame.returned = true;
|
||||||
|
break :blk .uninitialized;
|
||||||
|
},
|
||||||
|
opcode.break_opcode => blk: {
|
||||||
|
frame.broke = true;
|
||||||
|
break :blk .uninitialized;
|
||||||
|
},
|
||||||
|
opcode.continue_opcode, opcode.noop_opcode => .uninitialized,
|
||||||
|
|
||||||
|
opcode.if_opcode => try self.ifElse(current, frame),
|
||||||
|
opcode.while_opcode => try self.whileLoop(current, frame),
|
||||||
|
opcode.store_opcode => try self.store(current, frame),
|
||||||
|
opcode.increment_opcode => try self.incDec(current, frame, 1),
|
||||||
|
opcode.decrement_opcode => try self.incDec(current, frame, -1),
|
||||||
|
|
||||||
|
opcode.add_opcode => try self.binary(current, frame, .add),
|
||||||
|
opcode.subtract_opcode => try self.binary(current, frame, .sub),
|
||||||
|
opcode.multiply_opcode => try self.binary(current, frame, .mul),
|
||||||
|
opcode.mod_opcode => try self.binary(current, frame, .mod),
|
||||||
|
opcode.and_opcode => try self.binary(current, frame, .band),
|
||||||
|
opcode.or_opcode => try self.binary(current, frame, .bor),
|
||||||
|
opcode.xor_opcode => try self.binary(current, frame, .bxor),
|
||||||
|
opcode.nand_opcode => try self.binary(current, frame, .nand),
|
||||||
|
opcode.nor_opcode => try self.binary(current, frame, .nor),
|
||||||
|
opcode.shift_left_opcode => try self.binary(current, frame, .shl),
|
||||||
|
opcode.shift_right_opcode => try self.binary(current, frame, .shr),
|
||||||
|
opcode.divide_opcode => try self.divide(current, frame),
|
||||||
|
|
||||||
|
opcode.land_opcode => try self.logic2(current, frame, .land),
|
||||||
|
opcode.lor_opcode => try self.logic2(current, frame, .lor),
|
||||||
|
opcode.lequal_opcode => try self.logic2(current, frame, .eq),
|
||||||
|
opcode.lgreater_opcode => try self.logic2(current, frame, .gt),
|
||||||
|
opcode.lless_opcode => try self.logic2(current, frame, .lt),
|
||||||
|
opcode.lnot_opcode => try self.lnot(current, frame),
|
||||||
|
|
||||||
|
opcode.not_opcode => blk: {
|
||||||
|
const v = try self.evaluateInteger(current, frame);
|
||||||
|
const r = ~v;
|
||||||
|
try self.storeTarget(current, frame, .{ .integer = r });
|
||||||
|
break :blk .{ .integer = r };
|
||||||
|
},
|
||||||
|
|
||||||
|
opcode.size_of_opcode => try self.sizeOf(current, frame),
|
||||||
|
opcode.index_opcode => try self.index(current, frame),
|
||||||
|
opcode.dereference_of_opcode => try self.dereferenceOf(current, frame),
|
||||||
|
opcode.to_integer_opcode => blk: {
|
||||||
|
const v = try self.evaluateInteger(current, frame);
|
||||||
|
try self.storeTarget(current, frame, .{ .integer = v });
|
||||||
|
break :blk .{ .integer = v };
|
||||||
|
},
|
||||||
|
opcode.to_buffer_opcode => try self.passThroughUnary(current, frame),
|
||||||
|
|
||||||
|
opcode.notify_opcode => try self.notify(current, frame),
|
||||||
|
|
||||||
|
opcode.extended_opcode_prefix => try self.ext(current, frame),
|
||||||
|
|
||||||
|
// CreateXField: source, index, name (bit widths differ by op)
|
||||||
|
opcode.create_bit_field_opcode => try self.createField(current, frame, 1),
|
||||||
|
opcode.create_byte_field_opcode => try self.createField(current, frame, 8),
|
||||||
|
opcode.create_word_field_opcode => try self.createField(current, frame, 16),
|
||||||
|
opcode.create_dword_field_opcode => try self.createField(current, frame, 32),
|
||||||
|
opcode.create_qword_field_opcode => try self.createField(current, frame, 64),
|
||||||
|
|
||||||
|
else => error.Unsupported,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- name references ----------------------------------------------------
|
||||||
|
|
||||||
|
fn nameReference(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const name_path = try current.nameString();
|
||||||
|
const node = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice()) orelse
|
||||||
|
return .uninitialized; // unknown name -> treat as uninitialised
|
||||||
|
switch (node.kind) {
|
||||||
|
.method => {
|
||||||
|
var argbuf: [7]Object = undefined;
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < node.arg_count and i < argbuf.len) : (i += 1) argbuf[i] = try self.term(current, frame);
|
||||||
|
return self.invoke(node, argbuf[0..@min(node.arg_count, argbuf.len)]);
|
||||||
|
},
|
||||||
|
.field => return .{ .integer = try self.readField(node) },
|
||||||
|
.name => return self.invoke(node, &.{}),
|
||||||
|
else => return .{ .reference = node },
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- data objects -------------------------------------------------------
|
||||||
|
|
||||||
|
fn readConstant(self: *Interpreter, current: *Cursor, n: usize) Error!u64 {
|
||||||
|
_ = self;
|
||||||
|
const bytes = try current.take(n);
|
||||||
|
var v: u64 = 0;
|
||||||
|
for (bytes, 0..) |b, i| v |= @as(u64, b) << @intCast(i * 8);
|
||||||
|
return v;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn readString(self: *Interpreter, current: *Cursor) Error!Object {
|
||||||
|
const start = current.i;
|
||||||
|
while (current.peek()) |c| {
|
||||||
|
current.i += 1;
|
||||||
|
if (c == 0) break;
|
||||||
|
}
|
||||||
|
const raw = current.b[start .. current.i - 1];
|
||||||
|
const s = try self.arena.dupe(u8, raw);
|
||||||
|
return .{ .string = s };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn buffer(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const start = current.i;
|
||||||
|
const len = try current.packageLength();
|
||||||
|
const end = @min(start + len, current.b.len);
|
||||||
|
const size = try self.evaluateInteger(current, frame);
|
||||||
|
const data = current.b[@min(current.i, end)..end];
|
||||||
|
const bytes = try self.arena.alloc(u8, @intCast(size));
|
||||||
|
@memset(bytes, 0);
|
||||||
|
@memcpy(bytes[0..@min(bytes.len, data.len)], data[0..@min(bytes.len, data.len)]);
|
||||||
|
current.i = end;
|
||||||
|
return .{ .buffer = bytes };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn package(self: *Interpreter, current: *Cursor, frame: *Frame, variable: bool) Error!Object {
|
||||||
|
const start = current.i;
|
||||||
|
const len = try current.packageLength();
|
||||||
|
const end = @min(start + len, current.b.len);
|
||||||
|
const count: usize = if (variable) @intCast(try self.evaluateInteger(current, frame)) else try current.byte();
|
||||||
|
const elems = try self.arena.alloc(Object, count);
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < count and current.i < end) : (i += 1) elems[i] = try self.term(current, frame);
|
||||||
|
while (i < count) : (i += 1) elems[i] = .uninitialized;
|
||||||
|
current.i = end;
|
||||||
|
return .{ .package = elems };
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- operators ----------------------------------------------------------
|
||||||
|
|
||||||
|
const BinaryOperation = enum { add, sub, mul, mod, band, bor, bxor, nand, nor, shl, shr };
|
||||||
|
|
||||||
|
fn binary(self: *Interpreter, current: *Cursor, frame: *Frame, kind: BinaryOperation) Error!Object {
|
||||||
|
const a = try self.evaluateInteger(current, frame);
|
||||||
|
const b = try self.evaluateInteger(current, frame);
|
||||||
|
const r: u64 = switch (kind) {
|
||||||
|
.add => a +% b,
|
||||||
|
.sub => a -% b,
|
||||||
|
.mul => a *% b,
|
||||||
|
.mod => if (b == 0) return error.DivByZero else a % b,
|
||||||
|
.band => a & b,
|
||||||
|
.bor => a | b,
|
||||||
|
.bxor => a ^ b,
|
||||||
|
.nand => ~(a & b),
|
||||||
|
.nor => ~(a | b),
|
||||||
|
.shl => if (b >= 64) 0 else a << @intCast(b),
|
||||||
|
.shr => if (b >= 64) 0 else a >> @intCast(b),
|
||||||
|
};
|
||||||
|
try self.storeTarget(current, frame, .{ .integer = r });
|
||||||
|
return .{ .integer = r };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn divide(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const a = try self.evaluateInteger(current, frame);
|
||||||
|
const b = try self.evaluateInteger(current, frame);
|
||||||
|
if (b == 0) return error.DivByZero;
|
||||||
|
try self.storeTarget(current, frame, .{ .integer = a % b }); // remainder target
|
||||||
|
try self.storeTarget(current, frame, .{ .integer = a / b }); // quotient target
|
||||||
|
return .{ .integer = a / b };
|
||||||
|
}
|
||||||
|
|
||||||
|
const LogicOperation = enum { land, lor, eq, gt, lt };
|
||||||
|
|
||||||
|
fn logic2(self: *Interpreter, current: *Cursor, frame: *Frame, kind: LogicOperation) Error!Object {
|
||||||
|
const a = try self.evaluateInteger(current, frame);
|
||||||
|
const b = try self.evaluateInteger(current, frame);
|
||||||
|
const r = switch (kind) {
|
||||||
|
.land => a != 0 and b != 0,
|
||||||
|
.lor => a != 0 or b != 0,
|
||||||
|
.eq => a == b,
|
||||||
|
.gt => a > b,
|
||||||
|
.lt => a < b,
|
||||||
|
};
|
||||||
|
return .{ .integer = if (r) ~@as(u64, 0) else 0 };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn lnot(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
// 0x92 0x93/94/95 are the compound comparisons.
|
||||||
|
const b = current.peek() orelse return error.Truncated;
|
||||||
|
switch (b) {
|
||||||
|
opcode.lnot.not_equal => {
|
||||||
|
current.i += 1;
|
||||||
|
const x = try self.evaluateInteger(current, frame);
|
||||||
|
const y = try self.evaluateInteger(current, frame);
|
||||||
|
return .{ .integer = if (x != y) ~@as(u64, 0) else 0 };
|
||||||
|
},
|
||||||
|
opcode.lnot.less_equal => {
|
||||||
|
current.i += 1;
|
||||||
|
const x = try self.evaluateInteger(current, frame);
|
||||||
|
const y = try self.evaluateInteger(current, frame);
|
||||||
|
return .{ .integer = if (x <= y) ~@as(u64, 0) else 0 };
|
||||||
|
},
|
||||||
|
opcode.lnot.greater_equal => {
|
||||||
|
current.i += 1;
|
||||||
|
const x = try self.evaluateInteger(current, frame);
|
||||||
|
const y = try self.evaluateInteger(current, frame);
|
||||||
|
return .{ .integer = if (x >= y) ~@as(u64, 0) else 0 };
|
||||||
|
},
|
||||||
|
else => {
|
||||||
|
const x = try self.evaluateInteger(current, frame);
|
||||||
|
return .{ .integer = if (x == 0) ~@as(u64, 0) else 0 };
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn incDec(self: *Interpreter, current: *Cursor, frame: *Frame, delta: i64) Error!Object {
|
||||||
|
// Operand is a SuperName that is both read and written.
|
||||||
|
const save = current.i;
|
||||||
|
const current_value = try self.term(current, frame);
|
||||||
|
const v = try current_value.asInteger();
|
||||||
|
const r = if (delta > 0) v +% 1 else v -% 1;
|
||||||
|
var tcur = Cursor{ .b = current.b, .i = save };
|
||||||
|
try self.storeInto(&tcur, frame, .{ .integer = r });
|
||||||
|
return .{ .integer = r };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn sizeOf(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const o = try self.term(current, frame);
|
||||||
|
return .{ .integer = switch (o) {
|
||||||
|
.buffer => |b| b.len,
|
||||||
|
.string => |s| s.len,
|
||||||
|
.package => |p| p.len,
|
||||||
|
else => 0,
|
||||||
|
} };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn passThroughUnary(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const o = try self.term(current, frame);
|
||||||
|
try self.storeTarget(current, frame, o);
|
||||||
|
return o;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn index(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const source = try self.term(current, frame);
|
||||||
|
const element_index: usize = @intCast(try self.evaluateInteger(current, frame));
|
||||||
|
// Optional target (a reference); we don't materialise references, so store
|
||||||
|
// the indexed value if a target is present.
|
||||||
|
const value: Object = switch (source) {
|
||||||
|
.buffer => |b| .{ .integer = if (element_index < b.len) b[element_index] else 0 },
|
||||||
|
.package => |p| if (element_index < p.len) p[element_index] else .uninitialized,
|
||||||
|
.string => |s| .{ .integer = if (element_index < s.len) s[element_index] else 0 },
|
||||||
|
else => .uninitialized,
|
||||||
|
};
|
||||||
|
try self.storeTarget(current, frame, value);
|
||||||
|
return value;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn dereferenceOf(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const o = try self.term(current, frame);
|
||||||
|
return switch (o) {
|
||||||
|
.reference => |n| self.invoke(n, &.{}),
|
||||||
|
else => o,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- control flow -------------------------------------------------------
|
||||||
|
|
||||||
|
fn ifElse(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const start = current.i;
|
||||||
|
const end = @min(start + try current.packageLength(), current.b.len);
|
||||||
|
const cond = try self.evaluateInteger(current, frame);
|
||||||
|
if (cond != 0) {
|
||||||
|
var body = Cursor{ .b = current.b[0..end], .i = current.i };
|
||||||
|
try self.executeList(&body, frame);
|
||||||
|
current.i = end;
|
||||||
|
// Skip a trailing Else.
|
||||||
|
if (current.peek() == opcode.else_opcode) {
|
||||||
|
current.i += 1;
|
||||||
|
const es = current.i;
|
||||||
|
const ee = @min(es + try current.packageLength(), current.b.len);
|
||||||
|
current.i = ee;
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
current.i = end;
|
||||||
|
if (current.peek() == opcode.else_opcode) {
|
||||||
|
current.i += 1;
|
||||||
|
const es = current.i;
|
||||||
|
const ee = @min(es + try current.packageLength(), current.b.len);
|
||||||
|
var body = Cursor{ .b = current.b[0..ee], .i = current.i };
|
||||||
|
try self.executeList(&body, frame);
|
||||||
|
current.i = ee;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return .uninitialized;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn whileLoop(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const start = current.i;
|
||||||
|
const end = @min(start + try current.packageLength(), current.b.len);
|
||||||
|
const pred_at = current.i;
|
||||||
|
var guard: usize = 0;
|
||||||
|
while (guard < 100_000) : (guard += 1) {
|
||||||
|
var pc = Cursor{ .b = current.b[0..end], .i = pred_at };
|
||||||
|
const cond = try self.evaluateInteger(&pc, frame);
|
||||||
|
if (cond == 0) break;
|
||||||
|
var body = Cursor{ .b = current.b[0..end], .i = pc.i };
|
||||||
|
try self.executeList(&body, frame);
|
||||||
|
if (frame.returned) break;
|
||||||
|
if (frame.broke) {
|
||||||
|
frame.broke = false;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
current.i = end;
|
||||||
|
return .uninitialized;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- store --------------------------------------------------------------
|
||||||
|
|
||||||
|
fn store(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const value = try self.term(current, frame);
|
||||||
|
try self.storeInto(current, frame, value);
|
||||||
|
return value;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A Store *target* that may be NullName (no store).
|
||||||
|
fn storeTarget(self: *Interpreter, current: *Cursor, frame: *Frame, value: Object) Error!void {
|
||||||
|
if (current.peek() == 0x00) {
|
||||||
|
current.i += 1; // NullName
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
try self.storeInto(current, frame, value);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Notify(SuperName, NotifyValue): resolve the named device, evaluate the
|
||||||
|
/// code, and record the pair for the caller to dispatch. AML control flow
|
||||||
|
/// continues (Notify returns nothing).
|
||||||
|
fn notify(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const lead = current.peek() orelse return error.Truncated;
|
||||||
|
var target: ?*Node = null;
|
||||||
|
if (isNameStart(lead)) {
|
||||||
|
const name_path = try current.nameString();
|
||||||
|
target = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice());
|
||||||
|
} else {
|
||||||
|
// A non-name SuperName (Local/Arg holding a reference).
|
||||||
|
const obj = try self.term(current, frame);
|
||||||
|
if (obj == .reference) target = obj.reference;
|
||||||
|
}
|
||||||
|
const code = try self.evaluateInteger(current, frame);
|
||||||
|
if (target) |node| {
|
||||||
|
if (self.notify_count < self.notify_queue.len) {
|
||||||
|
self.notify_queue[self.notify_count] = .{ .node = node, .code = code };
|
||||||
|
self.notify_count += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return .uninitialized;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The Notify events the last `evaluate` produced. Valid until the next
|
||||||
|
/// `evaluate` clears the queue.
|
||||||
|
pub fn takeNotifications(self: *Interpreter) []const NotifyEvent {
|
||||||
|
return self.notify_queue[0..self.notify_count];
|
||||||
|
}
|
||||||
|
|
||||||
|
fn storeInto(self: *Interpreter, current: *Cursor, frame: *Frame, value: Object) Error!void {
|
||||||
|
const lead = current.peek() orelse return error.Truncated;
|
||||||
|
if (isNameStart(lead)) {
|
||||||
|
const name_path = try current.nameString();
|
||||||
|
const node = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice()) orelse return;
|
||||||
|
if (self.fields.get(node)) |buffer_field| {
|
||||||
|
try self.writeBufferField(buffer_field, try value.asInteger());
|
||||||
|
} else if (node.kind == .field) {
|
||||||
|
try self.writeField(node, try value.asInteger());
|
||||||
|
} else {
|
||||||
|
try self.dynamic_overrides.put(self.arena, node, value);
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
_ = try current.byte();
|
||||||
|
switch (lead) {
|
||||||
|
0x00 => {}, // NullName
|
||||||
|
opcode.local0_opcode...opcode.local7_opcode => frame.locals[lead - opcode.local0_opcode] = value,
|
||||||
|
opcode.arg0_opcode...opcode.arg6_opcode => frame.args[lead - opcode.arg0_opcode] = value,
|
||||||
|
opcode.index_opcode => {
|
||||||
|
const source = try self.term(current, frame);
|
||||||
|
const element_index: usize = @intCast(try self.evaluateInteger(current, frame));
|
||||||
|
switch (source) {
|
||||||
|
.buffer => |b| if (element_index < b.len) {
|
||||||
|
b[element_index] = @truncate(try value.asInteger());
|
||||||
|
},
|
||||||
|
.package => |p| if (element_index < p.len) {
|
||||||
|
p[element_index] = value;
|
||||||
|
},
|
||||||
|
else => {},
|
||||||
|
}
|
||||||
|
},
|
||||||
|
else => return error.Unsupported,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- CreateField (buffer patching) --------------------------------------
|
||||||
|
|
||||||
|
fn createField(self: *Interpreter, current: *Cursor, frame: *Frame, bit_width: u32) Error!Object {
|
||||||
|
const source = try self.term(current, frame); // source buffer (as a reference or value)
|
||||||
|
const bit_index = try self.evaluateInteger(current, frame);
|
||||||
|
const name_path = try current.nameString();
|
||||||
|
const node = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice()) orelse return .uninitialized;
|
||||||
|
|
||||||
|
// Bind the new name to the source buffer's node so stores land in it.
|
||||||
|
const buffer_node: *Node = switch (source) {
|
||||||
|
.reference => |n| n,
|
||||||
|
else => return .uninitialized,
|
||||||
|
};
|
||||||
|
// Materialise the buffer into `dynamic_overrides` so patches persist and are returned.
|
||||||
|
if (self.dynamic_overrides.get(buffer_node) == null) {
|
||||||
|
const value = try self.invoke(buffer_node, &.{});
|
||||||
|
try self.dynamic_overrides.put(self.arena, buffer_node, value);
|
||||||
|
}
|
||||||
|
const byte_off: usize = @intCast(bit_index / 8);
|
||||||
|
try self.fields.put(self.arena, node, .{ .buffer = buffer_node, .byte_off = byte_off, .bit_width = bit_width });
|
||||||
|
return .uninitialized;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn writeBufferField(self: *Interpreter, buffer_field: BufferField, value: u64) Error!void {
|
||||||
|
const obj = self.dynamic_overrides.get(buffer_field.buffer) orelse return;
|
||||||
|
const bytes = switch (obj) {
|
||||||
|
.buffer => |b| b,
|
||||||
|
else => return,
|
||||||
|
};
|
||||||
|
const byte_count = (buffer_field.bit_width + 7) / 8;
|
||||||
|
var k: usize = 0;
|
||||||
|
while (k < byte_count and buffer_field.byte_off + k < bytes.len) : (k += 1) {
|
||||||
|
bytes[buffer_field.byte_off + k] = @truncate(value >> @intCast(k * 8));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- OperationRegion field access ---------------------------------------
|
||||||
|
|
||||||
|
fn readField(self: *Interpreter, field: *Node) Error!u64 {
|
||||||
|
const region = field.region orelse return error.Unsupported;
|
||||||
|
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
||||||
|
const base = try self.regionBase(region);
|
||||||
|
const start_byte = base + field.bit_offset / 8;
|
||||||
|
const shift: u7 = @intCast(field.bit_offset % 8);
|
||||||
|
const total = @as(usize, shift) + field.bit_width;
|
||||||
|
const byte_count = (total + 7) / 8;
|
||||||
|
var raw: u128 = 0;
|
||||||
|
var k: usize = 0;
|
||||||
|
while (k < byte_count) : (k += 1) {
|
||||||
|
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
||||||
|
}
|
||||||
|
const masked = (raw >> shift) & bitMask(field.bit_width);
|
||||||
|
return @truncate(masked);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn writeField(self: *Interpreter, field: *Node, value: u64) Error!void {
|
||||||
|
const region = field.region orelse return error.Unsupported;
|
||||||
|
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
||||||
|
const base = try self.regionBase(region);
|
||||||
|
const start_byte = base + field.bit_offset / 8;
|
||||||
|
const shift: u7 = @intCast(field.bit_offset % 8);
|
||||||
|
const total = @as(usize, shift) + field.bit_width;
|
||||||
|
const byte_count = (total + 7) / 8;
|
||||||
|
// Read-modify-write byte by byte.
|
||||||
|
var raw: u128 = 0;
|
||||||
|
var k: usize = 0;
|
||||||
|
while (k < byte_count) : (k += 1) {
|
||||||
|
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
||||||
|
}
|
||||||
|
const mask = bitMask(field.bit_width) << shift;
|
||||||
|
raw = (raw & ~mask) | ((@as(u128, value) << shift) & mask);
|
||||||
|
k = 0;
|
||||||
|
while (k < byte_count) : (k += 1) {
|
||||||
|
try self.writeRegionByte(region.region_space, start_byte + k, @truncate(raw >> @intCast(k * 8)));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn regionBase(self: *Interpreter, region: *Node) Error!u64 {
|
||||||
|
var current = Cursor{ .b = region.region_offset_aml };
|
||||||
|
var frame = Frame{ .scope = region.parent orelse self.namespace.root };
|
||||||
|
return (try self.term(¤t, &frame)).asInteger();
|
||||||
|
}
|
||||||
|
|
||||||
|
fn readRegionByte(self: *Interpreter, space: u8, address: u64) Error!u8 {
|
||||||
|
switch (space) {
|
||||||
|
0 => { // SystemMemory
|
||||||
|
const virtual = self.hal.mapMmio(address & ~@as(u64, 0xFFF), 0x1000, true);
|
||||||
|
const p: *align(1) const volatile u8 = @ptrFromInt(virtual + (address & 0xFFF));
|
||||||
|
return p.*;
|
||||||
|
},
|
||||||
|
1 => return @truncate(self.hal.pioRead(1, @intCast(address & 0xFFFF))), // SystemIO
|
||||||
|
else => return error.Unsupported,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn writeRegionByte(self: *Interpreter, space: u8, address: u64, value: u8) Error!void {
|
||||||
|
switch (space) {
|
||||||
|
0 => {
|
||||||
|
const virtual = self.hal.mapMmio(address & ~@as(u64, 0xFFF), 0x1000, true);
|
||||||
|
const p: *align(1) volatile u8 = @ptrFromInt(virtual + (address & 0xFFF));
|
||||||
|
p.* = value;
|
||||||
|
},
|
||||||
|
1 => self.hal.pioWrite(1, @intCast(address & 0xFFFF), value),
|
||||||
|
else => return error.Unsupported,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- extended opcodes ---------------------------------------------------
|
||||||
|
|
||||||
|
fn ext(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const e = try current.byte();
|
||||||
|
switch (e) {
|
||||||
|
opcode.extended.debug => return .uninitialized,
|
||||||
|
opcode.extended.revision => return .{ .integer = 2 },
|
||||||
|
opcode.extended.timer => return .{ .integer = 0 },
|
||||||
|
// Mutex/Event ops are no-ops in this single-threaded evaluator.
|
||||||
|
opcode.extended.acquire => {
|
||||||
|
_ = try self.term(current, frame); // mutex SuperName
|
||||||
|
_ = try current.take(2); // timeout
|
||||||
|
return .{ .integer = 0 }; // acquired
|
||||||
|
},
|
||||||
|
opcode.extended.release, opcode.extended.reset, opcode.extended.signal => {
|
||||||
|
_ = try self.term(current, frame);
|
||||||
|
return .uninitialized;
|
||||||
|
},
|
||||||
|
opcode.extended.wait => {
|
||||||
|
_ = try self.term(current, frame);
|
||||||
|
_ = try self.term(current, frame);
|
||||||
|
return .{ .integer = 0 };
|
||||||
|
},
|
||||||
|
opcode.extended.sleep, opcode.extended.stall => {
|
||||||
|
_ = try self.term(current, frame);
|
||||||
|
return .uninitialized;
|
||||||
|
},
|
||||||
|
else => return error.Unsupported,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn evaluateInteger(self: *Interpreter, current: *Cursor, frame: *Frame) Error!u64 {
|
||||||
|
return (try self.term(current, frame)).asInteger();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
fn bitMask(width: u32) u128 {
|
||||||
|
if (width >= 128) return ~@as(u128, 0);
|
||||||
|
return (@as(u128, 1) << @intCast(width)) - 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn isNameStart(b: u8) bool {
|
||||||
|
return (b >= opcode.name_char_start and b <= opcode.name_char_end) or
|
||||||
|
b == opcode.name_char_underscore or
|
||||||
|
b == opcode.root_char or
|
||||||
|
b == opcode.parent_prefix_char or
|
||||||
|
b == opcode.dual_name_prefix or
|
||||||
|
b == opcode.multi_name_prefix;
|
||||||
|
}
|
||||||
@@ -18,7 +18,7 @@ pub const NodeKind = enum {
|
|||||||
mutex,
|
mutex,
|
||||||
event,
|
event,
|
||||||
processor,
|
processor,
|
||||||
power_res,
|
power_resource,
|
||||||
thermal_zone,
|
thermal_zone,
|
||||||
alias,
|
alias,
|
||||||
external,
|
external,
|
||||||
@@ -28,12 +28,12 @@ pub const NodeKind = enum {
|
|||||||
pub const Node = struct {
|
pub const Node = struct {
|
||||||
/// The 4-byte NameSeg identifying this node within its parent. The root uses
|
/// The 4-byte NameSeg identifying this node within its parent. The root uses
|
||||||
/// all-zero.
|
/// all-zero.
|
||||||
seg: [4]u8 = .{ 0, 0, 0, 0 },
|
segment: [4]u8 = .{ 0, 0, 0, 0 },
|
||||||
kind: NodeKind = .other,
|
kind: NodeKind = .other,
|
||||||
/// For Method / External: the declared argument count (0..7). Used to resolve
|
/// For Method / External: the declared argument count (0..7). Used to resolve
|
||||||
/// how many TermArgs a method invocation consumes.
|
/// how many TermArgs a method invocation consumes.
|
||||||
arg_count: u8 = 0,
|
arg_count: u8 = 0,
|
||||||
/// For Name: the AML bytes of its DataRefObject (so a value like a sleep
|
/// For Name: the AML bytes of its DataReferenceObject (so a value like a sleep
|
||||||
/// state's (`_Sx`) Package can be parsed on demand). For Method: the AML bytes of the body,
|
/// state's (`_Sx`) Package can be parsed on demand). For Method: the AML bytes of the body,
|
||||||
/// interpreted on demand by the evaluator. Empty otherwise.
|
/// interpreted on demand by the evaluator. Empty otherwise.
|
||||||
value: []const u8 = &.{},
|
value: []const u8 = &.{},
|
||||||
@@ -78,49 +78,49 @@ pub const Namespace = struct {
|
|||||||
return self.root.subtreeCount();
|
return self.root.subtreeCount();
|
||||||
}
|
}
|
||||||
|
|
||||||
fn findChild(parent: *Node, seg: [4]u8) ?*Node {
|
fn findChild(parent: *Node, segment: [4]u8) ?*Node {
|
||||||
var c = parent.first_child;
|
var c = parent.first_child;
|
||||||
while (c) |child| : (c = child.next_sibling) {
|
while (c) |child| : (c = child.next_sibling) {
|
||||||
if (std.mem.eql(u8, &child.seg, &seg)) return child;
|
if (std.mem.eql(u8, &child.segment, &segment)) return child;
|
||||||
}
|
}
|
||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The direct child of `node` named `seg`, or null. Unlike `resolve`, this does
|
/// The direct child of `node` named `segment`, or null. Unlike `resolve`, this does
|
||||||
/// not apply the search-rule walk-up — it looks only at immediate children (for
|
/// not apply the search-rule walk-up — it looks only at immediate children (for
|
||||||
/// reading a device's own hardware ID (`_HID`) / current resource settings (`_CRS`)).
|
/// reading a device's own hardware ID (`_HID`) / current resource settings (`_CRS`)).
|
||||||
pub fn childOf(node: *Node, seg: [4]u8) ?*Node {
|
pub fn childOf(node: *Node, segment: [4]u8) ?*Node {
|
||||||
return findChild(node, seg);
|
return findChild(node, segment);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn newChild(self: *Namespace, parent: *Node, seg: [4]u8, kind: NodeKind) !*Node {
|
fn newChild(self: *Namespace, parent: *Node, segment: [4]u8, kind: NodeKind) !*Node {
|
||||||
const n = try self.allocator.create(Node);
|
const n = try self.allocator.create(Node);
|
||||||
n.* = .{ .seg = seg, .kind = kind, .parent = parent };
|
n.* = .{ .segment = segment, .kind = kind, .parent = parent };
|
||||||
// Append at the tail so a dump reads in declaration order.
|
// Append at the tail so a dump reads in declaration order.
|
||||||
if (parent.first_child == null) {
|
if (parent.first_child == null) {
|
||||||
parent.first_child = n;
|
parent.first_child = n;
|
||||||
} else {
|
} else {
|
||||||
var cur = parent.first_child.?;
|
var current = parent.first_child.?;
|
||||||
while (cur.next_sibling) |sib| cur = sib;
|
while (current.next_sibling) |sib| current = sib;
|
||||||
cur.next_sibling = n;
|
current.next_sibling = n;
|
||||||
}
|
}
|
||||||
return n;
|
return n;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Create a Field unit node directly under `scope` (field units live in the
|
/// Create a Field unit node directly under `scope` (field units live in the
|
||||||
/// scope of the Field/IndexField/BankField, not under the region).
|
/// scope of the Field/IndexField/BankField, not under the region).
|
||||||
pub fn newFieldUnit(self: *Namespace, scope: *Node, seg: [4]u8) !*Node {
|
pub fn newFieldUnit(self: *Namespace, scope: *Node, segment: [4]u8) !*Node {
|
||||||
return self.findOrCreate(scope, seg, .field);
|
return self.findOrCreate(scope, segment, .field);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn findOrCreate(self: *Namespace, parent: *Node, seg: [4]u8, kind: NodeKind) !*Node {
|
fn findOrCreate(self: *Namespace, parent: *Node, segment: [4]u8, kind: NodeKind) !*Node {
|
||||||
if (findChild(parent, seg)) |existing| {
|
if (findChild(parent, segment)) |existing| {
|
||||||
// Reopening a scope (e.g. Scope(\_SB) after Device \_SB) keeps the more
|
// Reopening a scope (e.g. Scope(\_SB) after Device \_SB) keeps the more
|
||||||
// specific kind rather than downgrading to a plain scope.
|
// specific kind rather than downgrading to a plain scope.
|
||||||
if (existing.kind == .scope and kind != .scope) existing.kind = kind;
|
if (existing.kind == .scope and kind != .scope) existing.kind = kind;
|
||||||
return existing;
|
return existing;
|
||||||
}
|
}
|
||||||
return self.newChild(parent, seg, kind);
|
return self.newChild(parent, segment, kind);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The node a definition's NameString names, creating any intermediate scopes.
|
/// The node a definition's NameString names, creating any intermediate scopes.
|
||||||
@@ -131,16 +131,16 @@ pub const Namespace = struct {
|
|||||||
current: *Node,
|
current: *Node,
|
||||||
rooted: bool,
|
rooted: bool,
|
||||||
parents: u8,
|
parents: u8,
|
||||||
segs: []const [4]u8,
|
segments: []const [4]u8,
|
||||||
kind: NodeKind,
|
kind: NodeKind,
|
||||||
) !*Node {
|
) !*Node {
|
||||||
var base = startNode(self, current, rooted, parents);
|
var base = startNode(self, current, rooted, parents);
|
||||||
if (segs.len == 0) return base;
|
if (segments.len == 0) return base;
|
||||||
var i: usize = 0;
|
var i: usize = 0;
|
||||||
while (i + 1 < segs.len) : (i += 1) {
|
while (i + 1 < segments.len) : (i += 1) {
|
||||||
base = try self.findOrCreate(base, segs[i], .scope);
|
base = try self.findOrCreate(base, segments[i], .scope);
|
||||||
}
|
}
|
||||||
return self.findOrCreate(base, segs[segs.len - 1], kind);
|
return self.findOrCreate(base, segments[segments.len - 1], kind);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Resolve a NameString *reference* to an existing node, or null. A single
|
/// Resolve a NameString *reference* to an existing node, or null. A single
|
||||||
@@ -151,22 +151,22 @@ pub const Namespace = struct {
|
|||||||
current: *Node,
|
current: *Node,
|
||||||
rooted: bool,
|
rooted: bool,
|
||||||
parents: u8,
|
parents: u8,
|
||||||
segs: []const [4]u8,
|
segments: []const [4]u8,
|
||||||
) ?*Node {
|
) ?*Node {
|
||||||
if (segs.len == 0) return null;
|
if (segments.len == 0) return null;
|
||||||
|
|
||||||
if (!rooted and parents == 0 and segs.len == 1) {
|
if (!rooted and parents == 0 and segments.len == 1) {
|
||||||
// Search rule: this scope, then each ancestor up to the root.
|
// Search rule: this scope, then each ancestor up to the root.
|
||||||
var scope: ?*Node = current;
|
var scope: ?*Node = current;
|
||||||
while (scope) |s| : (scope = s.parent) {
|
while (scope) |s| : (scope = s.parent) {
|
||||||
if (findChild(s, segs[0])) |n| return n;
|
if (findChild(s, segments[0])) |n| return n;
|
||||||
}
|
}
|
||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
|
||||||
var base = startNode(self, current, rooted, parents);
|
var base = startNode(self, current, rooted, parents);
|
||||||
for (segs) |seg| {
|
for (segments) |segment| {
|
||||||
base = findChild(base, seg) orelse return null;
|
base = findChild(base, segment) orelse return null;
|
||||||
}
|
}
|
||||||
return base;
|
return base;
|
||||||
}
|
}
|
||||||
@@ -0,0 +1,137 @@
|
|||||||
|
//! AML opcode constants — the full ACPI Machine Language opcode table.
|
||||||
|
//!
|
||||||
|
//! Single-byte opcodes are plain values. Extended opcodes are a two-byte sequence
|
||||||
|
//! `ext_prefix` (0x5B) followed by a byte listed under `ext`. A few comparison
|
||||||
|
//! opcodes are `lnot_opcode` (0x92) followed by a second byte (see `lnot`).
|
||||||
|
|
||||||
|
// --- name / path characters -------------------------------------------------
|
||||||
|
pub const zero_opcode = 0x00;
|
||||||
|
pub const one_opcode = 0x01;
|
||||||
|
pub const alias_opcode = 0x06;
|
||||||
|
pub const name_opcode = 0x08;
|
||||||
|
pub const byte_prefix = 0x0A;
|
||||||
|
pub const word_prefix = 0x0B;
|
||||||
|
pub const dword_prefix = 0x0C;
|
||||||
|
pub const string_prefix = 0x0D;
|
||||||
|
pub const qword_prefix = 0x0E;
|
||||||
|
pub const scope_opcode = 0x10;
|
||||||
|
pub const buffer_opcode = 0x11;
|
||||||
|
pub const package_opcode = 0x12;
|
||||||
|
pub const var_package_opcode = 0x13;
|
||||||
|
pub const method_opcode = 0x14;
|
||||||
|
pub const external_opcode = 0x15;
|
||||||
|
|
||||||
|
pub const dual_name_prefix = 0x2E;
|
||||||
|
pub const multi_name_prefix = 0x2F;
|
||||||
|
pub const extended_opcode_prefix = 0x5B;
|
||||||
|
pub const root_char = 0x5C;
|
||||||
|
pub const parent_prefix_char = 0x5E;
|
||||||
|
pub const name_char_underscore = 0x5F;
|
||||||
|
|
||||||
|
pub const digit_char_start = 0x30;
|
||||||
|
pub const digit_char_end = 0x39;
|
||||||
|
pub const name_char_start = 0x41; // 'A'
|
||||||
|
pub const name_char_end = 0x5A; // 'Z'
|
||||||
|
|
||||||
|
// --- locals / args ----------------------------------------------------------
|
||||||
|
pub const local0_opcode = 0x60;
|
||||||
|
pub const local7_opcode = 0x67;
|
||||||
|
pub const arg0_opcode = 0x68;
|
||||||
|
pub const arg6_opcode = 0x6E;
|
||||||
|
|
||||||
|
// --- store / references / arithmetic ---------------------------------------
|
||||||
|
pub const store_opcode = 0x70;
|
||||||
|
pub const ref_of_opcode = 0x71;
|
||||||
|
pub const add_opcode = 0x72;
|
||||||
|
pub const concat_opcode = 0x73;
|
||||||
|
pub const subtract_opcode = 0x74;
|
||||||
|
pub const increment_opcode = 0x75;
|
||||||
|
pub const decrement_opcode = 0x76;
|
||||||
|
pub const multiply_opcode = 0x77;
|
||||||
|
pub const divide_opcode = 0x78;
|
||||||
|
pub const shift_left_opcode = 0x79;
|
||||||
|
pub const shift_right_opcode = 0x7A;
|
||||||
|
pub const and_opcode = 0x7B;
|
||||||
|
pub const nand_opcode = 0x7C;
|
||||||
|
pub const or_opcode = 0x7D;
|
||||||
|
pub const nor_opcode = 0x7E;
|
||||||
|
pub const xor_opcode = 0x7F;
|
||||||
|
pub const not_opcode = 0x80;
|
||||||
|
pub const find_set_left_bit_opcode = 0x81;
|
||||||
|
pub const find_set_right_bit_opcode = 0x82;
|
||||||
|
pub const dereference_of_opcode = 0x83;
|
||||||
|
pub const concat_resource_opcode = 0x84;
|
||||||
|
pub const mod_opcode = 0x85;
|
||||||
|
pub const notify_opcode = 0x86;
|
||||||
|
pub const size_of_opcode = 0x87;
|
||||||
|
pub const index_opcode = 0x88;
|
||||||
|
pub const match_opcode = 0x89;
|
||||||
|
pub const create_dword_field_opcode = 0x8A;
|
||||||
|
pub const create_word_field_opcode = 0x8B;
|
||||||
|
pub const create_byte_field_opcode = 0x8C;
|
||||||
|
pub const create_bit_field_opcode = 0x8D;
|
||||||
|
pub const object_type_opcode = 0x8E;
|
||||||
|
pub const create_qword_field_opcode = 0x8F;
|
||||||
|
|
||||||
|
pub const land_opcode = 0x90;
|
||||||
|
pub const lor_opcode = 0x91;
|
||||||
|
pub const lnot_opcode = 0x92; // may be followed by a second byte (see `lnot`)
|
||||||
|
pub const lequal_opcode = 0x93;
|
||||||
|
pub const lgreater_opcode = 0x94;
|
||||||
|
pub const lless_opcode = 0x95;
|
||||||
|
pub const to_buffer_opcode = 0x96;
|
||||||
|
pub const to_decimal_string_opcode = 0x97;
|
||||||
|
pub const to_hex_string_opcode = 0x98;
|
||||||
|
pub const to_integer_opcode = 0x99;
|
||||||
|
pub const to_string_opcode = 0x9C;
|
||||||
|
pub const copy_object_opcode = 0x9D;
|
||||||
|
pub const mid_opcode = 0x9E;
|
||||||
|
pub const continue_opcode = 0x9F;
|
||||||
|
pub const if_opcode = 0xA0;
|
||||||
|
pub const else_opcode = 0xA1;
|
||||||
|
pub const while_opcode = 0xA2;
|
||||||
|
pub const noop_opcode = 0xA3;
|
||||||
|
pub const return_opcode = 0xA4;
|
||||||
|
pub const break_opcode = 0xA5;
|
||||||
|
pub const break_point_opcode = 0xCC;
|
||||||
|
pub const ones_opcode = 0xFF;
|
||||||
|
|
||||||
|
/// Second bytes of the `lnot_opcode` (0x92) compound comparison opcodes.
|
||||||
|
pub const lnot = struct {
|
||||||
|
pub const not_equal = 0x93; // LNotEqualOp: 0x92 0x93
|
||||||
|
pub const less_equal = 0x94; // LLessEqualOp: 0x92 0x94
|
||||||
|
pub const greater_equal = 0x95; // LGreaterEqualOp: 0x92 0x95
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Second bytes of extended opcodes (prefixed by `extended_opcode_prefix`, 0x5B).
|
||||||
|
pub const extended = struct {
|
||||||
|
pub const mutex = 0x01;
|
||||||
|
pub const event = 0x02;
|
||||||
|
pub const conditional_reference_of = 0x12;
|
||||||
|
pub const create_field = 0x13;
|
||||||
|
pub const load_table = 0x1F;
|
||||||
|
pub const load = 0x20;
|
||||||
|
pub const stall = 0x21;
|
||||||
|
pub const sleep = 0x22;
|
||||||
|
pub const acquire = 0x23;
|
||||||
|
pub const signal = 0x24;
|
||||||
|
pub const wait = 0x25;
|
||||||
|
pub const reset = 0x26;
|
||||||
|
pub const release = 0x27;
|
||||||
|
pub const from_bcd = 0x28;
|
||||||
|
pub const to_bcd = 0x29;
|
||||||
|
pub const unload = 0x2A;
|
||||||
|
pub const revision = 0x30;
|
||||||
|
pub const debug = 0x31;
|
||||||
|
pub const fatal = 0x32;
|
||||||
|
pub const timer = 0x33;
|
||||||
|
pub const operation_region = 0x80;
|
||||||
|
pub const field = 0x81;
|
||||||
|
pub const device = 0x82;
|
||||||
|
pub const processor = 0x83;
|
||||||
|
pub const power_resource = 0x84;
|
||||||
|
pub const thermal_zone = 0x85;
|
||||||
|
pub const index_field = 0x86;
|
||||||
|
pub const bank_field = 0x87;
|
||||||
|
pub const data_region = 0x88;
|
||||||
|
};
|
||||||
@@ -0,0 +1,517 @@
|
|||||||
|
//! Recursive-descent AML parser. Walks the entire byte stream — including method
|
||||||
|
//! bodies — building the ACPI namespace as it goes. It does not *evaluate*
|
||||||
|
//! anything (no OperationRegion reads, no arithmetic); it parses structure so the
|
||||||
|
//! cursor stays aligned and every named object is recorded.
|
||||||
|
//!
|
||||||
|
//! The one genuine ambiguity in AML is method invocation: a bare NameString in an
|
||||||
|
//! operand position is a call whose argument count is only known from the method's
|
||||||
|
//! (earlier) declaration. Because we build the namespace in the same in-order pass,
|
||||||
|
//! `resolve` finds that declaration and tells us how many operands to consume.
|
||||||
|
//!
|
||||||
|
//! Safety net: every object delimited by a PkgLength (Scope/Device/Method/If/While/
|
||||||
|
//! Field/Buffer/Package/…) is parsed within its known extent, and the cursor is
|
||||||
|
//! snapped to that extent afterwards. So a mis-resolved invocation can only desync
|
||||||
|
//! *within* one such object; the enclosing walk realigns at the boundary.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const opcode = @import("opcodes.zig");
|
||||||
|
const Namespace = @import("namespace.zig").Namespace;
|
||||||
|
const Node = @import("namespace.zig").Node;
|
||||||
|
const NodeKind = @import("namespace.zig").NodeKind;
|
||||||
|
|
||||||
|
pub const Error = error{ Truncated, Malformed } || std.mem.Allocator.Error;
|
||||||
|
|
||||||
|
const maximum_segments = 64;
|
||||||
|
|
||||||
|
/// A parsed NameString: an optional root anchor or some parent hops, then a list
|
||||||
|
/// of 4-byte segments.
|
||||||
|
const NamePath = struct {
|
||||||
|
rooted: bool = false,
|
||||||
|
parents: u8 = 0,
|
||||||
|
segments: [maximum_segments][4]u8 = undefined,
|
||||||
|
count: usize = 0,
|
||||||
|
|
||||||
|
fn slice(self: *const NamePath) []const [4]u8 {
|
||||||
|
return self.segments[0..self.count];
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const Parser = struct {
|
||||||
|
aml: []const u8,
|
||||||
|
position: usize = 0,
|
||||||
|
namespace: *Namespace,
|
||||||
|
|
||||||
|
pub fn init(aml: []const u8, namespace: *Namespace) Parser {
|
||||||
|
return .{ .aml = aml, .namespace = namespace };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Parse the whole block as a TermList under the namespace root. Returns the
|
||||||
|
/// number of bytes consumed — equal to `aml.len` for a clean full traversal.
|
||||||
|
pub fn parseAll(self: *Parser) usize {
|
||||||
|
self.termList(self.aml.len, self.namespace.root);
|
||||||
|
return self.position;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- cursor primitives --------------------------------------------------
|
||||||
|
|
||||||
|
fn eof(self: *Parser) bool {
|
||||||
|
return self.position >= self.aml.len;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn peek(self: *Parser) ?u8 {
|
||||||
|
return if (self.eof()) null else self.aml[self.position];
|
||||||
|
}
|
||||||
|
|
||||||
|
fn readByte(self: *Parser) Error!u8 {
|
||||||
|
if (self.eof()) return error.Truncated;
|
||||||
|
const b = self.aml[self.position];
|
||||||
|
self.position += 1;
|
||||||
|
return b;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn skip(self: *Parser, n: usize) Error!void {
|
||||||
|
if (self.position + n > self.aml.len) return error.Truncated;
|
||||||
|
self.position += n;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn skipCString(self: *Parser) Error!void {
|
||||||
|
while (true) {
|
||||||
|
const b = try self.readByte();
|
||||||
|
if (b == 0) return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// AML PkgLength: the lead byte's top two bits give how many extra bytes
|
||||||
|
/// follow; the value counts from the start of the PkgLength field.
|
||||||
|
fn readPackageLength(self: *Parser) Error!usize {
|
||||||
|
const lead = try self.readByte();
|
||||||
|
const follow: usize = lead >> 6;
|
||||||
|
if (follow == 0) return lead & 0x3F;
|
||||||
|
var value: usize = lead & 0x0F;
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < follow) : (i += 1) {
|
||||||
|
const b = try self.readByte();
|
||||||
|
value |= @as(usize, b) << @intCast(4 + i * 8);
|
||||||
|
}
|
||||||
|
return value;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn readNameSegment(self: *Parser) Error![4]u8 {
|
||||||
|
if (self.position + 4 > self.aml.len) return error.Truncated;
|
||||||
|
const segment = self.aml[self.position..][0..4].*;
|
||||||
|
self.position += 4;
|
||||||
|
return segment;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn readNameString(self: *Parser) Error!NamePath {
|
||||||
|
var name_path = NamePath{};
|
||||||
|
// A NameString is either root-anchored or parent-relative, not both.
|
||||||
|
if (self.peek() == opcode.root_char) {
|
||||||
|
name_path.rooted = true;
|
||||||
|
self.position += 1;
|
||||||
|
} else {
|
||||||
|
while (self.peek() == opcode.parent_prefix_char) : (self.position += 1) name_path.parents += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
const lead = self.peek() orelse return name_path;
|
||||||
|
switch (lead) {
|
||||||
|
0x00 => self.position += 1, // NullName
|
||||||
|
opcode.dual_name_prefix => {
|
||||||
|
self.position += 1;
|
||||||
|
try self.appendSegment(&name_path);
|
||||||
|
try self.appendSegment(&name_path);
|
||||||
|
},
|
||||||
|
opcode.multi_name_prefix => {
|
||||||
|
self.position += 1;
|
||||||
|
const count = try self.readByte();
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < count) : (i += 1) try self.appendSegment(&name_path);
|
||||||
|
},
|
||||||
|
else => {
|
||||||
|
if (isNameStart(lead)) try self.appendSegment(&name_path);
|
||||||
|
},
|
||||||
|
}
|
||||||
|
return name_path;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn appendSegment(self: *Parser, name_path: *NamePath) Error!void {
|
||||||
|
const segment = try self.readNameSegment();
|
||||||
|
if (name_path.count < maximum_segments) {
|
||||||
|
name_path.segments[name_path.count] = segment;
|
||||||
|
name_path.count += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- term list / object -------------------------------------------------
|
||||||
|
|
||||||
|
/// Parse objects until `end`, then snap to `end`. Any parse error resyncs to
|
||||||
|
/// the boundary rather than propagating — containment for the rare desync.
|
||||||
|
fn termList(self: *Parser, end: usize, scope: *Node) void {
|
||||||
|
while (self.position < end) {
|
||||||
|
self.object(scope) catch break;
|
||||||
|
}
|
||||||
|
self.position = end;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Parse exactly one object/term at the cursor. Used for both TermObjs and
|
||||||
|
/// operands (TermArg / SuperName / Target all reduce to "one object" for the
|
||||||
|
/// purpose of advancing the cursor).
|
||||||
|
fn object(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const lead = self.peek() orelse return error.Truncated;
|
||||||
|
if (isNameStart(lead)) return self.nameInvocation(scope);
|
||||||
|
|
||||||
|
_ = try self.readByte();
|
||||||
|
switch (lead) {
|
||||||
|
// constants and no-operand statements
|
||||||
|
opcode.zero_opcode, opcode.one_opcode, opcode.ones_opcode => {},
|
||||||
|
opcode.noop_opcode, opcode.continue_opcode, opcode.break_opcode, opcode.break_point_opcode => {},
|
||||||
|
opcode.local0_opcode...opcode.local7_opcode => {},
|
||||||
|
opcode.arg0_opcode...opcode.arg6_opcode => {},
|
||||||
|
|
||||||
|
// literal data
|
||||||
|
opcode.byte_prefix => try self.skip(1),
|
||||||
|
opcode.word_prefix => try self.skip(2),
|
||||||
|
opcode.dword_prefix => try self.skip(4),
|
||||||
|
opcode.qword_prefix => try self.skip(8),
|
||||||
|
opcode.string_prefix => try self.skipCString(),
|
||||||
|
|
||||||
|
// data containers (contents skipped via their PkgLength)
|
||||||
|
opcode.buffer_opcode, opcode.package_opcode, opcode.var_package_opcode => try self.skipPackage(),
|
||||||
|
|
||||||
|
// namespace modifiers / named objects
|
||||||
|
opcode.name_opcode => try self.parseName(scope),
|
||||||
|
opcode.alias_opcode => try self.parseAlias(scope),
|
||||||
|
opcode.scope_opcode => try self.parseScopeLike(scope, .scope),
|
||||||
|
opcode.method_opcode => try self.parseMethod(scope),
|
||||||
|
opcode.external_opcode => try self.parseExternal(scope),
|
||||||
|
opcode.extended_opcode_prefix => try self.parseExtended(scope),
|
||||||
|
|
||||||
|
// control flow
|
||||||
|
opcode.if_opcode => try self.parseIf(scope),
|
||||||
|
opcode.else_opcode => try self.parseElse(scope),
|
||||||
|
opcode.while_opcode => try self.parseWhile(scope),
|
||||||
|
opcode.return_opcode => try self.object(scope),
|
||||||
|
opcode.notify_opcode => try self.args(scope, 2),
|
||||||
|
|
||||||
|
// stores / references / unary+target
|
||||||
|
opcode.store_opcode => try self.args(scope, 2),
|
||||||
|
opcode.ref_of_opcode, opcode.dereference_of_opcode, opcode.size_of_opcode, opcode.object_type_opcode => try self.args(scope, 1),
|
||||||
|
opcode.increment_opcode, opcode.decrement_opcode => try self.args(scope, 1),
|
||||||
|
opcode.not_opcode, opcode.find_set_left_bit_opcode, opcode.find_set_right_bit_opcode => try self.args(scope, 2),
|
||||||
|
opcode.to_buffer_opcode, opcode.to_decimal_string_opcode, opcode.to_hex_string_opcode, opcode.to_integer_opcode => try self.args(scope, 2),
|
||||||
|
opcode.copy_object_opcode => try self.args(scope, 2),
|
||||||
|
|
||||||
|
// binary + target
|
||||||
|
opcode.add_opcode, opcode.subtract_opcode, opcode.multiply_opcode, opcode.mod_opcode => try self.args(scope, 3),
|
||||||
|
opcode.and_opcode, opcode.nand_opcode, opcode.or_opcode, opcode.nor_opcode, opcode.xor_opcode => try self.args(scope, 3),
|
||||||
|
opcode.shift_left_opcode, opcode.shift_right_opcode, opcode.concat_opcode, opcode.concat_resource_opcode, opcode.index_opcode => try self.args(scope, 3),
|
||||||
|
opcode.divide_opcode => try self.args(scope, 4),
|
||||||
|
opcode.to_string_opcode => try self.args(scope, 3),
|
||||||
|
opcode.mid_opcode => try self.args(scope, 4),
|
||||||
|
|
||||||
|
// logical
|
||||||
|
opcode.land_opcode, opcode.lor_opcode => try self.args(scope, 2),
|
||||||
|
opcode.lequal_opcode, opcode.lgreater_opcode, opcode.lless_opcode => try self.args(scope, 2),
|
||||||
|
opcode.lnot_opcode => try self.parseLnot(scope),
|
||||||
|
|
||||||
|
opcode.match_opcode => try self.parseMatch(scope),
|
||||||
|
|
||||||
|
// CreateXField: <source> <index> NameString
|
||||||
|
opcode.create_dword_field_opcode,
|
||||||
|
opcode.create_word_field_opcode,
|
||||||
|
opcode.create_byte_field_opcode,
|
||||||
|
opcode.create_bit_field_opcode,
|
||||||
|
opcode.create_qword_field_opcode,
|
||||||
|
=> try self.parseCreateField(scope, 2),
|
||||||
|
|
||||||
|
else => return error.Malformed,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Parse `n` operands.
|
||||||
|
fn args(self: *Parser, scope: *Node, n: usize) Error!void {
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < n) : (i += 1) try self.object(scope);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A NameString in operand/statement position: a method invocation (consuming
|
||||||
|
/// the callee's declared argument count) or a plain name reference.
|
||||||
|
fn nameInvocation(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
if (self.namespace.resolve(scope, name_path.rooted, name_path.parents, name_path.slice())) |node| {
|
||||||
|
if ((node.kind == .method or node.kind == .external) and node.arg_count > 0) {
|
||||||
|
try self.args(scope, node.arg_count);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Skip a PkgLength-delimited body wholesale (Buffer / Package / VarPackage):
|
||||||
|
/// the contents are pure data, never namespace declarations.
|
||||||
|
fn skipPackage(self: *Parser) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const len = try self.readPackageLength();
|
||||||
|
const end = start + len;
|
||||||
|
if (end > self.aml.len) return error.Truncated;
|
||||||
|
self.position = end;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- namespace objects --------------------------------------------------
|
||||||
|
|
||||||
|
fn parseName(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
const value_start = self.position;
|
||||||
|
try self.object(scope); // the DataReferenceObject value
|
||||||
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .name);
|
||||||
|
node.value = self.aml[value_start..self.position];
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseAlias(self: *Parser, scope: *Node) Error!void {
|
||||||
|
_ = try self.readNameString(); // source
|
||||||
|
const name_path = try self.readNameString(); // the alias name
|
||||||
|
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .alias);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseMethod(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
const flags = try self.readByte();
|
||||||
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .method);
|
||||||
|
node.arg_count = flags & 0x7;
|
||||||
|
// Capture the body for on-demand evaluation and skip it — objects declared
|
||||||
|
// inside a method are created at *runtime*, not at load, so they must not
|
||||||
|
// become permanent namespace nodes.
|
||||||
|
node.value = self.aml[self.position..@min(end, self.aml.len)];
|
||||||
|
self.position = end;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseExternal(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
_ = try self.readByte(); // object type
|
||||||
|
const arg_count = try self.readByte();
|
||||||
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .external);
|
||||||
|
node.arg_count = arg_count;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Scope / Device / ThermalZone: PkgLength, NameString, then a nested TermList.
|
||||||
|
fn parseScopeLike(self: *Parser, scope: *Node, kind: NodeKind) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), kind);
|
||||||
|
self.termList(end, node);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseProcessor(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
try self.skip(6); // ProcID(byte) + PblkAddress(dword) + PblkLen(byte)
|
||||||
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .processor);
|
||||||
|
self.termList(end, node);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parsePowerResource(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
try self.skip(3); // SystemLevel(byte) + ResourceOrder(word)
|
||||||
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .power_resource);
|
||||||
|
self.termList(end, node);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// OperationRegion: NameString, RegionSpace(byte), Offset(TermArg), Len(TermArg).
|
||||||
|
/// The offset/length expressions are kept as AML for lazy evaluation.
|
||||||
|
fn parseRegion(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
const space = try self.readByte();
|
||||||
|
const off_start = self.position;
|
||||||
|
try self.object(scope);
|
||||||
|
const off_end = self.position;
|
||||||
|
try self.object(scope);
|
||||||
|
const len_end = self.position;
|
||||||
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .region);
|
||||||
|
node.region_space = space;
|
||||||
|
node.region_offset_aml = self.aml[off_start..off_end];
|
||||||
|
node.region_len_aml = self.aml[off_end..len_end];
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseDataRegion(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
try self.args(scope, 3); // signature, oem id, oem table id (TermArgs)
|
||||||
|
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .region);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseMutex(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
try self.skip(1); // sync flags
|
||||||
|
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .mutex);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseEvent(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .event);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// CreateXField: `count` TermArgs then the new field's NameString.
|
||||||
|
fn parseCreateField(self: *Parser, scope: *Node, count: usize) Error!void {
|
||||||
|
try self.args(scope, count);
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .name);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Field / IndexField / BankField: a region/bank reference, flags, then a
|
||||||
|
/// FieldList whose NamedFields become nodes in the current scope. For a plain
|
||||||
|
/// Field, the first NameString is the backing region — captured so field units
|
||||||
|
/// carry a region + bit position the evaluator can read/write.
|
||||||
|
fn parseField(self: *Parser, scope: *Node, name_strings: u8, bank: bool) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
var region: ?*Node = null;
|
||||||
|
var i: u8 = 0;
|
||||||
|
while (i < name_strings) : (i += 1) {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
// Only a plain Field's single NameString denotes an OperationRegion.
|
||||||
|
if (name_strings == 1) region = self.namespace.resolve(scope, name_path.rooted, name_path.parents, name_path.slice());
|
||||||
|
}
|
||||||
|
if (bank) try self.object(scope); // bank value TermArg
|
||||||
|
const flags = try self.readByte();
|
||||||
|
self.fieldList(end, scope, region, flags & 0x0F);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn fieldList(self: *Parser, end: usize, scope: *Node, region: ?*Node, initial_access: u8) void {
|
||||||
|
var bit_offset: u32 = 0;
|
||||||
|
var access = initial_access;
|
||||||
|
while (self.position < end) {
|
||||||
|
const lead = self.peek() orelse break;
|
||||||
|
switch (lead) {
|
||||||
|
0x00 => { // ReservedField: advances the bit position
|
||||||
|
self.position += 1;
|
||||||
|
const width = self.readPackageLength() catch break;
|
||||||
|
bit_offset += @intCast(width);
|
||||||
|
},
|
||||||
|
0x01 => { // AccessField: AccessType (low nibble) + AccessAttrib
|
||||||
|
self.position += 1;
|
||||||
|
const at = self.readByte() catch break;
|
||||||
|
self.skip(1) catch break;
|
||||||
|
access = at & 0x0F;
|
||||||
|
},
|
||||||
|
0x02 => { // ConnectField: NameString | BufferData
|
||||||
|
self.position += 1;
|
||||||
|
self.object(scope) catch break;
|
||||||
|
},
|
||||||
|
0x03 => { // ExtendedAccessField: type + attrib + length
|
||||||
|
self.position += 1;
|
||||||
|
self.skip(3) catch break;
|
||||||
|
},
|
||||||
|
else => { // NamedField: NameSegment + PkgLength (bit width)
|
||||||
|
const segment = self.readNameSegment() catch break;
|
||||||
|
const width = self.readPackageLength() catch break;
|
||||||
|
const unit = self.namespace.newFieldUnit(scope, segment) catch break;
|
||||||
|
unit.region = region;
|
||||||
|
unit.bit_offset = bit_offset;
|
||||||
|
unit.bit_width = @intCast(width);
|
||||||
|
unit.access_type = access;
|
||||||
|
bit_offset += @intCast(width);
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
self.position = end;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- control flow -------------------------------------------------------
|
||||||
|
|
||||||
|
fn parseIf(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
try self.object(scope); // predicate
|
||||||
|
self.termList(end, scope);
|
||||||
|
if (self.peek() == opcode.else_opcode) {
|
||||||
|
self.position += 1;
|
||||||
|
try self.parseElse(scope);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseElse(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
self.termList(end, scope);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseWhile(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
try self.object(scope); // predicate
|
||||||
|
self.termList(end, scope);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseLnot(self: *Parser, scope: *Node) Error!void {
|
||||||
|
// 0x92 followed by 0x93/94/95 is a compound comparison (two operands);
|
||||||
|
// otherwise it is a plain LNot of one operand.
|
||||||
|
const b = self.peek() orelse return error.Truncated;
|
||||||
|
switch (b) {
|
||||||
|
opcode.lnot.not_equal, opcode.lnot.less_equal, opcode.lnot.greater_equal => {
|
||||||
|
self.position += 1;
|
||||||
|
try self.args(scope, 2);
|
||||||
|
},
|
||||||
|
else => try self.object(scope),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseMatch(self: *Parser, scope: *Node) Error!void {
|
||||||
|
try self.object(scope); // search package
|
||||||
|
try self.skip(1); // match opcode 1
|
||||||
|
try self.object(scope); // operand 1
|
||||||
|
try self.skip(1); // match opcode 2
|
||||||
|
try self.object(scope); // operand 2
|
||||||
|
try self.object(scope); // start index
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- extended opcodes (0x5B xx) -----------------------------------------
|
||||||
|
|
||||||
|
fn parseExtended(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const e = try self.readByte();
|
||||||
|
switch (e) {
|
||||||
|
opcode.extended.mutex => try self.parseMutex(scope),
|
||||||
|
opcode.extended.event => try self.parseEvent(scope),
|
||||||
|
opcode.extended.operation_region => try self.parseRegion(scope),
|
||||||
|
opcode.extended.data_region => try self.parseDataRegion(scope),
|
||||||
|
opcode.extended.field => try self.parseField(scope, 1, false),
|
||||||
|
opcode.extended.index_field => try self.parseField(scope, 2, false),
|
||||||
|
opcode.extended.bank_field => try self.parseField(scope, 2, true),
|
||||||
|
opcode.extended.device => try self.parseScopeLike(scope, .device),
|
||||||
|
opcode.extended.thermal_zone => try self.parseScopeLike(scope, .thermal_zone),
|
||||||
|
opcode.extended.processor => try self.parseProcessor(scope),
|
||||||
|
opcode.extended.power_resource => try self.parsePowerResource(scope),
|
||||||
|
|
||||||
|
opcode.extended.conditional_reference_of => try self.args(scope, 2), // SuperName, Target
|
||||||
|
opcode.extended.create_field => try self.parseCreateField(scope, 3),
|
||||||
|
opcode.extended.load_table => try self.args(scope, 6),
|
||||||
|
opcode.extended.load => try self.args(scope, 2), // NameString, Target
|
||||||
|
opcode.extended.stall, opcode.extended.sleep => try self.args(scope, 1),
|
||||||
|
opcode.extended.acquire => {
|
||||||
|
try self.object(scope); // mutex SuperName
|
||||||
|
try self.skip(2); // timeout WordData
|
||||||
|
},
|
||||||
|
opcode.extended.signal, opcode.extended.reset, opcode.extended.release, opcode.extended.unload => try self.args(scope, 1),
|
||||||
|
opcode.extended.wait => try self.args(scope, 2),
|
||||||
|
opcode.extended.from_bcd, opcode.extended.to_bcd => try self.args(scope, 2),
|
||||||
|
opcode.extended.fatal => {
|
||||||
|
try self.skip(5); // Type(byte) + Code(dword)
|
||||||
|
try self.object(scope); // Arg TermArg
|
||||||
|
},
|
||||||
|
opcode.extended.revision, opcode.extended.debug, opcode.extended.timer => {},
|
||||||
|
|
||||||
|
else => return error.Malformed,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
fn isNameStart(b: u8) bool {
|
||||||
|
return (b >= opcode.name_char_start and b <= opcode.name_char_end) or
|
||||||
|
b == opcode.name_char_underscore or
|
||||||
|
b == opcode.root_char or
|
||||||
|
b == opcode.parent_prefix_char or
|
||||||
|
b == opcode.dual_name_prefix or
|
||||||
|
b == opcode.multi_name_prefix;
|
||||||
|
}
|
||||||
@@ -0,0 +1,95 @@
|
|||||||
|
//! The **device ABI**: the flat, `extern` device types that cross the system_call
|
||||||
|
//! boundary — what `device_enumerate` hands a user-space driver, what
|
||||||
|
//! `device_register` takes back. This is the devices sub-project's *public
|
||||||
|
//! interface*, exposed as its own `device-abi` module the same way the VFS server
|
||||||
|
//! exposes `vfs-protocol` — so both the kernel and user space depend on the contract
|
||||||
|
//! by name, and neither reaches into the other's files.
|
||||||
|
//!
|
||||||
|
//! It is also the **single source of truth** for `DeviceClass` and `ResourceKind`:
|
||||||
|
//! the kernel's rich, pointer-based device tree (system/devices/device-model.zig,
|
||||||
|
//! which user space must never import) re-exports these, so the enum that a driver
|
||||||
|
//! matches on and the enum the kernel classifies with are the *same* type — no
|
||||||
|
//! hand-kept "mirror in order" to drift. The core kernel↔user ABI is [[abi]]; the
|
||||||
|
//! loader↔kernel handoff is [[boot-handoff]].
|
||||||
|
|
||||||
|
/// A coarse classification of a device, independent of the describing firmware.
|
||||||
|
/// Kept small on purpose; refine as real drivers arrive. `enum(u32)` because the
|
||||||
|
/// `@intFromEnum` value crosses the system_call boundary in `DeviceDescriptor.class`.
|
||||||
|
pub const DeviceClass = enum(u32) {
|
||||||
|
/// The synthetic root every discovered device hangs beneath.
|
||||||
|
root,
|
||||||
|
processor,
|
||||||
|
interrupt_controller,
|
||||||
|
timer,
|
||||||
|
/// A PCI(e) host bridge — the root of a PCI segment (owns an ECAM window).
|
||||||
|
pci_host_bridge,
|
||||||
|
/// A single PCI function.
|
||||||
|
pci_device,
|
||||||
|
/// A device named in the ACPI namespace (from the DSDT/SSDT), carrying a
|
||||||
|
/// hardware ID (`_HID`) and, where static, current resource settings (`_CRS`).
|
||||||
|
acpi_device,
|
||||||
|
/// The ACPI tables themselves, published as one node for the user-space acpi
|
||||||
|
/// service (docs/discovery.md): memory resources over the AML blobs,
|
||||||
|
/// a broad io_port grant for OperationRegion access, and the SCI interrupt.
|
||||||
|
/// The one node whose claimant is trusted to run firmware bytecode.
|
||||||
|
acpi_tables,
|
||||||
|
/// One interface of a USB device, registered by the xHCI bus driver. It owns
|
||||||
|
/// no MMIO — it is reached through its controller — so it carries no
|
||||||
|
/// resources; the (class, subclass, protocol) triple that says what it is
|
||||||
|
/// travels in the bus report's identity, not here.
|
||||||
|
usb_device,
|
||||||
|
unknown,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The kind of hardware resource a device occupies. `enum(u32)` for the same
|
||||||
|
/// boundary-crossing reason as `DeviceClass` (see `ResourceDescriptor.kind`).
|
||||||
|
pub const ResourceKind = enum(u32) {
|
||||||
|
/// A memory-mapped I/O window: `start` is the physical base, `len` its size.
|
||||||
|
memory,
|
||||||
|
/// A legacy I/O-port range: `start` is the first port, `len` the count.
|
||||||
|
io_port,
|
||||||
|
/// An interrupt: `start` is the global system interrupt (GSI), `len` is 1.
|
||||||
|
irq,
|
||||||
|
/// A range of bus numbers owned by a bridge: `start`..`start+len`.
|
||||||
|
bus_range,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One device resource, as handed to a user-space driver (flat, extern).
|
||||||
|
pub const ResourceDescriptor = extern struct {
|
||||||
|
kind: u64, // a ResourceKind value
|
||||||
|
start: u64,
|
||||||
|
len: u64,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const maximum_device_resources = 8;
|
||||||
|
|
||||||
|
/// `DeviceDescriptor.parent` for a device with no parent — a root of the device tree.
|
||||||
|
pub const no_parent: u64 = ~@as(u64, 0);
|
||||||
|
|
||||||
|
/// `DeviceDescriptor.pci_class` for a device that is not a PCI function. (Zero would be
|
||||||
|
/// ambiguous: 0x000000 is a real class code, "unclassified device".)
|
||||||
|
pub const no_pci_class: u64 = ~@as(u64, 0);
|
||||||
|
|
||||||
|
/// A device, as snapshotted for user space by `device_enumerate`. A driver scans
|
||||||
|
/// these to find the hardware it owns, claims it, and maps its MMIO.
|
||||||
|
///
|
||||||
|
/// `parent` makes the table a tree rather than a list, which is what a **bus driver**
|
||||||
|
/// needs: it claims the bus, finds the devices below it, and publishes any it
|
||||||
|
/// discovers itself with `device_register`. A registered child's resources must lie
|
||||||
|
/// within its parent's (the kernel enforces this) — that containment is what makes
|
||||||
|
/// delegation safe, since a device descriptor is otherwise a licence to map physical
|
||||||
|
/// memory.
|
||||||
|
pub const DeviceDescriptor = extern struct {
|
||||||
|
id: u64,
|
||||||
|
parent: u64, // a device id, or `no_parent`
|
||||||
|
class: u64, // a DeviceClass value
|
||||||
|
// The PCI class/subclass/prog-IF triple packed as 0xCCSSPP when this device is a PCI
|
||||||
|
// function, or `no_pci_class` otherwise. This is how a manager tells *what* a
|
||||||
|
// `pci_device` is (an xHCI controller, an AHCI controller) — decode the triple into
|
||||||
|
// names with the pci-class module.
|
||||||
|
pci_class: u64,
|
||||||
|
hid_len: u64,
|
||||||
|
resource_count: u64,
|
||||||
|
hid: [8]u8,
|
||||||
|
resources: [maximum_device_resources]ResourceDescriptor,
|
||||||
|
};
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user