Compare commits
131
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
446f655c69 | ||
|
|
a785efa4a3 | ||
|
|
767a2a9a7c | ||
|
|
1f2c60b3ec | ||
|
|
dfc7d6a609 | ||
|
|
738f6aa697 | ||
|
|
01e56e3f36 | ||
|
|
d5d15cefcb | ||
|
|
fd96a35eb9 | ||
|
|
e3fe3f3f45 | ||
|
|
60da667b42 | ||
|
|
36145e623b | ||
|
|
565415327d | ||
|
|
bf6bdb389d | ||
|
|
0628944b15 | ||
|
|
e6d0bb7ef0 | ||
|
|
5ca804d827 | ||
|
|
a299363b59 | ||
|
|
d8dd62c639 | ||
|
|
d106b6e8dc | ||
|
|
af2c766f42 | ||
|
|
d26262bf56 | ||
|
|
10b89c06ff | ||
|
|
a2a05d0b3d | ||
|
|
75d62660b0 | ||
|
|
a53c2b0193 | ||
|
|
bf481c080c | ||
|
|
3a78dcab3f | ||
|
|
470f93a83d | ||
|
|
7798706b41 | ||
|
|
ad40de03c2 | ||
|
|
d8778b4b70 | ||
|
|
79d859a111 | ||
|
|
37fb09f75e | ||
|
|
34ebeb968d | ||
|
|
3cc1d38dd0 | ||
|
|
36e804b848 | ||
|
|
be83a42d42 | ||
|
|
650a1b1595 | ||
|
|
d8c55c6f2f | ||
|
|
2ebfb0c3b0 | ||
|
|
888eaa74e1 | ||
|
|
ed76cbbc79 | ||
|
|
140229b88d | ||
|
|
cb2379fd06 | ||
|
|
1665b239b0 | ||
|
|
70ed0337f8 | ||
|
|
116b8f6c41 | ||
|
|
77901bbba6 | ||
|
|
78582d24d2 | ||
|
|
1cdffe21b1 | ||
|
|
4df90bc212 | ||
|
|
713e77354b | ||
|
|
abb7b1b634 | ||
|
|
f5f0e15769 | ||
|
|
8652b4a724 | ||
|
|
e8233127c7 | ||
|
|
88e92254e9 | ||
|
|
5725d35e5b | ||
|
|
8c95525793 | ||
|
|
5bba5d3363 | ||
|
|
80b72db676 | ||
|
|
aa0c97353a | ||
|
|
dd93204b44 | ||
|
|
c7b17aaa0e | ||
|
|
d7a154a596 | ||
|
|
1bf91115dd | ||
|
|
65244e3103 | ||
|
|
75ccfff171 | ||
|
|
2a583d55a8 | ||
|
|
d218d93f79 | ||
|
|
a5fe63c1dd | ||
|
|
6b3ae0c997 | ||
|
|
59104dd988 | ||
|
|
e499f500c3 | ||
|
|
2ab0d129a2 | ||
|
|
d702d2e9ae | ||
|
|
3ec2d1828a | ||
|
|
21657943a8 | ||
|
|
e612d948d2 | ||
|
|
4ef21fa083 | ||
|
|
125a3b4993 | ||
|
|
e7c7e7b94c | ||
|
|
a581712b09 | ||
|
|
8eb4210251 | ||
|
|
56110b0019 | ||
|
|
afbf10f7fc | ||
|
|
b61b7775b9 | ||
|
|
be81394be3 | ||
|
|
47610e8ee2 | ||
|
|
193fd71a50 | ||
|
|
1c2b3ae64d | ||
|
|
d19a0ae38d | ||
|
|
3d1de37d0e | ||
|
|
ceacc6b514 | ||
|
|
8754d4e46a | ||
|
|
15b70856c9 | ||
|
|
83881641ca | ||
|
|
b0f894f50c | ||
|
|
750a73f050 | ||
|
|
eabb81684a | ||
|
|
0a8c81b17a | ||
|
|
9316f9f1c3 | ||
|
|
7384a730be | ||
|
|
b9d9e1e523 | ||
|
|
37fb3cb0cf | ||
|
|
5d57e7b01c | ||
|
|
8a7235c725 | ||
|
|
f4590fcc19 | ||
|
|
bb7597ea0b | ||
|
|
f57a73e8a1 | ||
|
|
724de7bbd0 | ||
|
|
75bc429aa1 | ||
|
|
91aff2bc4a | ||
|
|
546dd44a2a | ||
|
|
7501bd1703 | ||
|
|
1b47de5058 | ||
|
|
26ac97df31 | ||
|
|
37f72a4df8 | ||
|
|
f02259cae0 | ||
|
|
5940864958 | ||
|
|
91f2cfa17b | ||
|
|
debe815a5c | ||
|
|
dba3939a0f | ||
|
|
43afe6bf2e | ||
|
|
ed7f542006 | ||
|
|
36c29d2d6d | ||
|
|
941ab091db | ||
|
|
cf7c6df41c | ||
|
|
26d2f5259c | ||
|
|
cb63d2e31b |
@@ -0,0 +1,16 @@
|
|||||||
|
# EditorConfig: https://editorconfig.org/
|
||||||
|
# Follows the Zig style guide: https://ziglang.org/documentation/0.16.0/#Style-Guide
|
||||||
|
|
||||||
|
root = true
|
||||||
|
|
||||||
|
[*]
|
||||||
|
charset = utf-8
|
||||||
|
end_of_line = lf
|
||||||
|
indent_style = space
|
||||||
|
indent_size = 4
|
||||||
|
trim_trailing_whitespace = true
|
||||||
|
insert_final_newline = true
|
||||||
|
|
||||||
|
[*.zig]
|
||||||
|
# "Line length: aim for 100; use common sense."
|
||||||
|
max_line_length = 100
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
*.zig text eol=lf
|
||||||
@@ -4,3 +4,6 @@ zig-out/
|
|||||||
|
|
||||||
# JetBrains IDE
|
# JetBrains IDE
|
||||||
.idea/
|
.idea/
|
||||||
|
|
||||||
|
.claude/
|
||||||
|
.github/
|
||||||
@@ -2,12 +2,31 @@
|
|||||||
Codename: Shodan
|
Codename: Shodan
|
||||||
Version: 1
|
Version: 1
|
||||||
|
|
||||||
A small operating system, written from scratch in Zig — a bootloader (`src/boot/`)
|
A small resilient operating system, written from scratch in Zig.
|
||||||
and a microkernel (`src/kernel/`), sharing a neutral handoff contract (`src/root.zig`).
|
|
||||||
It boots x86-64 via UEFI, and so far has a framebuffer console, a physical frame
|
## Zen of DanOS:
|
||||||
allocator, its own paging with W^X permissions, interrupt/exception handling, a
|
|
||||||
LAPIC timer, a kernel heap, a fixed-priority preemptive scheduler, and in-kernel IPC
|
- Resilient Micro-Kernel Architecture.
|
||||||
channels. See [`docs/`](docs/README.md) for how each piece works.
|
- Every process run in an isolated user space not kernel space.
|
||||||
|
- Processes cannot take down the entire OS with it when they die or is killed
|
||||||
|
- Stable public runtime library, private OS ABI.
|
||||||
|
- Keeps a stable runtime for user space processes between OS versions (great for backwards compatibility)
|
||||||
|
- Allows the underlying OS to be changed without effecting applications
|
||||||
|
- Provides a boundary to enable compatibility between OS's e.g. POSIX, MUSL etc
|
||||||
|
- Drivers are just isolated processes in user space.
|
||||||
|
- Thin binaries that can be restarted like applications.
|
||||||
|
- Useful during driver development.
|
||||||
|
- Drivers can claim MMIO / ports
|
||||||
|
- Driver resources (e.g. IRQ/Port/MMIO) claims are automatically cleaned up if the driver dies or is killed
|
||||||
|
- Drivers can also hook into the process lifecyle to clean up or reset hardware
|
||||||
|
- No legacy to deal with
|
||||||
|
- Zig code uses a clean coding style (Zen of Zig)
|
||||||
|
- Favor reading code over writing code.
|
||||||
|
- No magic numbers.
|
||||||
|
- No shortend names unless its for ABI compatibility or acronyms
|
||||||
|
- Inter-Process Communication (IPC)
|
||||||
|
- Publish and subscribe to Asynchronous Messages
|
||||||
|
- Talk to services and processes synchronously
|
||||||
|
|
||||||
## Prerequisites
|
## Prerequisites
|
||||||
|
|
||||||
@@ -30,8 +49,10 @@ channels. See [`docs/`](docs/README.md) for how each piece works.
|
|||||||
zig build
|
zig build
|
||||||
```
|
```
|
||||||
|
|
||||||
Produces the UEFI bootloader (`zig-out/bin/BOOTX64.efi`) and the kernel ELF
|
Produces a FHS-shaped `zig-out/` that *is* the danos filesystem and the boot volume:
|
||||||
(`zig-out/bin/kernel`).
|
the UEFI bootloader at `zig-out/EFI/BOOT/BOOTX64.efi`, the kernel at
|
||||||
|
`zig-out/system/kernel`, init at `zig-out/system/services/init`, drivers under
|
||||||
|
`zig-out/system/drivers/`, and the initial-ramdisk at `zig-out/boot/`.
|
||||||
|
|
||||||
## Run
|
## Run
|
||||||
|
|
||||||
@@ -58,7 +79,7 @@ straight into CI.
|
|||||||
|
|
||||||
## Documentation
|
## Documentation
|
||||||
|
|
||||||
Design notes explaining the *why* behind the code live in
|
Design notes explaining *why* behind the code live in
|
||||||
[`docs/`](docs/README.md) — start with [`docs/README.md`](docs/README.md).
|
[`docs/`](docs/README.md) — start with [`docs/README.md`](docs/README.md).
|
||||||
|
|
||||||
## Logo
|
## Logo
|
||||||
|
|||||||
+274
-53
@@ -1,15 +1,24 @@
|
|||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const uefi = std.os.uefi;
|
const uefi = std.os.uefi;
|
||||||
const elf = std.elf;
|
const elf = std.elf;
|
||||||
const danos = @import("danos");
|
const boot_handoff = @import("boot-handoff");
|
||||||
const BootInfo = danos.BootInfo;
|
const BootInformation = boot_handoff.BootInformation;
|
||||||
const GraphicsOutput = uefi.protocol.GraphicsOutput;
|
const GraphicsOutput = uefi.protocol.GraphicsOutput;
|
||||||
const EdidActive = uefi.protocol.edid.Active;
|
const EdidActive = uefi.protocol.edid.Active;
|
||||||
const MemoryMapSlice = uefi.tables.MemoryMapSlice;
|
const MemoryMapSlice = uefi.tables.MemoryMapSlice;
|
||||||
|
|
||||||
/// Name of the kernel ELF on the boot volume (installed to the ESP root by
|
// The boot volume is the FHS-shaped zig-out (see build.zig / docs/README.md), so the
|
||||||
/// build.zig). UEFI wants a UTF-16, null-terminated path.
|
// loader reads each artifact from its addressed FHS path. UEFI paths use backslashes;
|
||||||
const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("kernel");
|
// the FAT driver walks the components itself, so no per-directory dance is needed.
|
||||||
|
|
||||||
|
/// The kernel image: /system/kernel.
|
||||||
|
const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("system\\kernel");
|
||||||
|
|
||||||
|
/// The init program: /system/services/init.
|
||||||
|
const init_file_name = std.unicode.utf8ToUtf16LeStringLiteral("system\\services\\init");
|
||||||
|
|
||||||
|
/// The initial-ramdisk (the VFS server + drivers), in /boot.
|
||||||
|
const initial_ramdisk_file_name = std.unicode.utf8ToUtf16LeStringLiteral("boot\\initial-ramdisk.img");
|
||||||
|
|
||||||
/// Physical page size, and the sentinel UEFI uses to seek to end-of-file.
|
/// Physical page size, and the sentinel UEFI uses to seek to end-of-file.
|
||||||
const page_size = 4096;
|
const page_size = 4096;
|
||||||
@@ -33,8 +42,16 @@ fn boot() !noreturn {
|
|||||||
|
|
||||||
// Everything the kernel needs must be gathered *before* we exit boot
|
// Everything the kernel needs must be gathered *before* we exit boot
|
||||||
// services, since afterwards none of these calls are usable.
|
// services, since afterwards none of these calls are usable.
|
||||||
var boot_info: BootInfo = .{
|
var boot_information: BootInformation = .{
|
||||||
.framebuffer = try queryFramebuffer(bs),
|
// A missing GOP (a headless machine) is not fatal — hand the kernel a
|
||||||
|
// "no framebuffer" descriptor (base 0) and let it log to serial instead.
|
||||||
|
.framebuffer = queryFramebuffer(bs) catch boot_handoff.Framebuffer{
|
||||||
|
.base = 0,
|
||||||
|
.width = 0,
|
||||||
|
.height = 0,
|
||||||
|
.pitch = 0,
|
||||||
|
.format = .bgrx,
|
||||||
|
},
|
||||||
.memory_map = undefined, // filled by exitBootServices, just below
|
.memory_map = undefined, // filled by exitBootServices, just below
|
||||||
.kernel_segments = undefined, // filled by loadKernel
|
.kernel_segments = undefined, // filled by loadKernel
|
||||||
.kernel_segment_count = 0,
|
.kernel_segment_count = 0,
|
||||||
@@ -44,16 +61,38 @@ fn boot() !noreturn {
|
|||||||
.acpi_rsdp = if (acpiRootSystemDescriptorPointer()) |p| @intFromPtr(p) else 0,
|
.acpi_rsdp = if (acpiRootSystemDescriptorPointer()) |p| @intFromPtr(p) else 0,
|
||||||
};
|
};
|
||||||
|
|
||||||
const entry = try loadKernel(bs, &boot_info);
|
const entry = try loadKernel(bs, &boot_information);
|
||||||
|
|
||||||
|
// Best effort: a volume without /system/services/init still boots (kernel-only).
|
||||||
|
loadInit(bs, &boot_information) catch |err| {
|
||||||
|
log("danos: no /system/services/init (");
|
||||||
|
logBytes(@errorName(err));
|
||||||
|
log(") - booting without user space\r\n");
|
||||||
|
};
|
||||||
|
|
||||||
|
// Best effort: the initial_ramdisk (VFS server + drivers) is optional too.
|
||||||
|
loadInitialRamdisk(bs, &boot_information) catch |err| {
|
||||||
|
log("danos: no initial_ramdisk (");
|
||||||
|
logBytes(@errorName(err));
|
||||||
|
log(")\r\n");
|
||||||
|
};
|
||||||
|
|
||||||
|
// Build the page tables the kernel starts life on: identity + a physmap of
|
||||||
|
// low RAM, plus the higher-half kernel image once it links high. Allocated
|
||||||
|
// now, while boot services (and the memory map) are still stable — nothing
|
||||||
|
// is allocatable after ExitBootServices, and any allocation between fetching
|
||||||
|
// the map and exiting would invalidate the map key.
|
||||||
|
const cr3 = try buildBootstrapTables(bs, &boot_information);
|
||||||
|
|
||||||
log("danos: kernel loaded, exiting boot services\r\n");
|
log("danos: kernel loaded, exiting boot services\r\n");
|
||||||
boot_info.memory_map = try exitBootServices(bs);
|
boot_information.memory_map = try exitBootServices(bs);
|
||||||
|
|
||||||
// Hand control to the kernel. `danos.kernel_abi` is SysV, so the pointer is
|
// Switch onto our tables and jump to the kernel in one uninterruptible step.
|
||||||
// passed in RDI as the kernel expects — not RCX, which this UEFI binary's
|
// We load RDI explicitly (SystemV first arg) rather than trusting this UEFI
|
||||||
// default `.c` convention (Microsoft x64) would use.
|
// binary's Microsoft-x64 default, and jump straight to the (possibly
|
||||||
const kernel: *const fn (*const BootInfo) callconv(danos.kernel_abi) noreturn = @ptrFromInt(entry);
|
// higher-half) entry — the bootstrap tables map both the low loader code
|
||||||
kernel(&boot_info);
|
// executing this and the kernel's link address.
|
||||||
|
handoff(cr3, entry, &boot_information);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A display resolution in pixels.
|
/// A display resolution in pixels.
|
||||||
@@ -61,7 +100,7 @@ const Resolution = struct { width: u32, height: u32 };
|
|||||||
|
|
||||||
/// Switch the GPU to the monitor's native resolution (when we can determine it)
|
/// Switch the GPU to the monitor's native resolution (when we can determine it)
|
||||||
/// and read the resulting graphics mode into our own framebuffer description.
|
/// and read the resulting graphics mode into our own framebuffer description.
|
||||||
fn queryFramebuffer(bs: *uefi.tables.BootServices) !danos.Framebuffer {
|
fn queryFramebuffer(bs: *uefi.tables.BootServices) !boot_handoff.Framebuffer {
|
||||||
// Enumerate the handles carrying the Graphics Output Protocol. We go through
|
// Enumerate the handles carrying the Graphics Output Protocol. We go through
|
||||||
// handles (rather than locateProtocol) so we can also ask them for their EDID,
|
// handles (rather than locateProtocol) so we can also ask them for their EDID,
|
||||||
// which is what tells us the panel's native resolution.
|
// which is what tells us the panel's native resolution.
|
||||||
@@ -93,7 +132,7 @@ fn queryFramebuffer(bs: *uefi.tables.BootServices) !danos.Framebuffer {
|
|||||||
|
|
||||||
/// Map a GOP pixel format to ours. bit_mask / blt_only have no linear 32bpp
|
/// Map a GOP pixel format to ours. bit_mask / blt_only have no linear 32bpp
|
||||||
/// layout we can paint into, so they're rejected.
|
/// layout we can paint into, so they're rejected.
|
||||||
fn pixelFormat(fmt: GraphicsOutput.PixelFormat) !danos.PixelFormat {
|
fn pixelFormat(fmt: GraphicsOutput.PixelFormat) !boot_handoff.PixelFormat {
|
||||||
return switch (fmt) {
|
return switch (fmt) {
|
||||||
.red_green_blue_reserved_8_bit_per_color => .rgbx,
|
.red_green_blue_reserved_8_bit_per_color => .rgbx,
|
||||||
.blue_green_red_reserved_8_bit_per_color => .bgrx,
|
.blue_green_red_reserved_8_bit_per_color => .bgrx,
|
||||||
@@ -158,7 +197,7 @@ fn edidNative(edid: []const u8) ?Resolution {
|
|||||||
|
|
||||||
/// Open the kernel on the volume we booted from, read it into a pool buffer,
|
/// Open the kernel on the volume we booted from, read it into a pool buffer,
|
||||||
/// load its segments, and return the physical entry-point address.
|
/// load its segments, and return the physical entry-point address.
|
||||||
fn loadKernel(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !usize {
|
fn loadKernel(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !usize {
|
||||||
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
|
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
|
||||||
return error.NoLoadedImage;
|
return error.NoLoadedImage;
|
||||||
const device = loaded.device_handle orelse return error.NoBootDevice;
|
const device = loaded.device_handle orelse return error.NoBootDevice;
|
||||||
@@ -187,13 +226,190 @@ fn loadKernel(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !usize {
|
|||||||
read_total += n;
|
read_total += n;
|
||||||
}
|
}
|
||||||
|
|
||||||
return loadElf(bs, image, boot_info);
|
return loadElf(bs, image, boot_information);
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- bootstrap page tables -------------------------------------------------
|
||||||
|
// The kernel is (or will be) linked in the higher half but loaded low; the
|
||||||
|
// firmware's identity map doesn't cover the higher half, so the loader builds
|
||||||
|
// the first set of real page tables and switches CR3 before jumping in. They
|
||||||
|
// carry: an identity map of low RAM (so the loader's own code/stack executing
|
||||||
|
// the switch stays valid, and the low-linked kernel keeps working during the
|
||||||
|
// staged move), a physmap at boot_handoff.physmap_base (the kernel's permanent way to
|
||||||
|
// reach physical memory), and 4 KiB mappings of any higher-half kernel segment.
|
||||||
|
// The kernel later builds its own precise tables (paging.init) and abandons
|
||||||
|
// these; they leak as reserved LoaderData (~a handful of frames).
|
||||||
|
|
||||||
|
const pte_present: u64 = 1 << 0;
|
||||||
|
const pte_write: u64 = 1 << 1;
|
||||||
|
const pte_ps: u64 = 1 << 7; // page-size: a 2 MiB leaf at the PD level
|
||||||
|
const pte_address: u64 = 0x000F_FFFF_FFFF_F000;
|
||||||
|
const gib: u64 = 1 << 30;
|
||||||
|
|
||||||
|
/// A bump allocator over a pre-reserved block of zeroed frames, for page tables.
|
||||||
|
const TablePool = struct {
|
||||||
|
base: usize,
|
||||||
|
next: usize,
|
||||||
|
cap: usize,
|
||||||
|
|
||||||
|
fn alloc(self: *TablePool) !u64 {
|
||||||
|
if (self.next >= self.cap) return error.OutOfBootstrapFrames;
|
||||||
|
const frame = self.base + self.next * page_size;
|
||||||
|
self.next += 1;
|
||||||
|
@memset(@as(*[512]u64, @ptrFromInt(frame)), 0);
|
||||||
|
return frame;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn table(physical: u64) *[512]u64 {
|
||||||
|
return @ptrFromInt(physical);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Return the next-level table an entry points at, creating it if absent.
|
||||||
|
fn descend(self: *TablePool, entry: *u64) !u64 {
|
||||||
|
if (entry.* & pte_present != 0) return entry.* & pte_address;
|
||||||
|
const frame = try self.alloc();
|
||||||
|
entry.* = frame | pte_present | pte_write;
|
||||||
|
return frame;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn map2M(self: *TablePool, pml4: u64, virtual: u64, physical: u64) !void {
|
||||||
|
const pml4e = &table(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
|
const pdpt = try self.descend(pml4e);
|
||||||
|
const pdpte = &table(pdpt)[(virtual >> 30) & 0x1FF];
|
||||||
|
const pd = try self.descend(pdpte);
|
||||||
|
table(pd)[(virtual >> 21) & 0x1FF] = (physical & ~@as(u64, 0x1F_FFFF)) | pte_present | pte_write | pte_ps;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn map4K(self: *TablePool, pml4: u64, virtual: u64, physical: u64) !void {
|
||||||
|
const pml4e = &table(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
|
const pdpt = try self.descend(pml4e);
|
||||||
|
const pdpte = &table(pdpt)[(virtual >> 30) & 0x1FF];
|
||||||
|
const pd = try self.descend(pdpte);
|
||||||
|
const pde = &table(pd)[(virtual >> 21) & 0x1FF];
|
||||||
|
const pt = try self.descend(pde);
|
||||||
|
table(pt)[(virtual >> 12) & 0x1FF] = (physical & pte_address) | pte_present | pte_write;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Build the bootstrap tables and return the physical PML4 address (for CR3).
|
||||||
|
/// No NX bits are set anywhere, so EFER.NXE (still off here) is irrelevant.
|
||||||
|
fn buildBootstrapTables(bs: *uefi.tables.BootServices, boot_information: *const BootInformation) !u64 {
|
||||||
|
// 64 frames (256 KiB) — comfortably covers a PML4, two PDPTs, eight PDs for
|
||||||
|
// the 4 GiB identity+physmap ranges, plus the kernel image's PTs.
|
||||||
|
const pool_pages = 64;
|
||||||
|
const block = try bs.allocatePages(.any, .loader_data, pool_pages);
|
||||||
|
var pool = TablePool{ .base = @intFromPtr(block.ptr), .next = 0, .cap = pool_pages };
|
||||||
|
|
||||||
|
const pml4 = try pool.alloc();
|
||||||
|
|
||||||
|
// Identity + physmap for low RAM. 4 GiB covers all of QEMU's RAM and MMIO
|
||||||
|
// (LAPIC/IOAPIC/HPET/ECAM/framebuffer under q35); a machine with RAM or a
|
||||||
|
// framebuffer above 4 GiB would extend this — see the fb window below.
|
||||||
|
var address: u64 = 0;
|
||||||
|
while (address < 4 * gib) : (address += 2 << 20) {
|
||||||
|
try pool.map2M(pml4, address, address); // identity
|
||||||
|
try pool.map2M(pml4, boot_handoff.physicalToVirtual(address), address); // physmap
|
||||||
|
}
|
||||||
|
|
||||||
|
// A framebuffer above the 4 GiB window needs its own identity + physmap
|
||||||
|
// pages (the kernel touches fb.base before it builds its own tables).
|
||||||
|
const fb = boot_information.framebuffer;
|
||||||
|
if (fb.present() and fb.base + @as(u64, fb.pitch) * fb.height > 4 * gib) {
|
||||||
|
var p: u64 = fb.base & ~@as(u64, 0x1F_FFFF);
|
||||||
|
const fb_end = fb.base + @as(u64, fb.pitch) * fb.height;
|
||||||
|
while (p < fb_end) : (p += 2 << 20) {
|
||||||
|
try pool.map2M(pml4, p, p);
|
||||||
|
try pool.map2M(pml4, boot_handoff.physicalToVirtual(p), p);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Higher-half kernel segments (virtual != physical). While the kernel still links
|
||||||
|
// low its segments sit in the identity range and need no separate mapping
|
||||||
|
// (and 4 KiB-mapping them would collide with the 2 MiB identity leaves), so
|
||||||
|
// only map segments that actually live in the higher half.
|
||||||
|
for (boot_information.kernel_segments[0..boot_information.kernel_segment_count]) |seg| {
|
||||||
|
if (seg.virtual < boot_handoff.kernel_virt_base) continue;
|
||||||
|
var off: u64 = 0;
|
||||||
|
while (off < seg.pages * page_size) : (off += page_size) {
|
||||||
|
try pool.map4K(pml4, seg.virtual + off, seg.physical + off);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return pml4;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Switch onto `cr3` and jump to the kernel `entry` with `boot_information` in RDI,
|
||||||
|
/// interrupts off, in one block so nothing runs between the CR3 load and the
|
||||||
|
/// jump. The identity mapping keeps this low loader code valid across the CR3
|
||||||
|
/// load; the jump target is mapped (identity while low, higher-half once high).
|
||||||
|
fn handoff(cr3: u64, entry: usize, boot_information: *const BootInformation) noreturn {
|
||||||
|
asm volatile (
|
||||||
|
\\cli
|
||||||
|
\\movq %[cr3], %%cr3
|
||||||
|
\\movq %[bi], %%rdi
|
||||||
|
\\callq *%[entry]
|
||||||
|
:
|
||||||
|
: [cr3] "r" (cr3),
|
||||||
|
[bi] "r" (boot_information),
|
||||||
|
[entry] "r" (entry),
|
||||||
|
: .{ .memory = true });
|
||||||
|
unreachable;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read a whole file off the boot volume into a pool buffer that outlives the
|
||||||
|
/// loader. The buffer is deliberately NOT freed: it's LoaderData, which the
|
||||||
|
/// memory-map conversion classifies as reserved, so the kernel identity-maps it
|
||||||
|
/// and reads from there. Returns the buffer (pointer + length).
|
||||||
|
fn loadFile(bs: *uefi.tables.BootServices, name: [*:0]const u16) ![]u8 {
|
||||||
|
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
|
||||||
|
return error.NoLoadedImage;
|
||||||
|
const device = loaded.device_handle orelse return error.NoBootDevice;
|
||||||
|
const fs = (try bs.handleProtocol(uefi.protocol.SimpleFileSystem, device)) orelse
|
||||||
|
return error.NoFileSystem;
|
||||||
|
|
||||||
|
const root = try fs.openVolume();
|
||||||
|
defer _ = root.close() catch {};
|
||||||
|
|
||||||
|
const file = try root.open(name, .read, .{});
|
||||||
|
defer _ = file.close() catch {};
|
||||||
|
|
||||||
|
try file.setPosition(seek_end);
|
||||||
|
const size: usize = @intCast(try file.getPosition());
|
||||||
|
try file.setPosition(0);
|
||||||
|
if (size == 0) return error.EmptyFile;
|
||||||
|
|
||||||
|
const image = try bs.allocatePool(.loader_data, size); // survives the handoff
|
||||||
|
|
||||||
|
var read_total: usize = 0;
|
||||||
|
while (read_total < size) {
|
||||||
|
const n = try file.read(image[read_total..]);
|
||||||
|
if (n == 0) return error.UnexpectedEof;
|
||||||
|
read_total += n;
|
||||||
|
}
|
||||||
|
return image[0..size];
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Ferry the init program (/system/services/init) to the kernel. The kernel does the ELF
|
||||||
|
/// loading itself (into ring-3 mappings) — the loader just carries the bytes.
|
||||||
|
fn loadInit(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||||
|
const image = try loadFile(bs, init_file_name);
|
||||||
|
boot_information.init_base = @intFromPtr(image.ptr);
|
||||||
|
boot_information.init_len = image.len;
|
||||||
|
log("danos: /system/services/init loaded\r\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Ferry the initial_ramdisk (the VFS server + drivers) to the kernel, same as init.
|
||||||
|
fn loadInitialRamdisk(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||||
|
const image = try loadFile(bs, initial_ramdisk_file_name);
|
||||||
|
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
||||||
|
boot_information.initial_ramdisk_len = image.len;
|
||||||
|
log("danos: initial_ramdisk loaded\r\n");
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
||||||
/// record each segment's layout so the kernel can re-map itself with the right
|
/// record each segment's layout so the kernel can re-map itself with the right
|
||||||
/// permissions.
|
/// permissions.
|
||||||
fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !usize {
|
fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_information: *BootInformation) !usize {
|
||||||
if (image.len < @sizeOf(elf.Elf64_Ehdr)) return error.NotElf;
|
if (image.len < @sizeOf(elf.Elf64_Ehdr)) return error.NotElf;
|
||||||
const ehdr: *const elf.Elf64_Ehdr = @ptrCast(@alignCast(image.ptr));
|
const ehdr: *const elf.Elf64_Ehdr = @ptrCast(@alignCast(image.ptr));
|
||||||
|
|
||||||
@@ -210,7 +426,9 @@ fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !us
|
|||||||
|
|
||||||
// Reserve the exact physical pages this segment is linked at. This
|
// Reserve the exact physical pages this segment is linked at. This
|
||||||
// requires the segment's p_paddr to be free in the firmware memory map;
|
// requires the segment's p_paddr to be free in the firmware memory map;
|
||||||
// if it collides, adjust `image_base` in build.zig.
|
// if it collides, adjust `image_base` in build.zig. (Once the kernel
|
||||||
|
// links high — M2 step 4 — p_paddr becomes a separate low load address
|
||||||
|
// via the linker's AT(), and this stays a valid physical allocation.)
|
||||||
const mem_sz: usize = @intCast(phdr.p_memsz);
|
const mem_sz: usize = @intCast(phdr.p_memsz);
|
||||||
const pages = (mem_sz + page_size - 1) / page_size;
|
const pages = (mem_sz + page_size - 1) / page_size;
|
||||||
const dest: [*]align(page_size) uefi.Page = @ptrFromInt(phdr.p_paddr);
|
const dest: [*]align(page_size) uefi.Page = @ptrFromInt(phdr.p_paddr);
|
||||||
@@ -223,15 +441,18 @@ fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !us
|
|||||||
@memcpy(bytes[0..file_sz], image[off..][0..file_sz]);
|
@memcpy(bytes[0..file_sz], image[off..][0..file_sz]);
|
||||||
@memset(bytes[file_sz..mem_sz], 0);
|
@memset(bytes[file_sz..mem_sz], 0);
|
||||||
|
|
||||||
// Record it (identity-loaded: virtual == physical) for the kernel's VMM.
|
// Record the virtual link address and the physical load address so the
|
||||||
const n = boot_info.kernel_segment_count;
|
// kernel can map itself with the right permissions post-switch. They're
|
||||||
if (n < boot_info.kernel_segments.len) {
|
// equal while the kernel links low; they diverge once it links high.
|
||||||
boot_info.kernel_segments[n] = .{
|
const n = boot_information.kernel_segment_count;
|
||||||
.virt = phdr.p_vaddr,
|
if (n < boot_information.kernel_segments.len) {
|
||||||
|
boot_information.kernel_segments[n] = .{
|
||||||
|
.virtual = phdr.p_vaddr,
|
||||||
|
.physical = phdr.p_paddr,
|
||||||
.pages = pages,
|
.pages = pages,
|
||||||
.flags = phdr.p_flags,
|
.flags = phdr.p_flags,
|
||||||
};
|
};
|
||||||
boot_info.kernel_segment_count = n + 1;
|
boot_information.kernel_segment_count = n + 1;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -242,27 +463,27 @@ fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !us
|
|||||||
/// neutral form. Allocating the buffers can itself change the map (invalidating
|
/// neutral form. Allocating the buffers can itself change the map (invalidating
|
||||||
/// the key), so retry until it takes. Both buffers are LoaderData, which survives
|
/// the key), so retry until it takes. Both buffers are LoaderData, which survives
|
||||||
/// the exit, so the returned map stays valid for the kernel.
|
/// the exit, so the returned map stays valid for the kernel.
|
||||||
fn exitBootServices(bs: *uefi.tables.BootServices) !danos.MemoryMap {
|
fn exitBootServices(bs: *uefi.tables.BootServices) !boot_handoff.MemoryMap {
|
||||||
var attempts: usize = 0;
|
var attempts: usize = 0;
|
||||||
while (attempts < 8) : (attempts += 1) {
|
while (attempts < 8) : (attempts += 1) {
|
||||||
const info = try bs.getMemoryMapInfo();
|
const info = try bs.getMemoryMapInfo();
|
||||||
// Spare descriptors to absorb the growth from the allocations below.
|
// Spare descriptors to absorb the growth from the allocations below.
|
||||||
const cap = info.len + 8;
|
const cap = info.len + 8;
|
||||||
const map_buf = try bs.allocatePool(.loader_data, cap * info.descriptor_size);
|
const map_buffer = try bs.allocatePool(.loader_data, cap * info.descriptor_size);
|
||||||
const regions_buf = try bs.allocatePool(.loader_data, cap * @sizeOf(danos.MemoryRegion));
|
const regions_buffer = try bs.allocatePool(.loader_data, cap * @sizeOf(boot_handoff.MemoryRegion));
|
||||||
const map = bs.getMemoryMap(map_buf) catch {
|
const map = bs.getMemoryMap(map_buffer) catch {
|
||||||
_ = bs.freePool(map_buf.ptr) catch {};
|
_ = bs.freePool(map_buffer.ptr) catch {};
|
||||||
_ = bs.freePool(regions_buf.ptr) catch {};
|
_ = bs.freePool(regions_buffer.ptr) catch {};
|
||||||
continue;
|
continue;
|
||||||
};
|
};
|
||||||
bs.exitBootServices(uefi.handle, map.info.key) catch {
|
bs.exitBootServices(uefi.handle, map.info.key) catch {
|
||||||
_ = bs.freePool(map_buf.ptr) catch {};
|
_ = bs.freePool(map_buffer.ptr) catch {};
|
||||||
_ = bs.freePool(regions_buf.ptr) catch {};
|
_ = bs.freePool(regions_buffer.ptr) catch {};
|
||||||
continue;
|
continue;
|
||||||
};
|
};
|
||||||
// Boot services are gone; do not touch `bs` again. Converting the map is
|
// Boot services are gone; do not touch `bs` again. Converting the map is
|
||||||
// pure computation on memory we already hold, so it's safe here.
|
// pure computation on memory we already hold, so it's safe here.
|
||||||
return convertMemoryMap(map, regions_buf);
|
return convertMemoryMap(map, regions_buffer);
|
||||||
}
|
}
|
||||||
return error.ExitBootServicesFailed;
|
return error.ExitBootServicesFailed;
|
||||||
}
|
}
|
||||||
@@ -271,8 +492,8 @@ fn exitBootServices(bs: *uefi.tables.BootServices) !danos.MemoryMap {
|
|||||||
/// into `out` (sized for at least `map.info.len` regions). Adjacent regions of
|
/// into `out` (sized for at least `map.info.len` regions). Adjacent regions of
|
||||||
/// the same kind are coalesced. This is the loader's job precisely so the kernel
|
/// the same kind are coalesced. This is the loader's job precisely so the kernel
|
||||||
/// never sees UEFI's vocabulary — the same seam the framebuffer already uses.
|
/// never sees UEFI's vocabulary — the same seam the framebuffer already uses.
|
||||||
fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
fn convertMemoryMap(map: MemoryMapSlice, out: []u8) boot_handoff.MemoryMap {
|
||||||
const regions: [*]danos.MemoryRegion = @ptrCast(@alignCast(out.ptr));
|
const regions: [*]boot_handoff.MemoryRegion = @ptrCast(@alignCast(out.ptr));
|
||||||
// We're about to call boot-services memory `usable`, but our own stack lives
|
// We're about to call boot-services memory `usable`, but our own stack lives
|
||||||
// in it and the kernel starts out running on it. Keep the region holding the
|
// in it and the kernel starts out running on it. Keep the region holding the
|
||||||
// current stack pointer reserved so it's never handed out.
|
// current stack pointer reserved so it's never handed out.
|
||||||
@@ -289,16 +510,16 @@ fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
|||||||
if (d.number_of_pages == 0) continue;
|
if (d.number_of_pages == 0) continue;
|
||||||
var kind = classify(d);
|
var kind = classify(d);
|
||||||
// The descriptor we're executing on stays reserved (see rsp above).
|
// The descriptor we're executing on stays reserved (see rsp above).
|
||||||
const region_end = d.physical_start + d.number_of_pages * danos.page_size;
|
const region_end = d.physical_start + d.number_of_pages * page_size;
|
||||||
if (kind == .usable and rsp >= d.physical_start and rsp < region_end) kind = .reserved;
|
if (kind == .usable and rsp >= d.physical_start and rsp < region_end) kind = .reserved;
|
||||||
|
|
||||||
// Coalesce with the previous region if it's the same kind and contiguous.
|
// Coalesce with the previous region if it's the same kind and contiguous.
|
||||||
if (count > 0) {
|
if (count > 0) {
|
||||||
const prev = ®ions[count - 1];
|
const previous = ®ions[count - 1];
|
||||||
if (prev.kind == kind and
|
if (previous.kind == kind and
|
||||||
prev.base + prev.pages * danos.page_size == d.physical_start)
|
previous.base + previous.pages * page_size == d.physical_start)
|
||||||
{
|
{
|
||||||
prev.pages += d.number_of_pages;
|
previous.pages += d.number_of_pages;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -314,7 +535,7 @@ fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
|||||||
|
|
||||||
/// Map a UEFI descriptor to danos's neutral kind. A region that isn't
|
/// Map a UEFI descriptor to danos's neutral kind. A region that isn't
|
||||||
/// writeback-cacheable (`wb`) isn't backed by real RAM — it's device registers or
|
/// writeback-cacheable (`wb`) isn't backed by real RAM — it's device registers or
|
||||||
/// a reserved address-space window (e.g. PCIe config space) — so it's `mmio`
|
/// a reserved address-space window (e.g. PCIe configuration space) — so it's `mmio`
|
||||||
/// regardless of type. UEFI overloads `reserved_memory_type` for both reserved RAM
|
/// regardless of type. UEFI overloads `reserved_memory_type` for both reserved RAM
|
||||||
/// and such holes, and the cache attribute is what actually tells them apart.
|
/// and such holes, and the cache attribute is what actually tells them apart.
|
||||||
///
|
///
|
||||||
@@ -323,7 +544,7 @@ fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
|||||||
/// ever the firmware's (the one live piece, our stack, is reserved by the caller).
|
/// ever the firmware's (the one live piece, our stack, is reserved by the caller).
|
||||||
/// Anything unrecognised is `reserved` — the safe default; our own LoaderData (the
|
/// Anything unrecognised is `reserved` — the safe default; our own LoaderData (the
|
||||||
/// kernel image and these buffers) lands there and stays reserved.
|
/// kernel image and these buffers) lands there and stays reserved.
|
||||||
fn classify(d: *const uefi.tables.MemoryDescriptor) danos.MemoryKind {
|
fn classify(d: *const uefi.tables.MemoryDescriptor) boot_handoff.MemoryKind {
|
||||||
if (!d.attribute.wb) return .mmio;
|
if (!d.attribute.wb) return .mmio;
|
||||||
return switch (d.type) {
|
return switch (d.type) {
|
||||||
.conventional_memory, .boot_services_code, .boot_services_data => .usable,
|
.conventional_memory, .boot_services_code, .boot_services_data => .usable,
|
||||||
@@ -335,33 +556,33 @@ fn classify(d: *const uefi.tables.MemoryDescriptor) danos.MemoryKind {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Write a compile-time string to the console (best effort).
|
/// Write a compile-time string to the console (best effort).
|
||||||
fn log(comptime msg: []const u8) void {
|
fn log(comptime message: []const u8) void {
|
||||||
const out = uefi.system_table.con_out orelse return;
|
const out = uefi.system_table.con_out orelse return;
|
||||||
_ = out.outputString(std.unicode.utf8ToUtf16LeStringLiteral(msg)) catch {};
|
_ = out.outputString(std.unicode.utf8ToUtf16LeStringLiteral(message)) catch {};
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Write a runtime ASCII byte string (e.g. an @errorName) by widening to UTF-16.
|
/// Write a runtime ASCII byte string (e.g. an @errorName) by widening to UTF-16.
|
||||||
fn logBytes(bytes: []const u8) void {
|
fn logBytes(bytes: []const u8) void {
|
||||||
const out = uefi.system_table.con_out orelse return;
|
const out = uefi.system_table.con_out orelse return;
|
||||||
var buf: [128]u16 = undefined;
|
var buffer: [128]u16 = undefined;
|
||||||
var i: usize = 0;
|
var i: usize = 0;
|
||||||
for (bytes) |b| {
|
for (bytes) |b| {
|
||||||
if (i + 1 >= buf.len) break;
|
if (i + 1 >= buffer.len) break;
|
||||||
buf[i] = b;
|
buffer[i] = b;
|
||||||
i += 1;
|
i += 1;
|
||||||
}
|
}
|
||||||
buf[i] = 0;
|
buffer[i] = 0;
|
||||||
_ = out.outputString(buf[0..i :0].ptr) catch {};
|
_ = out.outputString(buffer[0..i :0].ptr) catch {};
|
||||||
}
|
}
|
||||||
|
|
||||||
fn acpiRootSystemDescriptorPointer() ?*const anyopaque {
|
fn acpiRootSystemDescriptorPointer() ?*const anyopaque {
|
||||||
const table_entries = uefi.system_table.number_of_table_entries;
|
const table_entries = uefi.system_table.number_of_table_entries;
|
||||||
const config_tables = uefi.system_table.configuration_table;
|
const configuration_tables = uefi.system_table.configuration_table;
|
||||||
const acpi2 = uefi.tables.ConfigurationTable.acpi_20_table_guid;
|
const acpi2 = uefi.tables.ConfigurationTable.acpi_20_table_guid;
|
||||||
const acpi1 = uefi.tables.ConfigurationTable.acpi_10_table_guid;
|
const acpi1 = uefi.tables.ConfigurationTable.acpi_10_table_guid;
|
||||||
|
|
||||||
for (0..table_entries) |i| {
|
for (0..table_entries) |i| {
|
||||||
const entry = config_tables[i];
|
const entry = configuration_tables[i];
|
||||||
if (entry.vendor_guid.eql(acpi2) or entry.vendor_guid.eql(acpi1)) {
|
if (entry.vendor_guid.eql(acpi2) or entry.vendor_guid.eql(acpi1)) {
|
||||||
return entry.vendor_table;
|
return entry.vendor_table;
|
||||||
}
|
}
|
||||||
@@ -48,50 +48,235 @@ fn timestamp(b: *std.Build) []const u8 {
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Build one user-space binary the same way for every program (init, and later
|
||||||
|
/// the VFS server + drivers): freestanding, ReleaseSmall, `.large` code model
|
||||||
|
/// (the image base is above 4 GiB — smaller models emit 32-bit relocations that
|
||||||
|
/// can't reach), linked against the `runtime` runtime library with the shared user
|
||||||
|
/// link script. Pinned to LLVM + LLD so the script's PHDRS (segment permissions)
|
||||||
|
/// are authoritative — the kernel's W^X user-ELF loader requires exact perms.
|
||||||
|
fn addUserBinary(
|
||||||
|
b: *std.Build,
|
||||||
|
target: std.Build.ResolvedTarget,
|
||||||
|
runtime_module: *std.Build.Module,
|
||||||
|
posix_module: *std.Build.Module,
|
||||||
|
mmio_module: *std.Build.Module,
|
||||||
|
xkeyboard_config_module: *std.Build.Module,
|
||||||
|
acpi_ids_module: *std.Build.Module,
|
||||||
|
name: []const u8,
|
||||||
|
root: []const u8,
|
||||||
|
) *std.Build.Step.Compile {
|
||||||
|
const exe = b.addExecutable(.{
|
||||||
|
.name = name,
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path(root),
|
||||||
|
.target = target,
|
||||||
|
.optimize = .ReleaseSmall,
|
||||||
|
.code_model = .large,
|
||||||
|
.single_threaded = true,
|
||||||
|
.sanitize_c = .off,
|
||||||
|
.stack_check = false,
|
||||||
|
.stack_protector = false,
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "runtime", .module = runtime_module },
|
||||||
|
// POSIX/C compatibility layer, available to any program that wants it
|
||||||
|
// (danos-native code uses `runtime` directly). See library/posix/.
|
||||||
|
.{ .name = "posix", .module = posix_module },
|
||||||
|
// Typed volatile MMIO + memory barriers, for drivers. See library/mmio/.
|
||||||
|
.{ .name = "mmio", .module = mmio_module },
|
||||||
|
// Keyboard layouts (keycode + modifiers -> keysym/character), available
|
||||||
|
// to any program that wants it. See library/xkeyboard-config/.
|
||||||
|
.{ .name = "xkeyboard-config", .module = xkeyboard_config_module },
|
||||||
|
// ACPI/PnP hardware-ID registry, so drivers name devices
|
||||||
|
// (HardwareId.ps2_keyboard) instead of magic "_HID" strings.
|
||||||
|
.{ .name = "acpi-ids", .module = acpi_ids_module },
|
||||||
|
},
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
exe.setLinkerScript(b.path("library/runtime/user.ld"));
|
||||||
|
exe.entry = .{ .symbol_name = "_start" };
|
||||||
|
exe.image_base = 0x7000_0000_0000;
|
||||||
|
exe.use_llvm = true;
|
||||||
|
exe.use_lld = true;
|
||||||
|
return exe;
|
||||||
|
}
|
||||||
|
|
||||||
pub fn build(b: *std.Build) void {
|
pub fn build(b: *std.Build) void {
|
||||||
ensureZigVersion();
|
ensureZigVersion();
|
||||||
|
|
||||||
const target = b.standardTargetOptions(.{});
|
const target = b.standardTargetOptions(.{});
|
||||||
const optimize = b.standardOptimizeOption(.{});
|
const optimize = b.standardOptimizeOption(.{});
|
||||||
|
|
||||||
// Shared handoff definitions (BootInfo, Framebuffer, ...). No target is set,
|
// The three shared contracts, each with its own audience so every import
|
||||||
// so the module inherits the target of whichever binary imports it — the
|
// declares which one it speaks (no target is set, so each inherits the target of
|
||||||
// freestanding kernel or the UEFI bootloader.
|
// whichever binary imports it). See docs/coding-standards.md.
|
||||||
const mod = b.addModule("danos", .{
|
// boot-handoff : loader <-> kernel (BootInformation, framebuffer, VM layout)
|
||||||
.root_source_file = b.path("src/root.zig"),
|
// abi : kernel <-> runtime, core (SystemCall, mmap prot flags, page_size)
|
||||||
|
// device-abi : kernel <-> user, devices (DeviceDescriptor, DeviceClass, ...)
|
||||||
|
const boot_handoff_module = b.addModule("boot-handoff", .{
|
||||||
|
.root_source_file = b.path("system/boot-handoff.zig"),
|
||||||
|
});
|
||||||
|
const abi_module = b.addModule("abi", .{
|
||||||
|
.root_source_file = b.path("system/abi.zig"),
|
||||||
|
});
|
||||||
|
// The devices sub-project's public interface (the flat wire types), exposed as
|
||||||
|
// its own module like vfs-protocol — importable by user space, unlike the
|
||||||
|
// kernel-internal device model it also feeds (system/devices/device-model.zig).
|
||||||
|
const device_abi_module = b.addModule("device-abi", .{
|
||||||
|
.root_source_file = b.path("system/devices/device-abi.zig"),
|
||||||
|
});
|
||||||
|
// PCI class-code decoding (class/subclass/prog-IF -> names). Pure reference data,
|
||||||
|
// shared by kernel discovery (the device-tree dump) and any user-space PCI tool.
|
||||||
|
const pci_class_module = b.addModule("pci-class", .{
|
||||||
|
.root_source_file = b.path("system/devices/pci-class.zig"),
|
||||||
|
});
|
||||||
|
// ACPI/PnP hardware-ID (_HID) names — the flat analog of pci-class for acpi_device
|
||||||
|
// nodes. Also shared reference data.
|
||||||
|
// The AML interpreter, a build module so the ring-3 acpi service can run the
|
||||||
|
// same parser the kernel does (docs/m19-m20-plan.md decision 1). Pure Zig,
|
||||||
|
// no kernel imports — one source, two builds.
|
||||||
|
const aml_module = b.addModule("aml", .{
|
||||||
|
.root_source_file = b.path("system/devices/aml/aml.zig"),
|
||||||
|
});
|
||||||
|
|
||||||
|
const acpi_ids_module = b.addModule("acpi-ids", .{
|
||||||
|
.root_source_file = b.path("system/devices/acpi-ids.zig"),
|
||||||
|
});
|
||||||
|
|
||||||
|
// Kernel tunables (maximum_cpus, stack sizes, tick rate). A dependency-free module of
|
||||||
|
// compile-time constants, imported wherever a knob is read; keeps the trade-offs
|
||||||
|
// in one place instead of scattered across the tree. See system/parameters.zig.
|
||||||
|
const parameters_module = b.addModule("parameters", .{
|
||||||
|
.root_source_file = b.path("system/parameters.zig"),
|
||||||
});
|
});
|
||||||
|
|
||||||
// Architecture-specific kernel code (CPU ops, entry, later GDT/IDT/paging).
|
// Architecture-specific kernel code (CPU ops, entry, later GDT/IDT/paging).
|
||||||
// The generic kernel imports this as "arch" and never names x86_64, so a new
|
// The generic kernel imports this as "architecture" and never names x86_64, so a new
|
||||||
// architecture is a matter of pointing this module at a different directory.
|
// architecture is a matter of pointing this module at a different directory.
|
||||||
const arch_mod = b.addModule("arch", .{
|
const architecture_module = b.addModule("architecture", .{
|
||||||
.root_source_file = b.path("src/kernel/arch/x86_64/cpu.zig"),
|
.root_source_file = b.path("system/kernel/architecture/x86_64/cpu.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
.{ .name = "danos", .module = mod }, // paging uses the shared BootInfo/memory-map types
|
.{ .name = "boot-handoff", .module = boot_handoff_module }, // paging uses BootInformation/memory-map + physicalToVirtual
|
||||||
|
.{ .name = "abi", .module = abi_module }, // paging works in page_size units
|
||||||
|
.{ .name = "parameters", .module = parameters_module }, // maximum_cpus, ist_stack_size, timer_hz
|
||||||
},
|
},
|
||||||
});
|
});
|
||||||
// CPU-exception stubs — real assembly, since they need cross-symbol
|
// CPU-exception stubs — real assembly, since they need cross-symbol
|
||||||
// jumps/calls that Zig inline asm can't express (see the file's header).
|
// jumps/calls that Zig inline asm can't express (see the file's header).
|
||||||
arch_mod.addAssemblyFile(b.path("src/kernel/arch/x86_64/isr.s"));
|
architecture_module.addAssemblyFile(b.path("system/kernel/architecture/x86_64/isr.s"));
|
||||||
|
// The AP bring-up trampoline: 16-/32-/64-bit mode-switch code that can't be
|
||||||
|
// inline asm (it runs relocated to a low page, not at its link address).
|
||||||
|
architecture_module.addAssemblyFile(b.path("system/kernel/architecture/x86_64/trampoline.s"));
|
||||||
|
|
||||||
// Firmware-agnostic device discovery. The generic kernel imports this as
|
// Firmware-agnostic device discovery. The generic kernel imports this as
|
||||||
// "platform" and asks it to enumerate hardware into a backend-neutral device
|
// "platform" and asks it to enumerate hardware into a backend-neutral device
|
||||||
// tree, never naming ACPI (or, later, device-tree) — the same discipline the
|
// tree, never naming ACPI (or, later, device-tree) — the same discipline the
|
||||||
// arch module applies to CPU code. The backend is selected at runtime from
|
// architecture module applies to CPU code. The backend is selected at runtime from
|
||||||
// the boot handoff (see src/device/platform.zig).
|
// the boot handoff (see system/devices/platform.zig).
|
||||||
const platform_mod = b.addModule("platform", .{
|
const platform_module = b.addModule("platform", .{
|
||||||
.root_source_file = b.path("src/device/platform.zig"),
|
.root_source_file = b.path("system/devices/platform.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
.{ .name = "danos", .module = mod }, // BootInfo (carries the ACPI RSDP)
|
.{ .name = "boot-handoff", .module = boot_handoff_module }, // BootInformation (carries the ACPI RSDP), physicalToVirtual
|
||||||
|
.{ .name = "abi", .module = abi_module }, // acpi.zig works in page_size units
|
||||||
|
.{ .name = "device-abi", .module = device_abi_module }, // device-model's DeviceClass/ResourceKind live here
|
||||||
|
.{ .name = "pci-class", .module = pci_class_module }, // decode PCI class codes in the device dump
|
||||||
|
.{ .name = "acpi-ids", .module = acpi_ids_module }, // decode ACPI _HID names in the device dump
|
||||||
|
.{ .name = "parameters", .module = parameters_module }, // maximum_cpus (the discovery pool)
|
||||||
},
|
},
|
||||||
});
|
});
|
||||||
|
|
||||||
// Compile-time config the kernel reads as `@import("build_options")`. The
|
// The VFS wire protocol: the vfs sub-project's public interface, exposed as its
|
||||||
|
// own module. Both the vfs server and the runtime's file layer (unistd/stdio)
|
||||||
|
// depend on this contract by name — neither reaches into the other's files. This
|
||||||
|
// is the first "protocol module" (see docs/driver-model.md); usb/block will
|
||||||
|
// expose theirs the same way.
|
||||||
|
const vfs_protocol_module = b.addModule("vfs-protocol", .{
|
||||||
|
.root_source_file = b.path("system/services/vfs/protocol.zig"),
|
||||||
|
});
|
||||||
|
|
||||||
|
// The input wire protocol: the input service's public interface, exposed as its own
|
||||||
|
// module the same way vfs-protocol is. Shared by the input service, the runtime's
|
||||||
|
// `input` helper (subscribe/publish), and every source and subscriber.
|
||||||
|
const input_protocol_module = b.addModule("input-protocol", .{
|
||||||
|
.root_source_file = b.path("system/services/input/protocol.zig"),
|
||||||
|
});
|
||||||
|
|
||||||
|
// The danos-native user-space runtime: system_call wrappers, the C-convention
|
||||||
|
// heap, IPC helpers, the process start shim, device access. This is the stable
|
||||||
|
// application ABI; POSIX compatibility is a separate library on top (see below).
|
||||||
|
// Compiled into every user binary (see addUserBinary), so it inherits each exe's
|
||||||
|
// `.large` code model — do NOT set a target/code_model here. It imports `abi`
|
||||||
|
// for the shared SystemCall numbers / mmap flags, `device-abi` for the device
|
||||||
|
// types its `device` helper wraps, and re-exports `vfs-protocol` for the VFS
|
||||||
|
// server. It never touches `boot-handoff` — user space has no business with the
|
||||||
|
// loader↔kernel handoff.
|
||||||
|
const runtime_module = b.addModule("runtime", .{
|
||||||
|
.root_source_file = b.path("library/runtime/runtime.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "abi", .module = abi_module },
|
||||||
|
.{ .name = "device-abi", .module = device_abi_module },
|
||||||
|
.{ .name = "vfs-protocol", .module = vfs_protocol_module },
|
||||||
|
.{ .name = "input-protocol", .module = input_protocol_module },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
// The device-manager protocol: hello + (M18.2) tree reports, exposed as its
|
||||||
|
// own module like the other protocol modules. Imported through the runtime.
|
||||||
|
const device_manager_protocol_module = b.addModule("device-manager-protocol", .{
|
||||||
|
.root_source_file = b.path("system/services/device-manager/device-manager-protocol.zig"),
|
||||||
|
});
|
||||||
|
runtime_module.addImport("device-manager-protocol", device_manager_protocol_module);
|
||||||
|
|
||||||
|
// The power protocol: system power's domain-named surface (docs/m21-plan.md).
|
||||||
|
const power_protocol_module = b.addModule("power-protocol", .{
|
||||||
|
.root_source_file = b.path("system/services/power/protocol.zig"),
|
||||||
|
});
|
||||||
|
runtime_module.addImport("power-protocol", power_protocol_module);
|
||||||
|
|
||||||
|
// Typed volatile MMIO register access + memory-ordering barriers, for drivers on
|
||||||
|
// top of an mmio_map grant. Depends only on `builtin` (arch-conditional barriers);
|
||||||
|
// no target set, so it inherits each driver's. See library/mmio/mmio.zig.
|
||||||
|
const mmio_module = b.addModule("mmio", .{
|
||||||
|
.root_source_file = b.path("library/mmio/mmio.zig"),
|
||||||
|
});
|
||||||
|
|
||||||
|
// Keyboard layouts compiled from the X11 xkeyboard-config database into native Zig
|
||||||
|
// (keycode + modifiers -> keysym/character). The `layouts` tables are generated by
|
||||||
|
// tools/make-xkeyboard-config.py; `xkeyboard-config` is the hand-written API over them.
|
||||||
|
// No target set, so each inherits its importer's. See library/xkeyboard-config/.
|
||||||
|
const xkb_layouts_module = b.addModule("layouts", .{
|
||||||
|
.root_source_file = b.path("library/xkeyboard-config/generated/layouts.zig"),
|
||||||
|
});
|
||||||
|
const xkeyboard_config_module = b.addModule("xkeyboard-config", .{
|
||||||
|
.root_source_file = b.path("library/xkeyboard-config/xkeyboard-config.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "layouts", .module = xkb_layouts_module },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
// The POSIX / C compatibility layer, a separate library layered strictly over the
|
||||||
|
// runtime (it calls the runtime's IPC/heap, never system calls directly). This is
|
||||||
|
// the one place POSIX/C spellings are allowed verbatim — see docs/coding-standards.md
|
||||||
|
// and library/posix/posix.zig.
|
||||||
|
const posix_module = b.addModule("posix", .{
|
||||||
|
.root_source_file = b.path("library/posix/posix.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "runtime", .module = runtime_module },
|
||||||
|
.{ .name = "vfs-protocol", .module = vfs_protocol_module },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
// The initial_ramdisk container format, shared by the kernel (unpacks it) and the
|
||||||
|
// build-time packer tools/make-initial-ramdisk.py (produces it). No dependencies.
|
||||||
|
const initial_ramdisk_module = b.addModule("initial-ramdisk", .{
|
||||||
|
.root_source_file = b.path("system/initial-ramdisk.zig"),
|
||||||
|
});
|
||||||
|
|
||||||
|
// Compile-time configuration the kernel reads as `@import("build_options")`. The
|
||||||
// QEMU test harness sets -Dtest-case=<name> to run one self-test at boot.
|
// QEMU test harness sets -Dtest-case=<name> to run one self-test at boot.
|
||||||
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see src/kernel/tests.zig)");
|
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see system/kernel/tests.zig)");
|
||||||
const build_options = b.addOptions();
|
const build_options = b.addOptions();
|
||||||
build_options.addOption(?[]const u8, "test_case", test_case);
|
build_options.addOption(?[]const u8, "test_case", test_case);
|
||||||
const build_options_mod = build_options.createModule();
|
const build_options_module = build_options.createModule();
|
||||||
|
|
||||||
// --- Kernel: freestanding x86_64 ELF, jumped to by the bootloader ---
|
// --- Kernel: freestanding x86_64 ELF, jumped to by the bootloader ---
|
||||||
// SSE2 is part of the x86_64 baseline and UEFI leaves it enabled at handoff,
|
// SSE2 is part of the x86_64 baseline and UEFI leaves it enabled at handoff,
|
||||||
@@ -106,62 +291,199 @@ pub fn build(b: *std.Build) void {
|
|||||||
const exe = b.addExecutable(.{
|
const exe = b.addExecutable(.{
|
||||||
.name = "kernel",
|
.name = "kernel",
|
||||||
.root_module = b.createModule(.{
|
.root_module = b.createModule(.{
|
||||||
.root_source_file = b.path("src/kernel/main.zig"),
|
.root_source_file = b.path("system/kernel/kernel.zig"),
|
||||||
.target = kernel_target,
|
.target = kernel_target,
|
||||||
.optimize = optimize,
|
.optimize = optimize,
|
||||||
.code_model = .small, // kernel is linked in the low 2 GiB (see image_base)
|
.code_model = .kernel, // kernel runs in the top 2 GiB (higher half)
|
||||||
.red_zone = false, // interrupts would corrupt the SysV red zone
|
.red_zone = false, // interrupts would corrupt the SystemV red zone
|
||||||
.single_threaded = true, // no scheduler yet; avoids pulling in TLS/atomics
|
.single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores
|
||||||
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
||||||
.stack_check = false, // stack-probe calls have no runtime to land in
|
.stack_check = false, // stack-probe calls have no runtime to land in
|
||||||
.stack_protector = false,
|
.stack_protector = false,
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
.{ .name = "danos", .module = mod },
|
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
||||||
.{ .name = "arch", .module = arch_mod },
|
.{ .name = "abi", .module = abi_module },
|
||||||
.{ .name = "platform", .module = platform_mod },
|
.{ .name = "device-abi", .module = device_abi_module },
|
||||||
.{ .name = "build_options", .module = build_options_mod },
|
.{ .name = "architecture", .module = architecture_module },
|
||||||
|
.{ .name = "platform", .module = platform_module },
|
||||||
|
.{ .name = "parameters", .module = parameters_module },
|
||||||
|
.{ .name = "build_options", .module = build_options_module },
|
||||||
|
.{ .name = "initial-ramdisk", .module = initial_ramdisk_module },
|
||||||
},
|
},
|
||||||
}),
|
}),
|
||||||
});
|
});
|
||||||
exe.setLinkerScript(b.path("src/kernel/arch/x86_64/linker.ld"));
|
exe.setLinkerScript(b.path("system/kernel/architecture/x86_64/linker.ld"));
|
||||||
exe.entry = .{ .symbol_name = "_start" };
|
exe.entry = .{ .symbol_name = "_start" };
|
||||||
// Physical address the bootloader loads the kernel to (identity-mapped under
|
// The self-hosted linker ignores parts of the linker script (PHDRS,
|
||||||
// UEFI). Overrides Zig's default image base so the linker script's layout is
|
// /DISCARD/, AT(), section order); the higher-half layout depends on the
|
||||||
// honoured; adjust here if it collides with firmware-reserved memory.
|
// script being authoritative, so pin the kernel to LLVM + LLD.
|
||||||
exe.image_base = 0x100000; // 1 MiB
|
exe.use_llvm = true;
|
||||||
|
exe.use_lld = true;
|
||||||
|
// Higher-half virtual base (matches KERNEL_VIRT_BASE in linker.ld); the
|
||||||
|
// linker's AT() clauses give each segment a low physical load address
|
||||||
|
// (.text at 1 MiB), which the loader allocates and copies into.
|
||||||
|
exe.image_base = 0xFFFFFFFF80100000;
|
||||||
|
|
||||||
b.installArtifact(exe);
|
// Everything installs into a FHS-shaped zig-out: it IS the danos filesystem *and*
|
||||||
|
// the boot volume. Each binary lands at its addressed, leaf-collapsed path — the
|
||||||
|
// kernel at zig-out/system/kernel (from system/kernel/kernel.zig), init at
|
||||||
|
// zig-out/system/services/init, and so on (see docs/README.md). The bootloader
|
||||||
|
// then loads these FHS paths off the volume.
|
||||||
|
const kernel_install = b.addInstallArtifact(exe, .{ .dest_dir = .{ .override = .{ .custom = "system" } } });
|
||||||
|
b.getInstallStep().dependOn(&kernel_install.step);
|
||||||
|
|
||||||
// Boot methods live in src/boot/, one per way of getting the kernel running.
|
// --- init: the first user-space program (a system service) ---
|
||||||
|
// Built by the shared user-binary recipe (see addUserBinary): freestanding,
|
||||||
|
// linked into the kernel's user region against the `runtime` runtime library, and
|
||||||
|
// started in ring 3 by the kernel's user-ELF loader.
|
||||||
|
const init_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "init", "system/services/init/init.zig");
|
||||||
|
const init_install = b.addInstallArtifact(init_exe, .{ .dest_dir = .{ .override = .{ .custom = "system/services" } } });
|
||||||
|
b.getInstallStep().dependOn(&init_install.step);
|
||||||
|
|
||||||
|
// --- initial_ramdisk: a bundle of extra user binaries (VFS server + drivers) ---
|
||||||
|
// Each is built by the same user-binary recipe, then packed into one image by
|
||||||
|
// the host-side make-initial-ramdisk tool. The bootloader ferries the image to the kernel,
|
||||||
|
// which unpacks it and spawns each program (system/initial-ramdisk.zig).
|
||||||
|
const vfs_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs", "system/services/vfs/vfs.zig");
|
||||||
|
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs-test", "system/services/vfs/vfs-test.zig");
|
||||||
|
const hpet_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "hpet", "system/drivers/hpet/hpet.zig");
|
||||||
|
const bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "bus", "system/drivers/bus/bus.zig");
|
||||||
|
const ps2_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-bus", "system/drivers/ps2-bus/ps2-bus.zig");
|
||||||
|
const ps2_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig");
|
||||||
|
const ps2_mouse_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig");
|
||||||
|
const usb_xhci_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-xhci-bus", "system/drivers/usb-xhci-bus/usb-xhci-bus.zig");
|
||||||
|
const pci_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "pci-bus", "system/drivers/pci-bus/pci-bus.zig");
|
||||||
|
// The PCI bus driver decodes each function's class triple to human names in its
|
||||||
|
// boot log (class/subclass/prog-IF), so pull in the shared pci-class reference.
|
||||||
|
pci_bus_exe.root_module.addImport("pci-class", pci_class_module);
|
||||||
|
// A test fixture, not a real driver: hellos to the device manager, then faults —
|
||||||
|
// what the driver-restart scenario drives the crash-loop cap with.
|
||||||
|
const crash_test_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "crash-test", "system/services/crash-test/crash-test.zig");
|
||||||
|
const device_list_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-list", "system/services/device-list/device-list.zig");
|
||||||
|
// The discovery service: one swappable process per firmware
|
||||||
|
// (docs/m19-m20-plan.md decision 7), bundled under the neutral ramdisk name
|
||||||
|
// "discovery" so the device manager never learns which firmware it is on.
|
||||||
|
// x86 boots describe hardware with ACPI; the Raspberry Pis hand over a
|
||||||
|
// flattened device tree — the aarch64 target flips the default when it
|
||||||
|
// lands (docs/arm.md). Both are placeholders until M20.1 (acpi) and the
|
||||||
|
// ARM bring-up (fdt).
|
||||||
|
const Discovery = enum { acpi, fdt };
|
||||||
|
const discovery = b.option(Discovery, "discovery", "Which discovery service fills the ramdisk's 'discovery' slot (default: acpi)") orelse Discovery.acpi;
|
||||||
|
const discovery_source: []const u8 = switch (discovery) {
|
||||||
|
.acpi => "system/services/acpi/acpi.zig",
|
||||||
|
.fdt => "system/services/fdt/fdt.zig",
|
||||||
|
};
|
||||||
|
const discovery_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "discovery", discovery_source);
|
||||||
|
if (discovery == .acpi) discovery_exe.root_module.addImport("aml", aml_module);
|
||||||
|
const device_manager_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-manager", "system/services/device-manager/device-manager.zig");
|
||||||
|
// Names the xHCI PCI class triple from the shared taxonomy instead of a bare 0x0C0330.
|
||||||
|
device_manager_exe.root_module.addImport("pci-class", pci_class_module);
|
||||||
|
// The input service and its exercisers: the fan-out server, a hardware-free synthetic
|
||||||
|
// source, and a subscriber that doubles as the `input` test's oracle. See docs/input.md.
|
||||||
|
const input_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input", "system/services/input/input.zig");
|
||||||
|
const input_source_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-source", "system/services/input-source/input-source.zig");
|
||||||
|
const input_test_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-test", "system/services/input-test/input-test.zig");
|
||||||
|
const args_echo_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "args-echo", "system/services/args-echo/args-echo.zig");
|
||||||
|
const process_test_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "process-test", "system/services/process-test/process-test.zig");
|
||||||
|
|
||||||
|
// Pack the user binaries into the initial_ramdisk image with the host-side Python tool
|
||||||
|
// (the container format is trivial, and Python sidesteps std API churn). Args:
|
||||||
|
// make-initial-ramdisk.py <out> [<name> <file>]... — one name/file pair per binary.
|
||||||
|
const mk_run = b.addSystemCommand(&.{"python3"});
|
||||||
|
mk_run.addFileArg(b.path("tools/make-initial-ramdisk.py"));
|
||||||
|
const initial_ramdisk_img = mk_run.addOutputFileArg("initial-ramdisk.img");
|
||||||
|
mk_run.addArg("vfs");
|
||||||
|
mk_run.addFileArg(vfs_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("vfs-test");
|
||||||
|
mk_run.addFileArg(vfstest_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("hpet");
|
||||||
|
mk_run.addFileArg(hpet_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("bus");
|
||||||
|
mk_run.addFileArg(bus_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("ps2-bus");
|
||||||
|
mk_run.addFileArg(ps2_bus_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("ps2-keyboard");
|
||||||
|
mk_run.addFileArg(ps2_keyboard_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("ps2-mouse");
|
||||||
|
mk_run.addFileArg(ps2_mouse_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("usb-xhci-bus");
|
||||||
|
mk_run.addFileArg(usb_xhci_bus_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("pci-bus");
|
||||||
|
mk_run.addFileArg(pci_bus_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("crash-test");
|
||||||
|
mk_run.addFileArg(crash_test_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("device-list");
|
||||||
|
mk_run.addFileArg(device_list_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("discovery");
|
||||||
|
mk_run.addFileArg(discovery_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("device-manager");
|
||||||
|
mk_run.addFileArg(device_manager_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("input");
|
||||||
|
mk_run.addFileArg(input_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("input-source");
|
||||||
|
mk_run.addFileArg(input_source_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("input-test");
|
||||||
|
mk_run.addFileArg(input_test_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("args-echo");
|
||||||
|
mk_run.addFileArg(args_echo_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("process-test");
|
||||||
|
mk_run.addFileArg(process_test_exe.getEmittedBin());
|
||||||
|
|
||||||
|
// Also install the packed binaries to their FHS homes, so zig-out is a true image
|
||||||
|
// of the filesystem — even though at boot they arrive inside the initial-ramdisk.
|
||||||
|
for ([_]struct { *std.Build.Step.Compile, []const u8 }{
|
||||||
|
.{ vfs_exe, "system/services" },
|
||||||
|
.{ device_manager_exe, "system/services" },
|
||||||
|
.{ input_exe, "system/services" },
|
||||||
|
.{ hpet_exe, "system/drivers" },
|
||||||
|
.{ bus_exe, "system/drivers" },
|
||||||
|
.{ ps2_bus_exe, "system/drivers" },
|
||||||
|
.{ ps2_keyboard_exe, "system/drivers" },
|
||||||
|
.{ ps2_mouse_exe, "system/drivers" },
|
||||||
|
.{ usb_xhci_bus_exe, "system/drivers" },
|
||||||
|
}) |entry| {
|
||||||
|
const step = b.addInstallArtifact(entry[0], .{ .dest_dir = .{ .override = .{ .custom = entry[1] } } });
|
||||||
|
b.getInstallStep().dependOn(&step.step);
|
||||||
|
}
|
||||||
|
|
||||||
|
// The initial-ramdisk itself installs to /boot (with the loaders).
|
||||||
|
const initial_ramdisk_install = b.addInstallFile(initial_ramdisk_img, "boot/initial-ramdisk.img");
|
||||||
|
b.getInstallStep().dependOn(&initial_ramdisk_install.step);
|
||||||
|
|
||||||
|
// Boot methods live in boot/, one per way of getting the kernel running.
|
||||||
// Each is its own binary/entry (a loader is built for its own target); today
|
// Each is its own binary/entry (a loader is built for its own target); today
|
||||||
// that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis.
|
// that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis.
|
||||||
const efiexe = b.addExecutable(.{
|
const efiexe = b.addExecutable(.{
|
||||||
.name = "BOOTX64",
|
.name = "BOOTX64",
|
||||||
.root_module = b.createModule(.{
|
.root_module = b.createModule(.{
|
||||||
.root_source_file = b.path("src/boot/efi.zig"),
|
.root_source_file = b.path("boot/efi.zig"),
|
||||||
.target = b.resolveTargetQuery(.{
|
.target = b.resolveTargetQuery(.{
|
||||||
.cpu_arch = .x86_64,
|
.cpu_arch = .x86_64,
|
||||||
.os_tag = .uefi,
|
.os_tag = .uefi,
|
||||||
}),
|
}),
|
||||||
.optimize = optimize,
|
.optimize = optimize,
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
.{ .name = "danos", .module = mod },
|
// The bootloader speaks only the handoff contract — never the user ABI.
|
||||||
|
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
||||||
},
|
},
|
||||||
}),
|
}),
|
||||||
});
|
});
|
||||||
|
|
||||||
b.installArtifact(efiexe);
|
// UEFI firmware requires the removable-media loader at exactly \EFI\BOOT\BOOTX64.efi,
|
||||||
|
// so that path is fixed by the firmware (it is /boot's EFI stub, conceptually).
|
||||||
|
const efi_install = b.addInstallArtifact(efiexe, .{ .dest_dir = .{ .override = .{ .custom = "EFI/BOOT" } } });
|
||||||
|
b.getInstallStep().dependOn(&efi_install.step);
|
||||||
|
|
||||||
// --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF ---
|
// --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF ---
|
||||||
// Firmware lives in different places per OS/distro, so probe the known
|
// Firmware lives in different places per OS/distro, so probe the known
|
||||||
// layouts (Arch, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
|
// layouts (Architecture, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
|
||||||
// that exists. Override with -Dovmf-code / -Dovmf-vars if yours is elsewhere.
|
// that exists. Override with -Dovmf-code / -Dovmf-vars if yours is elsewhere.
|
||||||
const ovmf_code = b.option(
|
const ovmf_code = b.option(
|
||||||
[]const u8,
|
[]const u8,
|
||||||
"ovmf-code",
|
"ovmf-code",
|
||||||
"Path to the OVMF_CODE firmware image",
|
"Path to the OVMF_CODE firmware image",
|
||||||
) orelse firstExisting(b.graph.io, &.{
|
) orelse firstExisting(b.graph.io, &.{
|
||||||
"/usr/share/edk2/x64/OVMF_CODE.4m.fd", // Arch
|
"/usr/share/edk2/x64/OVMF_CODE.4m.fd", // Architecture
|
||||||
"/usr/share/OVMF/OVMF_CODE_4M.fd", // Debian/Ubuntu
|
"/usr/share/OVMF/OVMF_CODE_4M.fd", // Debian/Ubuntu
|
||||||
"/usr/share/OVMF/OVMF_CODE.fd", // older Debian/Ubuntu
|
"/usr/share/OVMF/OVMF_CODE.fd", // older Debian/Ubuntu
|
||||||
"/usr/share/edk2-ovmf/x64/OVMF_CODE.fd", // Fedora
|
"/usr/share/edk2-ovmf/x64/OVMF_CODE.fd", // Fedora
|
||||||
@@ -173,7 +495,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
"ovmf-vars",
|
"ovmf-vars",
|
||||||
"Path to the OVMF_VARS firmware image (a writable copy is made)",
|
"Path to the OVMF_VARS firmware image (a writable copy is made)",
|
||||||
) orelse firstExisting(b.graph.io, &.{
|
) orelse firstExisting(b.graph.io, &.{
|
||||||
"/usr/share/edk2/x64/OVMF_VARS.4m.fd", // Arch
|
"/usr/share/edk2/x64/OVMF_VARS.4m.fd", // Architecture
|
||||||
"/usr/share/OVMF/OVMF_VARS_4M.fd", // Debian/Ubuntu
|
"/usr/share/OVMF/OVMF_VARS_4M.fd", // Debian/Ubuntu
|
||||||
"/usr/share/OVMF/OVMF_VARS.fd", // older Debian/Ubuntu
|
"/usr/share/OVMF/OVMF_VARS.fd", // older Debian/Ubuntu
|
||||||
"/usr/share/edk2-ovmf/x64/OVMF_VARS.fd", // Fedora
|
"/usr/share/edk2-ovmf/x64/OVMF_VARS.fd", // Fedora
|
||||||
@@ -181,15 +503,8 @@ pub fn build(b: *std.Build) void {
|
|||||||
"/usr/local/share/qemu/edk2-i386-vars.fd", // macOS Homebrew (Intel)
|
"/usr/local/share/qemu/edk2-i386-vars.fd", // macOS Homebrew (Intel)
|
||||||
});
|
});
|
||||||
|
|
||||||
// Assemble an EFI System Partition layout: esp/EFI/BOOT/BOOTX64.efi
|
// The FHS zig-out (installed above) *is* the boot volume — no separate ESP to
|
||||||
const efi_install = b.addInstallArtifact(efiexe, .{
|
// assemble. QEMU presents it to the guest as a FAT drive below.
|
||||||
.dest_dir = .{ .override = .{ .custom = "esp/EFI/BOOT" } },
|
|
||||||
});
|
|
||||||
// The bootloader loads the kernel by name from the volume root, so drop the
|
|
||||||
// kernel ELF at esp/kernel.
|
|
||||||
const kernel_install = b.addInstallArtifact(exe, .{
|
|
||||||
.dest_dir = .{ .override = .{ .custom = "esp" } },
|
|
||||||
});
|
|
||||||
|
|
||||||
// The firmware needs to write NVRAM, so give it a writable copy of the vars.
|
// The firmware needs to write NVRAM, so give it a writable copy of the vars.
|
||||||
const vars_copy = b.addSystemCommand(&.{ "cp", "-f", ovmf_vars });
|
const vars_copy = b.addSystemCommand(&.{ "cp", "-f", ovmf_vars });
|
||||||
@@ -197,6 +512,19 @@ pub fn build(b: *std.Build) void {
|
|||||||
|
|
||||||
const run_efi = b.addSystemCommand(&.{
|
const run_efi = b.addSystemCommand(&.{
|
||||||
"qemu-system-x86_64",
|
"qemu-system-x86_64",
|
||||||
|
"-device",
|
||||||
|
"qemu-xhci,id=xhci",
|
||||||
|
"-device",
|
||||||
|
"usb-mouse,bus=xhci.0",
|
||||||
|
"-device",
|
||||||
|
"usb-kbd,bus=xhci.0",
|
||||||
|
// "-usb",
|
||||||
|
// "-device",
|
||||||
|
// "usb-ehci,id=ehci",
|
||||||
|
// "-device",
|
||||||
|
// "usb-tablet,bus=usb-bus.0",
|
||||||
|
// "-device",
|
||||||
|
// "usb-mouse,bus=ehci.0",
|
||||||
"-machine",
|
"-machine",
|
||||||
"q35",
|
"q35",
|
||||||
"-m",
|
"-m",
|
||||||
@@ -206,10 +534,10 @@ pub fn build(b: *std.Build) void {
|
|||||||
});
|
});
|
||||||
run_efi.addArg("-drive");
|
run_efi.addArg("-drive");
|
||||||
run_efi.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
run_efi.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
||||||
// Present the ESP directory to the guest as a FAT drive.
|
// Present the FHS zig-out to the guest as a FAT drive — it is the boot volume.
|
||||||
run_efi.addArgs(&.{
|
run_efi.addArgs(&.{
|
||||||
"-drive",
|
"-drive",
|
||||||
b.fmt("format=raw,file=fat:rw:{s}/esp", .{b.install_path}),
|
b.fmt("format=raw,file=fat:rw:{s}", .{b.install_path}),
|
||||||
"-net",
|
"-net",
|
||||||
"none",
|
"none",
|
||||||
// Emulated display advertising 1280x720 as its native (EDID preferred)
|
// Emulated display advertising 1280x720 as its native (EDID preferred)
|
||||||
@@ -220,14 +548,19 @@ pub fn build(b: *std.Build) void {
|
|||||||
"-device",
|
"-device",
|
||||||
"VGA,edid=on,xres=1280,yres=720",
|
"VGA,edid=on,xres=1280,yres=720",
|
||||||
});
|
});
|
||||||
// Always capture the guest's serial0 (the kernel's machine-readable log) to a
|
// Capture the guest's serial0 (danos's machine-readable log) to the qemu-test
|
||||||
// timestamped file under zig-out, so each run leaves its own log behind.
|
// scratch area — a dev/host artifact, kept out of the FHS boot volume we mount.
|
||||||
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ b.install_path, timestamp(b) });
|
// (/var/log/system is reserved for the kernel's own logging system later.) One
|
||||||
|
// timestamped file per run.
|
||||||
|
const log_dir = b.fmt("{s}/qemu-test", .{b.install_path});
|
||||||
|
const make_log_dir = b.addSystemCommand(&.{ "mkdir", "-p", log_dir });
|
||||||
|
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
||||||
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
|
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
|
||||||
run_efi.step.dependOn(&efi_install.step);
|
// The whole FHS zig-out must be installed (and the scratch dir created) before we mount it.
|
||||||
run_efi.step.dependOn(&kernel_install.step);
|
run_efi.step.dependOn(b.getInstallStep());
|
||||||
|
run_efi.step.dependOn(&make_log_dir.step);
|
||||||
|
|
||||||
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/run-x86-64-serial0-<timestamp>.log");
|
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/qemu-test/run-x86-64-serial0-<timestamp>.log");
|
||||||
run_efi_step.dependOn(&run_efi.step);
|
run_efi_step.dependOn(&run_efi.step);
|
||||||
|
|
||||||
// const run_cmd = b.addRunArtifact(exe);
|
// const run_cmd = b.addRunArtifact(exe);
|
||||||
@@ -240,18 +573,51 @@ pub fn build(b: *std.Build) void {
|
|||||||
// }
|
// }
|
||||||
|
|
||||||
// Tests run on the host. The kernel and bootloader target freestanding/UEFI
|
// Tests run on the host. The kernel and bootloader target freestanding/UEFI
|
||||||
// and can't be executed natively, so only the shared module is unit-tested
|
// and can't be executed natively, so only the shared contracts are unit-tested
|
||||||
// here (compiled for the host rather than inheriting a freestanding target).
|
// here (compiled for the host rather than inheriting a freestanding target) —
|
||||||
const mod_tests = b.addTest(.{
|
// which also compile-checks that the three-way split stays self-consistent.
|
||||||
|
const test_step = b.step("test", "Run tests");
|
||||||
|
for ([_][]const u8{
|
||||||
|
"system/boot-handoff.zig",
|
||||||
|
"system/abi.zig",
|
||||||
|
"system/devices/device-abi.zig",
|
||||||
|
"system/devices/pci-class.zig", // class/subclass/prog-IF name decoding
|
||||||
|
"system/devices/acpi-ids.zig", // _HID name decoding
|
||||||
|
"system/devices/aml/aml.zig", // AML parse + interpret, incl. Notify dispatch (M21)
|
||||||
|
"system/devices/usb-abi.zig", // wire sizes + bit packings + set-up packet encodings
|
||||||
|
"system/devices/usb-ids.zig", // class/subclass/protocol code assignments
|
||||||
|
"library/mmio/mmio.zig", // barriers assemble + registers round-trip
|
||||||
|
"system/drivers/ps2-bus/scancode.zig", // set-2 decode + keyboard state machine
|
||||||
|
"system/drivers/ps2-bus/mouse-packet.zig", // 3-byte mouse packet assembly
|
||||||
|
}) |root| {
|
||||||
|
const mod_tests = b.addTest(.{
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path(root),
|
||||||
|
.target = target,
|
||||||
|
.optimize = optimize,
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
test_step.dependOn(&b.addRunArtifact(mod_tests).step);
|
||||||
|
}
|
||||||
|
|
||||||
|
// The xkeyboard-config keymap tests need its generated `layouts` import wired, so they
|
||||||
|
// don't fit the plain loop above. Its keycode->character assertions are the end-to-end
|
||||||
|
// proof that the xkb-data -> generator -> Zig-lookup pipeline is correct.
|
||||||
|
const xkb_tests = b.addTest(.{
|
||||||
.root_module = b.createModule(.{
|
.root_module = b.createModule(.{
|
||||||
.root_source_file = b.path("src/root.zig"),
|
.root_source_file = b.path("library/xkeyboard-config/xkeyboard-config.zig"),
|
||||||
.target = target,
|
.target = target,
|
||||||
.optimize = optimize,
|
.optimize = optimize,
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "layouts", .module = xkb_layouts_module },
|
||||||
|
},
|
||||||
}),
|
}),
|
||||||
});
|
});
|
||||||
|
test_step.dependOn(&b.addRunArtifact(xkb_tests).step);
|
||||||
|
|
||||||
const run_mod_tests = b.addRunArtifact(mod_tests);
|
// Convenience: `zig build gen-xkeyboard-config` regenerates the layout tables from the
|
||||||
|
// vendored data (offline). `fetch` (the network step) stays a manual script run.
|
||||||
const test_step = b.step("test", "Run tests");
|
const gen_xkb = b.addSystemCommand(&.{ "python3", "tools/make-xkeyboard-config.py", "generate" });
|
||||||
test_step.dependOn(&run_mod_tests.step);
|
const gen_xkb_step = b.step("gen-xkeyboard-config", "Regenerate library/xkeyboard-config/generated from the vendored data");
|
||||||
|
gen_xkb_step.dependOn(&gen_xkb.step);
|
||||||
}
|
}
|
||||||
|
|||||||
+129
-13
@@ -35,9 +35,41 @@ rather than restate it. Roughly in the order things happen at runtime:
|
|||||||
multitasking: kernel threads, the context switch, O(1) priority selection, and
|
multitasking: kernel threads, the context switch, O(1) priority selection, and
|
||||||
blocking (sleep, wait queues) — the leap to a running system.
|
blocking (sleep, wait queues) — the leap to a running system.
|
||||||
11. **[ipc.md](ipc.md) — inter-process communication.** Bounded blocking
|
11. **[ipc.md](ipc.md) — inter-process communication.** Bounded blocking
|
||||||
message-passing channels — the backbone the microkernel's isolated servers will
|
message-passing channels, then synchronous call/reply between *processes* over
|
||||||
talk over.
|
endpoints — the backbone the microkernel's isolated servers talk over.
|
||||||
12. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
12. **[syscall.md](syscall.md) — system calls.** How ring 3 asks the kernel for
|
||||||
|
something: the `syscall`/`sysret` fast path, the trap frame, and why the table is
|
||||||
|
deliberately tiny.
|
||||||
|
13. **[drivers.md](drivers.md) — writing a driver.** The payoff: a driver is an
|
||||||
|
ordinary ring-3 process that claims a device, maps its registers, and **sleeps
|
||||||
|
until its hardware interrupts it**. The claim is the capability; `irq_ack` is the
|
||||||
|
unmask.
|
||||||
|
14. **[driver-model.md](driver-model.md) — buses, classes and host controllers.** How
|
||||||
|
real driver stacks factor into three shapes and how families share code. The
|
||||||
|
three primitives it proposed are long since built (M13 capability passing,
|
||||||
|
M14 DMA + barriers, M15 MSI), and the driver *contract* on top of them —
|
||||||
|
hello, supervision, restart — is built too (device-manager.md, M18).
|
||||||
|
15. **[process-management.md](process-management.md) — process management.** The
|
||||||
|
microkernel's `ps`/`kill`/SIGCHLD: enumerate as a table snapshot, the
|
||||||
|
supervision link as the kill authority, and child-exit notifications over the
|
||||||
|
same endpoints IRQs arrive on.
|
||||||
|
16. **[process-lifecycle.md](process-lifecycle.md) — the process lifecycle.** Built
|
||||||
|
(M17): signals over IPC as the one lifecycle vocabulary every process speaks — the
|
||||||
|
POSIX.1-1990 words with message delivery instead of stack hijack, the stable
|
||||||
|
`runtime.process` interface, exit reasons, published exit events any stateful
|
||||||
|
service can subscribe to (the VFS releasing dead clients' handles), and the two
|
||||||
|
iron rules (cleanup is the kernel's job; kill is not a signal).
|
||||||
|
17. **[device-manager.md](device-manager.md) — the device manager.** Built (M18,
|
||||||
|
through the app surface): the
|
||||||
|
tree, the matcher, and the supervisor. Tree structure lives in the manager,
|
||||||
|
authority stays in the kernel; bus drivers report what they see; drivers are
|
||||||
|
restarted through the lifecycle vocabulary — the plan that turns
|
||||||
|
[resilience.md](resilience.md)'s restart goal into increments.
|
||||||
|
18. **[input.md](input.md) — the input module.** Broadcasting input events (keyboard,
|
||||||
|
mouse, joystick): why a synchronous rendezvous can't fan out to many listeners, the
|
||||||
|
asynchronous `ipc_send` primitive built to fix it, and the per-device subscribe/publish
|
||||||
|
service layered on top.
|
||||||
|
19. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
||||||
how `while (true) hlt` parks the CPU safely once there's nothing left to do.
|
how `while (true) hlt` parks the CPU safely once there's nothing left to do.
|
||||||
|
|
||||||
Start with the north star:
|
Start with the north star:
|
||||||
@@ -68,11 +100,19 @@ Cutting across all of these:
|
|||||||
- **[smp.md](smp.md) — multiple cores.** A design/research note on how microkernels
|
- **[smp.md](smp.md) — multiple cores.** A design/research note on how microkernels
|
||||||
(L4, seL4) handle SMP — big kernel lock vs per-CPU vs multikernel — and how the
|
(L4, seL4) handle SMP — big kernel lock vs per-CPU vs multikernel — and how the
|
||||||
right choice depends on whether danos is chasing real-time or resilience.
|
right choice depends on whether danos is chasing real-time or resilience.
|
||||||
|
- **[coding-standards.md](coding-standards.md) — coding standards.** The naming rule the
|
||||||
|
tree follows: non-acronyms are spelled out in full (`message`, not `msg`), files are
|
||||||
|
`kebab-case`, code follows Zig's case conventions, and the handful of exceptions
|
||||||
|
(POSIX/C ABI names, `init`/`len`/`ptr`, acronyms).
|
||||||
- **[sysv.md](sysv.md) — the calling convention.** What "the kernel is SysV" means,
|
- **[sysv.md](sysv.md) — the calling convention.** What "the kernel is SysV" means,
|
||||||
and why the loader→kernel boundary has to pin it (the RDI-vs-RCX handoff).
|
and why the loader→kernel boundary has to pin it (the RDI-vs-RCX handoff).
|
||||||
- **[testing.md](testing.md) — testing.** How the kernel is tested by booting it in
|
- **[testing.md](testing.md) — testing.** How the kernel is tested by booting it in
|
||||||
QEMU and asserting on its serial output — reproducibly, and structured so the
|
QEMU and asserting on its serial output — reproducibly, and structured so the
|
||||||
same tests run across architectures.
|
same tests run across architectures.
|
||||||
|
- **[logging.md](logging.md) — logging.** The multi-sink diagnostic log (serial,
|
||||||
|
0xE9 debugcon, file later) kept separate from the framebuffer display, plus the
|
||||||
|
robustness path: optional framebuffer, POST-code checkpoints, and a persistent
|
||||||
|
panic breadcrumb so the kernel survives — and can be diagnosed — with no output.
|
||||||
|
|
||||||
## How the pieces relate
|
## How the pieces relate
|
||||||
|
|
||||||
@@ -90,19 +130,95 @@ passing messages over **[IPC](ipc.md)** channels — runs, its CPU-specific bits
|
|||||||
behind the [arch](arch.md) boundary, and when idle, or on a panic, it **halts**
|
behind the [arch](arch.md) boundary, and when idle, or on a panic, it **halts**
|
||||||
([halting.md](halting.md)).
|
([halting.md](halting.md)).
|
||||||
|
|
||||||
|
Above that line the microkernel proper begins: **discovery** ([discovery.md](discovery.md),
|
||||||
|
[acpi.md](acpi.md)) learns what hardware exists, ring-3 processes ask the kernel for
|
||||||
|
things through the small **[syscall](syscall.md)** table, isolated servers reach each
|
||||||
|
other over IPC **endpoints** ([ipc.md](ipc.md)), and a **[driver](drivers.md)** claims
|
||||||
|
a device, maps its registers, and sleeps until the hardware interrupts it — which is
|
||||||
|
the whole reason for the arrangement ([vision.md](vision.md)).
|
||||||
|
|
||||||
|
## Repository layout
|
||||||
|
|
||||||
|
danos is a **monorepo of sub-projects**. Each service or driver is a directory that is
|
||||||
|
its own Zig module — it can hold as many files as it needs, and other sub-projects
|
||||||
|
reach it *by module name*, never by a path into its files. The source tree deliberately
|
||||||
|
**mirrors the runtime FHS** ([danos-file-system-hierarchy-FSH.md](danos-file-system-hierarchy-FSH.md)):
|
||||||
|
what you see under `system/` in the source is what a running danos represents under
|
||||||
|
`/system`.
|
||||||
|
|
||||||
|
**A sub-project is addressed by its directory; its entry point repeats the directory's
|
||||||
|
name.** `system/services/init/` contains `init.zig` (its root), and produces a binary
|
||||||
|
addressed as **`system/services/init`** — the repeated leaf resolves away:
|
||||||
|
|
||||||
|
| Source (root file) | Addressed as (module / binary / FHS path) |
|
||||||
|
|----------------------------------------|--------------------------------------------|
|
||||||
|
| `system/services/init/init.zig` | `system/services/init` → `/system/services/init` |
|
||||||
|
| `system/drivers/hpet/hpet.zig` | `system/drivers/hpet` → `/system/drivers/hpet` |
|
||||||
|
| `library/runtime/runtime.zig` | `library/runtime` (the `runtime` module) |
|
||||||
|
|
||||||
|
In **source**, a sub-project is a directory so it can hold many files — the entry is
|
||||||
|
`init/init.zig`, beside it `vfs/vfs-test.zig`, `vfs/protocol.zig`, and so on. When
|
||||||
|
**addressed or installed**, that collapses to the single canonical path: the `init`
|
||||||
|
binary installs to `/system/services/init` (a file at that path), not
|
||||||
|
`/system/services/init/init`. The repeated leaf exists only in source; the directory is
|
||||||
|
the identity, the entry file is its implementation. (Same idea as a Go package being its
|
||||||
|
directory, or a macOS `.app` bundle addressed by the bundle, not the executable within.)
|
||||||
|
A sub-project's extra files are reached through the module, never as separate paths.
|
||||||
|
|
||||||
|
```
|
||||||
|
system/ → /system danos's own internals (the self-representation)
|
||||||
|
boot-handoff.zig the loader↔kernel contract (the `boot-handoff` module)
|
||||||
|
abi.zig the private kernel↔runtime syscall ABI (the `abi` module)
|
||||||
|
parameters.zig initial-ramdisk.zig shared contracts
|
||||||
|
kernel/ IPC, memory, scheduling, the private syscall dispatch
|
||||||
|
architecture/x86_64/ the `architecture` module (never named by generic code)
|
||||||
|
devices/ the device model /system/devices reflects (+ aml/)
|
||||||
|
device-abi.zig the device wire types (the `device-abi` module)
|
||||||
|
drivers/ hpet/ bus/ one sub-project per driver → /system/drivers
|
||||||
|
services/ init/ vfs/ device-manager/ system servers → /system/services (vfs/ holds
|
||||||
|
vfs.zig, vfs-test.zig, protocol.zig)
|
||||||
|
library/ → /lib libraries, one sub-directory each
|
||||||
|
runtime/ the danos-native runtime — the stable application ABI
|
||||||
|
posix/ POSIX/C compatibility, layered over runtime
|
||||||
|
boot/ → /boot the loaders
|
||||||
|
tools/ test/ host-side build + QEMU test harness
|
||||||
|
```
|
||||||
|
|
||||||
|
A sub-project exposes its **public interface as a module**: `system/services/vfs/` owns
|
||||||
|
the VFS wire protocol (`protocol.zig`, the `vfs-protocol` module), which the POSIX
|
||||||
|
layer imports by name. `usb`/`block` drivers will expose their protocols the same way.
|
||||||
|
|
||||||
|
`library/posix/` is special: it is the **one place** POSIX/C spellings are allowed
|
||||||
|
verbatim (`stat`, `O_CREAT`, `fopen`, `errno`). Everywhere else follows the danos
|
||||||
|
naming rule with no exception — see [coding-standards.md](coding-standards.md). The
|
||||||
|
POSIX layer calls the runtime, never the kernel's system calls directly, so it never
|
||||||
|
appears in the private-ABI path.
|
||||||
|
|
||||||
## Source map
|
## Source map
|
||||||
|
|
||||||
| Area | Code |
|
| Area | Code |
|
||||||
|------|------|
|
|------|------|
|
||||||
| Boot methods (one per way of booting the kernel) | `src/boot/` — `efi.zig` (UEFI) → `BOOTX64.efi` |
|
| Boot methods (one per way of booting the kernel) | `boot/` — `efi.zig` (UEFI) → `BOOTX64.efi` |
|
||||||
| Kernel entry, panic, bring-up | `src/kernel/main.zig` |
|
| Kernel entry, panic, bring-up | `system/kernel/kernel.zig` |
|
||||||
| Shared loader↔kernel contract (`BootInfo`, `Framebuffer`, `MemoryMap`, ABI) | `src/root.zig` |
|
| Loader↔kernel handoff (`BootInfo`, `Framebuffer`, `MemoryMap`, VM layout) | `system/boot-handoff.zig` |
|
||||||
| Physical frame allocator | `src/kernel/pmm.zig` |
|
| Private kernel↔runtime syscall ABI (`SystemCall`, mmap prot flags, `page_size`) — the runtime speaks it, not apps | `system/abi.zig` |
|
||||||
| Kernel heap (`std.mem.Allocator`) | `src/kernel/heap.zig` |
|
| Device wire types (`DeviceDescriptor`, `DeviceClass`, …) | `system/devices/device-abi.zig` |
|
||||||
| Scheduler (fixed-priority preemptive; blocking, wait queues) | `src/kernel/sched.zig` |
|
| Physical frame allocator | `system/kernel/pmm.zig` |
|
||||||
| IPC channels (message passing) | `src/kernel/ipc.zig` |
|
| Kernel heap (`std.mem.Allocator`) | `system/kernel/heap.zig` |
|
||||||
| Framebuffer text console (mirrors to serial) | `src/kernel/console.zig` |
|
| Scheduler (fixed-priority preemptive; blocking, wait queues) | `system/kernel/scheduler.zig` |
|
||||||
| In-kernel test cases | `src/kernel/tests.zig` |
|
| Big kernel lock + interrupt-safe critical sections | `system/kernel/sync.zig` |
|
||||||
| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/timer, serial, linker script) | `src/kernel/arch/x86_64/` |
|
| IPC channels between kernel threads (message passing) | `system/kernel/ipc.zig` |
|
||||||
|
| IPC endpoints: cross-address-space call/reply, handles, notifications | `system/kernel/ipc-synchronous.zig` |
|
||||||
|
| User processes: ELF loading, address spaces, the syscall table | `system/kernel/process.zig` |
|
||||||
|
| Device tree + claim capability + `device_register` containment | `system/kernel/devices-broker.zig` |
|
||||||
|
| IRQ-as-IPC: routing a device interrupt to a driver's endpoint | `system/kernel/irq.zig` |
|
||||||
|
| Hardware discovery (ACPI/device tree) behind one neutral device model | `system/devices/` |
|
||||||
|
| Framebuffer text console (mirrors to serial) | `system/kernel/console.zig` |
|
||||||
|
| In-kernel test cases | `system/kernel/tests.zig` |
|
||||||
|
| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/IO-APIC/timer, serial, linker script) | `system/kernel/architecture/x86_64/` |
|
||||||
|
| danos-native runtime (`runtime`): syscall wrappers, heap, IPC, device access — the stable application ABI | `library/runtime/` |
|
||||||
|
| POSIX/C compatibility (`posix`): unistd, stdio — the one place POSIX names are allowed | `library/posix/` |
|
||||||
|
| System services (init, the VFS server + `protocol`, the device-manager) | `system/services/` |
|
||||||
|
| Device drivers, one sub-project each (`hpet` leaf driver, `bus` bus driver) | `system/drivers/` |
|
||||||
| Build + `run-x86-64` (QEMU/OVMF) | `build.zig` |
|
| Build + `run-x86-64` (QEMU/OVMF) | `build.zig` |
|
||||||
| QEMU integration test harness | `test/qemu_test.py` |
|
| QEMU integration test harness | `test/qemu_test.py` |
|
||||||
|
|||||||
+6
-6
@@ -16,13 +16,13 @@ the RSDT's address is a field *inside* the RSDP. The platform follows that point
|
|||||||
UEFI configuration table
|
UEFI configuration table
|
||||||
│ the loader reads the RSDP's physical address
|
│ the loader reads the RSDP's physical address
|
||||||
▼
|
▼
|
||||||
BootInfo.acpi_rsdp (u64, in the shared `danos` module) src/root.zig
|
BootInfo.acpi_rsdp (u64, in the loader↔kernel handoff) system/boot-handoff.zig
|
||||||
│ the kernel forwards the whole BootInfo
|
│ the kernel forwards the whole BootInfo
|
||||||
▼
|
▼
|
||||||
platform.discover(boot_info, …) src/device/platform.zig
|
platform.discover(boot_info, …) system/devices/platform.zig
|
||||||
│ reads boot_info.acpi_rsdp, hands it to the ACPI backend
|
│ reads boot_info.acpi_rsdp, hands it to the ACPI backend
|
||||||
▼
|
▼
|
||||||
acpi.discover(rsdp_phys, …) src/device/acpi.zig
|
acpi.discover(rsdp_phys, …) system/devices/acpi.zig
|
||||||
│ dereferences the RSDP, reads the pointer it contains
|
│ dereferences the RSDP, reads the pointer it contains
|
||||||
▼
|
▼
|
||||||
RSDP ──(a field in the struct)──► RSDT / XSDT ──► SDTs (MADT, MCFG, FADT, HPET, DSDT…)
|
RSDP ──(a field in the struct)──► RSDT / XSDT ──► SDTs (MADT, MCFG, FADT, HPET, DSDT…)
|
||||||
@@ -35,7 +35,7 @@ successor the **XSDT** (ACPI 2.0+) — which in turn lists every other SDT.
|
|||||||
## Step 1 — the loader finds the RSDP
|
## Step 1 — the loader finds the RSDP
|
||||||
|
|
||||||
Only the firmware knows where ACPI lives, so the RSDP must be grabbed while UEFI is
|
Only the firmware knows where ACPI lives, so the RSDP must be grabbed while UEFI is
|
||||||
still up. `acpiRootSystemDescriptorPointer()` in `src/boot/efi.zig` walks the UEFI
|
still up. `acpiRootSystemDescriptorPointer()` in `boot/efi.zig` walks the UEFI
|
||||||
**configuration table** for the ACPI GUID and returns the vendor pointer — the same
|
**configuration table** for the ACPI GUID and returns the vendor pointer — the same
|
||||||
"grab it before `ExitBootServices`" pattern as the [framebuffer](framebuffer.md) and
|
"grab it before `ExitBootServices`" pattern as the [framebuffer](framebuffer.md) and
|
||||||
the [memory map](memory-map.md).
|
the [memory map](memory-map.md).
|
||||||
@@ -44,11 +44,11 @@ the [memory map](memory-map.md).
|
|||||||
|
|
||||||
The loader can't just call the device module: the bootloader binary and the kernel
|
The loader can't just call the device module: the bootloader binary and the kernel
|
||||||
binary are compiled separately, and **the loader isn't linked against the `platform`
|
binary are compiled separately, and **the loader isn't linked against the `platform`
|
||||||
module at all** (it imports only the shared `danos` module). So instead of a call, it
|
module at all** (it imports only the `boot-handoff` contract). So instead of a call, it
|
||||||
deposits a value in the handoff struct:
|
deposits a value in the handoff struct:
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
// src/boot/efi.zig — while boot services are still up
|
// boot/efi.zig — while boot services are still up
|
||||||
.acpi_rsdp = if (acpiRootSystemDescriptorPointer()) |p| @intFromPtr(p) else 0,
|
.acpi_rsdp = if (acpiRootSystemDescriptorPointer()) |p| @intFromPtr(p) else 0,
|
||||||
```
|
```
|
||||||
|
|
||||||
|
|||||||
+13
-13
@@ -13,7 +13,7 @@ runtime dispatch. `build.zig` exposes one architecture's code as a module called
|
|||||||
|
|
||||||
```zig
|
```zig
|
||||||
const arch_mod = b.addModule("arch", .{
|
const arch_mod = b.addModule("arch", .{
|
||||||
.root_source_file = b.path("src/kernel/arch/x86_64/cpu.zig"),
|
.root_source_file = b.path("system/kernel/architecture/x86_64/cpu.zig"),
|
||||||
});
|
});
|
||||||
```
|
```
|
||||||
|
|
||||||
@@ -26,7 +26,7 @@ arch.halt(); // never says "x86_64"
|
|||||||
```
|
```
|
||||||
|
|
||||||
Adding a second architecture is then a build-time choice: create
|
Adding a second architecture is then a build-time choice: create
|
||||||
`src/kernel/arch/aarch64/`, and point the `arch` module at it when the target CPU is
|
`system/kernel/arch/aarch64/`, and point the `arch` module at it when the target CPU is
|
||||||
AArch64. `main.zig` and `console.zig` don't change. **That compiler-checked module
|
AArch64. `main.zig` and `console.zig` don't change. **That compiler-checked module
|
||||||
boundary _is_ the architecture interface** — when a new arch is missing a function
|
boundary _is_ the architecture interface** — when a new arch is missing a function
|
||||||
the generic kernel calls, the build fails and names exactly what's missing.
|
the generic kernel calls, the build fails and names exactly what's missing.
|
||||||
@@ -36,7 +36,7 @@ the generic kernel calls, the build fails and names exactly what's missing.
|
|||||||
The split follows a simple test: does it name a CPU instruction, a hardware
|
The split follows a simple test: does it name a CPU instruction, a hardware
|
||||||
register, or a memory-management structure? If so, it's arch-specific.
|
register, or a memory-management structure? If so, it's arch-specific.
|
||||||
|
|
||||||
| Arch-specific — `src/kernel/arch/x86_64/` | Generic — kernel core |
|
| Arch-specific — `system/kernel/architecture/x86_64/` | Generic — kernel core |
|
||||||
|---|---|
|
|---|---|
|
||||||
| `cpu.zig`: `halt()` (`hlt`), later GDT/IDT/paging | `console.zig` — pure pixel math, works anywhere |
|
| `cpu.zig`: `halt()` (`hlt`), later GDT/IDT/paging | `console.zig` — pure pixel math, works anywhere |
|
||||||
| `linker.ld` — link layout, load address | `main.zig` — `kmain` orchestration, panic handler |
|
| `linker.ld` — link layout, load address | `main.zig` — `kmain` orchestration, panic handler |
|
||||||
@@ -51,9 +51,9 @@ should end up on the generic side; the arch module stays small.
|
|||||||
There are really two independent questions, and it's worth not conflating them:
|
There are really two independent questions, and it's worth not conflating them:
|
||||||
|
|
||||||
- **CPU architecture** (x86_64 vs AArch64): instructions, MMU, interrupts →
|
- **CPU architecture** (x86_64 vs AArch64): instructions, MMU, interrupts →
|
||||||
`src/kernel/arch/<cpu>/`.
|
`system/kernel/arch/<cpu>/`.
|
||||||
- **Boot protocol** (UEFI vs Raspberry Pi firmware + device tree): handled
|
- **Boot protocol** (UEFI vs Raspberry Pi firmware + device tree): handled
|
||||||
*separately*, because loaders are their own binaries. `src/boot/efi.zig` builds
|
*separately*, because loaders are their own binaries. `boot/efi.zig` builds
|
||||||
`BOOTX64.efi`, a distinct executable from the kernel ELF. On a Pi there is no
|
`BOOTX64.efi`, a distinct executable from the kernel ELF. On a Pi there is no
|
||||||
separate loader at all — the firmware jumps straight into the kernel with a
|
separate loader at all — the firmware jumps straight into the kernel with a
|
||||||
device-tree pointer, so that entry work would live in the AArch64 arch code.
|
device-tree pointer, so that entry work would live in the AArch64 arch code.
|
||||||
@@ -61,29 +61,29 @@ There are really two independent questions, and it's worth not conflating them:
|
|||||||
|
|
||||||
## Current x86_64 contents
|
## Current x86_64 contents
|
||||||
|
|
||||||
- **`src/kernel/arch/x86_64/cpu.zig`** — the `arch` module root. Exposes `halt()` (see
|
- **`system/kernel/architecture/x86_64/cpu.zig`** — the `arch` module root. Exposes `halt()` (see
|
||||||
[halting.md](halting.md)), `init()` (bring up the descriptor tables),
|
[halting.md](halting.md)), `init()` (bring up the descriptor tables),
|
||||||
`enablePaging()`, `setFaultHandler`, `readCr2`/`readCr3`, and the `CpuState`
|
`enablePaging()`, `setFaultHandler`, `readCr2`/`readCr3`, and the `CpuState`
|
||||||
trap frame.
|
trap frame.
|
||||||
- **`src/kernel/arch/x86_64/gdt.zig`** / **`idt.zig`** / **`tss.zig`** — the GDT, IDT and
|
- **`system/kernel/architecture/x86_64/gdt.zig`** / **`idt.zig`** / **`tss.zig`** — the GDT, IDT and
|
||||||
TSS plus CPU-exception handling (see [interrupts.md](interrupts.md)).
|
TSS plus CPU-exception handling (see [interrupts.md](interrupts.md)).
|
||||||
- **`src/kernel/arch/x86_64/paging.zig`** — the kernel's page tables (see
|
- **`system/kernel/architecture/x86_64/paging.zig`** — the kernel's page tables (see
|
||||||
[paging.md](paging.md)).
|
[paging.md](paging.md)).
|
||||||
- **`src/kernel/arch/x86_64/apic.zig`** — the Local APIC and its timer, the source of
|
- **`system/kernel/architecture/x86_64/apic.zig`** — the Local APIC and its timer, the source of
|
||||||
device interrupts (see [device-interrupts.md](device-interrupts.md)).
|
device interrupts (see [device-interrupts.md](device-interrupts.md)).
|
||||||
- **`src/kernel/arch/x86_64/serial.zig`** / **`io.zig`** — the COM1 UART (the kernel's
|
- **`system/kernel/architecture/x86_64/serial.zig`** / **`io.zig`** — the COM1 UART (the kernel's
|
||||||
machine-readable log channel, see [testing.md](testing.md)) and the shared
|
machine-readable log channel, see [testing.md](testing.md)) and the shared
|
||||||
port-I/O + MSR primitives.
|
port-I/O + MSR primitives.
|
||||||
- **`src/kernel/arch/x86_64/isr.s`** — the exception stubs, the `lgdt`/`lidt`/`ltr` load
|
- **`system/kernel/architecture/x86_64/isr.s`** — the exception stubs, the `lgdt`/`lidt`/`ltr` load
|
||||||
helpers, and the context switch (`switch_context` / `task_trampoline`, see
|
helpers, and the context switch (`switch_context` / `task_trampoline`, see
|
||||||
[scheduling.md](scheduling.md)) — real assembly, since Zig inline asm can't
|
[scheduling.md](scheduling.md)) — real assembly, since Zig inline asm can't
|
||||||
express them.
|
express them.
|
||||||
- **`src/kernel/arch/x86_64/linker.ld`** — the kernel link layout (fixed low load
|
- **`system/kernel/architecture/x86_64/linker.ld`** — the kernel link layout (fixed low load
|
||||||
address, one PT_LOAD per permission set).
|
address, one PT_LOAD per permission set).
|
||||||
|
|
||||||
The kernel entry point `_start` currently still lives in the generic `main.zig` as
|
The kernel entry point `_start` currently still lives in the generic `main.zig` as
|
||||||
a thin trampoline into `kmain`. It's arch-adjacent (its calling convention is
|
a thin trampoline into `kmain`. It's arch-adjacent (its calling convention is
|
||||||
x86_64 [SysV](sysv.md), via the shared `danos.kernel_abi`), but it's three lines
|
x86_64 [SysV](sysv.md), via the shared `system.kernel_abi`), but it's three lines
|
||||||
and mostly generic, so it stays put for now. When AArch64 arrives — where entry means setting
|
and mostly generic, so it stays put for now. When AArch64 arrives — where entry means setting
|
||||||
up a stack and reading a device-tree pointer from a register — the entry work will
|
up a stack and reading a device-tree pointer from a register — the entry work will
|
||||||
be substantial and per-arch, and *that* is when we extract an entry interface into
|
be substantial and per-arch, and *that* is when we extract an entry interface into
|
||||||
|
|||||||
+4
-4
@@ -18,7 +18,7 @@ matters for understanding why. This page maps the landscape so the
|
|||||||
new ISA.
|
new ISA.
|
||||||
|
|
||||||
They are as different from each other as either is from x86-64: separate registers,
|
They are as different from each other as either is from x86-64: separate registers,
|
||||||
page-table formats, and calling conventions. Each needs its own `src/kernel/arch/<name>/`.
|
page-table formats, and calling conventions. Each needs its own `system/kernel/arch/<name>/`.
|
||||||
|
|
||||||
## The Raspberry Pi models
|
## The Raspberry Pi models
|
||||||
|
|
||||||
@@ -57,16 +57,16 @@ the DTB/ACPI tells you what devices exist.
|
|||||||
|
|
||||||
## What danos needs, layer by layer
|
## What danos needs, layer by layer
|
||||||
|
|
||||||
- **One CPU arch module: `src/kernel/arch/aarch64/`** — covering the Zero 2 W and Pi 3-5,
|
- **One CPU arch module: `system/kernel/arch/aarch64/`** — covering the Zero 2 W and Pi 3-5,
|
||||||
providing the same `arch` interface as x86_64: `halt`, context switch,
|
providing the same `arch` interface as x86_64: `halt`, context switch,
|
||||||
interrupt/exception vectors, page tables, a UART, a timer. No `src/kernel/arch/arm/` is
|
interrupt/exception vectors, page tables, a UART, a timer. No `system/kernel/arch/arm/` is
|
||||||
planned (see the decision above), so there's a single ARM backend to write.
|
planned (see the decision above), so there's a single ARM backend to write.
|
||||||
- **A device-tree boot path.** Since stock Pis boot via DTB, danos needs an entry
|
- **A device-tree boot path.** Since stock Pis boot via DTB, danos needs an entry
|
||||||
that parses the DTB's `/memory` and `/reserved-memory` into the neutral
|
that parses the DTB's `/memory` and `/reserved-memory` into the neutral
|
||||||
[`MemoryMap`](memory-map.md) — the same neutral handoff `efi.zig` produces, just
|
[`MemoryMap`](memory-map.md) — the same neutral handoff `efi.zig` produces, just
|
||||||
from a different source. This is where keeping boot-protocol knowledge on the
|
from a different source. This is where keeping boot-protocol knowledge on the
|
||||||
loader side (as we did for the UEFI memory-map classification) pays off.
|
loader side (as we did for the UEFI memory-map classification) pays off.
|
||||||
- **The UEFI loader mostly carries over.** `src/boot/efi.zig` is largely
|
- **The UEFI loader mostly carries over.** `boot/efi.zig` is largely
|
||||||
boot-*protocol* code (`std.os.uefi` protocol calls), not x86 code. Its only truly
|
boot-*protocol* code (`std.os.uefi` protocol calls), not x86 code. Its only truly
|
||||||
x86-specific bits are the ELF machine check (`.X86_64`) and the SysV calling
|
x86-specific bits are the ELF machine check (`.X86_64`) and the SysV calling
|
||||||
convention for the kernel jump. So an `aarch64`-UEFI target (QEMU `virt` + AAVMF)
|
convention for the kernel jump. So an `aarch64`-UEFI target (QEMU `virt` + AAVMF)
|
||||||
|
|||||||
@@ -0,0 +1,194 @@
|
|||||||
|
# Coding standards
|
||||||
|
|
||||||
|
Conventions for danos source. The overriding one, from which most of the rest follows:
|
||||||
|
|
||||||
|
> **Names are spelled out in full. An identifier is not abbreviated unless the
|
||||||
|
> abbreviation is an acronym.**
|
||||||
|
|
||||||
|
`interruptDispatch`, not `intDisp`. `message_len`, not `message_len` (`msg` expands, `len`
|
||||||
|
is a Zig idiom — see the exceptions). `devices_broker`, not `devices_broker`. `scheduler`, not
|
||||||
|
`sched`. The cost of a longer name is paid once, at the keyboard; the cost of a
|
||||||
|
cryptic one is paid every time the code is read, by everyone who reads it. In a
|
||||||
|
microkernel whose whole argument is that a human can hold each piece in their head,
|
||||||
|
that trade is not close.
|
||||||
|
|
||||||
|
## The rule, precisely
|
||||||
|
|
||||||
|
**Acronyms and initialisms stay.** They *are* the full name — expanding them would make
|
||||||
|
the code worse, not better. `IPC`, `MMIO`, `DMA`, `IRQ`, `TSS`, `GDT`, `IDT`, `APIC`,
|
||||||
|
`GSI`, `HPET`, `ACPI`, `PCI`, `EOI`, `BAR`, `ECAM`, `MSI`, `CPU`, `ELF`, `ABI`, `UEFI`,
|
||||||
|
`MMU`, `TLB`, `ISR`, `ISA`, `GAS`, `HAL`, `PMM`, `VMM`, `VFS`, `HID`, `HCD`, `SMP`,
|
||||||
|
`AML`, `MADT`, `MCFG`, `FADT`, `RSDP`, `XSDT`, `RSDT`, `GOP`, `EDID`, `TSC`, `PIT`,
|
||||||
|
`RTC`, `LAPIC`, `SIPI`. In code they carry whatever case the surrounding convention
|
||||||
|
demands: `Hal` the type, `hal` the variable, `mapMmio` the function.
|
||||||
|
|
||||||
|
**Everything else is spelled out.** If it's a word with letters removed, restore them:
|
||||||
|
|
||||||
|
| Abbreviation | Full |
|
||||||
|
|---|---|
|
||||||
|
| `proto` | `protocol` |
|
||||||
|
| `msg` | `message` |
|
||||||
|
| `desc` | `descriptor` |
|
||||||
|
| `res` | `resource` |
|
||||||
|
| `recv` | `receive` |
|
||||||
|
| `buf` | `buffer` |
|
||||||
|
| `cur` | `current` |
|
||||||
|
| `src` / `dst` | `source` / `destination` |
|
||||||
|
| `idx` | `index` |
|
||||||
|
| `addr` | `address` |
|
||||||
|
| `reg` | `register` |
|
||||||
|
| `prev` | `previous` |
|
||||||
|
| `cfg` / `config` | `configuration` |
|
||||||
|
| `arch` | `architecture` |
|
||||||
|
| `sched` | `scheduler` |
|
||||||
|
| `dev` | `device` |
|
||||||
|
| `sys` / `syscall` | `system` / `system_call` |
|
||||||
|
| `info` | `information` |
|
||||||
|
| `dt` | `device_tree` |
|
||||||
|
| `ep` | `endpoint` |
|
||||||
|
| `rt` | `runtime` |
|
||||||
|
| `func` | `function` |
|
||||||
|
| `phys` / `virt` | `physical` / `virtual` |
|
||||||
|
| `wq` | `wait_queue` |
|
||||||
|
|
||||||
|
This list is illustrative, not exhaustive. The rule is the rule; when you meet a new
|
||||||
|
abbreviation, expand it.
|
||||||
|
|
||||||
|
## Exceptions
|
||||||
|
|
||||||
|
Three, and only three.
|
||||||
|
|
||||||
|
1. **Foreign ABI names are spelled exactly as the ABI spells them — but only inside
|
||||||
|
the layer that *is* that ABI.** A function that *is* the C or POSIX interface keeps
|
||||||
|
its name: `fopen`, `fwrite`, `fread`, `malloc`, `calloc`, `realloc`, `free`,
|
||||||
|
`memcpy`, `mmap`, `munmap`, `open`, `read`, `write`, `close`, `lseek`, `stat`,
|
||||||
|
`errno`, `O_CREAT`. We don't get to rename `fwrite` to `fileWrite` — it wouldn't be
|
||||||
|
`fwrite` any more.
|
||||||
|
|
||||||
|
**This exception is scoped to one place: `library/posix/`.** A file under
|
||||||
|
`library/posix/` *is* the foreign ABI, so it keeps the ABI's spellings — that is the
|
||||||
|
whole rule for that directory. **Everywhere else, Zig/danos naming applies with no
|
||||||
|
POSIX exception**, so there is nothing to get wrong: if you're not in
|
||||||
|
`library/posix/`, expand it. A concept POSIX also has gets a danos name outside that
|
||||||
|
layer — the VFS wire protocol carries a `FileStatus`, not a `Stat`, and a `create`
|
||||||
|
flag, not `O_CREAT`; `library/posix/` is what maps `stat`→`status` and
|
||||||
|
`O_CREAT`→`create` at the boundary. (The `syscall` *wrappers* elsewhere are not an
|
||||||
|
exception to this — they wrap the private danos ABI, so they use danos names.)
|
||||||
|
|
||||||
|
2. **Zig idioms are spelled the way Zig spells them.** Three names are the language's,
|
||||||
|
not ours, and are left alone:
|
||||||
|
- **`init` / `deinit`** — the constructor convention (`std.ArrayList.init`), not a
|
||||||
|
shortening of "initialize".
|
||||||
|
- **`len` / `ptr`** — the slice field names (`slice.len`, `slice.ptr`). Our own
|
||||||
|
structs use bare `len`/`ptr` fields to mirror them, so a reader carries one
|
||||||
|
mental model. (Compounds still expand: a field is `message_len`, not
|
||||||
|
`message_length` — `len` is kept, `msg` is not.)
|
||||||
|
- The builtins (`@min`, `@max`, `@memcpy`) and `allocator.alloc` / `.create` are
|
||||||
|
Zig's spelling.
|
||||||
|
|
||||||
|
The rule governs the names *we* coin.
|
||||||
|
|
||||||
|
3. **Single-letter variables in a trivial local scope.** `for (items) |item, i|` may
|
||||||
|
keep `i`; a coordinate may be `x`, `y`. The moment the scope is big enough that the
|
||||||
|
letter's meaning isn't obvious on sight, give it a real name. When in doubt, name it.
|
||||||
|
|
||||||
|
That's all — no Unix-abbreviation exception. The source directories are full words
|
||||||
|
(`system`, `library`, not `src`/`lib`), and there is no daemon `d` suffix: a driver
|
||||||
|
lives in `system/drivers/` and a service in `system/services/`, so the *location*
|
||||||
|
already says what it is. Encoding the role in the name too (`busd`, `vfsd`) is
|
||||||
|
redundant — the program is just `bus`, `vfs`. Don't put in a name what its directory
|
||||||
|
already tells you.
|
||||||
|
|
||||||
|
## A note on collisions
|
||||||
|
|
||||||
|
Two identifiers can legitimately expand to the same word. When they do, keep both
|
||||||
|
meaningful by renaming one to its *specific* identity rather than the generic
|
||||||
|
expansion. Two cases resolved this way:
|
||||||
|
|
||||||
|
- The `config` module (compile-time tunables — `maximum_cpus`, `timer_hz`) would
|
||||||
|
collide with `cfg` (a `PlatformConfiguration` value) at `configuration`. The module
|
||||||
|
became **`parameters`**, which is what it holds.
|
||||||
|
- The kernel `device.zig` module would collide with `dev` (a device value) at
|
||||||
|
`device`. The module alias became **`device_model`**, which is what it is — the
|
||||||
|
device data model (`Device`, `DeviceTree`, `ResourceKind`).
|
||||||
|
- The `Namespace` module alias (`ns`/`nsp` across the AML files) collides with a
|
||||||
|
`Namespace` **instance**. Resolved by dropping the module alias entirely — the two
|
||||||
|
types it provided are imported directly (`const Node = @import("namespace.zig").Node;`)
|
||||||
|
— which frees `namespace` for the instance.
|
||||||
|
|
||||||
|
A related case is one abbreviation with two meanings. In the AML code, `op` means
|
||||||
|
**opcode** (`opcodes.zig`, the `*_opcode` constants) but `Op` in `BinaryOperation` /
|
||||||
|
`LogicOperation` means **operation** — distinguished by case. The per-opcode parser
|
||||||
|
handlers, formerly `opName`/`opField`, are `parseName`/`parseField`: they *parse* the
|
||||||
|
opcode's structure, which says what they do without overloading "op".
|
||||||
|
|
||||||
|
## Case and file names
|
||||||
|
|
||||||
|
Within those spelling rules, follow Zig's own conventions:
|
||||||
|
|
||||||
|
- **Types** — `PascalCase`: `DeviceDescriptor`, `Endpoint`, `WaitQueue`.
|
||||||
|
- **Functions** — `camelCase`: `mapUserDeviceInto`, `notifyFromIsr`.
|
||||||
|
- **Variables, fields, constants** — `snake_case`: `message_length`, `devices_broker`,
|
||||||
|
`notify_badge_bit`.
|
||||||
|
|
||||||
|
**File names are `kebab-case`.** A file named for a multi-word thing hyphenates it:
|
||||||
|
`device-tree.zig`, `ipc-synchronous.zig`, `vfs-protocol.zig`, `devices-broker.zig`. A
|
||||||
|
single word or acronym needs no hyphen: `scheduler.zig`, `paging.zig`, `apic.zig`,
|
||||||
|
`idt.zig`. (The module *alias* a file is imported under still follows the code
|
||||||
|
conventions above — `snake_case` — because it's an identifier, not a filename.)
|
||||||
|
|
||||||
|
**A sub-project's entry point repeats its directory's name** — `init/init.zig`,
|
||||||
|
`runtime/runtime.zig`, `hpet/hpet.zig` — and the sub-project is addressed by the
|
||||||
|
*directory* (`system/services/init`, `library/runtime`), with the repeated leaf
|
||||||
|
resolving away. See the repository-layout section of [README.md](README.md).
|
||||||
|
|
||||||
|
## Named values, not magic numbers
|
||||||
|
|
||||||
|
The naming rule has a twin: **a value with meaning gets a name, too.** The same
|
||||||
|
principle drives both — a reader should never have to leave the code to understand it.
|
||||||
|
An abbreviated *name* forces a reader to guess; a bare *number* forces them worse, out
|
||||||
|
to a spec or a header or a comment three files away, to learn what the value even *is*.
|
||||||
|
If `0x0C` is the PCI serial-bus class, the code says `BaseClass.serial_bus`, not `0x0C`;
|
||||||
|
if `0x04` is the ACPI IRQ resource descriptor, it says `SmallResourceType.irq`, not
|
||||||
|
`0x04`. The number is an implementation detail of the name — recorded once, where the
|
||||||
|
name is defined, and never spelled again at a use site.
|
||||||
|
|
||||||
|
**Prefer an `enum`** when the values form a set (device classes, AML opcodes, resource
|
||||||
|
descriptor types, states): the type then also says *which* set a value belongs to, and
|
||||||
|
the compiler rejects a value from the wrong one. A lone `pub const` with a descriptive
|
||||||
|
name suffices for a one-off (`const large_descriptor_bit = 0x80`). Reach for the enum
|
||||||
|
the moment code elsewhere compares against, packs, or produces the value — a packed PCI
|
||||||
|
class triple is written from named parts (`.serial_bus`, `.usb`, `.xhci`), never as
|
||||||
|
`0x0C_03_30` under a comment that decodes the bytes.
|
||||||
|
|
||||||
|
The exceptions are the numbers that carry no hidden meaning: `0` and `1` as plain zero
|
||||||
|
and one, an index step, a field width, a bit shift. `x + 1`, `buffer[0]`, and `<< 8`
|
||||||
|
need no christening — there is nothing to look up. The test is exactly the naming test:
|
||||||
|
*would a reader have to look this up to know what it means?* If yes, name it. This is
|
||||||
|
what `opcodes.zig`'s `*_opcode` constants, `acpi-ids`'s `HardwareId`, and `pci-class`'s
|
||||||
|
class enums already are — reference data defined once and named everywhere it is used.
|
||||||
|
|
||||||
|
## Why acronyms are the line
|
||||||
|
|
||||||
|
Because an acronym has no letters to restore. `MMIO` doesn't become "memory mapped
|
||||||
|
input output" in code — that expansion is what the acronym *is for*. But `msg` is just
|
||||||
|
`message` with three letters stolen, and stealing them buys nothing a reader wants. The
|
||||||
|
test for "is this an abbreviation I must expand" is simply: *is there a longer word this
|
||||||
|
is a clipped form of?* If yes, write the word. If it's an initialism standing in for a
|
||||||
|
phrase, leave it.
|
||||||
|
|
||||||
|
## Zen of Zig
|
||||||
|
|
||||||
|
* Communicate intent precisely.
|
||||||
|
* Edge cases matter.
|
||||||
|
* Favor reading code over writing code.
|
||||||
|
* Only one obvious way to do things.
|
||||||
|
* Runtime crashes are better than bugs.
|
||||||
|
* Compile errors are better than runtime crashes.
|
||||||
|
* Incremental improvements.
|
||||||
|
* Avoid local maximums.
|
||||||
|
* Reduce the amount one must remember.
|
||||||
|
* Focus on code rather than style.
|
||||||
|
* Resource allocation may fail; resource deallocation must succeed.
|
||||||
|
* Memory is a resource.
|
||||||
|
* Together we serve the users.
|
||||||
@@ -0,0 +1,121 @@
|
|||||||
|
# DanOS Filesystem Hierarchy Standard (DFHS)
|
||||||
|
|
||||||
|
Most modern Unix and Unix-like operating systems follow the FHS. DanOS has its own FHS structure which extends the unix FHS. This is provided by virtual file system driver (VFS).
|
||||||
|
|
||||||
|
## Directory structure
|
||||||
|
|
||||||
|
| Path | Description |
|
||||||
|
|------------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||||
|
| / | Primary hierarchy root and root directory of the entire file system hierarchy. |
|
||||||
|
| /bin | Essential command binaries that need to be available in single-user mode, including to bring up the system or repair it, for all users (e.g., cat, ls, cp). |
|
||||||
|
| /boot | Boot loader files (e.g., EFI, initial-ramdisk.img ). |
|
||||||
|
| /dev | POSIX Device files (e.g., /dev/null, /dev/disk0, /dev/tty, /dev/random). |
|
||||||
|
| /etc | Host-specific system-wide configuration files. |
|
||||||
|
| /home | Users' home directories, containing saved files, personal settings, etc. |
|
||||||
|
| /lib | Libraries essential for the binaries in /bin and /sbin. eg realtime, system, ipc etc. |
|
||||||
|
| /sbin | Essential system binaries (e.g init) |
|
||||||
|
| /srv | Site-specific data served by this system, such as data and scripts for web servers, data offered by FTP servers, and repositories for version control systems |
|
||||||
|
| /system | DanOS operating system files (similar idea to C:\Windows). A true representation of danos — its layout mirrors the source tree, so `/system` is what danos *is*. |
|
||||||
|
| /system/devices | danos virtual device tree e.g. similar to /sys on linux but with danos device tree conventions (the structures in the devices module) |
|
||||||
|
| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/hpet) |
|
||||||
|
| /system/services | system-service binaries — the VFS server, init, and other user-mode servers (e.g. /system/services/vfs, /system/services/init) |
|
||||||
|
| /system/kernel | the kernel image |
|
||||||
|
| /tmp | Directory for temporary files (see also /var/tmp). Often not preserved between system reboots and may be severely size-restricted. |
|
||||||
|
| /usr | Secondary hierarchy for read-only user data; contains the majority of (multi-)user utilities and applications. Should be shareable and read-only. |
|
||||||
|
| /var | Variable files: files whose content is expected to continually change during normal operation of the system, such as logs, spool files, and temporary e-mail files. |
|
||||||
|
|
||||||
|
## File types
|
||||||
|
|
||||||
|
POSIX specifies the long format of the ls command to represent the Unix file type as the first letter for an entry.
|
||||||
|
|
||||||
|
| type | symbol | Description |
|
||||||
|
|-------------------|--------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||||
|
| regular | - | An ordinary file holding an uninterpreted byte stream. Reads and writes are positional, and the file grows on demand (e.g., a binary in /bin, a config file in /etc). |
|
||||||
|
| directory | d | A container mapping names to other files. It may only be modified through directory operations, never written to directly. |
|
||||||
|
| symbolic link | l | A file whose contents are a path that is resolved in its place. The target need not exist, and may cross mount points. |
|
||||||
|
| FIFO special | p | A named pipe: an in-order byte stream between processes, where writers block until a reader opens the other end. |
|
||||||
|
| block special | b | A device node addressed in fixed-size blocks with the kernel free to buffer and reorder access (e.g., /dev/disk0). |
|
||||||
|
| character special | c | A device node addressed as an unbuffered byte stream, delivered to the driver in order (e.g., /dev/tty, /dev/null). |
|
||||||
|
| socket | s | A named endpoint for bidirectional message-passing between processes, bound to a path rather than an address. |
|
||||||
|
|
||||||
|
## /dev
|
||||||
|
|
||||||
|
`/dev` holds the names through which processes reach devices. It is deliberately not
|
||||||
|
the device tree: the tree — every node discovered by ACPI or PCI enumeration, with its
|
||||||
|
resources and its parent — lives under [/system/devices](#directory-structure) and is
|
||||||
|
addressed by device id. `/dev` is the much smaller set of devices that have a driver
|
||||||
|
willing to serve them, addressed by name.
|
||||||
|
|
||||||
|
A device node is not a file the VFS can read. The bytes live in a driver process
|
||||||
|
([drivers.md](drivers.md)), so opening a `/dev` name has to resolve to that driver's
|
||||||
|
IPC endpoint, and subsequent reads and writes are calls against it. This is what
|
||||||
|
`system/services/vfs/vfs.zig` reserves for M10 and what the `Stat.kind` field is for; **none of it is
|
||||||
|
implemented today.** The current VFS is a flat, in-memory ramfs of eight nodes, with no
|
||||||
|
directories at all and `kind` hardcoded to zero. The three sections below describe the
|
||||||
|
intended shape, and are honest about which parts the kernel can already support.
|
||||||
|
|
||||||
|
### Character devices
|
||||||
|
|
||||||
|
A character device is a byte stream with no addressable position: bytes are delivered
|
||||||
|
to the driver in the order written, and a read consumes what is there. Terminals,
|
||||||
|
serial lines, keyboards and mice are all of this shape. These are the natural first
|
||||||
|
device nodes in danos, because a character driver needs nothing the kernel doesn't
|
||||||
|
already provide — it claims its device, maps its registers with `mmio_map`, and blocks
|
||||||
|
on `replyWait` for either an interrupt or a client request. `system/drivers/hpet/hpet.zig` is already
|
||||||
|
that program, minus the client half.
|
||||||
|
|
||||||
|
The obstacle was never the file type; it is which hardware a ring-3 driver can reach.
|
||||||
|
Direct `in`/`out` from user space is still a #GP (no TSS I/O bitmap, IOPL never raised),
|
||||||
|
but a driver no longer needs it: **`io_read`/`io_write`** grant port access the same way
|
||||||
|
`mmio_map` grants memory — gated by `device_claim` and the device's discovered `io_port`
|
||||||
|
resource. So the 16550 UART at `0x3F8` and the PS/2 controller at `0x60`/`0x64` (and thus
|
||||||
|
`/dev/ttyS0` and a keyboard node) are now writable as ordinary ring-3 drivers; the
|
||||||
|
low-rate legacy hardware that needs port I/O is fine with a syscall per access. A
|
||||||
|
memory-mapped device such as the framebuffer, needing no port I/O at all, remains the
|
||||||
|
easiest first entry.
|
||||||
|
|
||||||
|
### Block devices
|
||||||
|
|
||||||
|
A block device is addressed in fixed-size blocks and, unlike a character device, the
|
||||||
|
layer above is free to buffer, reorder, coalesce and retry requests against it. Disks
|
||||||
|
and other persistent storage are the whole population of this class.
|
||||||
|
|
||||||
|
A block driver is now **writable, but not yet memory-safe.** Every storage controller
|
||||||
|
worth naming is a bus master: it is programmed by handing it the physical address of a
|
||||||
|
descriptor ring and left to read and write memory on its own. That ring is exactly what
|
||||||
|
**`dma_alloc`** now provides — physically contiguous, pinned, uncacheable, with its
|
||||||
|
physical address disclosed — and **`/lib/mmio`**'s barriers order the descriptor writes
|
||||||
|
against the doorbell, and **`msi_bind`** delivers completions. So an AHCI or NVMe driver
|
||||||
|
can be written today (the M14/M15 work in [driver-model.md](driver-model.md); the earlier
|
||||||
|
"cannot host a block driver at all" is no longer true).
|
||||||
|
|
||||||
|
What is *not* yet true is that it is safe. A device programmed with an arbitrary physical
|
||||||
|
address writes to arbitrary physical memory, and page tables do not sit between a device
|
||||||
|
and RAM — an IOMMU does. The IOMMU is now *detected* (M16), but no translation domains
|
||||||
|
are programmed, so granting a DMA-capable device to a driver process is still equivalent
|
||||||
|
to granting ring 0. Until per-device domains confine a driver's DMA to the buffers it
|
||||||
|
`dma_alloc`'d, a block driver works but forfeits the isolation that motivates user-space
|
||||||
|
drivers — enforcement is the next step, and lands with that first driver. A ramdisk over
|
||||||
|
the initial ramdisk remains the one block-shaped thing that needs no driver process at all.
|
||||||
|
|
||||||
|
### Pseudo-devices
|
||||||
|
|
||||||
|
A pseudo-device has the interface of a device and no hardware behind it: `/dev/null`
|
||||||
|
discarding writes and reading as end-of-file, `/dev/zero` reading as an endless run of
|
||||||
|
zero bytes, `/dev/full` failing writes with `ENOSPC`, `/dev/random` and `/dev/urandom`
|
||||||
|
yielding unpredictable bytes.
|
||||||
|
|
||||||
|
These are the only `/dev` entries danos can implement immediately, and they are the
|
||||||
|
sensible place to start, because they are exactly the entries that need no driver
|
||||||
|
process, no `device_claim`, no MMIO grant and no interrupt. The VFS server answers them
|
||||||
|
out of its own address space — `null` and `zero` are a few lines each in
|
||||||
|
`system/services/vfs/vfs.zig`'s `read` and `write` handlers. Doing so forces the two pieces of
|
||||||
|
structure that every later device node depends on and that the flat ramfs currently
|
||||||
|
lacks: a directory, so that `/dev/null` is a path rather than a name; and a populated
|
||||||
|
`Stat.kind`, so that a caller can tell a character device from a regular file.
|
||||||
|
|
||||||
|
`/dev/random` is the one that is not free. It needs an entropy source, and the honest
|
||||||
|
options on this kernel are `RDRAND`/`RDSEED` where CPUID advertises them, and the HPET
|
||||||
|
counter's low bits as a poor fallback. Neither is a seeded CSPRNG, and a `/dev/random`
|
||||||
|
that is merely unpredictable-looking is worse than none — nothing should be keyed from
|
||||||
|
it until it is a real one.
|
||||||
+34
-12
@@ -18,7 +18,7 @@ Interrupt delivery on modern x86 goes through the **APIC**, not the legacy 8259
|
|||||||
PIC. There are two halves; we only need one so far:
|
PIC. There are two halves; we only need one so far:
|
||||||
|
|
||||||
- The **Local APIC** (per-CPU, memory-mapped at physical `0xFEE00000`) handles the
|
- The **Local APIC** (per-CPU, memory-mapped at physical `0xFEE00000`) handles the
|
||||||
CPU's own timer and receives interrupts routed to it. `src/kernel/arch/x86_64/apic.zig`.
|
CPU's own timer and receives interrupts routed to it. `system/kernel/architecture/x86_64/apic.zig`.
|
||||||
- The **IO-APIC** routes *external* device lines (keyboard, etc.) to LAPIC vectors.
|
- The **IO-APIC** routes *external* device lines (keyboard, etc.) to LAPIC vectors.
|
||||||
Not needed for the timer — it'll arrive with the keyboard.
|
Not needed for the timer — it'll arrive with the keyboard.
|
||||||
|
|
||||||
@@ -89,7 +89,6 @@ if (state.vector < 32) {
|
|||||||
on_fault(state); // exception: report and halt (never returns)
|
on_fault(state); // exception: report and halt (never returns)
|
||||||
} else if (handlers[state.vector]) |handler| {
|
} else if (handlers[state.vector]) |handler| {
|
||||||
handler(); // device: run the registered handler
|
handler(); // device: run the registered handler
|
||||||
apic.eoi(); // ...acknowledge the LAPIC
|
|
||||||
}
|
}
|
||||||
// else: spurious/unhandled — deliberately no EOI
|
// else: spurious/unhandled — deliberately no EOI
|
||||||
```
|
```
|
||||||
@@ -100,10 +99,23 @@ Two things make device interrupts *return* where exceptions don't:
|
|||||||
flows back to `isr_common`, which restores every register it saved and executes
|
flows back to `isr_common`, which restores every register it saved and executes
|
||||||
`iretq` — resuming the interrupted instruction exactly. (This is why the stub
|
`iretq` — resuming the interrupted instruction exactly. (This is why the stub
|
||||||
saves *all* the general registers.)
|
saves *all* the general registers.)
|
||||||
2. **End-of-interrupt.** After handling, we write the LAPIC's EOI register. Miss
|
2. **End-of-interrupt.** Somewhere in there we write the LAPIC's EOI register. Miss
|
||||||
this and the LAPIC thinks we're still busy and never delivers the next
|
this and the LAPIC thinks we're still busy and never delivers the next
|
||||||
interrupt. It's the single most common "my timer fired once and stopped" bug.
|
interrupt. It's the single most common "my timer fired once and stopped" bug.
|
||||||
|
|
||||||
|
**Each handler issues its own EOI**, rather than the dispatcher doing it around the
|
||||||
|
call. That looks like a needless devolution while the timer is the only device, and
|
||||||
|
`apic.timerTick` indeed does nothing but `eoi()` before bumping its counter (early,
|
||||||
|
because the tick hook is the scheduler, which may switch tasks and not return
|
||||||
|
promptly — the LAPIC mustn't wait on it).
|
||||||
|
|
||||||
|
It stops looking needless with the second device. A *routed* interrupt — one arriving
|
||||||
|
through the I/O APIC from a real device line — must be **masked before it is
|
||||||
|
acknowledged**, because a level-triggered line is still asserted at EOI time and would
|
||||||
|
redeliver instantly, forever. Only the handler knows which discipline its source
|
||||||
|
needs, so only the handler can sequence it. See [drivers.md](drivers.md), where the
|
||||||
|
device is quieted by a driver in ring 3, long after the ISR has returned.
|
||||||
|
|
||||||
A device handler is a plain `fn () void` — a timer or keyboard handler doesn't need
|
A device handler is a plain `fn () void` — a timer or keyboard handler doesn't need
|
||||||
the interrupted registers. (Note: the stubs don't save the SSE/vector registers, so
|
the interrupted registers. (Note: the stubs don't save the SSE/vector registers, so
|
||||||
a handler must not use them; ours don't.)
|
a handler must not use them; ours don't.)
|
||||||
@@ -131,14 +143,24 @@ If the APIC weren't enabled, or `sti` were missing, or EOI were forgotten, the
|
|||||||
count would stay put and the test would fail. That it advances — while the CPU was
|
count would stay put and the test would fail. That it advances — while the CPU was
|
||||||
spinning in unrelated code — is the whole mechanism working end to end.
|
spinning in unrelated code — is the whole mechanism working end to end.
|
||||||
|
|
||||||
|
## Since (done elsewhere)
|
||||||
|
|
||||||
|
- **Preemption**: the timer handler is where the scheduler decides to switch — the
|
||||||
|
reason a *returning* interrupt matters. See [scheduling.md](scheduling.md).
|
||||||
|
- **`sleep()` / timeouts** built on the calibrated clock.
|
||||||
|
- **The I/O APIC, routed**: external device lines now reach a vector, and the
|
||||||
|
interrupt is delivered onward to a *user-space* driver as an IPC message. See
|
||||||
|
[drivers.md](drivers.md).
|
||||||
|
- **Uncacheable MMIO**: device grants are mapped `PCD|PWT` (strong-uncacheable) for
|
||||||
|
user drivers — see [paging.md](paging.md).
|
||||||
|
|
||||||
## What's next (not done here)
|
## What's next (not done here)
|
||||||
|
|
||||||
- **The keyboard**: bring up the IO-APIC, route its IRQ to a vector, and read
|
- **The keyboard**: the PS/2 controller is port-mapped (`0x60`/`0x64`), and port I/O is
|
||||||
scancodes from the PS/2 controller — the first *input* device.
|
now available to ring 3 via the claim-gated `io_read`/`io_write` syscalls
|
||||||
- **`sleep()` / timeouts** built on the calibrated clock (the monotonic
|
([drivers.md](drivers.md)) — so the first *input* device is unblocked; it just needs
|
||||||
`uptimeMs()` is in place).
|
writing (claim the controller, `irq_bind` GSI 1, read scancodes from `0x60`).
|
||||||
- **Uncacheable MMIO**: the LAPIC page is currently mapped writeback-cacheable like
|
- **MSI-X**: `msi_bind` gives one per-device edge-triggered vector (M15); MSI-X's
|
||||||
the rest of the identity map. QEMU tolerates it, but real hardware wants MMIO
|
multi-vector table (many queues per device, e.g. NVMe) is the remaining extension.
|
||||||
marked uncacheable (via the page's cache bits or an MTRR).
|
- **The LAPIC's own page** is still mapped writeback-cacheable like the rest of the
|
||||||
- **Preemption**: once there are tasks, the timer handler is where the scheduler
|
identity map. QEMU tolerates it; real hardware wants it uncacheable.
|
||||||
decides to switch — the reason a *returning* interrupt matters.
|
|
||||||
|
|||||||
@@ -0,0 +1,156 @@
|
|||||||
|
# The device manager
|
||||||
|
|
||||||
|
**Status: the protocol and supervision are built** (M18.1, 2026-07-13): `hello`
|
||||||
|
with its deadline, supervised spawn, restart with backoff, and the crash-loop
|
||||||
|
cap are in — usb-xhci-bus is the first conforming driver, and the
|
||||||
|
`driver-restart` scenario proves fault → backoff → re-claim → cap end to end.
|
||||||
|
Tree reports are built too (M18.2, 2026-07-13): the xHCI driver scans its
|
||||||
|
root-hub ports and reports each connected device (`child_added`); the manager
|
||||||
|
mirrors them and prunes a dead reporter's children, and the `usb-report`
|
||||||
|
scenario proves report → prune → respawn → re-report. The application surface is built (M18.3, 2026-07-13):
|
||||||
|
`enumerate` and `subscribe` over IPC, with `device-list` as the first client —
|
||||||
|
the manager is now the one answer to "what devices exist" for applications.
|
||||||
|
The primitives underneath are real ([process-management.md](process-management.md):
|
||||||
|
spawn/supervise/kill/exit-notification; [driver-model.md](driver-model.md): the device
|
||||||
|
table as a capability system; [drivers.md](drivers.md): claim/map/IRQ), and the first
|
||||||
|
per-device driver spawn works (the device manager matches the xHCI controller by PCI
|
||||||
|
class and spawns `usb-xhci-bus` with the device id as argv[1]). This document designs
|
||||||
|
the rest: the device manager as **the tree, the matcher, and the supervisor** — the
|
||||||
|
policy process that turns [resilience.md](resilience.md)'s restart goal into practice
|
||||||
|
for drivers.
|
||||||
|
|
||||||
|
How processes stop, reload, and report their deaths is deliberately **not** in this
|
||||||
|
document: that is the universal lifecycle every danos process speaks —
|
||||||
|
[process-lifecycle.md](process-lifecycle.md), signals over IPC and the stable
|
||||||
|
`runtime.process` interface. The device manager is that design's first serious
|
||||||
|
customer, not its owner. Its own protocol contains nothing lifecycle-shaped; a
|
||||||
|
driver is stopped, health-checked, and buried exactly like any other process.
|
||||||
|
|
||||||
|
## The tree: structure in the manager, authority in the kernel
|
||||||
|
|
||||||
|
The device tree is two things fused: *information* (what exists, how it nests) and
|
||||||
|
*authority* (a descriptor is a licence to map physical memory). They separate:
|
||||||
|
|
||||||
|
- The **kernel keeps the capability system** — device, I/O-port, and interrupt
|
||||||
|
claims, resource containment on `device_register`, the
|
||||||
|
`mmio_map`/`irq_bind`/`msi_bind` gates — and **cleans all of it up when a process
|
||||||
|
dies** (settled; it is increment 1 of
|
||||||
|
[process-lifecycle.md](process-lifecycle.md)). The three invariants in
|
||||||
|
[driver-model.md](driver-model.md) stay exactly where they are. A device manager
|
||||||
|
that could mint MMIO mappings by its own say-so would be a second kernel, and a
|
||||||
|
buggy one would un-earn everything the microkernel bought.
|
||||||
|
- The **device manager owns the tree as data** — identity, topology, naming, driver
|
||||||
|
matching, hotplug events, and being the one process everything else asks about
|
||||||
|
devices. Firmware discovery seeds it (today via the kernel's snapshot); **bus
|
||||||
|
drivers grow it** by reporting what they see; applications query and watch it.
|
||||||
|
`device_enumerate` fades to a manager-internal (then deleted) seam.
|
||||||
|
|
||||||
|
Long-term, discovery itself leaves the kernel — but not *into* the manager. PCI
|
||||||
|
enumeration is a **pci-bus driver**: the manager spawns it against the host bridge
|
||||||
|
(already a device with the ECAM window as a resource), it scans, it reports functions
|
||||||
|
like any bus reports children. ACPI becomes an **acpi service** that interprets the
|
||||||
|
tables and reports the namespace. The manager only orchestrates and merges. Moving
|
||||||
|
AML interpretation out of ring 0 is its own project on its own track; nothing here
|
||||||
|
depends on when it lands.
|
||||||
|
|
||||||
|
## The protocol
|
||||||
|
|
||||||
|
A `device-manager-protocol` module (the vfs-protocol pattern): extern-struct
|
||||||
|
messages, a version in the handshake, reserved fields everywhere. The manager is a
|
||||||
|
well-known endpoint (`ipc.register(.device_manager)`); the badge tells it who is
|
||||||
|
talking; the same endpoint receives its children's exit notifications — one loop,
|
||||||
|
one world.
|
||||||
|
|
||||||
|
| Direction | Message | Purpose |
|
||||||
|
|---|---|---|
|
||||||
|
| driver → manager | `hello { version, role, device_id }` | confirms the argv assignment, starts the deadline clock |
|
||||||
|
| bus → manager | `child_added { parent, identity, resources }` | one node the bus discovered |
|
||||||
|
| bus → manager | `child_removed { id }` | unplug, or the bus lost it |
|
||||||
|
| app → manager | `enumerate` | snapshot of the tree (read-only) |
|
||||||
|
| app → manager | `subscribe` | receive published add/remove events |
|
||||||
|
|
||||||
|
`hello` is the one deadline the manager enforces itself: spawned and silent past the
|
||||||
|
deadline means wrong binary, wrong protocol version, or wedged before main — apply
|
||||||
|
the stop sequence and the restart policy. Everything else lifecycle-shaped
|
||||||
|
(terminate, the common `ping` liveness call, exit reasons) arrives through
|
||||||
|
[process-lifecycle.md](process-lifecycle.md)'s vocabulary, not this protocol.
|
||||||
|
|
||||||
|
Assignment stays argv (`usb-xhci-bus <device id>`) for now — simple, and it works.
|
||||||
|
The step after `hello` exists is delegation: the manager claims (or is granted) the
|
||||||
|
devices and passes the claim to the driver over IPC (the M13 capability-transfer
|
||||||
|
mechanism), replacing first-come-first-served `device_claim` with policy. Identity in
|
||||||
|
`child_added` is per-bus: PCI children carry the class triple (`pci_class`, as the
|
||||||
|
xHCI match already uses); USB children carry the (class, subclass, protocol) triple
|
||||||
|
from usb-ids.zig — each bus's native language, decoded by the shared ids modules.
|
||||||
|
|
||||||
|
## Supervision and restart
|
||||||
|
|
||||||
|
Every driver is spawned with the manager's exit endpoint (`spawnSupervised` — built).
|
||||||
|
On a death notification:
|
||||||
|
|
||||||
|
1. **Read the reason** ([process-lifecycle.md](process-lifecycle.md) increment 2).
|
||||||
|
Clean exit → it meant to; don't restart. Fault or missed `hello` deadline →
|
||||||
|
restart with **backoff**, and a crash-loop cap (three fast deaths → mark failed,
|
||||||
|
stop respawning, log loudly; a later `reload` to the manager can retry).
|
||||||
|
2. **Prune the subtree** the dead bus driver reported. Its children describe
|
||||||
|
protocol state (xHCI slot ids, transfer rings) that died with the process;
|
||||||
|
keeping the nodes would be keeping a lie. Watchers receive `child_removed` — the
|
||||||
|
input service losing, then regaining, a keyboard is the *honest* description of
|
||||||
|
what happened. The restarted instance rediscovers and re-reports.
|
||||||
|
3. **The claim is already free** because the kernel released it at death — the
|
||||||
|
restarted instance claims the same controller and comes up.
|
||||||
|
|
||||||
|
Who supervises the supervisor: **init** (PID 1), which already supervises the
|
||||||
|
services it starts. If the manager dies, drivers keep running (they hold their
|
||||||
|
claims; the kernel doesn't care who their supervisor was — though their exit
|
||||||
|
notifications now dangle harmlessly). The restarted manager re-learns the world:
|
||||||
|
kernel snapshot, then a re-`hello` round — drivers answer a broadcast or are stopped
|
||||||
|
and respawned. Full state handoff is deliberately not attempted.
|
||||||
|
|
||||||
|
## Thin drivers, class protocols
|
||||||
|
|
||||||
|
The [driver-model.md](driver-model.md) three-shape split, restated as processes:
|
||||||
|
|
||||||
|
- A **bus driver** (usb-xhci-bus) owns its controller — claim, MMIO, IRQ/MSI, DMA
|
||||||
|
rings — and offers a *transfer* protocol ("submit a control transfer to device N",
|
||||||
|
built from the usb-abi request constructors) plus tree reports to the manager.
|
||||||
|
- A **class driver** (usb-hid, usb-storage) owns nothing: it is matched to a reported
|
||||||
|
child by its identity triple, speaks the bus's transfer protocol downward and its
|
||||||
|
service's protocol upward — HID reports to the input service, blocks to the block
|
||||||
|
service. It works unchanged over any controller.
|
||||||
|
- **Services** (input, display, block) aggregate class drivers and face applications.
|
||||||
|
|
||||||
|
Each arrow is a protocol module. The manager routes none of the data plane — it
|
||||||
|
introduces the parties (matching), supervises them (lifecycle), and gets out of the
|
||||||
|
way.
|
||||||
|
|
||||||
|
## Increments
|
||||||
|
|
||||||
|
Increments 1–4 are the lifecycle prerequisites and live in
|
||||||
|
[process-lifecycle.md](process-lifecycle.md) (claim cleanup on death, exit reasons,
|
||||||
|
published exit events, signals + `runtime.process`). On top of those:
|
||||||
|
|
||||||
|
5. **device-manager-protocol**: `hello`, supervised spawn with restart policy;
|
||||||
|
usb-xhci-bus becomes the first conforming driver.
|
||||||
|
6. **Tree reports**: `child_added`/`child_removed`; the manager mirrors; xHCI reports
|
||||||
|
the mouse and keyboard QEMU already hangs off it.
|
||||||
|
7. **App surface**: `enumerate`/`subscribe` over IPC; `device_enumerate` retreats
|
||||||
|
to a manager-internal seam.
|
||||||
|
8. **Discovery migration** — DONE (M19–M20, 2026-07-13): pci-bus driver (M19)
|
||||||
|
then the acpi service (M20) moved enumeration to ring 3; the kernel seeds
|
||||||
|
only the host bridge and the acpi-tables node. See
|
||||||
|
[m19-m20-plan.md](m19-m20-plan.md).
|
||||||
|
|
||||||
|
## Settled questions (2026-07-12)
|
||||||
|
|
||||||
|
- **Stateful buses**: pruning the subtree on bus-driver death is right for USB. A
|
||||||
|
future storage bus with in-flight writes wants drain-before-terminate — which is
|
||||||
|
exactly the `deadline_ms` parameter `stop()` already has; a per-driver deadline
|
||||||
|
is one value in the manager's policy table when such a bus arrives. No design
|
||||||
|
change.
|
||||||
|
- **Manager death**: drivers survive the manager; the restarted manager re-learns
|
||||||
|
the world (above). Checkpointing driver state with the manager is deferred until
|
||||||
|
something demonstrates the need.
|
||||||
|
- **Matching stays code until the third bus.** `driverFor`/`pciDriverFor` are
|
||||||
|
honest at two bus types; the third triggers the manifest (a driver declares what
|
||||||
|
it binds: a PCI class triple, a USB class triple, an ACPI `_HID`).
|
||||||
@@ -167,3 +167,27 @@ free; discovery on x86 is partly about *finding* what ARM just tells you.
|
|||||||
- [ipc.md](ipc.md) — the channels that interrupts-as-messages and the device manager
|
- [ipc.md](ipc.md) — the channels that interrupts-as-messages and the device manager
|
||||||
will ride on.
|
will ride on.
|
||||||
- [vision.md](vision.md) — why drivers belong in isolated user space at all.
|
- [vision.md](vision.md) — why drivers belong in isolated user space at all.
|
||||||
|
|
||||||
|
## Update (M19.3, 2026-07-13): PCI enumeration left the kernel
|
||||||
|
|
||||||
|
The kernel now seeds only the `pci_host_bridge` node (ECAM window, MMIO
|
||||||
|
apertures derived from the memory map's holes, bus range, and the 16-bit I/O
|
||||||
|
window). The per-function walk moved to the ring-3 `pci-bus` driver
|
||||||
|
([device-manager.md](device-manager.md)): it claims the bridge, repeats the
|
||||||
|
ECAM scan through its mmio grant, and `device_register`s what it finds, which
|
||||||
|
the device manager mirrors and matches. The ACPI namespace walk follows in M20;
|
||||||
|
the static tables (MADT, HPET, MCFG, FADT + `\\_S5`) stay kernel-side.
|
||||||
|
|
||||||
|
## Update (M20.3, 2026-07-13): ACPI enumeration left the kernel too
|
||||||
|
|
||||||
|
The kernel no longer folds the AML namespace's Device objects into the device
|
||||||
|
tree. It still parses the *static* tables (MADT for SMP, HPET for the tick, MCFG
|
||||||
|
for the host bridge, FADT) and still builds the AML namespace — but only to read
|
||||||
|
the `\\_S5` sleep type for poweroff. Device discovery is the ring-3 **acpi
|
||||||
|
service** ([device-manager.md](device-manager.md)): it claims the `acpi-tables`
|
||||||
|
node the kernel publishes (the AML blobs, a broad io_port grant, the SCI),
|
||||||
|
re-parses the same blobs with the shared AML module, evaluates `_STA`/`_CRS`,
|
||||||
|
and registers + reports each `_HID` device — the device manager matches drivers
|
||||||
|
(ps2-bus) from those reports. With M19's pci-bus driver, discovery now runs
|
||||||
|
entirely in user space; the kernel seeds only the host bridge and the
|
||||||
|
acpi-tables node.
|
||||||
|
|||||||
@@ -0,0 +1,367 @@
|
|||||||
|
# The driver model: buses, classes, and host controllers
|
||||||
|
|
||||||
|
[drivers.md](drivers.md) shows how to write *a* driver — claim a device, map its
|
||||||
|
registers, sleep on its interrupt. That's enough for a leaf device like the HPET. It is
|
||||||
|
not enough for a disk, a keyboard, or a network card, because those hang off a
|
||||||
|
*controller*, on a *bus*, speaking a *protocol*, and no single process should have to
|
||||||
|
know all three.
|
||||||
|
|
||||||
|
Real driver stacks factor into three shapes. This document is about what each one is,
|
||||||
|
what the kernel must give it, how they share code — and precisely which primitive each
|
||||||
|
is still blocked on.
|
||||||
|
|
||||||
|
## Three shapes
|
||||||
|
|
||||||
|
| Shape | Owns | Reaches hardware by | Talks to |
|
||||||
|
|---|---|---|---|
|
||||||
|
| **Host controller driver** (HCD) | a controller — an xHCI PCI function, an AHCI port block | `mmio_map` + `irq_bind` + DMA | the devices behind it, in its bus's language |
|
||||||
|
| **Bus driver** | a bus — a PCI bridge, a USB hub | `device_register`, to publish what it finds | class drivers, over IPC |
|
||||||
|
| **Class / protocol driver** | *nothing* | *nothing* | its bus driver, over IPC |
|
||||||
|
|
||||||
|
The last row is the surprising one and the whole point. A USB keyboard driver touches
|
||||||
|
no registers, takes no interrupts, and maps no memory. It sends HID protocol messages
|
||||||
|
to whatever published the device, and it works identically whether the controller
|
||||||
|
below is xHCI, EHCI, or a Raspberry Pi's DWC2. That is what buys you drivers that
|
||||||
|
outlive the hardware they were written for.
|
||||||
|
|
||||||
|
In practice **HCD and bus driver are usually the same process**. An xHCI driver is a
|
||||||
|
host controller driver (it owns the PCI function, its BARs, its interrupt, its DMA
|
||||||
|
rings) *and* a bus driver (it enumerates USB devices and publishes them). Splitting
|
||||||
|
them is a fiction; what matters is that both *roles* have kernel support, because a
|
||||||
|
plain bus driver with no controller — a USB hub — is also a real thing.
|
||||||
|
|
||||||
|
## The device table is the spine
|
||||||
|
|
||||||
|
danos already has the right central structure. `system/kernel/devices-broker.zig` holds a table of
|
||||||
|
`DeviceDesc`, each with a parent, a class, and a set of resources. Firmware discovery
|
||||||
|
seeds it ([discovery.md](discovery.md)); `device_register` grows it.
|
||||||
|
|
||||||
|
Three invariants make it a capability system rather than a directory:
|
||||||
|
|
||||||
|
1. **A claim is exclusive.** `device_claim(id)` succeeds once. Everything downstream —
|
||||||
|
`mmio_map`, `irq_bind`, `device_register` — checks `devices_broker.ownerOf(id) == me`.
|
||||||
|
2. **A descriptor is a licence to map physical memory.** Whoever claims a device may
|
||||||
|
map its `.memory` resources and bind its `.irq` resources. This is why
|
||||||
|
`device_register` cannot be a free-for-all.
|
||||||
|
3. **Therefore: containment.** Every resource of a registered child must lie inside a
|
||||||
|
resource of the same kind on its parent (`devices_broker.contains`). A bus driver can only
|
||||||
|
ever *subdivide* what it already holds. Without this, `device_register` would be a
|
||||||
|
syscall named "map any physical page you like."
|
||||||
|
|
||||||
|
Containment is transitive by construction: a grandchild is contained in its child,
|
||||||
|
which is contained in the bus. Nothing can be laundered through a chain.
|
||||||
|
|
||||||
|
Note that firmware topology does **not** obey containment, and isn't asked to — a PCI
|
||||||
|
function's BAR is not inside its host bridge's `bus_range`, because a bus-number range
|
||||||
|
is not an address window. Discovery is trusted; user space is not.
|
||||||
|
|
||||||
|
### What a bus driver looks like
|
||||||
|
|
||||||
|
`system/drivers/bus/bus.zig` is the smallest honest one. Its "bus" is the HPET's register block and
|
||||||
|
its "devices" are the block's comparators:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
_ = dev.claim(bus.id); // 1. own the bus
|
||||||
|
const base = dev.mmioMap(bus.id, 0).?; // 2. enumerate it — from the hardware
|
||||||
|
const n = ((cap.* >> 8) & 0x1F) + 1; // GENERAL_CAP says how many children
|
||||||
|
|
||||||
|
for (0..n) |i| { // 3. publish each child
|
||||||
|
var child = std.mem.zeroes(dev.DeviceDesc);
|
||||||
|
child.class = @intFromEnum(dev.DeviceClass.timer);
|
||||||
|
child.resource_count = 1;
|
||||||
|
child.resources[0] = .{ .kind = memory,
|
||||||
|
.start = bus_mmio.start + 0x100 + 0x20 * i,
|
||||||
|
.len = 0x20 };
|
||||||
|
_ = dev.register(bus.id, &child).?; // kernel checks containment
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Each child is left **unclaimed**, which is the handoff: a comparator driver can now
|
||||||
|
`device_claim` one and `mmio_map` it, and will see only its own 0x20-byte window. A child
|
||||||
|
whose window escapes the bus is refused — `bus` asserts that, and the `bus` test
|
||||||
|
asserts the kernel's table upholds it.
|
||||||
|
|
||||||
|
A USB device has *no* resources at all: `resource_count = 0`, because it's addressed
|
||||||
|
through its controller, not by MMIO. That case is allowed and is the common one.
|
||||||
|
|
||||||
|
## Families: sharing code between drivers
|
||||||
|
|
||||||
|
A "family" is two modules, not one:
|
||||||
|
|
||||||
|
- **A logic module** — the parts of the bus that every driver on it re-derives. Config
|
||||||
|
space walking and BAR decode for PCI. Descriptor parsing, control transfers, and hub
|
||||||
|
protocol for USB.
|
||||||
|
- **A protocol module** — the IPC message types that let a class driver talk to
|
||||||
|
*whatever* published its device. This is the part that makes class drivers portable.
|
||||||
|
|
||||||
|
danos already has one of each: `library/runtime/device.zig` is a logic module,
|
||||||
|
[`system/services/vfs/protocol.zig`](system/services/vfs/protocol.zig) is a protocol module shared by `system/services/vfs/vfs.zig`
|
||||||
|
and its clients. The pattern generalises directly:
|
||||||
|
|
||||||
|
```
|
||||||
|
library/
|
||||||
|
runtime/ module "runtime" — syscalls, heap, ipc, device, stdio
|
||||||
|
mmio/ module "mmio" — volatile register access + barriers [M14]
|
||||||
|
bus/
|
||||||
|
pci/ module "pci" — ECAM, BAR decode, capability walk
|
||||||
|
usb/ module "usb" — descriptors, control transfers, hubs
|
||||||
|
proto/
|
||||||
|
vfs/ module "vfs-protocol" (today: system/services/vfs/protocol.zig)
|
||||||
|
block/ module "block-protocol"
|
||||||
|
hid/ module "hid-protocol"
|
||||||
|
|
||||||
|
system/drivers/ one sub-project each → /system/drivers (no `d` suffix)
|
||||||
|
xhci/ HCD + bus driver imports runtime, pci, usb, mmio
|
||||||
|
usb-hid/ class driver imports runtime, usb, hid-protocol
|
||||||
|
block/ class driver imports runtime, block-protocol
|
||||||
|
```
|
||||||
|
|
||||||
|
The only build change needed: [`addUserBinary`](build.zig) currently takes exactly one
|
||||||
|
module (`rt_mod`) and injects it. It should take a slice of modules. That's a
|
||||||
|
five-line change, and it's the *entire* mechanism — Zig modules already give you
|
||||||
|
everything else.
|
||||||
|
|
||||||
|
The discipline that makes this work: **a class driver must not import a bus's logic
|
||||||
|
module.** `usbhid` imports `proto.hid` and `usb` (for descriptor types), never `pci`.
|
||||||
|
If a class driver needs `mmio`, it has become an HCD and should be one.
|
||||||
|
|
||||||
|
## What exists today
|
||||||
|
|
||||||
|
- **M10** — `device_enumerate`, `device_claim`, `mmio_map`. Strong-uncacheable device
|
||||||
|
grants, `device_grant` teardown.
|
||||||
|
- **M11** — `irq_bind` / `irq_ack`. IRQ delivered as an IPC notification; mask before
|
||||||
|
EOI; `irq_ack` is the unmask.
|
||||||
|
- **M12** — `parent` in `DeviceDesc`, `device_register` with resource containment.
|
||||||
|
- **M13** — capability passing. `ipc_call` / `ipc_reply_wait` grew a `send_cap` argument
|
||||||
|
and a `received_cap` return (r8): an endpoint travels with a message, installed into
|
||||||
|
the receiver's handle table (shared, refcount-bumped — a copy, not a move). A full
|
||||||
|
table fails `-ENOSPC` and does not half-deliver. This is the "open" primitive — a bus
|
||||||
|
driver mints a per-device endpoint and hands it to a class driver. The runtime exposes
|
||||||
|
`callCap` and `replyWait(..., send_cap)`; no class driver consumes it yet.
|
||||||
|
- **M14** — DMA memory + the memory-ordering layer. `/lib/mmio` gives drivers typed
|
||||||
|
volatile access and `mb`/`rmb`/`wmb` (per-arch); `dma_alloc`/`dma_free` grant
|
||||||
|
physically-contiguous, pinned, uncacheable, reclaim-on-teardown buffers with the
|
||||||
|
physical address exposed (`pmm.allocContiguous`, a DMA arena, `mapUserDmaInto`).
|
||||||
|
`dma_below_4g` caps the address for legacy engines; `dma_write_combining` is accepted
|
||||||
|
but falls back to coherent until PAT is programmed. hpet is refactored onto `/lib/mmio`;
|
||||||
|
no DMA driver consumes `dma_alloc` yet.
|
||||||
|
- **M15** — interrupts for PCI devices, the MSI half. Discovery now gives every PCI
|
||||||
|
function its 4 KiB ECAM config space as resource 0 (unblocking the capability walk
|
||||||
|
with no new syscall), and `msi_bind(device_id, endpoint) -> address, data` allocates a
|
||||||
|
per-device edge-triggered vector, delivered as an IPC notification with no mask and no
|
||||||
|
ack cycle. Legacy INTx (`_PRT` parsing + shared lines) is deliberately skipped — MSI
|
||||||
|
is the real answer. QEMU's HPET has no MSI, so delivery is proven with a self-IPI; the
|
||||||
|
first PCI driver is the first real consumer.
|
||||||
|
- **Port I/O** — `io_read`/`io_write(device_id, resource_index, offset, width[, value])`:
|
||||||
|
a claimed device's `io_port` resource lets a driver read/write its ports, gated exactly
|
||||||
|
like `mmio_map` gates memory (direct ring-3 `in`/`out` stays a #GP). This is what makes
|
||||||
|
a PS/2 or 16550 driver possible; the low-rate legacy hardware that needs it is fine with
|
||||||
|
a syscall per access. `io_port` resources were recorded by discovery and ignored — now
|
||||||
|
they're used.
|
||||||
|
- **M16 (detection)** — the IOMMU is now *found*: discovery parses the ACPI DMAR table,
|
||||||
|
maps the first VT-d unit, and reads its version + capabilities (`iommu_present` in the
|
||||||
|
platform info). This is detection only — **no translation domains are programmed, so
|
||||||
|
DMA is still unprotected** (the caveat below). Enforcement lands with the first DMA
|
||||||
|
driver, which is what there is to protect and test against. Proven in the `iommu` test,
|
||||||
|
booted with an emulated `intel-iommu`.
|
||||||
|
- **`system_spawn`** — a user-space supervisor starts a driver:
|
||||||
|
`system_spawn(name, arguments)` loads a binary bundled in the initial-ramdisk as a
|
||||||
|
fresh ring-3 process; `name` becomes the child's argv[0] and the optional
|
||||||
|
NUL-separated `arguments` blob its argv[1..], delivered on a SysV entry stack
|
||||||
|
([sysv.md](sysv.md)). This is what
|
||||||
|
turned the device manager from "log the match" into "run the driver": the kernel now
|
||||||
|
spawns only `init`, `init` spawns the services, and the **device-manager** discovers
|
||||||
|
the hardware and spawns each driver ([drivers.md](drivers.md)). Ungated for now — a
|
||||||
|
spawn capability is future work.
|
||||||
|
|
||||||
|
So: **bus drivers work now, and they're started by the device manager, not the kernel.**
|
||||||
|
HCDs and class drivers do not work yet. Here is exactly why, and exactly what would fix it.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
# Proposed ABI
|
||||||
|
|
||||||
|
## M13 — capability passing, for class drivers ✅ done
|
||||||
|
|
||||||
|
*Implemented as described below (see "What exists today"). The signatures landed
|
||||||
|
verbatim: `send_cap` in r9, `received_cap` returned in r8, `-ENOSPC` on a full receiver
|
||||||
|
table with no delivery. The rest of this section is the original design note.*
|
||||||
|
|
||||||
|
**The blocker.** A class driver has to reach *its* device. Today the only way to find
|
||||||
|
an endpoint is the name registry: `ipc_register(service_id, h)` / `ipc_lookup(id)`,
|
||||||
|
where `ServiceId` is a global integer namespace with `max_services = 8`. You cannot
|
||||||
|
mint one endpoint per USB device that way, and there is no way for a bus driver to
|
||||||
|
*hand* a class driver an endpoint. M7 deferred this deliberately.
|
||||||
|
|
||||||
|
**The fix.** Let a message carry one handle. Sender names a handle in its own table;
|
||||||
|
the kernel installs the endpoint into the receiver's table (bumping `refcount`) and
|
||||||
|
tells the receiver the index it landed at.
|
||||||
|
|
||||||
|
```
|
||||||
|
ipc_call(h, msg, message_len, reply, reply_cap, send_cap) -> reply_len
|
||||||
|
ipc_reply_wait(h, reply, reply_len, recv, recv_cap, send_cap)
|
||||||
|
-> recv_len (rax), badge (rdx), received_cap (r8)
|
||||||
|
```
|
||||||
|
|
||||||
|
`send_cap` is a handle or `no_cap` (`~0`). `received_cap` is the index the transferred
|
||||||
|
endpoint was installed at in the receiver's table, or `no_cap`.
|
||||||
|
|
||||||
|
- Both calls grow from 5 args to 6, which fits: `syscall5` uses `rdi/rsi/rdx/r10/r8`,
|
||||||
|
leaving `r9`. `ipc_reply_wait` already returns two values via `setSyscallResult2`;
|
||||||
|
this needs a third (`setSyscallResult3`).
|
||||||
|
- If the receiver's handle table is full, the call fails `-ENOSPC` and **the message is
|
||||||
|
not delivered** — a half-delivered capability is worse than a failed send.
|
||||||
|
- `closeHandles` already drops references on exit, so the lifetime story is unchanged.
|
||||||
|
|
||||||
|
That single primitive gives you the standard `open` pattern:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
// class driver // bus driver
|
||||||
|
const h = ipc.lookup(.usb).?; const r = ipc.replyWait(ep, ...);
|
||||||
|
const dev_ep = ipc.callCap(h, // ... mint a per-device endpoint,
|
||||||
|
.{ .op = .open, .id = dev_id }); // reply with it as send_cap
|
||||||
|
// now dev_ep is a private channel to that one device
|
||||||
|
```
|
||||||
|
|
||||||
|
## M14 — DMA memory and the memory-ordering contract, for HCDs ✅ done
|
||||||
|
|
||||||
|
*Implemented: `/lib/mmio` (typed volatile access + `mb`/`rmb`/`wmb`, per-arch) and
|
||||||
|
`dma_alloc`/`dma_free` (contiguous, pinned, uncacheable, reclaim-on-teardown, physical
|
||||||
|
address exposed). `dma_write_combining` still falls back to coherent — real WC needs
|
||||||
|
PAT, a small follow-up. The rest of this section is the original design note.*
|
||||||
|
|
||||||
|
**The blocker.** An HCD is a DMA-engine programmer. It needs a descriptor ring the
|
||||||
|
device can read, which means memory that is (a) physically contiguous, (b) at a
|
||||||
|
physical address the driver knows, (c) of the right cacheability, and (d) pinned.
|
||||||
|
[`sysMmap`](system/kernel/process.zig) gives you *none* of the four: it calls `pmm.alloc()`
|
||||||
|
once per page, maps writeback-cached, and never reveals a physical address.
|
||||||
|
|
||||||
|
**The fix.**
|
||||||
|
|
||||||
|
```
|
||||||
|
dma_alloc(len, flags) -> vaddr (rax), paddr (rdx)
|
||||||
|
dma_free(vaddr, len) -> 0
|
||||||
|
|
||||||
|
flags: dma_coherent (1) uncacheable; the default and the only one that's portable
|
||||||
|
dma_wc (2) write-combining — needs PAT programmed; for framebuffers
|
||||||
|
dma_below_4g (4) for devices with 32-bit DMA addressing
|
||||||
|
```
|
||||||
|
|
||||||
|
Guarantees: page-aligned, physically contiguous, zeroed, pinned for the life of the
|
||||||
|
mapping, and the physical address is stable. It needs one thing the kernel lacks —
|
||||||
|
`pmm.allocContiguous(n, max_phys)`; today `pmm.alloc()` hands out one frame at a time
|
||||||
|
with no adjacency guarantee.
|
||||||
|
|
||||||
|
**The memory-ordering contract.** danos has, at the time of writing, **zero memory
|
||||||
|
barriers anywhere in the tree.** That is currently correct-by-accident and won't
|
||||||
|
survive the first DMA driver, or the first ARM boot.
|
||||||
|
|
||||||
|
`volatile` is not a barrier. In Zig it means: don't elide this access, and don't
|
||||||
|
reorder it against *other volatile* accesses. It says nothing about your *ordinary*
|
||||||
|
stores — the descriptor you just filled in normal WB memory — which LLVM may freely
|
||||||
|
sink past a volatile MMIO write. The canonical bug:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
ring[i] = descriptor; // ordinary store to WB RAM
|
||||||
|
doorbell.* = i; // volatile store to UC MMIO
|
||||||
|
// nothing stops the compiler reordering these; the device reads a stale descriptor
|
||||||
|
```
|
||||||
|
|
||||||
|
So the rules, which belong in `library/mmio.zig` and behind `arch`:
|
||||||
|
|
||||||
|
| Situation | Required |
|
||||||
|
|---|---|
|
||||||
|
| MMIO register read/write | `mmio.read` / `mmio.write` (volatile) |
|
||||||
|
| Fill DMA descriptor, then ring doorbell | `wmb()` between them |
|
||||||
|
| Woken by IRQ, then read what the device wrote | `rmb()` before the read |
|
||||||
|
| MMIO write that must complete before the next read | `mb()` |
|
||||||
|
|
||||||
|
And the per-arch lowering — the reason this must be an `arch` primitive and not a
|
||||||
|
sprinkling of `asm volatile`:
|
||||||
|
|
||||||
|
| | x86_64 | aarch64 |
|
||||||
|
|---|---|---|
|
||||||
|
| `mb()` | `mfence` | `dsb sy` |
|
||||||
|
| `rmb()` | `lfence` | `dsb ld` |
|
||||||
|
| `wmb()` | `sfence` | `dsb st` |
|
||||||
|
| DMA cache coherency | coherent; nothing to do | **not guaranteed**; needs non-cacheable buffers or cache maintenance |
|
||||||
|
|
||||||
|
x86 is forgiving here — TSO plus strong-uncacheable MMIO means you usually get away
|
||||||
|
with a compiler barrier alone. ARM is not, and [vision.md](vision.md) makes ARM the win
|
||||||
|
condition. Build the abstraction while there is one caller to fix.
|
||||||
|
|
||||||
|
(Zig note: `@fence` was **removed in 0.16**. Use `@atomicRmw(..., .seq_cst)` for a full
|
||||||
|
barrier, or per-arch inline asm — which is what `library/mmio.zig` should hide.)
|
||||||
|
|
||||||
|
## M15 — interrupts for PCI devices ✅ done (MSI)
|
||||||
|
|
||||||
|
*Implemented the MSI half: ECAM config space per PCI function (resource 0) and
|
||||||
|
`msi_bind` (per-device edge-triggered vector, delivered as a notification). Legacy INTx
|
||||||
|
`_PRT` parsing is skipped on purpose. `msi_bind` returns (address, data) as two values
|
||||||
|
rather than an out-struct. The rest of this section is the original design note.*
|
||||||
|
|
||||||
|
**The blocker, and it's a hard one.** No PCI device can take an interrupt today.
|
||||||
|
[`addBars`](system/devices/acpi.zig) records `.memory` and `.io_port` BARs and never an
|
||||||
|
`.irq`; there is no `_PRT` parsing anywhere in the tree. `hpet` only works because the
|
||||||
|
HPET advertises its own routing options in its own registers — a privilege no ordinary
|
||||||
|
device has.
|
||||||
|
|
||||||
|
**The fix, in two halves.**
|
||||||
|
|
||||||
|
*Legacy INTx*: parse `_PRT` from the DSDT to map (device, INTA–D) → GSI, and record it
|
||||||
|
as an `.irq` resource. Then `irq_bind` works unchanged. But INTx lines are **shared**,
|
||||||
|
and `irq.bound[gsi]` holds one endpoint. Sharing needs a list, and every driver on the
|
||||||
|
line must be polled on each interrupt — the reason everyone left INTx behind.
|
||||||
|
|
||||||
|
*MSI/MSI-X*, which is the real answer: per-device vectors, edge-triggered, unshared, no
|
||||||
|
mask/ack cycle, no 24-GSI ceiling. The kernel allocates a vector and hands the driver
|
||||||
|
the (address, data) pair to program into its own MSI capability:
|
||||||
|
|
||||||
|
```
|
||||||
|
msi_bind(dev_id, endpoint, out) -> 0 // out: extern struct { addr: u64, data: u32 }
|
||||||
|
```
|
||||||
|
|
||||||
|
The driver writes those into config space itself — which means it needs config space,
|
||||||
|
which means **discovery should give each `pci_device` a `.memory` resource for its
|
||||||
|
4 KiB ECAM slot**. That's a small change to `parseMcfg` and it unblocks the whole
|
||||||
|
capability walk (MSI, MSI-X, PCIe extended caps) without any new syscall.
|
||||||
|
|
||||||
|
Note QEMU's HPET reports `Tn_FSB_INT_DEL_CAP = 0` — no MSI — so `hpet` can never
|
||||||
|
exercise this path. The first MSI driver will be the first PCI driver.
|
||||||
|
|
||||||
|
## M16 — the IOMMU, and the honest caveat ◑ detection done, enforcement pending
|
||||||
|
|
||||||
|
*The IOMMU is now detected (DMAR parsed, VT-d unit mapped and read — see the `iommu`
|
||||||
|
test), but **enforcement is not built**: no translation domains are programmed, so the
|
||||||
|
caveat below still holds in full. Detection can't be taken further usefully until there
|
||||||
|
is a DMA driver to protect and QEMU's `intel-iommu` to test the protection against —
|
||||||
|
building the per-device domains alongside that first driver is both the natural order
|
||||||
|
and the only way to verify them. The rest of this section is the original caveat.*
|
||||||
|
|
||||||
|
Everything above is capability-gated at the *CPU*. None of it is gated at the *device*.
|
||||||
|
A driver that can program a bus-mastering engine can make that device write to any
|
||||||
|
physical address, because page tables sit between the CPU and RAM, not between a device
|
||||||
|
and RAM. Until VT-d/DMAR (or SMMU on ARM) is programmed from the DMAR table, **`device_claim`
|
||||||
|
on any DMA-capable device is equivalent to granting ring 0.**
|
||||||
|
|
||||||
|
This does not make the model useless — it's the same position Linux is in with the
|
||||||
|
IOMMU off, and every other guarantee (crash isolation, restart, no shared address
|
||||||
|
space) still holds. But "user-space drivers are memory-safe" is not true yet, and the
|
||||||
|
gap should be named rather than implied.
|
||||||
|
|
||||||
|
## Ordering
|
||||||
|
|
||||||
|
`M13` (capability passing) is independent of `M14`/`M15` and is the cheapest. It
|
||||||
|
unlocks class drivers, which are the shape with no hardware requirements at all — you
|
||||||
|
could write a real one against `bus`'s comparators tomorrow.
|
||||||
|
|
||||||
|
`M14` and `M15` together unlock the first HCD. `M14`'s barrier layer is worth landing
|
||||||
|
on its own regardless: it's small, obviously correct, and stops every future driver
|
||||||
|
from hand-rolling `*volatile` and getting ARM wrong.
|
||||||
|
|
||||||
|
## See also
|
||||||
|
|
||||||
|
- [drivers.md](drivers.md) — how to write one, concretely.
|
||||||
|
- [discovery.md](discovery.md) / [acpi.md](acpi.md) — where the device table comes from.
|
||||||
|
- [ipc.md](ipc.md) — endpoints, badges, and the notification path an IRQ arrives on.
|
||||||
|
- [resilience.md](resilience.md) — restart, the reason any of this is worth the trouble.
|
||||||
+384
@@ -0,0 +1,384 @@
|
|||||||
|
# Writing a driver
|
||||||
|
|
||||||
|
In a monolithic kernel a driver is a function call away from everything: it runs in
|
||||||
|
ring 0, dereferences any physical address, and its interrupt handler *is* the ISR. In
|
||||||
|
danos a driver is **an ordinary ring-3 process**. It has its own address space, it
|
||||||
|
can crash without taking the kernel with it, and — the point of this document — it
|
||||||
|
can be restarted ([resilience](resilience.md)).
|
||||||
|
|
||||||
|
That leaves three questions the kernel has to answer, because a process can't answer
|
||||||
|
them for itself:
|
||||||
|
|
||||||
|
1. **What hardware exists?** → `device_enumerate`, over the device table discovery built
|
||||||
|
([discovery](discovery.md), [acpi](acpi.md)).
|
||||||
|
2. **How do I touch its registers?** → `device_claim` + `mmio_map`: the kernel maps the
|
||||||
|
device's physical MMIO window into your address space, and from then on it's plain
|
||||||
|
memory. No syscall per register access.
|
||||||
|
3. **How do I find out it wants something?** → `irq_bind`: the interrupt is delivered
|
||||||
|
to you as an IPC notification. You block; the hardware wakes you.
|
||||||
|
|
||||||
|
A driver is, in one sentence, *a process that sleeps until its device has something to
|
||||||
|
say.*
|
||||||
|
|
||||||
|
## How a driver gets started: discover, match, spawn
|
||||||
|
|
||||||
|
Nothing in the kernel decides that the HPET needs the `hpet` driver — that is policy,
|
||||||
|
and policy lives in user space. Boot brings user space up as a three-level supervision
|
||||||
|
hierarchy, each level owning one job:
|
||||||
|
|
||||||
|
```
|
||||||
|
kernel ──spawns──► init (PID 1) ──spawns──► device-manager ──spawns──► hpet
|
||||||
|
| | |
|
||||||
|
spawns only init, the service supervisor: the driver supervisor: enumerates
|
||||||
|
publishes the starts the system /system/devices, matches each device
|
||||||
|
initial-ramdisk services (vfs, the to a driver, and system_spawn's it
|
||||||
|
so user space can device-manager). Its
|
||||||
|
system_spawn from it list is init policy.
|
||||||
|
```
|
||||||
|
|
||||||
|
The kernel launches exactly one process — `init` — and hands it nothing but the raw
|
||||||
|
ability to start more (`system_spawn(name, arguments)`, which loads a binary bundled
|
||||||
|
in the initial-ramdisk as a fresh ring-3 process — `name` becoming its argv[0],
|
||||||
|
the optional arguments its argv[1..], on a SysV entry stack, see sysv.md). Everything else is a user-space decision:
|
||||||
|
|
||||||
|
- **init** ([system/services/init](system/services/init/init.zig)) is the **service
|
||||||
|
supervisor**. It spawns the system services danos brings up at boot — today `vfs` and
|
||||||
|
the `device-manager` — from a small list. Drivers are deliberately *not* its job.
|
||||||
|
- **device-manager** ([system/services/device-manager](system/services/device-manager/device-manager.zig))
|
||||||
|
is the **driver supervisor**. It does the three steps a monolithic kernel would do in
|
||||||
|
its probe path, entirely from ring 3:
|
||||||
|
1. **Discover** — `device_enumerate` snapshots the device table the kernel built from
|
||||||
|
ACPI/PCI ([discovery](discovery.md)).
|
||||||
|
2. **Match** — for each device it looks up a driver by `DeviceClass`. The match policy
|
||||||
|
is a table (`driverFor`): today a static `timer → hpet` map; a fuller system reads
|
||||||
|
what each driver *binds* (a manifest under `/system/drivers`, or the driver
|
||||||
|
describing its own match).
|
||||||
|
3. **Spawn** — `system_spawn(driver_name, arguments)` starts the matched driver (the
|
||||||
|
arguments can carry *which* device it matched), which then claims
|
||||||
|
its device and runs the event loop below.
|
||||||
|
|
||||||
|
So "how is a driver discovered and configured" has two halves: **discovery** is the
|
||||||
|
kernel's device table, read by anyone; **configuration** is two user-space policies —
|
||||||
|
init's service list and the device-manager's match table. Both are hardcoded in their
|
||||||
|
respective programs today; the natural next step is to move them into `/etc` (see the
|
||||||
|
milestone notes in [driver-model.md](driver-model.md)). `system_spawn` is currently
|
||||||
|
ungated — any process may spawn any bundled binary — because there is no spawn
|
||||||
|
capability yet.
|
||||||
|
|
||||||
|
## The capability: claim before touch
|
||||||
|
|
||||||
|
The driver syscall numbers (`system/abi.zig`) with the device types they carry
|
||||||
|
(`system/devices/device-abi.zig`), dispatched in `system/kernel/process.zig`:
|
||||||
|
|
||||||
|
| # | Call | Meaning |
|
||||||
|
|---|------|---------|
|
||||||
|
| 11 | `device_enumerate(buf, max) -> total` | Snapshot the device table |
|
||||||
|
| 12 | `device_claim(id) -> ok` | Take **exclusive** ownership |
|
||||||
|
| 13 | `mmio_map(id, res_idx) -> vaddr` | Map a claimed device's register window |
|
||||||
|
| 14 | `irq_bind(id, res_idx, endpoint)` | Deliver that device's IRQ as a notification |
|
||||||
|
| 15 | `irq_ack(id, res_idx)` | Re-arm the IRQ after servicing the device |
|
||||||
|
| 16 | `device_register(parent_id, desc) -> id` | Publish a child of a device you claimed |
|
||||||
|
|
||||||
|
Notice that **nothing takes a physical address or an interrupt number.** Every call
|
||||||
|
names a device by id and a resource by index. That indirection is the entire security
|
||||||
|
model. If `mmio_map` took a physical address, any process could map the kernel's
|
||||||
|
memory; if `irq_bind` took a GSI, any process could bind the keyboard's line and
|
||||||
|
silently intercept it. Instead the kernel checks two things (`process.ownedGsi`, and
|
||||||
|
the same check at the top of `sysMmioMap`):
|
||||||
|
|
||||||
|
- `devices_broker.ownerOf(dev_id) == me` — you claimed it, and claims are exclusive
|
||||||
|
- the resource at `res_idx` is of the right *kind* — `memory` for `mmio_map`, `irq`
|
||||||
|
for `irq_bind`
|
||||||
|
|
||||||
|
The claim is the capability. Everything else follows from it.
|
||||||
|
|
||||||
|
## Registers: `mmio_map`
|
||||||
|
|
||||||
|
`mmio_map` walks the caller's page tables and installs the device's physical frames
|
||||||
|
with `present | user | writable | nx | pcd | pwt`
|
||||||
|
(`arch/x86_64/paging.zig:mapUserDeviceInto`). Two of those bits are load-bearing:
|
||||||
|
|
||||||
|
- **`pcd | pwt`** — strong-uncacheable. A device register is not memory; a cached read
|
||||||
|
would return a stale value and a write might never leave the CPU.
|
||||||
|
- **`device_grant`** (bit 9, one of the PTE's available bits) — marks the leaf as MMIO
|
||||||
|
rather than RAM, so `freeSubtree` skips `pmm.free` on it when the address space is
|
||||||
|
destroyed. Without this, killing a driver would hand the HPET's registers back to
|
||||||
|
the frame allocator as if they were free RAM. The `iopass` test guards it.
|
||||||
|
|
||||||
|
Grants land in their own arena, `0x0000_7100_0000_0000` (PML4[226]), so device pages
|
||||||
|
never widen an existing mapping.
|
||||||
|
|
||||||
|
Then you just… use it:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
const base = dev.mmioMap(dev_id, mmio_res) orelse return;
|
||||||
|
const counter: *volatile u64 = @ptrFromInt(base + 0xF0);
|
||||||
|
const now = counter.*; // a load, straight to the hardware. no kernel involved.
|
||||||
|
```
|
||||||
|
|
||||||
|
## Interrupts: the cycle, and why it has that shape
|
||||||
|
|
||||||
|
An interrupt handler in a microkernel has a problem. The code that knows how to quiet
|
||||||
|
the device is in ring 3, in another address space, and it will not run for
|
||||||
|
microseconds or milliseconds — after a context switch, when the scheduler gets to it.
|
||||||
|
But the CPU wants an EOI *now*, and a **level-triggered** line stays asserted until
|
||||||
|
the device is quieted. EOI a still-asserted line and the I/O APIC redelivers
|
||||||
|
immediately. Forever. The driver never gets to run at all.
|
||||||
|
|
||||||
|
The way out is to mask the line before acknowledging it:
|
||||||
|
|
||||||
|
```
|
||||||
|
kernel ISR irqMask(gsi) // line still asserted; stop it reaching a CPU
|
||||||
|
irqEoi() // now safe to tell the LAPIC we're done
|
||||||
|
notifyFromIsr() // wake the driver — it runs much later
|
||||||
|
|
||||||
|
driver replyWait() -> badge with the notify bit set
|
||||||
|
<clear the device's status register> // NOW the line deasserts
|
||||||
|
irq_ack(dev, res) // kernel unmasks: quiet, so it can't refire
|
||||||
|
|
||||||
|
```
|
||||||
|
|
||||||
|
`irq_ack` is not bookkeeping you could skip. **It is the unmask.** Forget it and the
|
||||||
|
interrupt fires exactly once, ever; call it before the device is quiet and you get an
|
||||||
|
interrupt storm. That single fact explains why `irq_bind` and `irq_ack` are two
|
||||||
|
syscalls and not one.
|
||||||
|
|
||||||
|
This is also why `interruptDispatch` (`arch/x86_64/idt.zig`) no longer issues the EOI
|
||||||
|
itself. It used to, before running the handler — correct for the LAPIC timer, and
|
||||||
|
impossible for a routed device line. Each handler now owns its EOI, because only the
|
||||||
|
handler knows which discipline its source needs.
|
||||||
|
|
||||||
|
### The driver side is an event loop, not a callback
|
||||||
|
|
||||||
|
`IPC_ReplyWait` returns *either* a client request *or* a notification, told apart by
|
||||||
|
the top bit of the badge (`ipc_sync.notify_badge_bit`). So a driver is one
|
||||||
|
single-threaded loop over both of its event sources:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
while (true) {
|
||||||
|
const r = ipc.replyWait(endpoint, reply, &recv);
|
||||||
|
if (r.isNotification()) { // r.source() is the GSI
|
||||||
|
service_device(); // clear the status register
|
||||||
|
_ = dev.irqAck(id, irq_res); // re-arm
|
||||||
|
} else {
|
||||||
|
handle_client_request(recv[0..r.len]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
No reentrancy, no "what am I allowed to call from an interrupt handler", no shared
|
||||||
|
state between ISR and task context. The interrupt is just a message.
|
||||||
|
|
||||||
|
Two properties worth knowing:
|
||||||
|
|
||||||
|
- **An interrupt taken while you're elsewhere is not lost.** If the driver is off in
|
||||||
|
an `ipc_call` to another server when the IRQ fires, `wakeLocked` finds nobody
|
||||||
|
waiting, but the badge is already on the endpoint's notify ring. The next
|
||||||
|
`replyWait` pops it (`ipc_sync.replyWait` checks `popNotify` before the sender FIFO).
|
||||||
|
- **Notifications coalesce, they don't count.** The ring is 8 deep and drops on
|
||||||
|
overflow. That's correct: an IRQ notification is a *level* ("the device wants
|
||||||
|
attention"), not a tally. Re-read the device's status register; never assume one
|
||||||
|
notification means exactly one event. Because the ISR masks the line until you
|
||||||
|
`irq_ack`, at most one badge per GSI can be outstanding — so the ring can only
|
||||||
|
overflow if you bind more than eight GSIs to a single endpoint. Don't.
|
||||||
|
|
||||||
|
## A whole driver
|
||||||
|
|
||||||
|
`system/drivers/hpet/hpet.zig` is ~150 lines and does all of it. The shape:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
const hpet = findHpet(buf) orelse return; // device_enumerate, look for
|
||||||
|
// class=timer with memory + irq
|
||||||
|
_ = dev.claim(hpet.dev_id); // the capability
|
||||||
|
const base = dev.mmioMap(hpet.dev_id, hpet.mmio).?;
|
||||||
|
const endpoint = ipc.createIpcEndpoint().?;
|
||||||
|
|
||||||
|
// program the hardware over the mapping we were just handed
|
||||||
|
reg(base, 0x100).* = level | int_enb | (hpet.gsi << 9); // timer 0 config
|
||||||
|
reg(base, 0x108).* = reg(base, 0xF0).* + period; // comparator
|
||||||
|
reg(base, 0x010).* |= 1; // ENABLE
|
||||||
|
|
||||||
|
_ = dev.irqBind(hpet.dev_id, hpet.irq, endpoint);
|
||||||
|
|
||||||
|
while (...) {
|
||||||
|
const r = ipc.replyWait(endpoint, &.{}, &recv); // blocked. not polling.
|
||||||
|
if (r.badge & notify_bit == 0) continue;
|
||||||
|
reg(base, 0x020).* = 1; // clear status -> deassert
|
||||||
|
reg(base, 0x108).* = reg(base, 0xF0).* + period; // re-arm
|
||||||
|
_ = dev.irqAck(hpet.dev_id, hpet.irq); // unmask
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
The HPET is a good first driver for a reason that isn't obvious. Its *counter* is a
|
||||||
|
clocksource — the only way to use it is to read it, so it proved `mmio_map` without
|
||||||
|
needing interrupts at all. Its *comparators* are a clockevent, and can be configured
|
||||||
|
**level-triggered** (`Tn_INT_TYPE_CNF`), which asserts a bit in `GENERAL_INT_STATUS`
|
||||||
|
that the driver must write-1-to-clear. That's a genuine deassert step, so the full
|
||||||
|
mask/ack cycle above is exercised for real rather than being decoration on an
|
||||||
|
edge-triggered line that would have been fine without it.
|
||||||
|
|
||||||
|
One wrinkle it also demonstrates: the ACPI HPET table carries **no interrupt number**.
|
||||||
|
Which I/O APIC inputs a comparator may drive is a bitmask in `Tn_INT_ROUTE_CAP`, in
|
||||||
|
the device's own registers. So discovery (`acpi.parseHpet`) maps the block, reads the
|
||||||
|
mask, and records one concrete GSI as an `irq` resource. The driver then programs
|
||||||
|
`Tn_INT_ROUTE_CNF` to raise exactly that line — and the kernel will only bind the one
|
||||||
|
it recorded. Hardware that describes itself at runtime still has to fit through a
|
||||||
|
static capability.
|
||||||
|
|
||||||
|
## Publishing children: `device_register`
|
||||||
|
|
||||||
|
A device that *contains other devices* — a PCI bridge, a USB hub, or the HPET's block
|
||||||
|
of comparators — needs a driver that enumerates it and tells the kernel what it found.
|
||||||
|
That's `device_register`, and it makes the device table a tree rather than a list
|
||||||
|
(`DeviceDesc.parent`).
|
||||||
|
|
||||||
|
```zig
|
||||||
|
var child = std.mem.zeroes(dev.DeviceDesc);
|
||||||
|
child.class = @intFromEnum(dev.DeviceClass.timer);
|
||||||
|
child.resource_count = 1;
|
||||||
|
child.resources[0] = .{ .kind = memory, .start = bus_base + 0x100, .len = 0x20 };
|
||||||
|
const child_id = dev.register(bus_id, &child).?;
|
||||||
|
```
|
||||||
|
|
||||||
|
The child is left **unclaimed**, which is the whole point: another process claims it and
|
||||||
|
`mmio_map`s it, and sees only that 0x20-byte window.
|
||||||
|
|
||||||
|
The rule the kernel enforces is **containment**: every resource of a child must lie
|
||||||
|
inside a resource of the same kind on its parent. Ranges must nest; an IRQ must match
|
||||||
|
exactly. This isn't bureaucracy — a `DeviceDesc` is a licence to map physical memory, so
|
||||||
|
without containment `device_register` would be a syscall for mapping any page you like. A
|
||||||
|
bus driver may only ever subdivide what it already owns.
|
||||||
|
|
||||||
|
A device with **no resources** is legal and common. A USB device is reached through its
|
||||||
|
controller, not by MMIO, so it gets `resource_count = 0`.
|
||||||
|
|
||||||
|
See [`system/drivers/bus/bus.zig`](../system/drivers/bus/bus.zig) for a complete one, and
|
||||||
|
[driver-model.md](driver-model.md) for how bus drivers, class drivers and host
|
||||||
|
controller drivers fit together.
|
||||||
|
|
||||||
|
## What the kernel does not do for you
|
||||||
|
|
||||||
|
- **It does not quiet your device.** That's the whole reason `irq_ack` exists.
|
||||||
|
- **It does not know your registers.** `mmio_map` hands you a base address; every
|
||||||
|
offset in this document came from the HPET spec, not from danos.
|
||||||
|
- **It does not serialise your driver.** Two clients calling one driver endpoint are
|
||||||
|
serialised by `replyWait`, but nothing stops your driver from being preempted.
|
||||||
|
|
||||||
|
## Limits, today
|
||||||
|
|
||||||
|
Worth knowing before you write the second driver:
|
||||||
|
|
||||||
|
Several things this list used to warn about are now available (see
|
||||||
|
[driver-model.md](driver-model.md)): **port I/O** (`io_read`/`io_write`, claim-gated by
|
||||||
|
the device's `io_port` resource — direct ring-3 `in`/`out` is still a #GP, so a PS/2 or
|
||||||
|
16550 driver goes through these), **DMA memory** (`dma_alloc`: contiguous, pinned,
|
||||||
|
uncacheable, physical address exposed), and **memory barriers** (`/lib/mmio`'s
|
||||||
|
`mb`/`rmb`/`wmb`). What remains:
|
||||||
|
|
||||||
|
- **Page granularity.** `mmio_map` rounds to 4 KiB. Two devices sharing a page means
|
||||||
|
granting one grants the other. A `device_register`ed child's *resource* can be narrower
|
||||||
|
than a page, but its *mapping* can't.
|
||||||
|
- **DMA is not contained.** A driver that can program a bus-mastering device can make
|
||||||
|
that device write to *any* physical address — page tables don't sit between a device
|
||||||
|
and RAM; an IOMMU does. The IOMMU is now *detected* (M16), but no translation domains
|
||||||
|
are programmed, so `device_claim` on a DMA-capable device is still effectively
|
||||||
|
equivalent to granting ring 0. This is the largest gap between the design's promise and
|
||||||
|
what it delivers; enforcement lands with the first DMA driver.
|
||||||
|
- **No `dev_release`.** A claim is never dropped (only IRQ/MSI bindings are, on exit), so
|
||||||
|
a device stays owned for the life of its driver — which blocks restart.
|
||||||
|
- **One endpoint per GSI**, so shared legacy PCI INTx lines can't be split between two
|
||||||
|
drivers. MSI/MSI-X — one vector per device, edge-triggered, unshared — is the real
|
||||||
|
answer, and QEMU's HPET doesn't offer it (`Tn_FSB_INT_DEL_CAP = 0`).
|
||||||
|
- **Polarity is hardcoded** active-high in `irq.bind`. A device whose MADT override
|
||||||
|
says active-low needs that threaded through from discovery.
|
||||||
|
- **14 device vectors** (33–46) and **24 GSIs**, bounded by the stubs `isr.s` emits and
|
||||||
|
by a single I/O APIC.
|
||||||
|
- **Don't bind more than 8 GSIs to one endpoint.** The notify ring is 8 deep and drops
|
||||||
|
on overflow. With one GSI per endpoint that's unreachable — the line is masked from
|
||||||
|
the ISR until `irq_ack`, so at most one badge is ever outstanding. Bind nine devices
|
||||||
|
to one endpoint, though, and a dropped badge leaves that line masked with nobody
|
||||||
|
left to ack it.
|
||||||
|
- **A faulting driver still kills the machine.** There is no per-process kill path: a
|
||||||
|
ring-3 page fault halts the kernel, so `releaseIrqs` runs only on a voluntary
|
||||||
|
`exit`. Fault isolation is the whole premise ([vision](vision.md)) and it is
|
||||||
|
[not built yet](resilience.md).
|
||||||
|
- **A dead driver's device is not reclaimed.** `releaseIrqs` unbinds and masks the
|
||||||
|
line on exit, but the claim is never released — restart is
|
||||||
|
[not built](resilience.md).
|
||||||
|
- **On real hardware, the mask/EOI cycle may need a remote-IRR flush.** Masking a
|
||||||
|
level-triggered redirection entry with remote-IRR set doesn't clear it on some
|
||||||
|
chipsets, and the line never fires again. QEMU clears it on EOI regardless, so the
|
||||||
|
tests can't see this. Linux flushes remote-IRR by toggling the entry to edge and
|
||||||
|
back. See the note at the top of `system/kernel/irq.zig`.
|
||||||
|
|
||||||
|
## Verifying it
|
||||||
|
|
||||||
|
The `hpet` test spawns `hpet` from the initial ramdisk and watches the serial log. The driver
|
||||||
|
prints `hpet: ok` only after being woken five times, and its loop's only exit is
|
||||||
|
through `replyWait` returning a notification — it cannot reach that line by polling.
|
||||||
|
|
||||||
|
The last check doesn't trust the driver's self-report at all: the kernel reads the I/O
|
||||||
|
APIC redirection entry back and asserts the line really is routed to a device vector,
|
||||||
|
really is level-triggered, and really was left unmasked by the driver's final
|
||||||
|
`irq_ack`.
|
||||||
|
|
||||||
|
```
|
||||||
|
$ python3 test/qemu_test.py hpet irqfree iopass
|
||||||
|
hpet ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
|
irqfree ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
|
iopass ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
|
```
|
||||||
|
|
||||||
|
Two companions cover what `hpet` can't, because it never exits:
|
||||||
|
|
||||||
|
- **`irqfree`** — the teardown path. Binds two owners to one shared endpoint, releases
|
||||||
|
one, and reads the I/O APIC back: the departing owner's line is masked, the sibling's
|
||||||
|
is not. That second half is why bindings are keyed on the owning *task* and not on
|
||||||
|
the endpoint pointer — endpoints are shared, so releasing "everything pointing at
|
||||||
|
this endpoint" would silently mask a live driver's device.
|
||||||
|
- **`iopass`** — the `device_grant` teardown rule, so destroying a driver's address
|
||||||
|
space never returns MMIO frames to the RAM pool.
|
||||||
|
|
||||||
|
## What's next (not done here)
|
||||||
|
|
||||||
|
The big driver-model pieces — capability passing (class drivers), DMA + barriers, MSI,
|
||||||
|
and IOMMU detection — are **now done** ([driver-model.md](driver-model.md), M13–M16), as
|
||||||
|
is **port I/O** (`io_read`/`io_write`, the claim-gated syscalls that make a PS/2 or 16550
|
||||||
|
driver possible). What's left is IOMMU *enforcement* (per-device domains — it waits on
|
||||||
|
the first DMA driver to protect and test against) and these smaller items:
|
||||||
|
|
||||||
|
- **Releasing a claim.** There is no `dev_release`, and `devices_broker` never drops a claim on
|
||||||
|
exit — only IRQ bindings are released. A dead driver's device stays owned forever,
|
||||||
|
which blocks restart.
|
||||||
|
- **Unregistering children.** `device_register` only appends. A USB device that is
|
||||||
|
unplugged cannot be removed, and a bus driver in a loop can exhaust the 64-entry
|
||||||
|
table.
|
||||||
|
- **Restart.** A supervisor that *spawns* drivers now exists — the device-manager starts
|
||||||
|
them with `system_spawn` — but a supervisor that *restarts* them does not. A driver that
|
||||||
|
dies should release its claim, have its device quiesced, and be respawned; today nothing
|
||||||
|
notices the death. Some pieces (`releaseIrqs`, `device_grant` teardown, the claim table)
|
||||||
|
exist, and `dev_release` (below) is the missing mechanism; the restart policy is the
|
||||||
|
resilience track ([resilience.md](resilience.md)).
|
||||||
|
- **Interrupt priority / threaded IRQ latency.** `notifyFromIsr` enqueues the woken
|
||||||
|
driver but doesn't preempt (`wakeLocked` deliberately leaves that to the caller), so
|
||||||
|
a woken driver waits for the next scheduling point.
|
||||||
|
|
||||||
|
## The driver contract (M17–M18)
|
||||||
|
|
||||||
|
Claiming and mapping is half of being a danos driver; the other half is the
|
||||||
|
**lifecycle and protocol contract**, and the runtime makes it nearly free:
|
||||||
|
|
||||||
|
- Build on `runtime.service.run` — one replyWait loop folding protocol
|
||||||
|
requests, signals, and notifications into callbacks. The harness answers the
|
||||||
|
universal zero-length ping and turns `terminate` into a clean exit for you
|
||||||
|
([process-lifecycle.md](process-lifecycle.md)).
|
||||||
|
- A driver spawned with an assignment (its device id as argv[1]) sends the
|
||||||
|
versioned `hello` to the device manager inside the deadline, and a **bus**
|
||||||
|
driver reports what it discovers with `child_added`
|
||||||
|
([device-manager.md](device-manager.md); usb-xhci-bus is the reference
|
||||||
|
implementation).
|
||||||
|
- Crash freely — that is the design. The kernel releases your claims, IRQ
|
||||||
|
bindings, and MSI vectors at death; the manager reads your exit reason,
|
||||||
|
prunes what you reported, restarts you with backoff, and your fresh instance
|
||||||
|
re-claims and re-reports. Never depend on your own cleanup running
|
||||||
|
(iron rule 1).
|
||||||
+25
-20
@@ -10,7 +10,7 @@ that hands us a working CPU, a memory map, and a screen, and then gets out of th
|
|||||||
way.
|
way.
|
||||||
|
|
||||||
The key thing to understand: **UEFI is not our OS, it's a stepping stone.** It
|
The key thing to understand: **UEFI is not our OS, it's a stepping stone.** It
|
||||||
exists to load *us*. Our `src/boot/efi.zig` is a UEFI *application* — a normal program
|
exists to load *us*. Our `boot/efi.zig` is a UEFI *application* — a normal program
|
||||||
that the firmware runs — and its entire purpose is to gather what the kernel needs
|
that the firmware runs — and its entire purpose is to gather what the kernel needs
|
||||||
and then jump into the kernel.
|
and then jump into the kernel.
|
||||||
|
|
||||||
@@ -20,14 +20,17 @@ UEFI boots by looking for a FAT-formatted partition called the **EFI System
|
|||||||
Partition (ESP)** and running a file at a well-known fallback path:
|
Partition (ESP)** and running a file at a well-known fallback path:
|
||||||
|
|
||||||
```
|
```
|
||||||
esp/EFI/BOOT/BOOTX64.efi <- the "removable media" default for x86-64
|
EFI/BOOT/BOOTX64.efi <- the "removable media" default for x86-64
|
||||||
```
|
```
|
||||||
|
|
||||||
That's exactly the layout `build.zig` assembles. It builds `src/boot/efi.zig` for the
|
The boot volume is the **FHS-shaped `zig-out`** itself (see the repository-layout note
|
||||||
`uefi` target, installs it to `esp/EFI/BOOT/BOOTX64.efi`, and drops the kernel ELF
|
in [README.md](README.md)): `build.zig` installs `boot/efi.zig` (built for the `uefi`
|
||||||
at `esp/kernel`. The `run-x86-64` step then points QEMU at OVMF (UEFI firmware for
|
target) to `zig-out/EFI/BOOT/BOOTX64.efi` — the one path UEFI firmware fixes — and lays
|
||||||
virtual machines) and presents that `esp/` directory to the guest as a FAT drive.
|
the rest out by FHS path: the kernel at `zig-out/system/kernel`, init at
|
||||||
The firmware finds `BOOTX64.efi` and runs it — that's our `main()`.
|
`zig-out/system/services/init`, the initial-ramdisk at `zig-out/boot/`. The
|
||||||
|
`run-x86-64` step points QEMU at OVMF (UEFI firmware for virtual machines) and presents
|
||||||
|
`zig-out` to the guest as a FAT drive. The firmware finds `BOOTX64.efi` and runs it —
|
||||||
|
that's our `main()`, which then loads the kernel and init from their FHS paths.
|
||||||
|
|
||||||
## Boot services: the firmware's API
|
## Boot services: the firmware's API
|
||||||
|
|
||||||
@@ -83,9 +86,9 @@ All of this *must* happen now, because after exit there's no GOP to ask. (See
|
|||||||
|
|
||||||
- Use the **LoadedImage** protocol to discover which device we booted from, then
|
- Use the **LoadedImage** protocol to discover which device we booted from, then
|
||||||
**SimpleFileSystem** to open that volume.
|
**SimpleFileSystem** to open that volume.
|
||||||
- Open the file named `danos`, seek to the end to learn its size, rewind, and read
|
- Open the kernel ELF at its FHS path (`system\kernel`), seek to the end to learn its
|
||||||
the whole ELF into a firmware-allocated pool buffer. (`read` may return short, so
|
size, rewind, and read the whole ELF into a firmware-allocated pool buffer. (`read`
|
||||||
we loop.)
|
may return short, so we loop.)
|
||||||
- Parse the ELF: validate the `\x7fELF` magic and the `x86_64` machine type, then
|
- Parse the ELF: validate the `\x7fELF` magic and the `x86_64` machine type, then
|
||||||
walk the program headers. For every `PT_LOAD` segment we:
|
walk the program headers. For every `PT_LOAD` segment we:
|
||||||
- reserve the exact physical pages it's linked at (`p_paddr`) via
|
- reserve the exact physical pages it's linked at (`p_paddr`) via
|
||||||
@@ -121,7 +124,7 @@ entirely ours.
|
|||||||
### 4. Jump to the kernel
|
### 4. Jump to the kernel
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
const kernel: *const fn (*const BootInfo) callconv(danos.kernel_abi) noreturn =
|
const kernel: *const fn (*const BootInfo) callconv(boot_handoff.kernel_abi) noreturn =
|
||||||
@ptrFromInt(entry);
|
@ptrFromInt(entry);
|
||||||
kernel(&boot_info);
|
kernel(&boot_info);
|
||||||
```
|
```
|
||||||
@@ -139,17 +142,19 @@ kernel is freestanding and uses the **SysV AMD64** convention (first argument in
|
|||||||
read garbage.
|
read garbage.
|
||||||
|
|
||||||
So both sides pin the convention explicitly to SysV via the shared
|
So both sides pin the convention explicitly to SysV via the shared
|
||||||
`danos.kernel_abi` (defined in `src/root.zig`). The loader's function-pointer type
|
`boot_handoff.kernel_abi` (defined in `system/boot-handoff.zig`). The loader's
|
||||||
and the kernel's `_start` both reference it, so the pointer lands in the register
|
function-pointer type and the kernel's `_start` both reference it, so the pointer lands
|
||||||
the kernel expects. This is the whole reason `kernel_abi` lives in the shared
|
in the register the kernel expects. This is the whole reason `kernel_abi` lives in the
|
||||||
`danos` module: it's a contract both binaries must agree on. See
|
shared `boot-handoff` module: it's a contract both binaries must agree on. See
|
||||||
[sysv.md](sysv.md) for what "SysV" means and where else it shows up.
|
[sysv.md](sysv.md) for what "SysV" means and where else it shows up.
|
||||||
|
|
||||||
## The handoff contract
|
## The handoff contract
|
||||||
|
|
||||||
The loader and kernel are two *separate* binaries built for two different targets,
|
The loader and kernel are two *separate* binaries built for two different targets,
|
||||||
so everything they exchange must have an identically-defined memory layout. That's
|
so everything they exchange must have an identically-defined memory layout. That's
|
||||||
what `src/root.zig` provides — imported by both as the `danos` module:
|
what `system/boot-handoff.zig` provides — imported by both as the `boot-handoff` module.
|
||||||
|
It is *only* the handoff: the kernel↔user ABI (`system/abi.zig`) and the device types
|
||||||
|
(`system/devices/device-abi.zig`) are separate contracts the bootloader never sees.
|
||||||
|
|
||||||
- `BootInfo` — the top-level struct passed to the kernel (currently just the
|
- `BootInfo` — the top-level struct passed to the kernel (currently just the
|
||||||
framebuffer; this is where future handoff data like the memory map will go).
|
framebuffer; this is where future handoff data like the memory map will go).
|
||||||
@@ -164,16 +169,16 @@ the loader writes are the bytes the kernel reads.
|
|||||||
```
|
```
|
||||||
power on
|
power on
|
||||||
-> UEFI firmware initialises hardware
|
-> UEFI firmware initialises hardware
|
||||||
-> finds esp/EFI/BOOT/BOOTX64.efi, runs it (our efi.zig main)
|
-> finds EFI/BOOT/BOOTX64.efi on the FHS volume, runs it (our efi.zig main)
|
||||||
-> grab boot services
|
-> grab boot services
|
||||||
-> queryFramebuffer (via GOP: EDID native res, setMode, describe fb)
|
-> queryFramebuffer (via GOP: EDID native res, setMode, describe fb)
|
||||||
-> loadKernel (read danos ELF, load PT_LOAD segments to 0x100000)
|
-> loadKernel (read system/kernel ELF, load PT_LOAD segments to 0x100000)
|
||||||
-> exitBootServices (retry until the memory-map key holds)
|
-> exitBootServices (retry until the memory-map key holds)
|
||||||
-> jump to e_entry, boot_info pointer in RDI
|
-> jump to e_entry, boot_info pointer in RDI
|
||||||
-> kernel _start (src/kernel/main.zig: framebuffer console, then halt)
|
-> kernel _start (system/kernel/kernel.zig: framebuffer console, then halt)
|
||||||
```
|
```
|
||||||
|
|
||||||
Bottom line: **UEFI's job is to give us a CPU, memory, and a framebuffer, then
|
Bottom line: **UEFI's job is to give us a CPU, memory, and a framebuffer, then
|
||||||
disappear.** `src/boot/efi.zig` is the thin bridge that collects those gifts into a
|
disappear.** `boot/efi.zig` is the thin bridge that collects those gifts into a
|
||||||
`BootInfo`, tears down the firmware, and jumps into the kernel — after which we're
|
`BootInfo`, tears down the firmware, and jumps into the kernel — after which we're
|
||||||
on our own.
|
on our own.
|
||||||
|
|||||||
@@ -3,12 +3,12 @@
|
|||||||
Once the kernel knows what RAM exists ([memory-map.md](memory-map.md)), it needs a
|
Once the kernel knows what RAM exists ([memory-map.md](memory-map.md)), it needs a
|
||||||
way to *hand out* that RAM: give me a free page of physical memory, and later,
|
way to *hand out* that RAM: give me a free page of physical memory, and later,
|
||||||
here's one back. That's the **physical frame allocator** (a "physical memory
|
here's one back. That's the **physical frame allocator** (a "physical memory
|
||||||
manager", hence `src/kernel/pmm.zig`). It deals only in fixed 4 KiB **frames** — the
|
manager", hence `system/kernel/pmm.zig`). It deals only in fixed 4 KiB **frames** — the
|
||||||
natural unit because that's the granularity the CPU's paging hardware maps — and
|
natural unit because that's the granularity the CPU's paging hardware maps — and
|
||||||
it is the primitive everything above it stands on: page tables, the kernel heap,
|
it is the primitive everything above it stands on: page tables, the kernel heap,
|
||||||
per-process memory all ultimately ask the frame allocator for pages.
|
per-process memory all ultimately ask the frame allocator for pages.
|
||||||
|
|
||||||
It's **generic kernel code**: it operates on the neutral `danos.MemoryRegion`
|
It's **generic kernel code**: it operates on the neutral `system.MemoryRegion`
|
||||||
array, so there's no UEFI in it and nothing architecture-specific beyond the 4 KiB
|
array, so there's no UEFI in it and nothing architecture-specific beyond the 4 KiB
|
||||||
page. (Contrast [arch.md](arch.md), which is where CPU-specific code lives.)
|
page. (Contrast [arch.md](arch.md), which is where CPU-specific code lives.)
|
||||||
|
|
||||||
@@ -33,7 +33,7 @@ RAM is 32768 frames — a **4 KiB bitmap, a single frame**. Even 64 GiB needs on
|
|||||||
|
|
||||||
## How it works
|
## How it works
|
||||||
|
|
||||||
State lives in `src/kernel/pmm.zig`: the `bitmap` slice, `total_frames`, `used_frames`,
|
State lives in `system/kernel/pmm.zig`: the `bitmap` slice, `total_frames`, `used_frames`,
|
||||||
and a `next_hint` marking where the next allocation scan should start.
|
and a `next_hint` marking where the next allocation scan should start.
|
||||||
|
|
||||||
### init(map) — building it from the memory map
|
### init(map) — building it from the memory map
|
||||||
|
|||||||
+2
-2
@@ -9,10 +9,10 @@ write a 32-bit value to the right address, and a pixel changes color. That's
|
|||||||
exactly what `Console.pixel` does:
|
exactly what `Console.pixel` does:
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
self.rowPtr(y)[x] = color; // src/kernel/console.zig
|
self.rowPtr(y)[x] = color; // system/kernel/console.zig
|
||||||
```
|
```
|
||||||
|
|
||||||
Our `Framebuffer` struct (`src/root.zig`) is the four facts you need to
|
Our `Framebuffer` struct (`system/boot-handoff.zig`) is the four facts you need to
|
||||||
address it:
|
address it:
|
||||||
|
|
||||||
| Field | Meaning |
|
| Field | Meaning |
|
||||||
|
|||||||
+3
-3
@@ -16,7 +16,7 @@ safely, until the machine is reset or powered off.
|
|||||||
## The core of it: `hlt`
|
## The core of it: `hlt`
|
||||||
|
|
||||||
Everything comes down to one x86 instruction. It's CPU-specific, so it lives in
|
Everything comes down to one x86 instruction. It's CPU-specific, so it lives in
|
||||||
the arch module, `src/kernel/arch/x86_64/cpu.zig` (see [arch.md](arch.md)), and the
|
the arch module, `system/kernel/architecture/x86_64/cpu.zig` (see [arch.md](arch.md)), and the
|
||||||
generic kernel calls it as `arch.halt()`:
|
generic kernel calls it as `arch.halt()`:
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
@@ -83,7 +83,7 @@ treats the call:
|
|||||||
signature for a kernel entry point — the bootloader jumps in and nothing ever
|
signature for a kernel entry point — the bootloader jumps in and nothing ever
|
||||||
jumps back out.
|
jumps back out.
|
||||||
|
|
||||||
You can see the chain in `src/kernel/main.zig`: `_start` is `noreturn`, it calls
|
You can see the chain in `system/kernel/kernel.zig`: `_start` is `noreturn`, it calls
|
||||||
`kmain` which is `noreturn`, which ends by calling `arch.halt()` which is
|
`kmain` which is `noreturn`, which ends by calling `arch.halt()` which is
|
||||||
`noreturn`. The "never returns" property is threaded all the way down.
|
`noreturn`. The "never returns" property is threaded all the way down.
|
||||||
|
|
||||||
@@ -106,7 +106,7 @@ There are three halt sites, and they're all the same idea:
|
|||||||
`arch.halt()`. A panic is unrecoverable here, so stopping the machine — rather
|
`arch.halt()`. A panic is unrecoverable here, so stopping the machine — rather
|
||||||
than limping on with corrupted state — is the safe response.
|
than limping on with corrupted state — is the safe response.
|
||||||
|
|
||||||
3. **Bootloader failure** — in `src/boot/efi.zig`, if `boot()` fails *before* handing
|
3. **Bootloader failure** — in `boot/efi.zig`, if `boot()` fails *before* handing
|
||||||
off to the kernel, `main` logs the error and parks the machine with the same
|
off to the kernel, `main` logs the error and parks the machine with the same
|
||||||
loop so the message stays on screen:
|
loop so the message stays on screen:
|
||||||
|
|
||||||
|
|||||||
+1
-1
@@ -7,7 +7,7 @@ top of both to provide what the rest of the kernel actually wants: `alloc(n)` /
|
|||||||
the thing that unlocks dynamic data structures — lists, hash maps, driver state,
|
the thing that unlocks dynamic data structures — lists, hash maps, driver state,
|
||||||
eventually a process table.
|
eventually a process table.
|
||||||
|
|
||||||
It's generic kernel code (`src/kernel/heap.zig`): the allocator logic is
|
It's generic kernel code (`system/kernel/heap.zig`): the allocator logic is
|
||||||
architecture-neutral, using `arch.mapPage` and the frame allocator underneath.
|
architecture-neutral, using `arch.mapPage` and the frame allocator underneath.
|
||||||
|
|
||||||
## A growable free-list allocator
|
## A growable free-list allocator
|
||||||
|
|||||||
+161
@@ -0,0 +1,161 @@
|
|||||||
|
# The input module: broadcasting input events
|
||||||
|
|
||||||
|
A keyboard driver has one keystroke and *many* programs that might want it — a shell, a
|
||||||
|
window server, a logger. None of them owns the hardware, and the driver should not know
|
||||||
|
who is listening. So between the drivers and the listeners sits the **input service**
|
||||||
|
(`system/services/input/`): drivers **publish** events to it, programs **subscribe**, and
|
||||||
|
it fans each event out to every interested subscriber. It is an ordinary ring-3 process
|
||||||
|
reached over IPC, like the [VFS server](../system/services/vfs/vfs.zig) — no kernel knows
|
||||||
|
what a key is.
|
||||||
|
|
||||||
|
## One service, several device classes
|
||||||
|
|
||||||
|
The service carries three device classes today — **keyboard**, **mouse**, and
|
||||||
|
**joystick/gamepad** — and is built to take more
|
||||||
|
([protocol.zig](../system/services/input/protocol.zig)). Each class has its own typed
|
||||||
|
event:
|
||||||
|
|
||||||
|
- `KeyEvent` — `key_down`/`key_up` (physical make/break) and `key_press` (a character was
|
||||||
|
produced, carrying the Unicode scalar); plus a layout-independent `keycode` and a
|
||||||
|
`modifiers` bitmask.
|
||||||
|
- `MouseEvent` — relative `motion` (`dx`/`dy`), `button_down`/`button_up`, and `scroll`.
|
||||||
|
- `JoystickEvent` — `axis` moves (a signed value on a `control` index) and
|
||||||
|
`button_down`/`button_up`.
|
||||||
|
|
||||||
|
All three travel in one **`InputEvent` envelope** tagged with a `DeviceKind`, so the
|
||||||
|
fan-out is a single code path and a subscriber can take a mix of classes on one stream.
|
||||||
|
Decode an envelope with `asKeyboard()` / `asMouse()` / `asJoystick()` (each returns null
|
||||||
|
unless the tag matches). A subscriber names the classes it wants with a **`device_mask`**,
|
||||||
|
and the service routes each event only to subscribers whose mask includes its class — so a
|
||||||
|
mouse-only listener never wakes for keystrokes.
|
||||||
|
|
||||||
|
## Why this needed a new kernel primitive
|
||||||
|
|
||||||
|
The interesting part is delivery, and it runs straight into the shape of danos IPC.
|
||||||
|
[ipc.md](ipc.md) describes a **synchronous rendezvous**: a server holds exactly one
|
||||||
|
pending reply (`Task.ipc_client`) and *must* answer it on its next `replyWait`. Two
|
||||||
|
consequences decide the whole design:
|
||||||
|
|
||||||
|
1. **You cannot block N subscribers waiting for "the next event".** A server can hold only
|
||||||
|
one caller at a time, so the natural "subscriber calls `next_event()` and blocks" API
|
||||||
|
is impossible for more than one subscriber. Delivery therefore has to be **push** — the
|
||||||
|
service reaching out to subscribers — not pull.
|
||||||
|
|
||||||
|
2. **A synchronous push can hang the whole service.** If the service delivered with
|
||||||
|
`ipc_call`, it would block until each subscriber replied. `ipc_call` has no timeout, and
|
||||||
|
the kernel does **not** wake a caller parked on a *dead* peer's endpoint (it only fails a
|
||||||
|
peer that was mid-reply — see [process.zig](../system/kernel/process.zig)
|
||||||
|
`releaseTaskResourcesLocked`). One subscriber that exits mid-delivery would wedge input
|
||||||
|
for everyone. That is the opposite of the resilience the microkernel is for.
|
||||||
|
|
||||||
|
The fix is the asynchronous send that [ipc.md](ipc.md) had already earmarked as future
|
||||||
|
work ("asynchronous / buffered send … for notifications between servers"):
|
||||||
|
|
||||||
|
```
|
||||||
|
ipc_send(handle, message_ptr, message_len) -> 0 / -errno
|
||||||
|
```
|
||||||
|
|
||||||
|
`ipc_send` copies a small payload into the endpoint's **bounded queue** and wakes a
|
||||||
|
receiver, then returns immediately — it never blocks and so can never hang on a dead or
|
||||||
|
slow subscriber. The receiver picks it up through the same `replyWait` it already runs:
|
||||||
|
the wake arrives as a **buffered message** — `notify_badge_bit | notify_message_bit` set in
|
||||||
|
the badge (distinguishing it from a bare IRQ/child-exit notification), the sender's task id
|
||||||
|
in the low bits, and the payload in the receive buffer, with no reply owed. The queue holds
|
||||||
|
16 messages per endpoint; a full queue **drops the oldest**, because a buffered message is
|
||||||
|
discrete data, not a coalescing "level" like an interrupt. See
|
||||||
|
[ipc-synchronous.zig](../system/kernel/ipc-synchronous.zig) (`sendLocked`, `popPost`, and
|
||||||
|
the `replyWait` receive loop).
|
||||||
|
|
||||||
|
This is the async counterpart of `ipc_call`, and the input service is its first consumer.
|
||||||
|
|
||||||
|
## How the pieces fit
|
||||||
|
|
||||||
|
```
|
||||||
|
keyboard/mouse driver, input-source input service subscriber(s)
|
||||||
|
----------------------------------- ------------- -------------
|
||||||
|
connectSource(); loop: replyWait: subscribeKeyboard()/…All:
|
||||||
|
publishKeyboardEvent(k) ─ ipc_call ─▶ publish → broadcast: createIpcEndpoint()
|
||||||
|
publishMouseEvent(m) for each sub whose callCap(subscribe,
|
||||||
|
publishJoystickEvent(j) mask matches event.device: send_cap = ep,
|
||||||
|
ipc_send(sub_ep) ──────▶ device_mask)
|
||||||
|
reply ok loop: next()
|
||||||
|
subscribe → store {ep cap, └─ replyWait(ep)
|
||||||
|
task id, device_mask} → InputEvent
|
||||||
|
```
|
||||||
|
|
||||||
|
- A **subscriber** calls `input.subscribe(mask)` — or a typed helper: `subscribeKeyboard()`,
|
||||||
|
`subscribeMouse()`, `subscribeJoystick()` (one class, `next()` returns the decoded event),
|
||||||
|
or `subscribeAll()` (every class, `next()` returns a tagged `InputEvent`)
|
||||||
|
([library/runtime/input.zig](../library/runtime/input.zig)). It creates its own endpoint
|
||||||
|
and hands it to the service as a **capability** (M13 capability passing — the input
|
||||||
|
service is that feature's first real user), along with its `device_mask`. Then it loops on
|
||||||
|
`next()`, a `replyWait` on that endpoint returning each pushed event.
|
||||||
|
- A **source** (a keyboard, mouse, or joystick driver) calls `input.connectSource()` and the
|
||||||
|
method for its class: `publishKeyboardEvent`, `publishMouseEvent`, or
|
||||||
|
`publishJoystickEvent`. Publishing is a short synchronous `ipc_call` the service answers at
|
||||||
|
once; the service's own fan-out is asynchronous, so publishing never blocks on a slow
|
||||||
|
subscriber.
|
||||||
|
- The **service** ([input.zig](../system/services/input/input.zig)) keeps a small subscriber
|
||||||
|
table (endpoint handle + owning task id + `device_mask`). On `publish` it `ipc_send`s the
|
||||||
|
event to every subscriber whose mask includes the event's device class. On `subscribe` it
|
||||||
|
stores the passed capability and mask and, as housekeeping, prunes any slot whose owning
|
||||||
|
process has exited (checked against `process_enumerate`) — not for correctness (an async
|
||||||
|
send to an orphaned endpoint is harmless) but to reclaim the slot.
|
||||||
|
|
||||||
|
Publisher and subscriber must be **separate processes**: a single thread that both
|
||||||
|
published and serviced its own subscription would deadlock (its `publish` call blocks until
|
||||||
|
the service delivers to its endpoint, which only the same thread could receive).
|
||||||
|
|
||||||
|
## Status and follow-ups
|
||||||
|
|
||||||
|
- **The keyboard is real.** The `ps2-bus` driver owns PNP0303, which carries *both* the
|
||||||
|
0x60/0x64 ports and IRQ1, so reading the hardware lives in the bus, not in
|
||||||
|
[keyboard.zig](../system/drivers/ps2-bus/keyboard.zig): the bus binds IRQ1 and, on each
|
||||||
|
interrupt, drains port 0x60, routing every byte by the status register's
|
||||||
|
auxiliary-output bit to whichever child driver **attached** for that device (an
|
||||||
|
`AttachRequest` to the well-known `ps2_bus` service, carrying the child's endpoint as a
|
||||||
|
capability; the bytes then arrive as asynchronous `ForwardedByte` messages, so the IRQ
|
||||||
|
path never blocks on a child). The keyboard driver decodes the stream — scancode **set 2**,
|
||||||
|
what the keyboard sends with the 8042's legacy translation off, decoded by
|
||||||
|
[scancode.zig](../system/drivers/ps2-bus/scancode.zig) into USB HID usage keycodes with
|
||||||
|
make/break, typematic-repeat, and modifier tracking (host-tested under `zig build test`) —
|
||||||
|
and publishes real `key_down`/`key_press`/`key_up` events.
|
||||||
|
- **Keycode → character** is wired in: the keyboard driver fills a `key_press` event's
|
||||||
|
`character` through [`library/xkeyboard-config`](../library/xkeyboard-config/README.md)
|
||||||
|
(`xkb.map(layout, keycode, mods)` → keysym + Unicode character), synthesizing the ASCII
|
||||||
|
control characters for Enter/Tab/Backspace/Escape, whose keysyms map to no Unicode. The
|
||||||
|
layout defaults to `us`; the bus can pass another as the driver's argv[2] — the seam for
|
||||||
|
a future settings source.
|
||||||
|
- **The mouse is real too.** IRQ12 is enumerated on the auxiliary device's own ACPI node
|
||||||
|
(PNP0F13), so the bus claims that node alongside the controller and routes both IRQs to
|
||||||
|
its one endpoint, acking whichever line the notification's badge names.
|
||||||
|
[mouse.zig](../system/drivers/ps2-bus/mouse.zig) attaches the way the keyboard does and
|
||||||
|
assembles the forwarded bytes with
|
||||||
|
[mouse-packet.zig](../system/drivers/ps2-bus/mouse-packet.zig) (three-byte stream-mode
|
||||||
|
packets: sync/overflow handling, nine-bit movement, screen-convention `dy` — host-tested
|
||||||
|
under `zig build test`) into `button_down`/`button_up` transitions and `motion` events.
|
||||||
|
**Follow-up:** the IntelliMouse magic-knock for a scroll wheel (four-byte packets) and
|
||||||
|
`scroll` events. The hardware-free `input-source` still rotates through all three classes
|
||||||
|
synthetically (including a joystick, which has no driver yet) via the
|
||||||
|
`input.synthetic*Event` helpers.
|
||||||
|
- **Drop-oldest under overflow** is a defined loss; the 16-slot ring absorbs normal bursts.
|
||||||
|
Real backpressure/flow-control is future work.
|
||||||
|
- **`publish` is unauthenticated** — any process may publish, consistent with the current
|
||||||
|
bring-up trust model (see [driver-model.md](driver-model.md)). A source capability is
|
||||||
|
future work.
|
||||||
|
|
||||||
|
## Verifying it
|
||||||
|
|
||||||
|
The `input` case (`python3 test/qemu_test.py input`, in
|
||||||
|
[tests.zig](../system/kernel/tests.zig) `inputTest`) boots the real kernel and spawns the
|
||||||
|
service, the synthetic source (which cycles keyboard, mouse, and joystick events), and a
|
||||||
|
subscriber that took all three classes. It passes only when the subscriber heartbeats
|
||||||
|
`input-test: ok` — proof that an event travelled source → service → subscriber over IPC,
|
||||||
|
exercising `ipc_send`, capability-passing subscription, and per-device routing. Each
|
||||||
|
serial line names the class received, so the log shows all three arriving on one stream.
|
||||||
|
|
||||||
|
## See also
|
||||||
|
|
||||||
|
- [ipc.md](ipc.md) — the synchronous rendezvous and the notification path `ipc_send` extends.
|
||||||
|
- [syscall.md](syscall.md) — the system-call surface, including `ipc_send`.
|
||||||
|
- [driver-model.md](driver-model.md) — class drivers, capability passing (M13), the trust model.
|
||||||
+20
-9
@@ -9,7 +9,7 @@ reboot is miserable.
|
|||||||
|
|
||||||
This is the machinery that catches those faults and prints what happened instead.
|
This is the machinery that catches those faults and prints what happened instead.
|
||||||
It's all x86_64-specific, so it lives behind the [arch](arch.md) boundary in
|
It's all x86_64-specific, so it lives behind the [arch](arch.md) boundary in
|
||||||
`src/kernel/arch/x86_64/`. Only the 32 CPU-defined exception vectors are wired up so far;
|
`system/kernel/architecture/x86_64/`. Only the 32 CPU-defined exception vectors are wired up so far;
|
||||||
device interrupts (timer, keyboard, via the APIC) come later, on the same IDT.
|
device interrupts (timer, keyboard, via the APIC) come later, on the same IDT.
|
||||||
|
|
||||||
## First the GDT
|
## First the GDT
|
||||||
@@ -20,7 +20,7 @@ IDT gate names a code-segment *selector* that must resolve in the current GDT. T
|
|||||||
firmware left a GDT in place, but we don't control it, so we install our own with
|
firmware left a GDT in place, but we don't control it, so we install our own with
|
||||||
known selectors: `0x08` kernel code, `0x10` kernel data.
|
known selectors: `0x08` kernel code, `0x10` kernel data.
|
||||||
|
|
||||||
`src/kernel/arch/x86_64/gdt.zig` holds three flat descriptors — a required null entry,
|
`system/kernel/architecture/x86_64/gdt.zig` holds three flat descriptors — a required null entry,
|
||||||
plus code and data — where the only bits that matter in long mode are the access
|
plus code and data — where the only bits that matter in long mode are the access
|
||||||
byte and the code segment's long-mode (`L`) flag. Loading it (`gdt_flush` in
|
byte and the code segment's long-mode (`L`) flag. Loading it (`gdt_flush` in
|
||||||
`isr.s`) does two things: `lgdt`, then reload the segment registers. The data
|
`isr.s`) does two things: `lgdt`, then reload the segment registers. The data
|
||||||
@@ -33,7 +33,7 @@ into CS:RIP.
|
|||||||
The **Interrupt Descriptor Table** maps each of 256 vectors to a handler. Each
|
The **Interrupt Descriptor Table** maps each of 256 vectors to a handler. Each
|
||||||
entry is a 16-byte *gate* holding the handler's address (split across three
|
entry is a 16-byte *gate* holding the handler's address (split across three
|
||||||
fields, a quirk of the format), the code selector (`0x08`), and flags: `0x8E`
|
fields, a quirk of the format), the code selector (`0x08`), and flags: `0x8E`
|
||||||
means present, ring 0, 64-bit interrupt gate. `src/kernel/arch/x86_64/idt.zig` builds the
|
means present, ring 0, 64-bit interrupt gate. `system/kernel/architecture/x86_64/idt.zig` builds the
|
||||||
table, points the first 32 vectors at their stubs, and loads it with `lidt`
|
table, points the first 32 vectors at their stubs, and loads it with `lidt`
|
||||||
(`idt_flush`).
|
(`idt_flush`).
|
||||||
|
|
||||||
@@ -49,7 +49,7 @@ hit a fault *while trying to deliver another fault* — very often because the
|
|||||||
current stack pointer is bad, so pushing the exception frame itself faulted. If
|
current stack pointer is bad, so pushing the exception frame itself faulted. If
|
||||||
the #DF handler then tried to push onto that same bad stack, it would fault a
|
the #DF handler then tried to push onto that same bad stack, it would fault a
|
||||||
third time and **triple-fault** — an instant reset. So the #DF gate is pointed at
|
third time and **triple-fault** — an instant reset. So the #DF gate is pointed at
|
||||||
**IST1**, a small dedicated stack (`src/kernel/arch/x86_64/tss.zig`) that's always valid.
|
**IST1**, a small dedicated stack (`system/kernel/architecture/x86_64/tss.zig`) that's always valid.
|
||||||
|
|
||||||
Bringing it up: fill in the TSS's IST1 pointer, publish the TSS through a
|
Bringing it up: fill in the TSS's IST1 pointer, publish the TSS through a
|
||||||
descriptor in the GDT (`gdt.setTss`), and load it into the task register with
|
descriptor in the GDT (`gdt.setTss`), and load it into the task register with
|
||||||
@@ -60,7 +60,7 @@ which is why the GDT grew from three entries to five.
|
|||||||
|
|
||||||
On an exception the CPU pushes a small frame (SS, RSP, RFLAGS, CS, RIP) and, for
|
On an exception the CPU pushes a small frame (SS, RSP, RFLAGS, CS, RIP) and, for
|
||||||
*some* vectors, an **error code**. That inconsistency is a nuisance, so each stub
|
*some* vectors, an **error code**. That inconsistency is a nuisance, so each stub
|
||||||
in `src/kernel/arch/x86_64/isr.s` normalises it: vectors that don't get a hardware error
|
in `system/kernel/architecture/x86_64/isr.s` normalises it: vectors that don't get a hardware error
|
||||||
code push a dummy `0`, then every stub pushes its **vector number** and jumps to a
|
code push a dummy `0`, then every stub pushes its **vector number** and jumps to a
|
||||||
shared tail, `isr_common`. The tail pushes all the general registers and calls the
|
shared tail, `isr_common`. The tail pushes all the general registers and calls the
|
||||||
Zig handler with a pointer to the whole thing.
|
Zig handler with a pointer to the whole thing.
|
||||||
@@ -80,11 +80,22 @@ inline). `build.zig` adds `isr.s` to the arch module.
|
|||||||
## Reporting a fault
|
## Reporting a fault
|
||||||
|
|
||||||
`isr_common` calls `exceptionHandler`, which forwards to a swappable `on_fault`
|
`isr_common` calls `exceptionHandler`, which forwards to a swappable `on_fault`
|
||||||
hook. The generic kernel installs a reporter (`onException` in `main.zig`) that
|
hook. The generic kernel installs a reporter (`onException` in `kernel.zig`) that
|
||||||
prints, in red, the exception name and vector, the error code, the faulting RIP
|
prints the exception name and vector, the error code, the faulting RIP
|
||||||
and RSP, and — for a page fault (#PF, vector 14) — the faulting address from
|
and RSP, and — for a page fault (#PF, vector 14) — the faulting address from
|
||||||
**CR2**. Then it halts. There's no fault *recovery* yet, so every exception is
|
**CR2**. What happens next depends on where the fault came from:
|
||||||
terminal; the point is that it's now **visible** instead of a silent reset.
|
|
||||||
|
- **User mode (CPL 3): kill the process, keep the machine.** The kernel is intact
|
||||||
|
(the CPU trapped onto the task's kernel stack), so the faulting process is
|
||||||
|
killed — address space, IRQ bindings, and IPC handles reclaimed; a client it
|
||||||
|
owed a reply to is failed with `-EPEER` — and the core reschedules. A crashing
|
||||||
|
driver takes itself down, never the OS. This is fault recovery step 2 of
|
||||||
|
[resilience.md](resilience.md). NMI, double fault, and machine check are
|
||||||
|
excluded: they report machine trouble regardless of what was running.
|
||||||
|
- **Kernel mode: halt this core.** The trusted base itself is broken, so there is
|
||||||
|
nothing safe to kill; the fault is still *contained* to the core (an
|
||||||
|
application-processor fault leaves the rest of the system running), and the
|
||||||
|
report makes it **visible** instead of a silent reset.
|
||||||
|
|
||||||
The hook is set before `arch.init()` in `kmain`, so a fault during setup is still
|
The hook is set before `arch.init()` in `kmain`, so a fault during setup is still
|
||||||
caught.
|
caught.
|
||||||
|
|||||||
+79
-12
@@ -6,12 +6,20 @@ just call each other — a request becomes a **message**. In a microkernel, what
|
|||||||
was a function call across a monolithic kernel is IPC, so it's a first-class
|
was a function call across a monolithic kernel is IPC, so it's a first-class
|
||||||
concern, not an afterthought.
|
concern, not an afterthought.
|
||||||
|
|
||||||
This first form is a **bounded blocking channel** (`src/kernel/ipc.zig`): a fixed-size
|
There are two layers, built a milestone apart:
|
||||||
ring buffer of messages with a producer/consumer rendezvous, built on the
|
|
||||||
scheduler's [wait queues](scheduling.md).
|
- **`system/kernel/ipc.zig`** — a bounded blocking channel between *kernel threads*,
|
||||||
|
described below. The primitive, and where the blocking discipline was worked out.
|
||||||
|
- **`system/kernel/ipc-synchronous.zig`** — synchronous call/reply between *processes*, across
|
||||||
|
address spaces. What user-space servers and drivers actually talk over. It's the
|
||||||
|
second half of this document.
|
||||||
|
|
||||||
## The channel
|
## The channel
|
||||||
|
|
||||||
|
The first form is a **bounded blocking channel** (`system/kernel/ipc.zig`): a fixed-size
|
||||||
|
ring buffer of messages with a producer/consumer rendezvous, built on the
|
||||||
|
scheduler's [wait queues](scheduling.md).
|
||||||
|
|
||||||
`Channel(T, capacity)` is generic over the message type and buffer size. It holds a
|
`Channel(T, capacity)` is generic over the message type and buffer size. It holds a
|
||||||
ring buffer, a count, and two wait queues:
|
ring buffer, a count, and two wait queues:
|
||||||
|
|
||||||
@@ -42,16 +50,75 @@ full and empty over and over, so both the blocking-send and blocking-recv paths
|
|||||||
exercised heavily. The messages arrive intact and in order (their sum is the
|
exercised heavily. The messages arrive intact and in order (their sum is the
|
||||||
expected `5050`), and neither task busy-waits — they block and wake each other.
|
expected `5050`), and neither task busy-waits — they block and wake each other.
|
||||||
|
|
||||||
|
## Endpoints: call/reply across address spaces
|
||||||
|
|
||||||
|
A channel connects two kernel threads sharing one address space. Real servers are
|
||||||
|
*processes*, so the payload has to cross an address-space boundary. That's
|
||||||
|
`system/kernel/ipc-synchronous.zig`, and its shape is L4's: a synchronous **rendezvous** at an
|
||||||
|
`Endpoint`, with the message copied directly from the sender's pages to the receiver's
|
||||||
|
(`copyAcross` walks both sets of page tables through the physmap — no CR3 switch, no
|
||||||
|
bounce buffer).
|
||||||
|
|
||||||
|
Two syscalls carry it:
|
||||||
|
|
||||||
|
- **`ipc_call(h, msg, reply)`** — copy `msg` to the server, block until it replies.
|
||||||
|
- **`ipc_reply_wait(h, reply, recv)`** — reply to the client you're still holding (if
|
||||||
|
any), then block for the next request. One syscall, because a server's steady state
|
||||||
|
is *always* "finish the last one, wait for the next".
|
||||||
|
|
||||||
|
An endpoint is reached by **handle** — a small integer index into the process's handle
|
||||||
|
table (`Task.handles`), exactly like a file descriptor, and just as unforgeable. The
|
||||||
|
bootstrap problem (how do you get the first handle?) is solved by a tiny name registry:
|
||||||
|
a server calls `ipc_register(service_id, h)` under a well-known small integer, and a
|
||||||
|
client calls `ipc_lookup(service_id)`.
|
||||||
|
|
||||||
|
The server never learns the client's identity beyond a **badge**, delivered alongside
|
||||||
|
the message: the caller's task id.
|
||||||
|
|
||||||
|
### Interrupts are messages too
|
||||||
|
|
||||||
|
`notifyFromIsr` posts an *asynchronous* notification to an endpoint — no payload, no
|
||||||
|
reply owed — and wakes whoever is blocked in `reply_wait`. Its badge has the top bit
|
||||||
|
set (`notify_badge_bit`), which is how a driver's single event loop distinguishes "a
|
||||||
|
client wants something" from "the hardware wants something". Notifications sit in a
|
||||||
|
small coalescing ring on the endpoint, so an interrupt taken while the driver was busy
|
||||||
|
elsewhere is not lost.
|
||||||
|
|
||||||
|
This is what makes a user-space driver possible at all, and it's the subject of
|
||||||
|
[drivers.md](drivers.md).
|
||||||
|
|
||||||
## What's next (not done here)
|
## What's next (not done here)
|
||||||
|
|
||||||
- **Across address spaces.** Today both endpoints are kernel threads sharing the
|
|
||||||
kernel's memory, so the message is copied within one address space. When user
|
|
||||||
mode arrives, the same channel carries messages between *isolated* processes,
|
|
||||||
copying the payload across the boundary — which is where IPC earns its place as
|
|
||||||
the microkernel's backbone.
|
|
||||||
- **Synchronous call/reply.** A request/response pattern (send-and-wait-for-reply)
|
|
||||||
on top of channels, the shape most driver/service calls take.
|
|
||||||
- **Interrupts as messages.** A hardware interrupt delivered to the driver task
|
|
||||||
that owns the device, as an IPC message.
|
|
||||||
- **Priority inheritance** through IPC, so a high-priority client blocked on a
|
- **Priority inheritance** through IPC, so a high-priority client blocked on a
|
||||||
low-priority server doesn't suffer unbounded priority inversion.
|
low-priority server doesn't suffer unbounded priority inversion.
|
||||||
|
- **Handle transfer.** A server can't hand a client a handle to a third endpoint, so
|
||||||
|
every capability is either well-known (the registry) or inherited — there's no way
|
||||||
|
to delegate one.
|
||||||
|
- **Asynchronous / buffered send** for the cases where a rendezvous is the wrong
|
||||||
|
shape (logging, notifications between servers). *Landed as `ipc_send`* — a
|
||||||
|
non-blocking post to an endpoint's bounded payload queue, delivered through
|
||||||
|
`reply_wait` as a buffered message (badge bit `notify_message_bit`). Built for, and
|
||||||
|
first used by, the [input service](input.md)'s keyboard-event broadcast, where a
|
||||||
|
synchronous push would let one dead subscriber hang the fan-out. A full queue drops
|
||||||
|
the oldest (discrete messages, not a coalescing level like the notification ring).
|
||||||
|
- **A bounded reply.** `MSG_MAX` is 256 bytes and the copy runs under the big kernel
|
||||||
|
lock; a bulk transfer wants shared pages, not a copy.
|
||||||
|
|
||||||
|
## Lifecycle conventions over IPC (M17)
|
||||||
|
|
||||||
|
Three conventions from [process-lifecycle.md](process-lifecycle.md) ride the
|
||||||
|
notification mechanism:
|
||||||
|
|
||||||
|
- **Signals** arrive as notifications on the endpoint a process nominated with
|
||||||
|
`signal_bind` (`runtime.process.bindSignals`): badge = the signal bit plus the
|
||||||
|
coalesced pending mask (`runtime.process.signalsFrom` decodes). Statements,
|
||||||
|
never questions; no payload, no reply.
|
||||||
|
- **One-shot timers** (`timer_bind`, `runtime.system.timerOnce`) land as a
|
||||||
|
timer-bit notification — the timed wait: a service arms a deadline and keeps
|
||||||
|
serving, instead of blocking in sleep.
|
||||||
|
- **The universal ping**: a **zero-length request is the liveness probe**,
|
||||||
|
answered with a zero-length reply by the service harness itself
|
||||||
|
(`runtime.service.run`). No protocol's requests start at length zero, so the
|
||||||
|
encoding cannot collide, and a wedged service simply fails to answer — which
|
||||||
|
is the diagnosis. Deep health ("can I reach my hardware?") stays a per-service
|
||||||
|
protocol message.
|
||||||
|
|||||||
+104
@@ -0,0 +1,104 @@
|
|||||||
|
# Logging: the diagnostic log vs. the display
|
||||||
|
|
||||||
|
danos separates two things that are easy to conflate: the **diagnostic log** — the
|
||||||
|
machine-readable stream of *what the kernel is doing* — and the **display**, the
|
||||||
|
framebuffer surface the OS draws on. They are different concerns with different
|
||||||
|
lifetimes, so they're different code paths.
|
||||||
|
|
||||||
|
The guiding rule: **output is a diagnostic convenience, never a correctness
|
||||||
|
dependency.** The kernel must boot and run correctly with *zero* output channels —
|
||||||
|
no serial, no screen. Logging that can take the kernel down isn't robust; it's a
|
||||||
|
liability. This is the same [resilience](resilience.md) posture the rest of the
|
||||||
|
kernel follows.
|
||||||
|
|
||||||
|
## The log is multi-sink
|
||||||
|
|
||||||
|
`system/kernel/log.zig` is the diagnostic log. It fans a message out to a set of
|
||||||
|
registered **sinks**, each best-effort and self-guarding:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
log.addSink(arch.serialWrite); // the serial UART
|
||||||
|
if (arch.debugconPresent()) log.addSink(arch.debugconWrite); // 0xE9 debug console
|
||||||
|
// later: log.addSink(fs.logWrite); // a file on a ramdisk / USB / SSD
|
||||||
|
log.write("…"); log.print("x={d}\n", .{x});
|
||||||
|
```
|
||||||
|
|
||||||
|
Properties that matter:
|
||||||
|
|
||||||
|
- **No allocation.** The sink table is a fixed array, so the log works before the
|
||||||
|
heap is up and inside a panic.
|
||||||
|
- **Best-effort.** A sink whose device is absent is a no-op (e.g. writing to a
|
||||||
|
missing UART just goes nowhere — the TX-wait is bounded so it can't hang). A
|
||||||
|
message reaches whatever channels exist; if none do, the kernel runs on, silent.
|
||||||
|
- **Order-independent.** Every registered sink gets every message. Adding the file
|
||||||
|
logger later is one `addSink` call and **zero** changes to call sites.
|
||||||
|
|
||||||
|
## The framebuffer is *not* a log sink
|
||||||
|
|
||||||
|
The framebuffer is a general graphics surface, **not inherently a text terminal**.
|
||||||
|
Today `system/kernel/console.zig` paints a text grid on it as a *bootstrap* console, but
|
||||||
|
that's a stop-gap: once the driver machinery exists the framebuffer becomes a proper
|
||||||
|
**graphics device driver**, and the text crutch goes away. So the log must not assume
|
||||||
|
it — routing the verbose log through a pixel console would bake in "the OS is text".
|
||||||
|
|
||||||
|
Instead the two paths are explicit:
|
||||||
|
|
||||||
|
```
|
||||||
|
verbose diagnostics ──► log ──► serial, debugcon, (file later)
|
||||||
|
user status / panics ──► status() ──► log (above) + framebuffer (if present)
|
||||||
|
```
|
||||||
|
|
||||||
|
A handful of user-facing lines (`kernel initialised`, a panic) go through
|
||||||
|
`main.zig`'s `status()` / `statusPrint()`, which write to the log **and** paint the
|
||||||
|
framebuffer when one is present. Everything else uses `log.*` and never touches the
|
||||||
|
screen. `console.write` is a no-op when the firmware gave us no framebuffer.
|
||||||
|
|
||||||
|
## Optional framebuffer (headless machines)
|
||||||
|
|
||||||
|
A framebuffer is not guaranteed — a headless server exposes no UEFI Graphics Output
|
||||||
|
Protocol. That used to be *fatal* (the loader failed the boot). Now the loader hands
|
||||||
|
over a "no framebuffer" descriptor (`base == 0`) rather than failing, and
|
||||||
|
`Framebuffer.present()` (in `system/boot-handoff.zig`) gates every on-screen path. A headless,
|
||||||
|
serial-less machine boots and runs correctly — it just goes quiet.
|
||||||
|
|
||||||
|
## Last-resort channels (no text output at all)
|
||||||
|
|
||||||
|
Two signals bypass the sink list, because they must survive even a total
|
||||||
|
output-channel failure:
|
||||||
|
|
||||||
|
- **`log.checkpoint(code)`** — a one-byte **POST code** to I/O port `0x80` (a POST
|
||||||
|
card or BMC shows it). `main.zig` emits one at each boot milestone (`cp_paging`,
|
||||||
|
`cp_heap`, …) and on a fault/panic, so "where did it hang?" is answerable with no
|
||||||
|
text output whatsoever. Writing `0x80` is universally safe.
|
||||||
|
- **`log.recordPanic(msg)`** — stamps the panic message into a fixed record
|
||||||
|
(`log.panic_record`, with a `magic` written last). A post-mortem — an attached
|
||||||
|
debugger, a RAM dump, or a future file/pstore reader — recovers *what killed it*
|
||||||
|
even though nothing was on screen.
|
||||||
|
|
||||||
|
The panic and CPU-exception handlers fan out to every sink, emit a POST code, and
|
||||||
|
drop the breadcrumb — they never assume a console.
|
||||||
|
|
||||||
|
## The 0xE9 debug console
|
||||||
|
|
||||||
|
Port `0xE9` is the Bochs/QEMU debug console. It's detected safely: the port returns
|
||||||
|
`0xE9` when read if present, and `0xFF` on real hardware, so `debugconPresent()`
|
||||||
|
only enables the sink when it's really there. Under QEMU it's captured with
|
||||||
|
`-debugcon file:…`, giving CI a log channel independent of `-serial`.
|
||||||
|
|
||||||
|
## The robustness spectrum
|
||||||
|
|
||||||
|
The result handles every combination — framebuffer only, serial only, both, or
|
||||||
|
**neither**. With no channels at all the kernel still boots and runs; port-`0x80`
|
||||||
|
checkpoints track progress and the panic breadcrumb captures failures. *Runs blind
|
||||||
|
but correct* is the goal, not *always has output*.
|
||||||
|
|
||||||
|
## Related
|
||||||
|
|
||||||
|
- [framebuffer.md](framebuffer.md) — the display surface itself (pitch, format), the
|
||||||
|
thing that becomes a graphics device driver.
|
||||||
|
- [efi.md](efi.md) — where the loader captures (or, headless, doesn't capture) the
|
||||||
|
framebuffer before `ExitBootServices`.
|
||||||
|
- [device-interrupts.md](device-interrupts.md) — the serial UART bring-up the log's
|
||||||
|
primary sink rides on.
|
||||||
|
- [resilience.md](resilience.md) — why "never let a missing peripheral take the
|
||||||
|
kernel down" is a core design stance.
|
||||||
@@ -0,0 +1,210 @@
|
|||||||
|
# M17–M18 execution plan: process lifecycle + device manager
|
||||||
|
|
||||||
|
**Archived — completed 2026-07-13** (every item checked; suite ended 54/54).
|
||||||
|
Kept as the record of how M17–M18 landed; the successor is
|
||||||
|
[m19-m20-plan.md](m19-m20-plan.md).
|
||||||
|
|
||||||
|
The operational plan for building [process-lifecycle.md](process-lifecycle.md)
|
||||||
|
(M17) and [device-manager.md](device-manager.md) increments 5–7 (M18). Design is
|
||||||
|
settled in those documents; this file is the build order — one phase at a time,
|
||||||
|
each phase green before the next starts. Delete or archive this file when M18
|
||||||
|
lands.
|
||||||
|
|
||||||
|
**Definition of green, every phase:** `zig build` clean, `zig build test` clean,
|
||||||
|
`python3 test/qemu_test.py` passes (existing scenarios plus the phase's new one),
|
||||||
|
and the relevant design doc's "known gaps" / status lines updated. Commit per
|
||||||
|
green phase (no co-author trailers).
|
||||||
|
|
||||||
|
**Workflow (settled 2026-07-12):** work happens in a dedicated git worktree, on
|
||||||
|
feature branches cut from `main` — `feat/process-lifecycle` (M17.1–17.4),
|
||||||
|
`feat/device-manager` (M18.1), `feat/usb-xhci-bus` (M18.2–18.3). When a branch's
|
||||||
|
phases are all green it is **auto-merged into `main`**; branches are kept after
|
||||||
|
merge, not deleted. Merges and branches are pushed to origin. Phase 0 (once):
|
||||||
|
commit the design docs, merge the outstanding `feat/usb` work into `main`, and
|
||||||
|
run the existing QEMU suite green before any new work starts.
|
||||||
|
|
||||||
|
**Numbering note:** continues the milestone sequence (driver track ended at M16).
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
The loop marks a phase `[x]` in the same commit that lands it. A phase is marked
|
||||||
|
only when its definition of green holds.
|
||||||
|
|
||||||
|
- [x] **Phase 0** — baseline: docs committed, feat/usb merged to main, pushed;
|
||||||
|
`usb-xhci-libary.zig` renamed to `usb-xhci-library.zig`; existing QEMU
|
||||||
|
suite green from the worktree (48/48, 2026-07-12).
|
||||||
|
- [x] **M17.1** — kernel releases claims/MSI on death (claims: `releaseAllOwnedBy`
|
||||||
|
in the reap; MSI was already swept by `irq.releaseOwner`; `claim-release`
|
||||||
|
test; suite 49/49)
|
||||||
|
- [x] **M17.2** — exit reasons (`ExitReason` recorded at exit/fault/kill before
|
||||||
|
the notification; `process_exit_reason` supervisor-gated;
|
||||||
|
`runtime.process.exitReason`; kernel + ring-3 assertions; suite 49/49)
|
||||||
|
- [x] **M17.3** — published exit events + VFS subscriber (`process_subscribe`,
|
||||||
|
bounded ref-counted table, publish on every death;
|
||||||
|
`runtime.process.subscribeExits`; VFS handles carry owners and are swept on
|
||||||
|
the owner's death; `vfs-client-death` test; suite 50/50)
|
||||||
|
- [x] **M17.4** — signals, timer notifications, `runtime.process`, the service
|
||||||
|
harness (signal_bind/process_signal + coalescing pending mask; timer_bind
|
||||||
|
on the tick; bindSignals/signalsFrom/sendSignal/stop + timerOnce;
|
||||||
|
runtime.service.run with the zero-length ping; VFS converted; `signals`
|
||||||
|
scenario; suite 51/51)
|
||||||
|
- [x] **merge** `feat/process-lifecycle` → main, push (merged 2026-07-13)
|
||||||
|
- [x] **M18.1** — device-manager protocol: hello + restart policy
|
||||||
|
(device-manager-protocol module; the manager as a harness service:
|
||||||
|
supervised spawns, hello deadline via timer sweep, restart with
|
||||||
|
300/600/1200ms backoff, exit reasons deciding restart-vs-stopped,
|
||||||
|
crash-loop cap; usb-xhci-bus first conforming driver; crash-test fixture
|
||||||
|
re-proving claim release each respawn; `driver-restart` scenario;
|
||||||
|
maximum_tasks 16→32 — the sweep was overflowing the pool; suite 52/52)
|
||||||
|
- [x] **merge** `feat/device-manager` → main, push (merged 2026-07-13)
|
||||||
|
- [x] **M18.2** — xHCI port scan + tree reports (child_added/child_removed in
|
||||||
|
the protocol; the manager's child mirror with death-pruning; xHCI maps the
|
||||||
|
register BAR — resource 0 is ECAM — reads CAPLENGTH/HCSPARAMS1, scans
|
||||||
|
PORTSC, reports connected ports with speed-class identity; `usb-report`
|
||||||
|
scenario proves report → prune → respawn → re-report; suite 53/53)
|
||||||
|
- [x] **M18.3** — app surface: enumerate/subscribe over IPC (subscriber
|
||||||
|
endpoint rides as the call's capability; events are the same structs the
|
||||||
|
buses send); device-list first client; protocol capped at the kernel's
|
||||||
|
IPC MESSAGE_MAXIMUM (256); the startUserTask debug print removed — it
|
||||||
|
sheared concurrent serial lines and was the scenario-flake root cause;
|
||||||
|
`device-list` scenario; suite 54/54)
|
||||||
|
- [x] **merge** `feat/usb-xhci-bus` → main, push (merged 2026-07-13) — **plan complete**
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## M17.1 — the kernel releases a dead process's claims
|
||||||
|
|
||||||
|
The cleanup half of iron rule 1; the prerequisite for every restart story.
|
||||||
|
|
||||||
|
- `system/kernel/devices-broker.zig`: `releaseAllOwnedBy(owner: u32)` — clear
|
||||||
|
every `claimed[]` slot holding this task id.
|
||||||
|
- `system/kernel/process.zig`: call it from the reap path, alongside the existing
|
||||||
|
IRQ-binding release (the ordering comment there says why IRQs go first — claims
|
||||||
|
slot in after them, before the exit notification).
|
||||||
|
- MSI vectors: find where `msi_bind` records per-device vectors (interrupts
|
||||||
|
module) and release those by owner in the same pass.
|
||||||
|
- Docs: remove the claims bullet from process-management.md "Known gaps".
|
||||||
|
|
||||||
|
**Test:** new QEMU scenario `claim-release` — a test child claims an unclaimed
|
||||||
|
device, is killed, is respawned, and claims the same device again successfully;
|
||||||
|
assert both claims in the serial log. Kernel-side unit coverage in
|
||||||
|
`system/kernel/tests.zig` for `releaseAllOwnedBy` (claim two devices as two owners,
|
||||||
|
release one owner, verify exactly its claims freed).
|
||||||
|
|
||||||
|
## M17.2 — exit reasons
|
||||||
|
|
||||||
|
- `system/abi.zig`: `ExitReason` (exited, aborted, segmentation_fault,
|
||||||
|
illegal_instruction, arithmetic_fault, killed).
|
||||||
|
- Kernel: record the reason at every death site — clean exit path, each fault
|
||||||
|
class in `onException`, the kill path. Bounded recent-exits table (ids are never
|
||||||
|
reused, so a small ring keyed by id is enough).
|
||||||
|
- New system call `process_exit_reason(id)` — supervisor-gated, like kill; returns
|
||||||
|
the recorded reason or `-ESRCH` once evicted.
|
||||||
|
- `library/runtime/process.zig`: `ExitReason` + `exitReason(id: u32)`.
|
||||||
|
- Docs: remove the no-exit-status bullet from process-management.md.
|
||||||
|
|
||||||
|
**Test:** extend the `supervision` scenario — three children: one exits cleanly,
|
||||||
|
one faults (the fault-recovery pattern), one is killed; the supervisor asserts all
|
||||||
|
three reasons.
|
||||||
|
|
||||||
|
## M17.3 — published exit events
|
||||||
|
|
||||||
|
- Kernel: bounded subscriber table (endpoints); new system call
|
||||||
|
`process_subscribe(endpoint)` (ungated, like `process_enumerate`); every death
|
||||||
|
posts `notify_exit_bit | id` to each subscriber — the same post the supervisor
|
||||||
|
path already uses.
|
||||||
|
- `library/runtime/process.zig`: `subscribeExits(endpoint)`.
|
||||||
|
- VFS becomes the first subscriber: on an exit event, release every handle keyed
|
||||||
|
by that task id (badges already are task ids). Log the release.
|
||||||
|
- Docs: note the convention in ipc.md (exit events reuse the exit-notification
|
||||||
|
badge encoding).
|
||||||
|
|
||||||
|
**Test:** new QEMU scenario `vfs-client-death` — a client opens a file and is
|
||||||
|
killed without closing; assert the VFS logs the handle release and its open-handle
|
||||||
|
count returns to baseline.
|
||||||
|
|
||||||
|
## M17.4 — signals and the service harness
|
||||||
|
|
||||||
|
- Kernel: per-task pending mask + bound endpoint; system calls
|
||||||
|
`signal_bind(endpoint)` and `process_signal(id, signal)` (supervisor-or-self
|
||||||
|
gated); delivery posts `notify_signal_bit | pending mask`, coalescing; pending
|
||||||
|
signals with no bound endpoint pend silently.
|
||||||
|
- `library/runtime/process.zig`: `Signal`, `SignalSet`, `bindSignals`,
|
||||||
|
`signalsFrom`, `sendSignal`, `stop(id, deadline_ms)` (terminate → wait for exit
|
||||||
|
notification → kill). Implement `terminate`, `reload`, `user_1`, `user_2`;
|
||||||
|
`interrupt`/`quit` are enum members with no sender yet; `alarm` stays unbuilt.
|
||||||
|
- Kernel: **one-shot timer notifications** — `timer_bind(endpoint, ms)` posts a
|
||||||
|
notification badge when the deadline lands (IRQ-as-IPC again, on the timer
|
||||||
|
wheel `sleep` already uses). This is the missing timed-wait primitive:
|
||||||
|
`replyWait` blocks forever and `sleep` blocks the whole process, but `stop()`'s
|
||||||
|
escalation, the device manager's `hello` deadline (M18.1), and restart backoff
|
||||||
|
all need a deadline while staying responsive. It is also the mechanism `alarm`
|
||||||
|
gets for free later.
|
||||||
|
- New `library/runtime/service.zig`: the harness — `run(callbacks)` owning the
|
||||||
|
replyWait loop, folding protocol messages, signals, and child-exit notifications
|
||||||
|
into `init` / `on_message` / `on_reload` / `on_terminate`; answers the common
|
||||||
|
`ping` automatically. Define the reserved `ping` request encoding here and
|
||||||
|
document it in ipc.md (one obvious encoding; smallest that cannot collide with
|
||||||
|
existing protocols).
|
||||||
|
- Convert one existing service (input-source or hpet) to the harness as proof it
|
||||||
|
subtracts code rather than adding it.
|
||||||
|
|
||||||
|
**Test:** extend `supervision` — a harness-built child: `sendSignal(reload)`
|
||||||
|
observed in its log, `ping` answered, `stop()` produces a clean exit with reason
|
||||||
|
`exited`; a second child that ignores signals (no bind) is killed by `stop()`'s
|
||||||
|
deadline with reason `killed`.
|
||||||
|
|
||||||
|
## M18.1 — device-manager protocol: hello + restart policy
|
||||||
|
|
||||||
|
- New `system/services/device-manager/device-manager-protocol.zig` module
|
||||||
|
(vfs-protocol pattern): `hello { version, role, device_id }`; version constant;
|
||||||
|
reserved fields.
|
||||||
|
- Device manager: register the `.device_manager` endpoint; spawn drivers with its
|
||||||
|
exit endpoint; enforce the hello deadline; restart policy — backoff, crash-loop
|
||||||
|
cap (three fast deaths → mark failed, log, stop), reasons from M17.2 deciding
|
||||||
|
restart vs not.
|
||||||
|
- usb-xhci-bus: adopt the harness + send hello. hpet/ps2-bus follow only if the
|
||||||
|
conversion is mechanical; otherwise they keep working unconverted (the manager
|
||||||
|
only enforces hello on drivers spawned with an assignment).
|
||||||
|
- build.zig: test-loop entry for the protocol module if it grows pure logic.
|
||||||
|
|
||||||
|
**Test:** new QEMU scenario `driver-restart` — the xHCI driver takes a test-only
|
||||||
|
argv flag to fault after hello on its first run; assert: fault, exit reason
|
||||||
|
recorded, manager respawns with backoff, second run claims the controller
|
||||||
|
(M17.1) and hellos clean. Assert the crash-loop cap by a driver that always
|
||||||
|
faults (a tiny test driver, not xhci).
|
||||||
|
|
||||||
|
## M18.2 — bus tree reports
|
||||||
|
|
||||||
|
- Protocol: `child_added { parent, identity, resources }` / `child_removed { id }`.
|
||||||
|
- usb-xhci-bus: bring-up to **port scan only** — map the MMIO window (claimed in
|
||||||
|
M16-era work), controller reset/start per xHCI spec, walk the port registers,
|
||||||
|
report one `child_added` per connected port with speed + port number as
|
||||||
|
identity. **No transfer rings, no descriptors** — reading device/interface
|
||||||
|
descriptors (and therefore USB class triples for matching) is the follow-on USB
|
||||||
|
track, not this plan.
|
||||||
|
- Device manager: mirror reports into its tree; prune the subtree (emitting
|
||||||
|
`child_removed`) when a bus driver dies; assert re-report on restart.
|
||||||
|
|
||||||
|
**Test:** QEMU already attaches usb-kbd + usb-mouse on xhci.0 — assert two
|
||||||
|
`child_added` events reach the manager and appear in its tree dump; kill the
|
||||||
|
driver, assert two `child_removed` then two fresh `child_added` after respawn.
|
||||||
|
|
||||||
|
## M18.3 — the application surface
|
||||||
|
|
||||||
|
- Protocol: `enumerate` (tree snapshot) + `subscribe` (published add/remove
|
||||||
|
events, input-service pattern).
|
||||||
|
- A small client (`device-list`, the `ps` analog) exercising both; the manager
|
||||||
|
becomes the one answer to "what devices exist" for user space.
|
||||||
|
`device_enumerate` stays for drivers/kernel seeding — its retreat is tied to the
|
||||||
|
discovery migration, out of this plan.
|
||||||
|
|
||||||
|
**Test:** QEMU scenario — `device-list` shows the tree including USB children;
|
||||||
|
during a driver restart the subscribing client logs remove + add events.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
**Explicitly out of scope** (own tracks, after M18): discovery migration (pci-bus
|
||||||
|
driver, acpi service, retiring the kernel scan), USB control transfers +
|
||||||
|
descriptors + class-driver matching, the musl layer, `interrupt`/`quit` senders
|
||||||
|
(needs a console), job control.
|
||||||
@@ -0,0 +1,196 @@
|
|||||||
|
# M19–M20 execution plan: discovery migration
|
||||||
|
|
||||||
|
The operational plan for [device-manager.md](device-manager.md)'s increment 8:
|
||||||
|
discovery leaves the kernel — a **pci-bus driver** (M19) and an **acpi service**
|
||||||
|
(M20), with the kernel's device enumeration retired behind them. Same rules as
|
||||||
|
[m17-m18-plan.md](m17-m18-plan.md): one phase at a time, each green before the
|
||||||
|
next; this file is the build order and the checklist.
|
||||||
|
|
||||||
|
**Definition of green, every phase:** `zig build` clean, `zig build test` clean,
|
||||||
|
`python3 test/qemu_test.py` passes (existing scenarios plus the phase's new
|
||||||
|
one), and the relevant design doc updated. Commit per green phase (no co-author
|
||||||
|
trailers). The full suite is the regression net — the existing
|
||||||
|
`driver-restart` / `usb-report` / `device-list` / `input` scenarios must stay
|
||||||
|
green *through* the migration, which is the whole point: the system must not be
|
||||||
|
able to tell who enumerated it.
|
||||||
|
|
||||||
|
**Workflow:** dedicated worktree; branches off `main` — `feat/pci-bus`
|
||||||
|
(M19.0–19.3), `feat/acpi-service` (M20.1–20.3); auto-merge to main when a
|
||||||
|
branch is green; keep branches; push everything.
|
||||||
|
|
||||||
|
## Settled decisions (2026-07-13 — veto before the loop starts)
|
||||||
|
|
||||||
|
1. **What "retiring the kernel scan" means.** The kernel keeps, forever, the
|
||||||
|
parses it needs before user space exists: RSDP/XSDT location, MADT (SMP),
|
||||||
|
the HPET table (the tick), FADT + the AML `\_S5` evaluation (poweroff — the
|
||||||
|
power tests prove it), and MCFG (the host bridge node). What retires is
|
||||||
|
**device enumeration**: the ECAM function walk (M19.3) and the DSDT/SSDT
|
||||||
|
namespace walk that builds device nodes (M20.3). The AML module stays a
|
||||||
|
shared build module compiled into both the kernel (for `\_S5`) and the acpi
|
||||||
|
service (for everything else) — same source, two builds, no fork.
|
||||||
|
2. **Bridge apertures come from the firmware memory map, not AML.** Registered
|
||||||
|
PCI functions carry BAR resources, and containment demands the bridge own
|
||||||
|
windows that cover them. The apertures are derived kernel-side from the
|
||||||
|
boot memory map's MMIO holes (regions that are neither RAM nor tables) —
|
||||||
|
mechanical, AML-free, and available at boot regardless of what later moved
|
||||||
|
to user space. (The bridge today carries only ECAM + bus range; this is the
|
||||||
|
prerequisite M19.0 exists for.)
|
||||||
|
3. **`device_register` becomes idempotent on exact match.** A re-registration
|
||||||
|
with identical (parent, class, resources) returns the existing id instead
|
||||||
|
of appending. The kernel table has no unregister, so without this a
|
||||||
|
restarted registering bus would duplicate its children on every respawn —
|
||||||
|
idempotence makes restart-and-re-report safe for every future bus, not just
|
||||||
|
PCI.
|
||||||
|
4. **The manager matches from reports.** `ChildAdded` gains a `device_id`
|
||||||
|
field (the kernel-registered id, `no_device` for unregistered leaves like
|
||||||
|
USB ports). After the M19.3 flip, PCI driver matching keys off reported
|
||||||
|
identity (the class triple) instead of the manager's boot-time snapshot —
|
||||||
|
the snapshot match remains only for what the kernel still seeds. One flip
|
||||||
|
phase changes both sides at once so no device is ever matched twice.
|
||||||
|
5. **The acpi service's authority is one node.** The kernel publishes an
|
||||||
|
`acpi-tables` device: memory resources covering the table blobs plus a
|
||||||
|
broad `io_port` resource — the documented trust grant to exactly one
|
||||||
|
process (AML OperationRegions reach EC/PM ports; the claim-gated
|
||||||
|
io_read/io_write calls already exist). The service claims it, maps the
|
||||||
|
tables, and runs the shared AML module in ring 3 behind a `Hal` backed by
|
||||||
|
`mmio_map` + `io_read`/`io_write`.
|
||||||
|
6. **Both new processes are protocol drivers** under the manager: hello,
|
||||||
|
supervision, restart with backoff — all inherited from M18.1 for free.
|
||||||
|
Registration idempotence (decision 3) is what makes their restarts sound.
|
||||||
|
7. **Firmware neutrality is the contract** (2026-07-13). The generic layer is
|
||||||
|
everything at and above the device-manager protocol — descriptors,
|
||||||
|
containment, reports, matching, supervision — and none of it may become
|
||||||
|
x86-specific. Discovery is one swappable process per firmware: the acpi
|
||||||
|
service on x86; an **fdt service** on the Raspberry Pis (claims a
|
||||||
|
`devicetree-blob` node, reports children from the flattened device tree —
|
||||||
|
pure data, no bytecode, no port grant, strictly simpler than ACPI). The
|
||||||
|
manager owns the tree as *data* and touches no hardware, ever — AML runs in
|
||||||
|
a crashable, supervised discoverer precisely so a firmware-bytecode fault
|
||||||
|
can never take down the supervisor. Two consequences recorded now:
|
||||||
|
`DeviceDescriptor`'s 8-byte `hid` cannot hold an FDT `compatible` string
|
||||||
|
("brcm,bcm2835-aux-uart") — identity widens before the fdt service exists;
|
||||||
|
and cross-firmware surfaces are named by **domain, not firmware** (M21
|
||||||
|
defines a *power* protocol, not an "ACPI events" protocol — PSCI/mailbox
|
||||||
|
sources feed the same subscribers on ARM). **Landed early (2026-07-13):**
|
||||||
|
both services exist as placeholders (system/services/acpi, system/services/
|
||||||
|
fdt) and the build's `-Ddiscovery=acpi|fdt` option fills the ramdisk's
|
||||||
|
neutral `discovery` slot — the manager will spawn "discovery" by that name
|
||||||
|
in M20.3 and never learn which firmware it is on.
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
- [x] **M19.0** — prerequisites (bridge apertures from the memory map's
|
||||||
|
*gaps* — the single-hole rule died on OVMF's flash at the top of 4 GiB,
|
||||||
|
caught by the new every-BAR-contained assert in `discovery`; idempotent
|
||||||
|
`device_register` proven in `bus`; `ChildAdded.device_id`;
|
||||||
|
m17-m18-plan.md archived; suite 54/54).
|
||||||
|
- [x] **M19.1** — pci-bus driver, scan only (claims the bridge, maps ECAM
|
||||||
|
through its grant, brute-force walk with the multifunction rule; the
|
||||||
|
manager matches pci_host_bridge → pci-bus per device with the full
|
||||||
|
protocol contract; `pci-scan` builds its expected marker from the
|
||||||
|
kernel's own count — equivalence on the first run; suite 55/55).
|
||||||
|
- [x] **M19.2** — register + report (BAR probe mirrored byte-for-byte from the
|
||||||
|
kernel's addBars so dedupe returns the kernel's node ids during
|
||||||
|
coexistence; the bridge gained the io_port aperture I/O BARs need;
|
||||||
|
reports carry the registered device_id; pci-scan drills a forced restart
|
||||||
|
and asserts the PCI node count never grows — plus harness hardening: a
|
||||||
|
failing case now preserves its serial as <case>-failed-serial.log, and
|
||||||
|
the heavy scenarios run at 150s; suite 55/55).
|
||||||
|
- [x] **M19.3** — the flip: kernel `enumeratePci`/`addBars`/`PciHeader` all
|
||||||
|
deleted (bridge node stays); manager matches PCI drivers from reported
|
||||||
|
identity, deduped by registered id. Surfaced and fixed a real SMP race the
|
||||||
|
flip created — ring-3 device_register made the broker table concurrent, so
|
||||||
|
mmio_map's lock-free read intermittently tore hpet's resource length
|
||||||
|
(user fault) and overflowed `r.len-1` into a kernel panic; now the broker
|
||||||
|
read is under the big lock and the arithmetic is guarded, and pci-bus
|
||||||
|
skips size-0 BARs. discovery.md updated; suite 55/55 (driver-restart
|
||||||
|
hammered 6×).
|
||||||
|
- [x] **merge** `feat/pci-bus` → main, push (merged 2026-07-13).
|
||||||
|
- [x] **M20.1** — acpi service, parse only: the AML interpreter is now a build
|
||||||
|
module compiled into both kernel and service; the kernel publishes the
|
||||||
|
`acpi-tables` node (AML blobs as memory resources, the broad io_port grant,
|
||||||
|
the SCI); the service claims it, maps the blobs, runs the shared parser in
|
||||||
|
ring 3, and self-verifies its Device count against the kernel's (34 = 34,
|
||||||
|
deterministic via argv, no log-scraping); the manager spawns `discovery`
|
||||||
|
at startup. Parse-only touches no hardware. Suite 56/56.
|
||||||
|
|
||||||
|
- [x] **M20.2** — register + report: the service evaluates `_STA`/`_CRS` in
|
||||||
|
ring 3 (interpreter Hal = port I/O over the claimed node; a scratch page
|
||||||
|
backs SystemMemory maps so a stray region can't fault it) and registers +
|
||||||
|
reports each present `_HID` device under `acpi-tables`. Containment: the
|
||||||
|
broker's irq check became range-based (len-1 == the old equality) so the
|
||||||
|
node's broad irq window covers children's legacy lines; io ports fall in
|
||||||
|
the broad io grant. ChildAdded gained `hid`. Matching stays off. The
|
||||||
|
`acpi-report` scenario asserts the PS/2 keyboard (3 resources) and mouse
|
||||||
|
(1 resource) among the reports. Suite 57/57.
|
||||||
|
- [x] **M20.3** — the flip: the kernel's `wireAcpiDevices` call is gone (the
|
||||||
|
device-building helpers are retained-but-dead pending a focused sweep,
|
||||||
|
spawned as a task; static tables + `\_S5` + the acpi-tables node stay).
|
||||||
|
The manager matches ps2-bus from ACPI `_HID` reports; the service
|
||||||
|
registers all devices before reporting any (no keyboard-before-mouse
|
||||||
|
race). The `acpi-ps2` scenario proves report → spawn → ps2-bus attaches
|
||||||
|
its keyboard; `ioport` retargeted to the acpi-tables I/O window (the
|
||||||
|
kernel-built PS/2 node is gone). Suite 58/58.
|
||||||
|
- [x] **merge** `feat/acpi-service` → main, push (merged 2026-07-13) — **discovery migration complete**.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Phase notes
|
||||||
|
|
||||||
|
**M19.0 apertures:** the boot memory map already crosses the handoff
|
||||||
|
([boot-handoff]), but discovery never sees it today — expect a small
|
||||||
|
pass-through (kernel init hands the map to the platform layer) before the
|
||||||
|
holes computation, which belongs where the bridge node is built
|
||||||
|
(`parseMcfg`). Sanity-check on QEMU q35: the xHCI BAR (`0xc0000000`-region
|
||||||
|
values seen in the M18 logs) must land inside a derived aperture, asserted in
|
||||||
|
the kernel unit test.
|
||||||
|
|
||||||
|
**M19.1 scanning without owning config access twice:** the driver reads config
|
||||||
|
space through its ECAM mmio_map grant of the *bridge* window — the same bytes
|
||||||
|
the kernel walk read. Vendor-id `0xFFFF` skip, header-type multifunction rule,
|
||||||
|
no bridge recursion (matches the kernel's current single-segment walk).
|
||||||
|
|
||||||
|
**M19.2 BAR sizing:** the classic size probe (write all-ones, read mask,
|
||||||
|
restore) is deferred — the BARs' current programmed values and types are
|
||||||
|
enough for containment-checked registration at bring-up; sizing lands with the
|
||||||
|
first driver that needs to *move* a BAR. Log what is registered so the
|
||||||
|
scenario can assert it.
|
||||||
|
|
||||||
|
**M19.3 what the manager still seeds from the snapshot:** everything the
|
||||||
|
kernel still enumerates (timers, ACPI nodes until M20.3). The PCI arm of
|
||||||
|
`pciDriverFor` switches source; `driverFor` doesn't move until M20.3.
|
||||||
|
|
||||||
|
**M20.1 spawn and identity (pre-settled 2026-07-13):** the manager spawns
|
||||||
|
`discovery` by its neutral ramdisk name at startup, as an ordinary protocol
|
||||||
|
driver (hello, supervision) — from M20.1 on, on every boot. For reporting ACPI
|
||||||
|
devices, `ChildAdded` gains `hid: [8]u8` (EISA ids fit; zero = none):
|
||||||
|
firmware *string* identity travels beside the numeric `identity` field until
|
||||||
|
the FDT-driven widening replaces both (decision 7).
|
||||||
|
|
||||||
|
**M20.1 Hal in ring 3:** `mapMmio` → `device.mmioMap` over the claimed
|
||||||
|
acpi-tables node (plus a table-offset map for blobs); `pioRead`/`pioWrite` →
|
||||||
|
`device.ioRead`/`ioWrite` against its io_port resource. The interpreter cannot
|
||||||
|
tell it moved — that is the assertion of `acpi-parse`.
|
||||||
|
|
||||||
|
**M20.2 containment for `_CRS`:** io ports fall inside the node's broad
|
||||||
|
io_port resource; MMIO windows (HPET, LAPIC ranges some firmwares list) fall
|
||||||
|
inside the memory-map holes added to the node in M20.1. Anything that doesn't
|
||||||
|
fit is logged and skipped, loudly — bring-up honesty over silent drops.
|
||||||
|
|
||||||
|
**M20.3 ps2 ordering:** ps2-bus binds nodes the acpi service now reports, so
|
||||||
|
its spawn moves behind the report (the manager's matching handles this once
|
||||||
|
the source flips); the `input` scenario proves the keyboard still types.
|
||||||
|
|
||||||
|
**Explicitly out of scope:** PCI bridge recursion (single segment, flat bus
|
||||||
|
walk stays); BAR reprogramming/sizing; disk/PCIe hotplug; interrupt routing
|
||||||
|
changes (`_PRT` stays wherever it is today); the USB descriptor track;
|
||||||
|
multi-segment ECAM; per-device power states (D-states, `_PSx`/`_PRx`,
|
||||||
|
suspend/resume — a future *lifecycle-vocabulary* extension, since "suspend"
|
||||||
|
has the shape of a signal every driver must answer, and it has no consumer
|
||||||
|
until laptop sleep); CPU P/C-states.
|
||||||
|
|
||||||
|
## M21 — ACPI events + system power — DONE
|
||||||
|
|
||||||
|
Built and merged (docs/m21-plan.md, 2026-07-13): the SCI + power button, Notify/GPE
|
||||||
|
dispatch, and orderly shutdown (init's stop cascade into a ring-3 S5 write).
|
||||||
|
See that plan for the phase record.
|
||||||
@@ -0,0 +1,139 @@
|
|||||||
|
# M21 execution plan: ACPI events + system power
|
||||||
|
|
||||||
|
The operational plan for the event side of the acpi service and orderly
|
||||||
|
shutdown — the capstone [m19-m20-plan.md](m19-m20-plan.md) previewed. Same
|
||||||
|
rules as its predecessors: one phase at a time, each green before the next;
|
||||||
|
this file is the build order and the checklist.
|
||||||
|
|
||||||
|
**Definition of green, every phase:** `zig build` clean, `zig build test`
|
||||||
|
clean, `python3 test/qemu_test.py` passes (existing scenarios plus the
|
||||||
|
phase's new one), and the relevant design doc updated. Commit per green phase
|
||||||
|
(no co-author trailers). Failing cases preserve their serial logs
|
||||||
|
(`<case>-failed-serial.log`).
|
||||||
|
|
||||||
|
**Workflow:** dedicated worktree; branch `feat/power-events` off `main`;
|
||||||
|
auto-merge to main when the branch is green; keep the branch; push everything.
|
||||||
|
|
||||||
|
## Settled decisions (2026-07-13, approved)
|
||||||
|
|
||||||
|
1. **S5 is executed by the acpi service from ring 3.** No new syscall: the
|
||||||
|
broad port grant (M20 decision 5) already made this physically possible —
|
||||||
|
the service holds the PM1 control ports in its io grant and derives `_S5`
|
||||||
|
from its own namespace (`aml.sleepState`). Formalizing it adds no
|
||||||
|
authority. The kernel keeps `power.zig` for its own test paths and
|
||||||
|
panic-time use.
|
||||||
|
2. **The power surface is domain-named** (decision 7 of the last plan): a
|
||||||
|
`power-protocol` module + `ServiceId.power = 5`, registered by the acpi
|
||||||
|
service — on ARM, a PSCI/mailbox service registers the same id and
|
||||||
|
subscribers never know the difference. Messages: `subscribe` (endpoint as
|
||||||
|
the call's capability, the input/manager pattern), `shutdown` (accepted
|
||||||
|
only from PID 1 — init), and events published as buffered messages:
|
||||||
|
`power_button`, `lid`, `ac`, `battery`, generic `notify` with a code.
|
||||||
|
3. **The service learns event ports from its own FADT copy**: the kernel adds
|
||||||
|
the FADT as one more memory resource on the acpi-tables node; the service
|
||||||
|
tells it apart from the AML blobs by signature ("FACP" header — the blob
|
||||||
|
resources are header-stripped bytecode and start with no signature). The
|
||||||
|
kernel's own FADT parse is untouched.
|
||||||
|
4. **The acpi service converts to the harness** (`runtime.service.run`):
|
||||||
|
protocol messages (subscribe/shutdown), the SCI notification, and the
|
||||||
|
existing report flow fold into one loop — the shape it was always meant
|
||||||
|
to have.
|
||||||
|
5. **GPE/Notify correctness is proven by host unit tests** (synthetic AML
|
||||||
|
with a Notify inside a method body; aml.zig joins the `zig build test`
|
||||||
|
loop). The QEMU scenario proves the power button — a *fixed* event,
|
||||||
|
deterministically injectable via QMP `system_powerdown` — because QEMU
|
||||||
|
cannot raise GPEs deterministically on this config. Battery/AC/lid and the
|
||||||
|
embedded controller (`_Qxx`) are interface-complete here and validated on
|
||||||
|
real hardware (the laptop) later.
|
||||||
|
|
||||||
|
## Ground truth the phases build on (verified 2026-07-13)
|
||||||
|
|
||||||
|
- `system/devices/power.zig` `shutdown()` is the kernel's S5 write
|
||||||
|
(SLP_TYP|SLP_EN to PM1a/PM1b control); there is no power syscall.
|
||||||
|
- init (`system/services/init/init.zig`) spawns vfs/input/device-manager
|
||||||
|
fire-and-forget — no child ids kept, no signals, no event loop. The whole
|
||||||
|
stop toolkit exists in `runtime.process` (stop/sendSignal/bindSignals).
|
||||||
|
- `test/qemu_test.py` has no QMP channel (serial is a one-way file).
|
||||||
|
- The kernel parses PM1 *control* blocks and SCI_INT from the FADT; the PM1
|
||||||
|
**event** blocks (offsets 56/60, len at 88) and **GPE0/GPE1** blocks
|
||||||
|
(offsets 80/84, lens 92/93) are unparsed — the service reads them from its
|
||||||
|
FADT copy (decision 3).
|
||||||
|
- The acpi-tables node carries the SCI as its only `len == 1` irq resource
|
||||||
|
(the broad window is len 256) — that is how the service finds it to
|
||||||
|
`irqBind`.
|
||||||
|
- `notify_opcode = 0x86` exists in `system/devices/aml/opcodes.zig` but the
|
||||||
|
interpreter never handles it — a GPE `_Lxx` body containing Notify fails
|
||||||
|
evaluation today. Everything else a GPE handler needs (field access,
|
||||||
|
control flow, method calls) is proven by the ring-3 `_STA`/`_CRS` work.
|
||||||
|
- The dead-code sweep (spawned task) also edits `system/devices/acpi.zig`;
|
||||||
|
M21.0 checks whether it landed and rebases before touching that file.
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
- [x] **M21.0** — baseline (dead-code sweep confirmed landed on main — no
|
||||||
|
acpi.zig conflict; `feat/power-events` cut; QMP channel in the harness:
|
||||||
|
always-on unix socket, client with the capabilities handshake, per-case
|
||||||
|
`qmp_after` hook, and a hook-must-deliver pass gate that the smoke case
|
||||||
|
now proves with a harmless query-status; suite 58/58).
|
||||||
|
- [x] **M21.1** — SCI + the power button (kernel appends the FADT as an
|
||||||
|
acpi-tables memory resource, tagged by its "FACP" header; `power-protocol`
|
||||||
|
module + `ServiceId.power = 5`; the acpi service converted to
|
||||||
|
`runtime.service.run`, registers `.power`, reads PM1 event/control + GPE
|
||||||
|
ports from its FADT copy, enables ACPI mode if SCI_EN is clear, binds the
|
||||||
|
SCI (the len-1 irq), sets PWRBTN_EN; the SCI handler clears PM1_STS,
|
||||||
|
logs `power: button pressed`, publishes `power_button`, acks. Scenario
|
||||||
|
`power-button` injects a real `system_powerdown` via QMP; initial-ramdisk
|
||||||
|
timeout 30→60s for the service's added boot work; suite 59/59).
|
||||||
|
- [x] **M21.2** — Notify + GPE dispatch (interpreter handles `notify_opcode`
|
||||||
|
into a bounded queue, cleared per-evaluate, drained via
|
||||||
|
`takeNotifications`; the service walks GPE status/enable bytes, evaluates
|
||||||
|
`\_GPE._Lxx`/`_Exx` per active bit, maps notified nodes to events
|
||||||
|
(battery/ac/lid/generic), clears GPE_STS write-1, acks. EC `_Qxx` out.
|
||||||
|
Host unit test with hand-encoded AML proves the queue; aml.zig joined the
|
||||||
|
`zig build test` loop. QEMU raises no GPEs — suite is regression net,
|
||||||
|
59/59).
|
||||||
|
- [x] **M21.3** — orderly shutdown (init supervises its children on one
|
||||||
|
endpoint that also carries signals, power events, and a re-arming
|
||||||
|
heartbeat timer; on `power_button` or a `terminate` signal it logs
|
||||||
|
`init: shutting down`, runs `stop(child, 2000, endpoint)` in reverse
|
||||||
|
order, then requests `.power` shutdown; the acpi service honors shutdown
|
||||||
|
from a subscriber — init is the one subscriber, a soft gate that survives
|
||||||
|
testing where PID 1 isn't init — and writes SLP_TYP|SLP_EN from ring 3.
|
||||||
|
`orderly-shutdown` scenario proves button → shutting-down → S5 → QEMU
|
||||||
|
exit; suite 60/60).
|
||||||
|
- [ ] **merge** `feat/power-events` → main, push, keep the branch — **loop
|
||||||
|
ends here**.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Phase notes
|
||||||
|
|
||||||
|
**M21.0 QMP:** open the unix socket after Popen, complete the
|
||||||
|
`qmp_capabilities` handshake, then send the hook's command (for these
|
||||||
|
scenarios: `{"execute": "system_powerdown"}`). The socket is additive — no
|
||||||
|
existing case may notice it. Note e3fe3f3 recently reworked how the harness
|
||||||
|
boots; adapt to its current shape rather than the pre-rework description.
|
||||||
|
|
||||||
|
**M21.1 SCI details:** PM1_STS is at the event block base (write-1-to-clear);
|
||||||
|
PM1_EN at base + block_len/2; PWRBTN bit is 8 in both. If PM1b exists, mirror
|
||||||
|
reads/writes to both blocks. Enable ACPI mode only when SCI_EN (PM1 control
|
||||||
|
bit 0) is clear — OVMF boots may already have it set. The publish path reuses
|
||||||
|
the manager's subscriber table pattern (bounded, drop-on-failed-send).
|
||||||
|
|
||||||
|
**M21.2 GPE walk:** GPE0_STS bytes live at the GPE0 block base, GPE0_EN in
|
||||||
|
the block's upper half; for a set+enabled bit n, the handler method is
|
||||||
|
`_L%02X` (level) or `_E%02X` (edge) under `\_GPE`. Evaluate, drain the notify
|
||||||
|
queue, clear the status bit, ack. A missing handler method is clear-and-log,
|
||||||
|
not an error.
|
||||||
|
|
||||||
|
**M21.3 ordering:** init subscribes with retries — the acpi service registers
|
||||||
|
`.power` well after init starts. The stop sequence runs vfs last (other
|
||||||
|
services may flush through it). The S5 write mirrors `power.zig`'s
|
||||||
|
`sleepValue` (SLP_TYP bits [12:10], SLP_EN bit 13); if the write returns, log
|
||||||
|
`power: S5 write did not take` so the scenario fails loudly instead of
|
||||||
|
hanging.
|
||||||
|
|
||||||
|
**Explicitly out of scope:** the embedded controller and `_Qxx` queries,
|
||||||
|
battery `_BST`/`_BIF` evaluation beyond the interface stubs, lid/AC on QEMU
|
||||||
|
(no emulation), reboot over the power protocol, S3 sleep, per-device D-states
|
||||||
|
(a future lifecycle-vocabulary extension), thermal zones.
|
||||||
+3
-3
@@ -27,7 +27,7 @@ danos's own neutral format, and the kernel only ever sees that.**
|
|||||||
|
|
||||||
## The neutral format
|
## The neutral format
|
||||||
|
|
||||||
Defined in `src/root.zig`, the shared loader↔kernel contract:
|
Defined in `system/boot-handoff.zig`, the shared loader↔kernel contract:
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
pub const MemoryKind = enum(u32) {
|
pub const MemoryKind = enum(u32) {
|
||||||
@@ -68,7 +68,7 @@ pub const BootInfo = extern struct {
|
|||||||
|
|
||||||
## The loader side (UEFI)
|
## The loader side (UEFI)
|
||||||
|
|
||||||
Two functions in `src/boot/efi.zig`, called from `exitBootServices`:
|
Two functions in `boot/efi.zig`, called from `exitBootServices`:
|
||||||
|
|
||||||
- **`classify`** maps each UEFI descriptor to a `MemoryKind`:
|
- **`classify`** maps each UEFI descriptor to a `MemoryKind`:
|
||||||
`conventional_memory` **and** `boot_services_code`/`boot_services_data → usable`;
|
`conventional_memory` **and** `boot_services_code`/`boot_services_data → usable`;
|
||||||
@@ -121,7 +121,7 @@ The kernel receives a plain array and reads it with zero UEFI knowledge:
|
|||||||
|
|
||||||
```zig
|
```zig
|
||||||
const mm = boot_info.memory_map;
|
const mm = boot_info.memory_map;
|
||||||
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
const regions = @as([*]const system.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
||||||
for (regions) |r| {
|
for (regions) |r| {
|
||||||
if (r.kind == .usable) usable_pages += r.pages;
|
if (r.kind == .usable) usable_pages += r.pages;
|
||||||
}
|
}
|
||||||
|
|||||||
+49
-14
@@ -7,7 +7,7 @@ which live in memory we'd like to reclaim and don't control), switches CR3 onto
|
|||||||
them, and — crucially — maps with **real permissions**.
|
them, and — crucially — maps with **real permissions**.
|
||||||
|
|
||||||
It's x86_64-specific (the 4-level table format is an Intel/AMD thing), so it lives
|
It's x86_64-specific (the 4-level table format is an Intel/AMD thing), so it lives
|
||||||
behind the [arch](arch.md) boundary in `src/kernel/arch/x86_64/paging.zig`.
|
behind the [arch](arch.md) boundary in `system/kernel/architecture/x86_64/paging.zig`.
|
||||||
|
|
||||||
## The format
|
## The format
|
||||||
|
|
||||||
@@ -17,20 +17,54 @@ the final 4 KiB page. Each entry holds a physical address plus flag bits —
|
|||||||
present, writable, and (bit 63) **no-execute**. danos maps everything with 4 KiB
|
present, writable, and (bit 63) **no-execute**. danos maps everything with 4 KiB
|
||||||
pages: precise, and the extra table memory is negligible against available RAM.
|
pages: precise, and the extra table memory is negligible against available RAM.
|
||||||
|
|
||||||
|
## Higher half: the address-space layout
|
||||||
|
|
||||||
|
danos is a **higher-half kernel**. The kernel is linked to run at
|
||||||
|
`0xFFFF_FFFF_8000_0000` but loaded low (the linker script's `AT()` gives each
|
||||||
|
segment a physical load address at 1 MiB up; the bootloader maps the high link
|
||||||
|
address to the low load address in its bootstrap tables and jumps in). The entire
|
||||||
|
**low canonical half is reserved for user space**; the kernel lives in the top half
|
||||||
|
alongside a **physmap** — a straight window onto all of physical memory at
|
||||||
|
`physmap_base + phys`. Wherever the kernel needs to touch a physical address (a
|
||||||
|
page-table frame, an ACPI table, a device register), it adds that constant:
|
||||||
|
`system.physToVirt(phys)`. The layout constants live in `system/boot-handoff.zig`:
|
||||||
|
|
||||||
|
| region | virtual base | PML4 slot |
|
||||||
|
|--------|--------------|-----------|
|
||||||
|
| user image + stack | `0x0000_7000_0000_0000` | 224 (low half) |
|
||||||
|
| kernel heap | `0xFFFF_8000_0000_0000` | 256 |
|
||||||
|
| physmap (all RAM + MMIO windows) | `0xFFFF_8800_0000_0000` + phys | 272 |
|
||||||
|
| kernel image | `0xFFFF_FFFF_8000_0000` | 511 |
|
||||||
|
|
||||||
|
The bootloader builds temporary **bootstrap tables** (identity + a 4 GiB physmap +
|
||||||
|
the high kernel) so it can switch CR3 and jump to the high entry; the kernel then
|
||||||
|
builds its own precise tables below and abandons them. Because both use the same
|
||||||
|
`physmap_base`, any physmap pointer minted before the switch stays valid after it.
|
||||||
|
|
||||||
## What gets mapped, and with what permissions
|
## What gets mapped, and with what permissions
|
||||||
|
|
||||||
The address space is built in three passes (`init`):
|
The address space is built in four passes (`init`):
|
||||||
|
|
||||||
1. **All RAM, identity-mapped RW + NX.** Every non-MMIO region from the
|
1. **All RAM in the physmap, RW + NX.** Every non-MMIO region from the
|
||||||
[memory map](memory-map.md) is mapped virtual == physical, read-write and
|
[memory map](memory-map.md) is mapped at `physToVirt(phys)`, read-write and
|
||||||
*non-executable*. Identity mapping keeps everything already running valid across
|
*non-executable*. There is **no low/identity mapping** — the low half is user
|
||||||
the CR3 switch (the frame allocator addresses frames by physical address, page
|
space. (Frames the kernel touches while still building these tables are reached
|
||||||
tables are reached the same way, the stack stays put).
|
through the loader's bootstrap physmap, which covers the low 4 GiB; both the
|
||||||
2. **The framebuffer and the Local APIC**, the device memory we actually touch,
|
frame allocator and the table builder scan low-address-up, so those frames stay
|
||||||
also RW + NX. Everything else — unbacked address space, other MMIO — is simply
|
under that limit.)
|
||||||
left unmapped, so a stray access faults instead of silently succeeding.
|
2. **The framebuffer and the Local APIC**, the device memory the kernel touches
|
||||||
3. **The kernel's own segments, overlaid with their true ELF permissions.** This is
|
directly, as physmap windows (RW + NX). Other MMIO is mapped on demand by
|
||||||
|
`mapMmio`, also into the physmap; everything else is left unmapped, so a stray
|
||||||
|
access faults instead of silently succeeding.
|
||||||
|
3. **The kernel's own segments, overlaid with their true ELF permissions**, at
|
||||||
|
their high link addresses mapped to their low physical load addresses. This is
|
||||||
the interesting part.
|
the interesting part.
|
||||||
|
4. **Every higher-half PML4 entry pre-created** (an empty PDPT where none exists
|
||||||
|
yet). The kernel half is then a fixed set of top-level slots, so a per-process
|
||||||
|
address space can share it by copying `PML4[256..512)` once — growth beneath
|
||||||
|
those slots (heap, on-demand MMIO) propagates to every address space because
|
||||||
|
they share the PDPTs. `init` asserts no new higher-half PML4 entry appears
|
||||||
|
afterward.
|
||||||
|
|
||||||
### W^X from the ELF program headers
|
### W^X from the ELF program headers
|
||||||
|
|
||||||
@@ -55,9 +89,10 @@ reserved bit and fault.
|
|||||||
|
|
||||||
### The null guard
|
### The null guard
|
||||||
|
|
||||||
Page 0 is deliberately left unmapped. A null (or near-null) pointer dereference now
|
The whole low half is unmapped except for explicit user mappings, so page 0 (and
|
||||||
takes a page fault instead of quietly reading or writing real memory — turning a
|
every near-null address) is unmapped by construction. A null (or near-null) pointer
|
||||||
whole class of silent bugs into an immediate, located crash.
|
dereference in the kernel takes a page fault instead of quietly reading or writing
|
||||||
|
real memory — turning a whole class of silent bugs into an immediate, located crash.
|
||||||
|
|
||||||
## Switching on, and the on-demand API
|
## Switching on, and the on-demand API
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,327 @@
|
|||||||
|
# Process lifecycle: signals over IPC
|
||||||
|
|
||||||
|
**Status: increments 1–4 built** (2026-07-12): claim release on death, exit
|
||||||
|
reasons, published exit events, and signals + one-shot timers + the service
|
||||||
|
harness are all in — the interface below is as-built. The primitives underneath
|
||||||
|
predate this design ([process-management.md](process-management.md):
|
||||||
|
spawn, the supervision link, kill, child-exit notifications); this document designs
|
||||||
|
the layer above them — the standard vocabulary a danos process speaks about its own
|
||||||
|
life, and the stable `runtime.process` interface that carries it. Nothing here is
|
||||||
|
device- or driver-specific: a driver, the VFS, and a user application all stop,
|
||||||
|
reload, and die the same way. The device manager is simply this design's first
|
||||||
|
serious customer ([device-manager.md](device-manager.md)).
|
||||||
|
|
||||||
|
**"POSIX" in this document means the concepts, never the letter of the standard.**
|
||||||
|
danos borrows the ideas and the hard-won lessons (what SIGTERM *means*, why SIGPIPE
|
||||||
|
was a mistake) without inheriting the mechanism, the API, or the names. The naming
|
||||||
|
rule is danos's own and it is strict: plain words that communicate intent
|
||||||
|
(`terminate`, `reload`, `exited`) and the IPC vocabulary the system already speaks
|
||||||
|
(`bind`, `subscribe`, `publish`, `endpoint`) — never `SIG*`, never a second word for
|
||||||
|
a concept that already has one. Literal POSIX arrives later and lives elsewhere: a
|
||||||
|
**musl-based C layer** (growing out of library/posix) that wires C programs to the
|
||||||
|
danos runtime — musl's syscall surface retargeted at danos system calls and IPC
|
||||||
|
protocols (files onto the VFS protocol, `sigaction`/`wait` onto this lifecycle,
|
||||||
|
sockets onto whatever networking becomes). Ported programs see POSIX; the system
|
||||||
|
underneath never does.
|
||||||
|
|
||||||
|
## Why a standard vocabulary
|
||||||
|
|
||||||
|
A supervisor can only manage processes it has never heard of if "please exit" means
|
||||||
|
the same thing to all of them. That is the one thing POSIX signals got deeply right:
|
||||||
|
`SIGTERM` means the same thing to nginx and to a five-line script, which is why
|
||||||
|
process supervision on Unix (init systems, container runtimes) is possible at all.
|
||||||
|
danos wants that property from day one, because supervision-and-restart is the
|
||||||
|
system's core motivation ([resilience.md](resilience.md)).
|
||||||
|
|
||||||
|
What POSIX got wrong — for a system like this — is the **delivery mechanism**:
|
||||||
|
asynchronous control-flow hijack. A Unix handler runs on a stolen stack at an
|
||||||
|
arbitrary instruction boundary, which is why the async-signal-safe function list
|
||||||
|
exists, why `errno` must be saved, and why the canonical signal bug is a SIGTERM
|
||||||
|
handler innocently calling `printf` mid-`malloc`. That entire bug class comes from
|
||||||
|
the mechanism, not the vocabulary, and none of it is worth importing.
|
||||||
|
|
||||||
|
A microkernel already has the right channel: **a signal is a message.** QNX delivers
|
||||||
|
POSIX signals over its message passing; seL4 has notification objects; Erlang turned
|
||||||
|
"death is a message to whoever linked" into a reliability philosophy. danos has
|
||||||
|
already done it once without naming it: a child's death arrives as a notification
|
||||||
|
badge on the supervisor's endpoint — the microkernel's SIGCHLD, the IRQ-as-IPC
|
||||||
|
pattern reused. Signals are the same pattern reused a third time.
|
||||||
|
|
||||||
|
## The mechanism
|
||||||
|
|
||||||
|
- **`signal_bind(endpoint)`** — a process nominates the endpoint its signals arrive
|
||||||
|
on, exactly as `irq_bind` nominates where a device's interrupts land. The runtime
|
||||||
|
does this at startup for any program that opts in.
|
||||||
|
- **`process_signal(id, signal)`** — posts the signal as an asynchronous
|
||||||
|
notification to the target's bound endpoint: badge = `notify_badge_bit |
|
||||||
|
notify_signal_bit | pending signals`. Non-blocking for the sender, always.
|
||||||
|
- **Pending signals coalesce** in a per-process bitmask until the target next waits
|
||||||
|
— exactly like interrupt notifications, and exactly POSIX's own semantics for
|
||||||
|
non-realtime signals (two pending SIGTERMs are one SIGTERM). The bitmask *is* the
|
||||||
|
design: signals carry no payload. Anything with a payload is a protocol message.
|
||||||
|
- **Authority**: the supervisor may signal its children — the same link that is
|
||||||
|
already the kill authority. A process may signal itself. Anything broader waits
|
||||||
|
for transferable process handles.
|
||||||
|
- **No binding, no problem**: a process that never calls `signal_bind` is not
|
||||||
|
broken — its signals pend unread and only `process_kill` works on it. Simple
|
||||||
|
programs stay simple; the vocabulary is opt-in, the kill authority is not.
|
||||||
|
|
||||||
|
Because delivery is a message into the process's own event loop, there is no
|
||||||
|
async-signal-safe list in danos: a handler is ordinary code running at a point the
|
||||||
|
process chose. The bug class is gone by construction, not by discipline.
|
||||||
|
|
||||||
|
## The vocabulary: POSIX.1-1990, sorted honestly
|
||||||
|
|
||||||
|
The full 1990 set, and what each becomes. Two intrinsically problematic cases get a
|
||||||
|
defense below the table.
|
||||||
|
|
||||||
|
| POSIX.1-1990 | danos disposition | Notes |
|
||||||
|
|---|---|---|
|
||||||
|
| SIGTERM | signal `terminate` | finish up and exit; the supervisor's polite half |
|
||||||
|
| SIGHUP | signal `reload` | re-read configuration / re-scan |
|
||||||
|
| SIGINT | signal `interrupt` | interactive interrupt; meaningful once a console can send it, in the vocabulary now so numbering is stable |
|
||||||
|
| SIGQUIT | signal `quit` | as SIGINT, without the core-dump baggage |
|
||||||
|
| SIGALRM | signal `alarm` | timer expiry as a message; the Unix SIGALRM+`longjmp` timeout hacks are impossible here. In the vocabulary, unbuilt: no consumer yet, and when one appears it is runtime sugar over the existing timer — zero kernel work |
|
||||||
|
| SIGUSR1, SIGUSR2 | signals `user_1`, `user_2` | service-defined |
|
||||||
|
| SIGCHLD | **already exists** — the exit notification | the badge carries the child id, dodging the classic coalescing bug (Unix code must loop `waitpid`) |
|
||||||
|
| SIGKILL | `process_kill` — kernel mechanism | its definition is "cannot be handled"; it was never really a signal |
|
||||||
|
| SIGABRT | exit reason `abort` | `abort()` is synchronous self-termination, not an event |
|
||||||
|
| SIGSEGV, SIGILL, SIGFPE | exit reasons, **never delivered** | see below |
|
||||||
|
| SIGPIPE | **an error return**, not a signal | see below |
|
||||||
|
| SIGSTOP, SIGTSTP, SIGTTIN, SIGTTOU, SIGCONT | deferred | job control needs terminals, sessions, and process groups; stop/continue is scheduler territory |
|
||||||
|
|
||||||
|
**The fault signals (SIGSEGV, SIGILL, SIGFPE) are intrinsically wrong for messages.**
|
||||||
|
They are *synchronous* — raised at a specific faulting instruction, not "sometime
|
||||||
|
soon". A message cannot be delivered to a process whose next instruction re-faults;
|
||||||
|
it never reaches its event loop to read it. POSIX only makes fault handlers "work"
|
||||||
|
via the async hijack (run the handler *instead of* the instruction), and even there,
|
||||||
|
returning from a SIGSEGV handler without curing the cause is undefined behavior.
|
||||||
|
danos's architecture already has the better answer: fault → the kernel kills the
|
||||||
|
process ([resilience.md](resilience.md) step 2, built) → the supervisor reads the
|
||||||
|
reason → restart. Recovery is restart, not a handler. This is also truer to the 1990
|
||||||
|
standard than handling is: the standard's default action for all three was
|
||||||
|
"terminate the process".
|
||||||
|
|
||||||
|
**SIGPIPE deserves special contempt.** Its default kills a process that writes to a
|
||||||
|
closed pipe — which is why "the whole server died because one client disconnected"
|
||||||
|
is roughly every network daemon's first production bug, and why every mature codebase
|
||||||
|
contains the same fix: ignore SIGPIPE, handle the `EPIPE` error return. danos made
|
||||||
|
the right choice natively already — a reply owed to a dead peer fails with `-EPEER`.
|
||||||
|
Errors from operations are error returns from those operations. The posix layer can
|
||||||
|
synthesize SIGPIPE for ported code that expects it.
|
||||||
|
|
||||||
|
### Statements, not questions
|
||||||
|
|
||||||
|
A signal and a protocol message both travel over IPC — the difference is the
|
||||||
|
**contract**, not the transport. danos IPC has two primitives, both already in
|
||||||
|
daily use: the **asynchronous notification** (a badge — bits that coalesce into a
|
||||||
|
pending mask; the sender never blocks; no payload, *no reply path*; how IRQs and
|
||||||
|
exit events arrive) and the **synchronous call** (a rendezvous — payload both
|
||||||
|
ways, the caller waits for the reply; how VFS requests work). A signal is the
|
||||||
|
first kind: a *statement*. `terminate` wants no reply — the exit notification is
|
||||||
|
its acknowledgement.
|
||||||
|
|
||||||
|
A health probe is the second kind: a *question*, worthless without its answer —
|
||||||
|
and the answer's absence within a deadline is the very thing being measured.
|
||||||
|
Asked as a signal it has no reply channel (a coalescing bit can't carry an answer,
|
||||||
|
and the authority rule forbids a child signalling its supervisor back); asked as a
|
||||||
|
call, the timeout-is-the-diagnosis semantics come free. So there is no `health`
|
||||||
|
signal. Liveness is the common **`ping`**: a reserved request every harness-run
|
||||||
|
service answers automatically on its main endpoint — still free for the service
|
||||||
|
author, still one obvious way — and a supervisor's probe is a `ping` call with a
|
||||||
|
deadline.
|
||||||
|
|
||||||
|
## The two iron rules
|
||||||
|
|
||||||
|
1. **Cleanup is the kernel's job.** A process can die with no warning — fault,
|
||||||
|
kill, power. Correctness must never depend on a `terminate` handler running. On
|
||||||
|
any death the kernel releases the address space, IPC handles, IRQ bindings, and
|
||||||
|
owed replies (built), and must also release **device, I/O-port, and interrupt
|
||||||
|
claims and MSI vectors** (the known gap in
|
||||||
|
[process-management.md](process-management.md); increment 1). A signal handler is
|
||||||
|
for *graceful* work — flushing, deregistering, saving — never for *necessary*
|
||||||
|
work.
|
||||||
|
2. **Kill is not a signal, and exit reasons are load-bearing.** The standard stop
|
||||||
|
sequence is *terminate → deadline → `process_kill`*; the unhandleable kill stays
|
||||||
|
a kernel mechanism. And a supervisor deciding whether to restart must know *how*
|
||||||
|
the child died: clean exit (meant to — don't restart), fault (restart with
|
||||||
|
backoff), killed (the supervisor did it). The exit notification today carries
|
||||||
|
only the id; it grows a reason. Restart policy cannot be written without it.
|
||||||
|
|
||||||
|
## Who learns of a death
|
||||||
|
|
||||||
|
A death has three audiences, and conflating them is how systems end up with either
|
||||||
|
zombie state or privileged snooping:
|
||||||
|
|
||||||
|
1. **The supervisor** — gets the exit notification on the endpoint it gave at spawn
|
||||||
|
(built), which grows the `ExitReason` (increment 2). The supervisor is the only
|
||||||
|
audience that needs the *reason*, because it is the only one deciding whether to
|
||||||
|
restart.
|
||||||
|
2. **The peer owed a reply** — already built: a client that dies mid-request fails
|
||||||
|
the server's reply with `-EPEER`; a server that dies fails its waiting clients
|
||||||
|
the same way. This covers the *synchronous* case only.
|
||||||
|
3. **The subscribers** — the new piece, and it is the input service's
|
||||||
|
publish/subscribe shape ([input.md](input.md)) applied to exits. A stateful
|
||||||
|
service accumulates per-client state across many requests: the VFS holds a dead
|
||||||
|
client's open file handles, the input service holds its subscriptions, a future
|
||||||
|
network stack holds its sockets. None of these are the client's supervisor, and
|
||||||
|
none learn anything from a failed reply if the client simply never calls again.
|
||||||
|
So the kernel **publishes every exit** to whoever subscribed:
|
||||||
|
`process_subscribe(endpoint)` adds a subscriber, and each death posts a
|
||||||
|
notification to every subscriber (badge = `notify_exit_bit | process id` — the
|
||||||
|
same encoding supervisors already decode, the IRQ-as-IPC pattern once more). The
|
||||||
|
subscriber filters for ids it holds state for and releases what the dead client
|
||||||
|
held. Correlating is free of bookkeeping: an IPC sender's badge already *is* its
|
||||||
|
task id (`runtime.ipc.Received`), so the id a service has been keying client
|
||||||
|
state by all along is the id the exit event carries.
|
||||||
|
|
||||||
|
Subscription, not broadcast-to-everyone: only processes that asked receive
|
||||||
|
events, the kernel keeps a bounded subscriber table, and delivery is the same
|
||||||
|
non-blocking coalescing notification as everything else — a dying process never
|
||||||
|
waits on its mourners. Subscribing is ungated, like `process_enumerate`: what is
|
||||||
|
running (and dying) is not a secret between cooperating processes. Subscribers
|
||||||
|
do not receive the exit reason — the VFS does not care *why* the client died.
|
||||||
|
|
||||||
|
This is the service-side mirror of iron rule 1: **a service must never depend on
|
||||||
|
its clients cleaning up after themselves.** Handle release on client death is the
|
||||||
|
service's job, triggered by the published exit event — never by a courtesy
|
||||||
|
"closing now" message that a crashed client will never send.
|
||||||
|
|
||||||
|
## The stable interface: `runtime.process`
|
||||||
|
|
||||||
|
`runtime.process` already owns what a process receives at birth (`Init`, the
|
||||||
|
argv contract). It grows to own the other end of life.
|
||||||
|
|
||||||
|
**The runtime is the stable interface; the numbers are not.** danos applications do
|
||||||
|
not make system calls — they call the runtime library, and the system-call numbers,
|
||||||
|
notification bits, and signal bit positions beneath it are a **private kernel ↔
|
||||||
|
runtime contract** that may change at any time (settled 2026-07-12). This is why
|
||||||
|
the runtime exists. Today kernel and runtime ship from one tree in one image, so
|
||||||
|
"stability" is simply building them together. When driver binaries start shipping
|
||||||
|
as separately-versioned applications — the whole point of the restart design — the
|
||||||
|
binary's embedded runtime version becomes compatibility metadata (the same idea as
|
||||||
|
the protocol version in the device manager's `hello`), and the kernel refuses what
|
||||||
|
it cannot serve. Signals therefore need no reserved numbering scheme: the enum
|
||||||
|
below is vocabulary, not ABI.
|
||||||
|
|
||||||
|
```zig
|
||||||
|
/// The signal vocabulary. The value is the bit position in the pending mask — a
|
||||||
|
/// private kernel/runtime detail, free to change while they ship together.
|
||||||
|
pub const Signal = enum(u5) {
|
||||||
|
terminate = 0, // SIGTERM: finish up and exit
|
||||||
|
reload = 1, // SIGHUP: re-read configuration
|
||||||
|
interrupt = 2, // SIGINT
|
||||||
|
quit = 3, // SIGQUIT
|
||||||
|
alarm = 4, // SIGALRM
|
||||||
|
user_1 = 5, // SIGUSR1
|
||||||
|
user_2 = 6, // SIGUSR2
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A decoded pending mask: the coalesced set of signals a notification delivered.
|
||||||
|
pub const SignalSet = struct {
|
||||||
|
pending: u32,
|
||||||
|
pub fn has(set: SignalSet, signal: Signal) bool { ... }
|
||||||
|
pub fn iterate(set: SignalSet) Iterator { ... }
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Nominate `endpoint` as this process's signal endpoint (signal_bind). The
|
||||||
|
/// runtime's service harness calls this; a bare program may call it directly and
|
||||||
|
/// fold signals into its own replyWait loop.
|
||||||
|
pub fn bindSignals(endpoint: usize) bool { ... }
|
||||||
|
|
||||||
|
/// Decode a received badge into signals, or null if the badge is not a signal
|
||||||
|
/// notification (mirrors ipc.Received.isChildExit).
|
||||||
|
pub fn signalsFrom(badge: usize) ?SignalSet { ... }
|
||||||
|
|
||||||
|
/// Send `signal` to process `id`. Supervisor-gated, like kill; non-blocking.
|
||||||
|
pub fn sendSignal(id: u32, signal: Signal) bool { ... }
|
||||||
|
|
||||||
|
/// The standard stop sequence: terminate, wait up to `deadline_ms` for the exit
|
||||||
|
/// notification, then process_kill. The one call a supervisor needs.
|
||||||
|
pub fn stop(id: u32, deadline_ms: u64) void { ... }
|
||||||
|
|
||||||
|
/// Subscribe `endpoint` to published exit events (process_subscribe). Every
|
||||||
|
/// process death posts an asynchronous notification: badge = notify_exit_bit |
|
||||||
|
/// process id — the same encoding a supervisor's exit notification uses, decoded
|
||||||
|
/// by the same ipc.Received helpers. For stateful services: release what the dead
|
||||||
|
/// client held (file handles, subscriptions, sockets). Ungated, like
|
||||||
|
/// process_enumerate.
|
||||||
|
pub fn subscribeExits(endpoint: usize) bool { ... }
|
||||||
|
|
||||||
|
/// How a process ended — queried after the exit notification (the kernel records
|
||||||
|
/// it first, so the two never race). What restart policy reads. (Built in M17.2.)
|
||||||
|
pub const ExitReason = enum(u8) {
|
||||||
|
exited, // returned from main / clean exit
|
||||||
|
aborted, // abort() — deliberate self-termination (SIGABRT's ghost; reserved)
|
||||||
|
segmentation_fault, // SIGSEGV's ghost
|
||||||
|
illegal_instruction, // SIGILL's ghost
|
||||||
|
arithmetic_fault, // SIGFPE's ghost
|
||||||
|
protection_fault, // general protection fault
|
||||||
|
fault, // any other CPU exception
|
||||||
|
killed, // process_kill
|
||||||
|
};
|
||||||
|
```
|
||||||
|
|
||||||
|
Two deliberate absences. There is no `mask`/`block` API — a process that is not
|
||||||
|
ready for a signal simply has not waited on its endpoint yet; the pending mask *is*
|
||||||
|
the blocked set. And there is no per-signal handler registration at this layer —
|
||||||
|
dispatch is the process's own `switch` over `SignalSet`, or the service harness's
|
||||||
|
callbacks (`on_terminate`, `on_reload`) for programs that want defaults.
|
||||||
|
|
||||||
|
### The service harness
|
||||||
|
|
||||||
|
`runtime.service` owns the `replyWait` loop and folds every event source — signals,
|
||||||
|
child exits, protocol messages — into callbacks, with the vocabulary's defaults:
|
||||||
|
`terminate` returns from the loop (clean exit), the common `ping` is answered automatically,
|
||||||
|
`reload` is ignored unless overridden. One loop, no locking, nothing reentrant. A
|
||||||
|
service author writes domain logic; the lifecycle contract is satisfied by the
|
||||||
|
harness. A process that bypasses the harness and ignores its signals meets the
|
||||||
|
deadline-then-kill escalation — you cannot force a process to implement an
|
||||||
|
interface, but you can make compliance free and non-compliance fatal.
|
||||||
|
|
||||||
|
### The musl layer later
|
||||||
|
|
||||||
|
The POSIX C layer is a **musl port**: musl's arch/syscall layer retargeted so that
|
||||||
|
what musl believes are kernel syscalls become danos runtime calls and IPC — `open`
|
||||||
|
and `read` onto the VFS protocol, `kill`/`sigaction`/`waitpid` onto this document's
|
||||||
|
vocabulary, `exit` onto the runtime's exit path. `sigaction` handlers registered
|
||||||
|
through it are invoked by the runtime's loop when the signal message arrives —
|
||||||
|
synchronous underneath, async-looking to ported code, delivered at wait boundaries
|
||||||
|
the way most Unix programs already experience signals (at syscalls). No stack hijack
|
||||||
|
ever happens, `SA_RESTART` semantics come free because nothing was interrupted, and
|
||||||
|
SIGPIPE can be synthesized from `-EPEER` for the programs that expect it. C programs
|
||||||
|
get POSIX; danos-native programs never pay for it.
|
||||||
|
|
||||||
|
## Increments
|
||||||
|
|
||||||
|
1. **Kernel: release device/port/IRQ claims and MSI vectors on death** — the
|
||||||
|
cleanup half of iron rule 1, and the prerequisite for any restart story. Test:
|
||||||
|
kill a claiming driver, spawn it again, the claim succeeds.
|
||||||
|
2. **Exit reason in the death notification** (`ExitReason` above).
|
||||||
|
3. **Exit events**: `process_subscribe` in the kernel (bounded subscriber table,
|
||||||
|
publishes on every death), `runtime.process.subscribeExits`; the VFS becomes the
|
||||||
|
first subscriber — releasing a dead client's handles is its proof test.
|
||||||
|
4. **Signals**: `signal_bind` + `process_signal` + the pending mask in the kernel;
|
||||||
|
`runtime.process` grows the interface above; the service harness handles
|
||||||
|
`terminate` and answers the common `ping`; `stop()` for supervisors.
|
||||||
|
|
||||||
|
[device-manager.md](device-manager.md) builds directly on all four.
|
||||||
|
|
||||||
|
## Settled questions (2026-07-12)
|
||||||
|
|
||||||
|
- **Signal numbering is not ABI**: the runtime is the stable interface; the numbers
|
||||||
|
beneath it are a private kernel ↔ runtime contract (see "The stable interface").
|
||||||
|
- **Liveness is a `ping` call, not a signal**: signals are statements, questions
|
||||||
|
are synchronous calls (see "Statements, not questions"). A service wanting *deep*
|
||||||
|
health ("can I reach my hardware?") defines its own protocol message on top.
|
||||||
|
- **Process handles: deferred.** Pids + the supervisor gate cover everything
|
||||||
|
planned; transferable handles (Fuchsia-style, delegating signalling without
|
||||||
|
delegating kill) wait for the capability table to grow types beyond endpoints.
|
||||||
|
- **`alarm`: in the vocabulary, unbuilt.** No consumer yet; when one appears it is
|
||||||
|
runtime sugar over the existing timer (arm a timer that posts your own signal) —
|
||||||
|
zero kernel work, so deferring costs nothing.
|
||||||
|
- **Subscription granularity: all exits**, subscriber-side filtering — one
|
||||||
|
subscription per service, a bounded kernel table. Per-id subscriptions only if
|
||||||
|
event volume ever matters (hundreds of processes, not before).
|
||||||
|
- **Client identity across the exit boundary: no convention needed** — an IPC
|
||||||
|
sender's badge already is its task id (see "Who learns of a death").
|
||||||
@@ -0,0 +1,120 @@
|
|||||||
|
# Process Management
|
||||||
|
|
||||||
|
How danos lists, supervises, and kills processes — the microkernel answer to
|
||||||
|
`ps`, `kill`, and `SIGCHLD`/`wait`.
|
||||||
|
|
||||||
|
## Why system calls, not `/proc`
|
||||||
|
|
||||||
|
Unix systems sit on a spectrum. Classic BSD/macOS list processes through
|
||||||
|
syscalls (`sysctl(KERN_PROC)`) and kill through `kill(2)`; Linux renders the
|
||||||
|
process table as `/proc` for *reading* but still kills through a syscall; Plan 9
|
||||||
|
made the file tree the whole interface (`echo kill > /proc/n/ctl`). Microkernels
|
||||||
|
mostly abandon ambient PIDs: Minix and QNX route everything through a user-space
|
||||||
|
process-manager server, and Fuchsia/seL4 control processes only through handles.
|
||||||
|
|
||||||
|
danos rules out `/proc` **as the primitive**: here a `/proc` would be served by
|
||||||
|
the VFS server — a user process — which would put the VFS in the path of process
|
||||||
|
control. If the VFS (or anything under it) hangs, nothing could be listed or
|
||||||
|
killed, *including the hung VFS*. The control plane for processes must not
|
||||||
|
depend on a process. So the primitives are kernel system calls; a read-only
|
||||||
|
`/proc` rendering can be layered on later, and a POSIX-style process-manager
|
||||||
|
server can be built *from* these primitives when one is needed.
|
||||||
|
|
||||||
|
## The three primitives
|
||||||
|
|
||||||
|
### `process_enumerate(buffer, maximum) -> total`
|
||||||
|
|
||||||
|
A snapshot of the task table into a caller buffer of `abi.ProcessDescriptor`
|
||||||
|
(id, supervisor, state, priority, name) — the exact shape of
|
||||||
|
`device_enumerate`, so `ps` is a user program over a snapshot, not a kernel
|
||||||
|
service. The total may exceed what fit; call again with a larger buffer. Kernel
|
||||||
|
tasks are included with an empty name — an honest listing shows the idle tasks
|
||||||
|
too. Ungated and read-only: what is running is not a secret between cooperating
|
||||||
|
bring-up processes.
|
||||||
|
|
||||||
|
### `system_spawn(..., exit_endpoint) -> child id`, and the supervision link
|
||||||
|
|
||||||
|
`system_spawn` records the caller as the child's **supervisor** and returns the
|
||||||
|
child's process id (ids are monotonic, never reused — a stale id can only miss).
|
||||||
|
That link is the kill authority: it answers "who may kill process 7?" without
|
||||||
|
inventing users or permissions, the same way a device *claim* is the capability
|
||||||
|
for `mmio_map`. It composes with the supervision hierarchy the device manager
|
||||||
|
already forms: init supervises the services it starts, the device manager
|
||||||
|
supervises the drivers it matches. (A transferable process *handle* — Fuchsia
|
||||||
|
style — can replace the id once the handle table grows types beyond endpoints.)
|
||||||
|
|
||||||
|
`exit_endpoint` (a handle, or `abi.no_cap`) is the supervisor's death-watch: when
|
||||||
|
the child ends — clean exit, CPU fault, or `process_kill` — the kernel posts an
|
||||||
|
asynchronous notification to that endpoint, exactly like a bound IRQ. The badge
|
||||||
|
carries `abi.notify_badge_bit | abi.notify_exit_bit | child_id`, so one endpoint
|
||||||
|
supervises many children and can even share with IRQ notifications. This is the
|
||||||
|
microkernel's SIGCHLD: no new mechanism, just the IRQ-as-IPC pattern reused, and
|
||||||
|
a supervisor's event loop (`ipc.replyWait`) already knows how to receive it. The
|
||||||
|
child holds a reference to the endpoint from birth, so the notification cannot
|
||||||
|
dangle even if the supervisor dies first.
|
||||||
|
|
||||||
|
### `process_kill(id) -> 0 / -ESRCH / -EPERM`
|
||||||
|
|
||||||
|
Only the supervisor may kill; kernel tasks are not killable processes. Like a
|
||||||
|
signal, delivery is prompt but asynchronous — 0 means the kill is accepted and
|
||||||
|
irrevocable; the exit notification confirms completion.
|
||||||
|
|
||||||
|
## How a kill lands (the kernel mechanics)
|
||||||
|
|
||||||
|
Everything below runs under the big kernel lock, where task states cannot move.
|
||||||
|
|
||||||
|
- **Target ready or blocked** (not on any core): reaped on the killer's own
|
||||||
|
call. The reap releases what death always releases (IRQ bindings first, then
|
||||||
|
a client the target still owed a reply to is failed with `-EPEER`, IPC handles
|
||||||
|
closed, the exit notification posted last) — plus the unlinking only a
|
||||||
|
*remote* death needs: out of the ready queue, out of an endpoint's sender FIFO
|
||||||
|
(`Task.ipc_wait_endpoint`), out of a receive wait queue (`Task.wait_queue`),
|
||||||
|
and out of any server's owed-reply slot, so nothing ever dequeues a dangling
|
||||||
|
pointer. Destroying the address space is safe because no core can have it
|
||||||
|
loaded: every switch away from a task loads the next task's tables.
|
||||||
|
- **Target running on another core**: it cannot be torn down mid-instruction,
|
||||||
|
so it is condemned (`Task.kill_pending`) and dies at whichever comes first:
|
||||||
|
- its next **system_call entry** — checked before dispatch, so a condemned
|
||||||
|
process cannot spawn, claim, or message anything on its way out;
|
||||||
|
- its core's next **timer tick** — but only when the task is not inside one
|
||||||
|
of its own system calls (`Task.in_system_call`): the tick may have
|
||||||
|
interrupted kernel code mid-operation, where teardown would leak whatever
|
||||||
|
the operation held. User-mode execution is always a safe kill point. The
|
||||||
|
tick-time terminate abandons the interrupt frame exactly like the fault
|
||||||
|
path (the LAPIC is acknowledged before the tick hook runs);
|
||||||
|
- any core's tick finding it **blocked or ready** (it entered a syscall and
|
||||||
|
parked after being condemned) — reaped by the same remote-reap path.
|
||||||
|
|
||||||
|
A pure user-mode spin loop that never makes a system call therefore dies
|
||||||
|
within one tick; nothing a process does can outrun the kill.
|
||||||
|
|
||||||
|
The scheduler stays below the process layer: finishing a kill (IRQ bindings,
|
||||||
|
handles, the notification) is called *up* through two hooks process.zig
|
||||||
|
registers at boot (`terminate_current_hook`, `reap_task_hook`), mirroring how
|
||||||
|
the architecture layer calls up into `tick`.
|
||||||
|
|
||||||
|
## Known gaps (bring-up honesty)
|
||||||
|
|
||||||
|
- ~~Device claims are not released on death~~ Closed (M17.1): every path out of a
|
||||||
|
process releases its device claims alongside its IRQ and MSI bindings
|
||||||
|
(`releaseTaskResourcesLocked`), so a restarted driver can claim its hardware
|
||||||
|
again — the cleanup half of [process-lifecycle.md](process-lifecycle.md)'s iron
|
||||||
|
rule 1. The `claim-release` test proves the kill → release → re-claim cycle.
|
||||||
|
- Kernel stacks of dead tasks are leaked, as on every exit path (no reaper yet).
|
||||||
|
- ~~There is no exit status in the notification~~ Closed (M17.2): the kernel
|
||||||
|
records how every process ends — exited, a fault class, or killed — before it
|
||||||
|
posts the exit notification, and the supervisor reads it with
|
||||||
|
`process_exit_reason` (`runtime.process.exitReason`). This is the input to
|
||||||
|
restart policy ([process-lifecycle.md](process-lifecycle.md)); an exit *code*
|
||||||
|
for the clean case can still ride alongside later.
|
||||||
|
- Enumerate writes through the caller's raw pointer under the bring-up trust
|
||||||
|
model, like `device_enumerate` (an unmapped page is a self-DoS, not an
|
||||||
|
isolation break).
|
||||||
|
|
||||||
|
## Tests
|
||||||
|
|
||||||
|
`process-list` (enumerate), `process-kill` (kernel-level kill paths, refusals,
|
||||||
|
notifications), `supervision` (the whole user-side surface via the process-test
|
||||||
|
service: spawn supervised → enumerate → kill blocked and spinning children →
|
||||||
|
notifications → gone), `claim-release` (a killed claim-holder's device is
|
||||||
|
claimable again). See test/qemu_test.py.
|
||||||
+16
-3
@@ -1,6 +1,17 @@
|
|||||||
# Resilience: fault isolation and live restart
|
# Resilience: fault isolation and live restart
|
||||||
|
|
||||||
A design/research note, not built yet. This is the property danos is really chasing:
|
Steps 1–4 of the ordering below are **built** (M17–M18, 2026-07-13): user-mode
|
||||||
|
isolation; fault → kill the process → keep the core (`onException`; the
|
||||||
|
`fault-recovery` test); the supervisor notification **with exit reasons**
|
||||||
|
([process-lifecycle.md](process-lifecycle.md) — clean exit, fault class, or
|
||||||
|
killed, recorded before the notice posts); and the **restart policy itself**
|
||||||
|
([device-manager.md](device-manager.md)): the device manager supervises every
|
||||||
|
driver, restarts crashes with backoff, caps crash loops, and re-claims work
|
||||||
|
because the kernel releases a dead process's claims. The `driver-restart` and
|
||||||
|
`usb-report` scenarios prove kill → release → respawn → re-claim → re-report
|
||||||
|
end to end. What remains of this document's ladder is scope, not mechanism:
|
||||||
|
more of the system moved into restartable processes (the discovery migration,
|
||||||
|
[m19-m20-plan.md](m19-m20-plan.md), is the next rung). This is the property danos is really chasing:
|
||||||
**if a part of the OS breaks, isolate it, and re-initialise it — without rebooting.**
|
**if a part of the OS breaks, isolate it, and re-initialise it — without rebooting.**
|
||||||
A crashed driver gets restarted; a wedged service gets killed and brought back. It's
|
A crashed driver gets restarted; a wedged service gets killed and brought back. It's
|
||||||
the reason the [microkernel](vision.md) shape was chosen, and it's a *separate* goal
|
the reason the [microkernel](vision.md) shape was chosen, and it's a *separate* goal
|
||||||
@@ -111,9 +122,11 @@ Honest boundaries:
|
|||||||
## Suggested ordering
|
## Suggested ordering
|
||||||
|
|
||||||
1. **User mode + address-space isolation** — the shared prerequisite (also on the
|
1. **User mode + address-space isolation** — the shared prerequisite (also on the
|
||||||
path for everything else).
|
path for everything else). **Done.**
|
||||||
2. **Kernel: fault → kill process → notify.** Turn today's "halt on fault" into
|
2. **Kernel: fault → kill process → notify.** Turn today's "halt on fault" into
|
||||||
"confine to the process and report it."
|
"confine to the process and report it." **Done** (the kill and reclaim; the
|
||||||
|
supervisor notification waits for step 3's supervisor). A killed server's
|
||||||
|
pending client is unblocked with `-EPEER` rather than hung.
|
||||||
3. **A minimal supervisor server** that can (re)start a process.
|
3. **A minimal supervisor server** that can (re)start a process.
|
||||||
4. **Resource cleanup on death** — reclaim memory/MMIO/IPC/IRQ, via caps or a grant
|
4. **Resource cleanup on death** — reclaim memory/MMIO/IPC/IRQ, via caps or a grant
|
||||||
table.
|
table.
|
||||||
|
|||||||
+23
-2
@@ -6,8 +6,8 @@ ready task always runs, and tasks at the same priority take turns. That model is
|
|||||||
chosen for [real-time](vision.md) — it's predictable (you can reason about which
|
chosen for [real-time](vision.md) — it's predictable (you can reason about which
|
||||||
task runs when) and its decisions are O(1), unlike a fair-share scheduler.
|
task runs when) and its decisions are O(1), unlike a fair-share scheduler.
|
||||||
|
|
||||||
The scheduler proper (`src/kernel/sched.zig`) is generic; the context switch and new-task
|
The scheduler proper (`system/kernel/sched.zig`) is generic; the context switch and new-task
|
||||||
stack setup are architecture-specific (`src/kernel/arch/x86_64/`, see [arch](arch.md)).
|
stack setup are architecture-specific (`system/kernel/architecture/x86_64/`, see [arch](arch.md)).
|
||||||
|
|
||||||
## Tasks
|
## Tasks
|
||||||
|
|
||||||
@@ -67,6 +67,27 @@ exist, which is what a real-time scheduler needs.
|
|||||||
- **Round-robin within a level.** When a task is descheduled it goes to the *back*
|
- **Round-robin within a level.** When a task is descheduled it goes to the *back*
|
||||||
of its level's queue, so equal-priority tasks share the CPU fairly.
|
of its level's queue, so equal-priority tasks share the CPU fairly.
|
||||||
|
|
||||||
|
## Affinity: pinning a task to a core
|
||||||
|
|
||||||
|
By default a task runs on **any** core — the ready queue above is global, and any
|
||||||
|
idle core pulls the highest-priority task from it (work-conserving; see
|
||||||
|
[smp.md](smp.md)). A task can instead be **pinned** to one core with
|
||||||
|
`spawnOn(entry, priority, cpu)`, giving it an *affinity*: it will only ever run
|
||||||
|
there, never migrating.
|
||||||
|
|
||||||
|
Mechanically, each core has its **own** pinned queue (same 8-level FIFO + bitmap)
|
||||||
|
alongside the global one. A pinned task is enqueued only into its core's pinned
|
||||||
|
queue; selection compares the top of the global queue and the running core's pinned
|
||||||
|
queue and takes the higher priority (still O(1) — two bit-scans and a compare), with
|
||||||
|
a pinned task winning an equal-priority tie so it can't be starved by global work.
|
||||||
|
Because every queue is mutated under the [big kernel lock](smp.md), one core enqueuing
|
||||||
|
into another core's pinned queue is safe.
|
||||||
|
|
||||||
|
This is the *explicit-affinity* model (no surprise migration mid-deadline), which is
|
||||||
|
the more real-time-predictable direction. `spawnOn` refuses to pin to an offline or
|
||||||
|
out-of-range core — it creates the task unpinned instead, so it still runs somewhere
|
||||||
|
rather than stranding in a queue no core services, and returns whether the pin took.
|
||||||
|
|
||||||
## Sleeping and the idle task
|
## Sleeping and the idle task
|
||||||
|
|
||||||
A task can **block** — give up the CPU until an event, rather than busy-wait
|
A task can **block** — give up the CPU until an event, rather than busy-wait
|
||||||
|
|||||||
+114
-7
@@ -16,10 +16,14 @@ danos specifics):
|
|||||||
- **Real-time** — whether timing is *predictable*. Comes from bounded operations
|
- **Real-time** — whether timing is *predictable*. Comes from bounded operations
|
||||||
(our O(1) scheduler), not from core count.
|
(our O(1) scheduler), not from core count.
|
||||||
|
|
||||||
danos is uniprocessor today: one global `current` task, one set of ready queues, one
|
danos now runs on multiple cores. The firmware starts only the **bootstrap processor
|
||||||
timer. Even on an 8-core CPU, the firmware starts only the **bootstrap processor
|
(BSP)**; the kernel wakes the other cores (**application processors**, APs) with
|
||||||
(BSP)**; the other cores (**application processors**, APs) sit parked until the
|
INIT–SIPI–SIPI, brings each up into 64-bit long mode with its own descriptor tables,
|
||||||
kernel wakes them, which it doesn't yet.
|
LAPIC and timer, and drops it into the scheduler. Tasks run **genuinely in parallel** —
|
||||||
|
the `smp` self-test confirms worker tasks executing on all four cores at once under
|
||||||
|
QEMU `-smp 4`. Shared kernel state (scheduler queues, IPC) is serialised behind a big
|
||||||
|
kernel lock. What's left is refinement, not first-light: per-core run queues, IPIs,
|
||||||
|
and thread-to-core affinity (see [Implementation status](#implementation-status)).
|
||||||
|
|
||||||
## The common microkernel instinct: don't share kernel state
|
## The common microkernel instinct: don't share kernel state
|
||||||
|
|
||||||
@@ -114,14 +118,20 @@ active reconsideration in favour of resilience — see [vision.md](vision.md).)
|
|||||||
Whatever the top goal, the *sequence* is the same and seL4 validates starting simple:
|
Whatever the top goal, the *sequence* is the same and seL4 validates starting simple:
|
||||||
|
|
||||||
1. **Enumerate cores** — needs [device discovery](discovery.md) (ACPI MADT on x86,
|
1. **Enumerate cores** — needs [device discovery](discovery.md) (ACPI MADT on x86,
|
||||||
device tree on ARM). SMP is a concrete consumer of that work.
|
device tree on ARM). SMP is a concrete consumer of that work. **Done on x86:** the
|
||||||
|
MADT parse records every usable Local APIC — with the `apic_id` an AP wake targets —
|
||||||
|
and `platform.cpus()` returns the list (see [discovery.md](discovery.md)). The boot
|
||||||
|
log reports the count; the ARM (device-tree) path still needs it.
|
||||||
2. **Wake the APs** — INIT–SIPI–SIPI on x86; PSCI/spin-tables on ARM. Each core brings
|
2. **Wake the APs** — INIT–SIPI–SIPI on x86; PSCI/spin-tables on ARM. Each core brings
|
||||||
up its own tables, timer, and idle task.
|
up its own tables, timer, and idle task. **Done on x86** — cores climb to long mode,
|
||||||
|
set up their own GDT/TSS, and enter the scheduler; tasks run in parallel across all
|
||||||
|
cores ([status](#implementation-status)).
|
||||||
3. **Start with a big kernel lock.** It's a legitimate first design, not a shortcut —
|
3. **Start with a big kernel lock.** It's a legitimate first design, not a shortcut —
|
||||||
philosophically aligned with a tiny kernel, and it lets the single-core correctness
|
philosophically aligned with a tiny kernel, and it lets the single-core correctness
|
||||||
model you already have (the interrupt-flag discipline in
|
model you already have (the interrupt-flag discipline in
|
||||||
[scheduling.md](scheduling.md)) stay largely intact: one lock around kernel entry
|
[scheduling.md](scheduling.md)) stay largely intact: one lock around kernel entry
|
||||||
instead of rethinking every critical section.
|
instead of rethinking every critical section. **Done** — see
|
||||||
|
`system/kernel/sync.zig`.
|
||||||
4. **Later, if contention bites,** evolve toward **per-core run queues + explicit
|
4. **Later, if contention bites,** evolve toward **per-core run queues + explicit
|
||||||
affinity** (the Fiasco.OC direction) — also the more real-time-predictable model.
|
affinity** (the Fiasco.OC direction) — also the more real-time-predictable model.
|
||||||
5. **Placement stays a user-space policy** — the kernel runs a thread on the core it's
|
5. **Placement stays a user-space policy** — the kernel runs a thread on the core it's
|
||||||
@@ -130,6 +140,103 @@ Whatever the top goal, the *sequence* is the same and seL4 validates starting si
|
|||||||
Big-lock-first → per-core-later. The affinity/MCS depth is only worth it if real-time
|
Big-lock-first → per-core-later. The affinity/MCS depth is only worth it if real-time
|
||||||
turns out to be the actual goal.
|
turns out to be the actual goal.
|
||||||
|
|
||||||
|
## Implementation status
|
||||||
|
|
||||||
|
The "wake + schedule" build (real parallel task execution) is going in as a sequence
|
||||||
|
of green checkpoints — each step keeps the single-core test suite passing before the
|
||||||
|
next lands.
|
||||||
|
|
||||||
|
**Done:**
|
||||||
|
|
||||||
|
- **Core enumeration** — the MADT parse records every usable Local APIC (with its
|
||||||
|
`apic_id`, which an AP wake targets); `platform.cpus()` returns the list. See
|
||||||
|
[discovery.md](discovery.md).
|
||||||
|
- **The big kernel lock** (`system/kernel/sync.zig`) — one coarse spinlock guarding the
|
||||||
|
scheduler queues and IPC, always held with local interrupts disabled. It is held
|
||||||
|
*across* a context switch and released by whichever task resumes (the hand-off
|
||||||
|
rule); `task_trampoline` releases it for a freshly-spawned task. `scheduler.zig` and
|
||||||
|
`ipc.zig` run every critical section under it. Uncontended on one core, so behaviour
|
||||||
|
is identical to the old interrupt-flag model.
|
||||||
|
- **Per-CPU state** — a `PerCpu` struct (running task, idle task, APIC id) per core,
|
||||||
|
its pointer kept in the x86 **GS base** (`IA32_GS_BASE`; no `swapgs`, since there's
|
||||||
|
no user mode yet). The old global `current` is now `thisCpu().current`. The ready
|
||||||
|
queues stay **global** under the lock — work-conserving, so any idle core will pull
|
||||||
|
the highest-priority ready task; per-core queues are a later optimisation.
|
||||||
|
- **AP wake to long mode** — `arch.startSecondary` drives INIT–SIPI–SIPI (via the
|
||||||
|
LAPIC ICR) to wake each parked core one at a time. A woken core starts in 16-bit
|
||||||
|
real mode at a low page and runs the [trampoline](../system/kernel/architecture/x86_64/trampoline.s)
|
||||||
|
up through protected mode into 64-bit long mode, then lands in `smp.zig:apEntry`,
|
||||||
|
publishes its per-CPU pointer, and reports in. Verified in QEMU with `-smp 4`:
|
||||||
|
all four cores report `online`.
|
||||||
|
|
||||||
|
The trampoline earns its complexity from four hardware facts:
|
||||||
|
- a STARTUP IPI vectors a core to physical `vector << 12` (a *byte* vector), so the
|
||||||
|
trampoline must live **below 1 MiB** — the kernel reserves that page from the frame
|
||||||
|
allocator at boot, before paging/heap draw down the scarce low frames;
|
||||||
|
- the blanket RAM identity map is **NX** (W^X), but the AP fetches the trampoline
|
||||||
|
from it under paging, so that one page is made executable for bring-up;
|
||||||
|
- the blob is copied to a page whose address isn't known at link time, so it is
|
||||||
|
**position-independent**: it derives its own base from `CS` and, crucially,
|
||||||
|
addresses data *segment-relative in real mode* (where the segment base already
|
||||||
|
supplies the page base) but *base-register-relative in protected/long mode* (flat
|
||||||
|
segments, base 0). Getting that distinction wrong was the first bug found;
|
||||||
|
- an AP starts with a bare `CR0`/`CR4`, but the kernel is built **with SSE** (the
|
||||||
|
x86_64 baseline) and the compiler emits SSE for things as ordinary as a struct
|
||||||
|
copy — so the trampoline must set `CR4.OSFXSR`/`OSXMMEXCPT` and fix `CR0.EM`/`MP`,
|
||||||
|
or the first SSE instruction on the AP `#UD`s. The BSP inherited those bits from
|
||||||
|
UEFI; the AP has to set them itself. This was the second bug — it masqueraded as a
|
||||||
|
fault in `lgdt` (the first kernel code after entry that the compiler vectorised).
|
||||||
|
|
||||||
|
- **Per-core tables + scheduler entry** — each AP loads **its own GDT** (with its own
|
||||||
|
TSS descriptor) and **its own TSS** (its own IST/`rsp0` stack), loads the shared
|
||||||
|
IDT, enables its LAPIC and timer, then calls the generic `secondaryMain`: it turns
|
||||||
|
its bring-up context into the core's idle task (as task 0 is for the BSP), marks the
|
||||||
|
core online, and enters the run loop. With interrupts on, each core's own timer tick
|
||||||
|
preempts its idle context into whatever the global ready queue offers — so all cores
|
||||||
|
pull real work in parallel. The `smp` test spawns CPU-bound workers and confirms they
|
||||||
|
execute on all four cores at once, and `fault-ap-df` pins a #DF to an AP and checks
|
||||||
|
that core catches it on **its own** IST (a broken per-core TSS would triple-fault) —
|
||||||
|
reported as "core N: …", so a fault is always attributed to the core it happened on,
|
||||||
|
and is contained to that core (the rest of the system keeps running).
|
||||||
|
- **Thread affinity** — `spawnOn(entry, priority, cpu)` pins a task to a core (its own
|
||||||
|
per-core pinned queue, merged with the global queue at selection; see
|
||||||
|
[scheduling.md](scheduling.md#affinity-pinning-a-task-to-a-core)). The `affinity`
|
||||||
|
test confirms a pinned task never migrates. This is the mechanism the fault-on-AP
|
||||||
|
test rides on, and the *explicit-affinity* real-time-predictable model.
|
||||||
|
- **Right-sized footprint** — the per-CPU ceiling (`system.max_cpus`, one constant
|
||||||
|
shared by discovery, the scheduler, and the per-core GDT/TSS) is generous (128), but
|
||||||
|
the *large* per-core resources — the kernel and IST (double-fault) stacks — are
|
||||||
|
**heap-allocated at bring-up**, only for cores that actually come online. Only the
|
||||||
|
BSP's IST stack is static, because it must exist before the frame allocator does.
|
||||||
|
This kept the kernel image small (a static `[128][16 KiB]` IST array would have been
|
||||||
|
2 MiB of `.bss`); it's a few tens of KiB instead.
|
||||||
|
- **`single_threaded` off** — the kernel was built `single_threaded = true`, which
|
||||||
|
compiles `std.atomic` down to plain non-atomic ops. Harmless on one core, but it
|
||||||
|
quietly breaks the big kernel lock across cores; it's now `false`.
|
||||||
|
- **Re-armable wake + retry** — the trampoline frame is reserved for the system's
|
||||||
|
life, but kept **inert between wakes**: zeroed and non-executable, armed (blob
|
||||||
|
copied in, page made executable) only for the moment a core is actually climbing,
|
||||||
|
then disarmed again. So there's never a dormant executable page, and a core can be
|
||||||
|
(re)woken at any time — `arch.startSecondary` is one self-contained attempt (arm →
|
||||||
|
INIT–SIPI–SIPI → disarm), and its `INIT` resets a wedged core, so retrying just
|
||||||
|
works. Boot retries a non-responding core up to three times; the same primitive is
|
||||||
|
the groundwork a future **power manager** would drive to bring cores up (and,
|
||||||
|
eventually, its counterpart to take them offline — which additionally needs the
|
||||||
|
core's tasks migrated off first).
|
||||||
|
|
||||||
|
**Next (refinement, not first-light):**
|
||||||
|
|
||||||
|
- **IPIs** — cross-core wake/preempt. Not needed for correctness: an idle core wakes
|
||||||
|
on its own timer tick and pulls ready work then; IPIs only cut that latency from
|
||||||
|
≤1 ms to near-instant.
|
||||||
|
- **Per-core run queues** — the Fiasco.OC direction, if the single global queue's lock
|
||||||
|
contention ever bites. (Thread *affinity* already exists — see above; this is the
|
||||||
|
further step of giving each core its own primary run queue for load distribution.)
|
||||||
|
- **Fault recovery** — today a fault halts (only) the faulting core. Turning that into
|
||||||
|
"kill the task, keep the core running" is the [resilience](resilience.md) track (it
|
||||||
|
needs the task's lock/resource state handled), and for taking a core fully offline,
|
||||||
|
its tasks migrated first.
|
||||||
|
|
||||||
## Further reading
|
## Further reading
|
||||||
|
|
||||||
**Microkernel SMP & scheduling**
|
**Microkernel SMP & scheduling**
|
||||||
|
|||||||
@@ -1,6 +1,16 @@
|
|||||||
# System Calls
|
# System Calls
|
||||||
System calls (syscalls) are the bridge between your programs and the operating system's restricted core (kernel).
|
System calls (syscalls) are the bridge between your programs and the operating system's restricted core (kernel).
|
||||||
|
|
||||||
|
> **Status:** danos has real user processes (M3). User programs enter the kernel
|
||||||
|
> via the `syscall` instruction (STAR/LSTAR/SFMASK set per core; the entry stub in
|
||||||
|
> `isr.s` does the `swapgs` + kernel-stack switch and reuses the interrupt
|
||||||
|
> dispatcher). The `int 0x80` gate is kept alongside as a minimal test path. The
|
||||||
|
> current call set is still a placeholder — `0 = exit(code)`, `1 = ping`,
|
||||||
|
> `2 = write(ptr, len)`, `3 = sleep(ms)` (see `system/kernel/process.zig`); the
|
||||||
|
> handler dispatches on whether the caller is a scheduled process (its own address
|
||||||
|
> space) or a borrowed test thread. The microkernel set below (IPC_Call /
|
||||||
|
> IPC_ReplyWait / Yield) replaces it once a second user server exists.
|
||||||
|
|
||||||
## The Mechanism of a Syscall
|
## The Mechanism of a Syscall
|
||||||
|
|
||||||
A system call follows a highly orchestrated, 6-step sequence to ensure hardware safety and process security:
|
A system call follows a highly orchestrated, 6-step sequence to ensure hardware safety and process security:
|
||||||
@@ -37,6 +47,8 @@ Everything else---including`read()`,`write()`,`malloc()`, and`fork()`---will run
|
|||||||
- **What it does:**Used strictly by your background user-space servers (like your disk driver or filesystem). It sends a reply to the last client that called it, and immediately puts the server to sleep until the next request arrives.[[1](https://news.ycombinator.com/item?id=33078441)]
|
- **What it does:**Used strictly by your background user-space servers (like your disk driver or filesystem). It sends a reply to the last client that called it, and immediately puts the server to sleep until the next request arrives.[[1](https://news.ycombinator.com/item?id=33078441)]
|
||||||
3. **`Yield()`/`Thread_Ctrl()`**
|
3. **`Yield()`/`Thread_Ctrl()`**
|
||||||
- **What it does:**Allows a thread to voluntarily give up its CPU time slice, or allows a root task to spawn/kill threads.
|
- **What it does:**Allows a thread to voluntarily give up its CPU time slice, or allows a root task to spawn/kill threads.
|
||||||
|
4. **`ipc_send(endpoint, message_buffer)`(Asynchronous Send)**
|
||||||
|
- **What it does:**Posts a small payload to an endpoint's bounded queue and returns *without* blocking — no rendezvous, no reply. The receiver picks it up through the same `IPC_ReplyWait`, as a buffered message. It is the async counterpart of `IPC_Call`, for one-to-many broadcasts where a synchronous rendezvous would let one dead or slow receiver hang the sender. The [input service](input.md) — keyboard-event fan-out — is its first user. A full queue drops the oldest message (a buffered message is discrete data, unlike a coalescing interrupt notification).
|
||||||
|
|
||||||
* * * * *
|
* * * * *
|
||||||
|
|
||||||
|
|||||||
+36
-4
@@ -1,6 +1,6 @@
|
|||||||
# SysV: the kernel's calling convention
|
# SysV: the kernel's calling convention
|
||||||
|
|
||||||
Several places in danos say "the kernel is SysV" — most visibly `src/root.zig`:
|
Several places in danos say "the kernel is SysV" — most visibly `system/boot-handoff.zig`:
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
pub const kernel_abi: std.builtin.CallingConvention = .{ .x86_64_sysv = .{} };
|
pub const kernel_abi: std.builtin.CallingConvention = .{ .x86_64_sysv = .{} };
|
||||||
@@ -52,19 +52,51 @@ argument arrives in **RCX**, not RDI.
|
|||||||
|
|
||||||
danos's two binaries default to different conventions:
|
danos's two binaries default to different conventions:
|
||||||
|
|
||||||
- `src/boot/efi.zig` is built for the UEFI target, so its default C convention is
|
- `boot/efi.zig` is built for the UEFI target, so its default C convention is
|
||||||
Microsoft x64 (first argument → RCX).
|
Microsoft x64 (first argument → RCX).
|
||||||
- The kernel is freestanding, so its convention is SysV (first argument → RDI).
|
- The kernel is freestanding, so its convention is SysV (first argument → RDI).
|
||||||
|
|
||||||
When the loader jumps to the kernel passing the `BootInfo` pointer, both sides have
|
When the loader jumps to the kernel passing the `BootInfo` pointer, both sides have
|
||||||
to agree *which register that pointer lands in*. Left to their defaults, the loader
|
to agree *which register that pointer lands in*. Left to their defaults, the loader
|
||||||
would place it in RCX while the kernel looked in RDI — and the kernel would read
|
would place it in RCX while the kernel looked in RDI — and the kernel would read
|
||||||
garbage. So both sides reference the same `danos.kernel_abi` (SysV): the loader's
|
garbage. So both sides reference the same `system.kernel_abi` (SysV): the loader's
|
||||||
function-pointer type and the kernel's `_start` both carry
|
function-pointer type and the kernel's `_start` both carry
|
||||||
`callconv(danos.kernel_abi)`, and the pointer reliably arrives in RDI. That is the
|
`callconv(system.kernel_abi)`, and the pointer reliably arrives in RDI. That is the
|
||||||
whole reason `kernel_abi` lives in the shared contract — see [efi.md](efi.md) for
|
whole reason `kernel_abi` lives in the shared contract — see [efi.md](efi.md) for
|
||||||
the handoff it governs.
|
the handoff it governs.
|
||||||
|
|
||||||
|
## The process-entry stack (argc/argv)
|
||||||
|
|
||||||
|
The SysV ABI also fixes what a *fresh process* finds on its stack — and danos
|
||||||
|
follows it, so its own runtime and any future C libc read arguments the same way.
|
||||||
|
At the first user instruction, `rsp` is 16-byte aligned and points at (addresses
|
||||||
|
growing upward):
|
||||||
|
|
||||||
|
```
|
||||||
|
rsp → argc u64
|
||||||
|
argv[0] … argv[argc-1] pointers into the strings area below
|
||||||
|
NULL argv terminator
|
||||||
|
NULL envp terminator (no environment yet)
|
||||||
|
{AT_PAGESZ, page size} auxiliary vector
|
||||||
|
{AT_NULL, 0} auxiliary-vector terminator
|
||||||
|
argv string bytes NUL-terminated
|
||||||
|
───────────────────────── stack top (stack_top_virtual)
|
||||||
|
```
|
||||||
|
|
||||||
|
The kernel builds this block at the top of the process's stack — 8 pages (32 KiB,
|
||||||
|
`parameters.user_stack_pages`) mapped RW+NX below a fixed top, with the page below
|
||||||
|
them left unmapped as a **guard**, so a stack overflow faults (killing only that
|
||||||
|
process) instead of silently corrupting the image
|
||||||
|
(`buildEntryStack` in `system/kernel/process.zig`); `argv[0]` is always the path
|
||||||
|
or initial-ramdisk name the process was spawned as, and `system_spawn`'s optional
|
||||||
|
argument blob becomes `argv[1..]`. The runtime's `_start`
|
||||||
|
(`library/runtime/start.zig`) hands the block to `rt_start`, which builds a
|
||||||
|
`runtime.process.Init` from it and passes that to the program's `main`
|
||||||
|
(`pub fn main(init: runtime.process.Init)`; a parameterless `main()` is also
|
||||||
|
accepted). A C runtime's `crt0` would walk
|
||||||
|
the identical layout unmodified — that's the compatibility being bought. The
|
||||||
|
`args` test proves the round trip.
|
||||||
|
|
||||||
## Where else it surfaces
|
## Where else it surfaces
|
||||||
|
|
||||||
- **The red zone → `red_zone = false`.** `build.zig` disables the red zone for the
|
- **The red zone → `red_zone = false`.** `build.zig` disables the red zone for the
|
||||||
|
|||||||
+7
-6
@@ -8,8 +8,9 @@ without a human staring at the screen.
|
|||||||
There are two layers:
|
There are two layers:
|
||||||
|
|
||||||
- **Host unit tests** (`zig build test`) — for pure, platform-independent logic in
|
- **Host unit tests** (`zig build test`) — for pure, platform-independent logic in
|
||||||
the shared `danos` module (the handoff layout in `src/root.zig`). These compile
|
the shared contracts (`system/boot-handoff.zig`, `system/abi.zig`,
|
||||||
for the host and run natively.
|
`system/devices/device-abi.zig`), which also compile-checks the three-way split
|
||||||
|
stays self-consistent. These compile for the host and run natively.
|
||||||
- **QEMU integration tests** (`python3 test/qemu_test.py`) — boot the real kernel
|
- **QEMU integration tests** (`python3 test/qemu_test.py`) — boot the real kernel
|
||||||
and check its behaviour. This is the interesting part.
|
and check its behaviour. This is the interesting part.
|
||||||
|
|
||||||
@@ -17,7 +18,7 @@ There are two layers:
|
|||||||
|
|
||||||
The framebuffer console draws pixels, which a test can't read without
|
The framebuffer console draws pixels, which a test can't read without
|
||||||
screen-scraping. So the kernel also writes everything to a **serial port**
|
screen-scraping. So the kernel also writes everything to a **serial port**
|
||||||
(`src/kernel/arch/x86_64/serial.zig`, a 16550 UART on COM1). `Console.write` mirrors every
|
(`system/kernel/architecture/x86_64/serial.zig`, a 16550 UART on COM1). `Console.write` mirrors every
|
||||||
byte to it, so all kernel output — boot log, memory summary, exception reports —
|
byte to it, so all kernel output — boot log, memory summary, exception reports —
|
||||||
appears on serial as plain text.
|
appears on serial as plain text.
|
||||||
|
|
||||||
@@ -29,7 +30,7 @@ a new architecture's UART is what makes the same tests run there.
|
|||||||
## In-kernel test cases
|
## In-kernel test cases
|
||||||
|
|
||||||
Building with `-Dtest-case=<name>` makes the kernel, after normal bring-up, run one
|
Building with `-Dtest-case=<name>` makes the kernel, after normal bring-up, run one
|
||||||
self-test from `src/kernel/tests.zig` instead of idling. Each case writes structured
|
self-test from `system/kernel/tests.zig` instead of idling. Each case writes structured
|
||||||
markers to serial:
|
markers to serial:
|
||||||
|
|
||||||
```
|
```
|
||||||
@@ -108,7 +109,7 @@ firmware, boot method, serial device). The cases are architecture-neutral —
|
|||||||
So bringing up a second architecture — an AArch64 Raspberry Pi is the motivating
|
So bringing up a second architecture — an AArch64 Raspberry Pi is the motivating
|
||||||
one — means:
|
one — means:
|
||||||
|
|
||||||
1. implement `src/kernel/arch/aarch64/` (CPU ops, its UART, exception vectors, page
|
1. implement `system/kernel/arch/aarch64/` (CPU ops, its UART, exception vectors, page
|
||||||
tables) behind the same `arch` interface,
|
tables) behind the same `arch` interface,
|
||||||
2. add an `aarch64` entry to `ARCHES` with its `qemu-system-aarch64` invocation,
|
2. add an `aarch64` entry to `ARCHES` with its `qemu-system-aarch64` invocation,
|
||||||
|
|
||||||
@@ -118,7 +119,7 @@ architectures".
|
|||||||
|
|
||||||
## Writing a new case
|
## Writing a new case
|
||||||
|
|
||||||
1. Add a function to `src/kernel/tests.zig` and dispatch it in `run` on its name.
|
1. Add a function to `system/kernel/tests.zig` and dispatch it in `run` on its name.
|
||||||
2. Emit `[PASS]/[FAIL]` lines and a `DANOS-TEST-RESULT:` line (non-faulting cases),
|
2. Emit `[PASS]/[FAIL]` lines and a `DANOS-TEST-RESULT:` line (non-faulting cases),
|
||||||
or trigger the condition and rely on the handler's output (faulting cases).
|
or trigger the condition and rely on the handler's output (faulting cases).
|
||||||
3. Add an entry to `CASES` in `test/qemu_test.py` with the regex that proves it.
|
3. Add an entry to `CASES` in `test/qemu_test.py` with the regex that proves it.
|
||||||
|
|||||||
+13
-4
@@ -81,11 +81,20 @@ prerequisites.
|
|||||||
(with boot-services memory reclaimed), [paging](paging.md) with W^X, [exceptions and
|
(with boot-services memory reclaimed), [paging](paging.md) with W^X, [exceptions and
|
||||||
interrupts](interrupts.md), a [calibrated timer + ns clock](device-interrupts.md), a
|
interrupts](interrupts.md), a [calibrated timer + ns clock](device-interrupts.md), a
|
||||||
[heap](heap.md), a [fixed-priority preemptive scheduler](scheduling.md) with blocking,
|
[heap](heap.md), a [fixed-priority preemptive scheduler](scheduling.md) with blocking,
|
||||||
and in-kernel [IPC channels](ipc.md) — plus a [test harness](testing.md).
|
in-kernel [IPC channels](ipc.md), SMP (all cores scheduling, with affinity), a
|
||||||
|
**higher-half kernel** with a physmap, and **user space**: per-process address
|
||||||
|
spaces, `syscall`/`sysret` with the `swapgs` discipline, a user-ELF loader, and
|
||||||
|
`/system/services/init` — a real user ELF built from `system/services/init/`, running at CPL 3 as PID 1 on its
|
||||||
|
own page tables — plus a [test harness](testing.md).
|
||||||
|
|
||||||
- **Isolation track** — **user mode + address-space isolation** (higher-half kernel,
|
- **Isolation track** — **user mode + address-space isolation**. *Done: a
|
||||||
ring 3, per-process page tables). The substrate everything else needs. *Next, and a
|
higher-half kernel with a physmap (the low half is user space), per-process
|
||||||
prerequisite for the resilience and driver tracks.*
|
address spaces with CR3 switched on context switch, the `swapgs` discipline,
|
||||||
|
`syscall`/`sysret`, a user-ELF loader, and `/system/services/init` running as a real
|
||||||
|
preemptive ring-3 process (PID 1). Remaining polish: an address-space/stack
|
||||||
|
reaper for exited tasks, SMAP + fault-recovering copy-in/out, the real IPC
|
||||||
|
syscalls (IPC_Call/IPC_ReplyWait — they arrive with the second user server),
|
||||||
|
and TLB shootdown once a process has more than one thread.*
|
||||||
- **Resilience track** — fault → kill → notify, a supervisor/reincarnation server,
|
- **Resilience track** — fault → kill → notify, a supervisor/reincarnation server,
|
||||||
resource cleanup on death, then a restartable driver as proof. Needs isolation.
|
resource cleanup on death, then a restartable driver as proof. Needs isolation.
|
||||||
See [resilience.md](resilience.md).
|
See [resilience.md](resilience.md).
|
||||||
|
|||||||
@@ -0,0 +1,80 @@
|
|||||||
|
//! /lib/mmio — typed volatile MMIO register access, plus the memory-ordering
|
||||||
|
//! barriers a device driver needs. Used by drivers on top of an `mmio_map` grant.
|
||||||
|
//!
|
||||||
|
//! **`volatile` is not a barrier.** In Zig it means only: don't elide this access, and
|
||||||
|
//! don't reorder it against *other volatile* accesses. It says nothing about ordinary
|
||||||
|
//! stores — the DMA descriptor you just filled in write-back RAM — which the compiler
|
||||||
|
//! (and, on weakly-ordered hardware, the CPU) may freely move past a volatile MMIO
|
||||||
|
//! write. The canonical bug:
|
||||||
|
//!
|
||||||
|
//! ring[i] = descriptor; // ordinary store to WB RAM
|
||||||
|
//! doorbell.* = i; // volatile store to UC MMIO
|
||||||
|
//! // nothing orders these; the device can read a stale descriptor
|
||||||
|
//!
|
||||||
|
//! Put a `wmb()` between them. The barriers lower per-architecture — which is the whole
|
||||||
|
//! reason they are a named primitive and not scattered `asm volatile`:
|
||||||
|
//!
|
||||||
|
//! x86_64 aarch64
|
||||||
|
//! mb() mfence dsb sy
|
||||||
|
//! rmb() lfence dsb ld
|
||||||
|
//! wmb() sfence dsb st
|
||||||
|
//!
|
||||||
|
//! x86 is forgiving (TSO + strong-uncacheable MMIO), so a compiler barrier usually
|
||||||
|
//! suffices; ARM is not, and ARM is the win condition (docs/vision.md) — so the
|
||||||
|
//! abstraction exists now, while there is one caller (hpet) to get right. See
|
||||||
|
//! docs/driver-model.md (M14) for the full ordering contract.
|
||||||
|
|
||||||
|
const builtin = @import("builtin");
|
||||||
|
|
||||||
|
/// Read a register of type `T` at absolute virtual address `addr` — a location inside
|
||||||
|
/// a device's `mmio_map` grant. `volatile`: never elided, never reordered against
|
||||||
|
/// another volatile access.
|
||||||
|
pub inline fn read(comptime T: type, addr: usize) T {
|
||||||
|
return @as(*const volatile T, @ptrFromInt(addr)).*;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write `value` of type `T` to the register at absolute virtual address `addr`.
|
||||||
|
pub inline fn write(comptime T: type, addr: usize, value: T) void {
|
||||||
|
@as(*volatile T, @ptrFromInt(addr)).* = value;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Full barrier: all loads and stores before it are globally visible before any after
|
||||||
|
/// it. Use when an MMIO write must complete before a following read.
|
||||||
|
pub inline fn mb() void {
|
||||||
|
switch (builtin.target.cpu.arch) {
|
||||||
|
.x86_64 => asm volatile ("mfence" ::: .{ .memory = true }),
|
||||||
|
.aarch64 => asm volatile ("dsb sy" ::: .{ .memory = true }),
|
||||||
|
else => @compileError("mmio.mb: unsupported architecture"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read barrier: loads before it complete before loads after it. Use after an IRQ
|
||||||
|
/// wake, before reading what the device wrote to shared memory.
|
||||||
|
pub inline fn rmb() void {
|
||||||
|
switch (builtin.target.cpu.arch) {
|
||||||
|
.x86_64 => asm volatile ("lfence" ::: .{ .memory = true }),
|
||||||
|
.aarch64 => asm volatile ("dsb ld" ::: .{ .memory = true }),
|
||||||
|
else => @compileError("mmio.rmb: unsupported architecture"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write barrier: stores before it become visible before stores after it. Use between
|
||||||
|
/// filling a DMA descriptor in RAM and ringing the device's doorbell.
|
||||||
|
pub inline fn wmb() void {
|
||||||
|
switch (builtin.target.cpu.arch) {
|
||||||
|
.x86_64 => asm volatile ("sfence" ::: .{ .memory = true }),
|
||||||
|
.aarch64 => asm volatile ("dsb st" ::: .{ .memory = true }),
|
||||||
|
else => @compileError("mmio.wmb: unsupported architecture"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
test "barriers emit and registers round-trip through a RAM cell" {
|
||||||
|
// The barriers must at least assemble for the host arch; ordering can't be unit
|
||||||
|
// tested, but a missing/mistyped mnemonic is caught here.
|
||||||
|
wmb();
|
||||||
|
rmb();
|
||||||
|
mb();
|
||||||
|
var cell: u64 = 0;
|
||||||
|
write(u64, @intFromPtr(&cell), 0xDEAD_BEEF);
|
||||||
|
try @import("std").testing.expectEqual(@as(u64, 0xDEAD_BEEF), read(u64, @intFromPtr(&cell)));
|
||||||
|
}
|
||||||
@@ -0,0 +1,13 @@
|
|||||||
|
//! DanOS's POSIX / C compatibility layer — `unistd`, `stdio`, and (later) the C
|
||||||
|
//! `errno` / `struct stat` / `extern "C"` surface. This is the *one* place POSIX and
|
||||||
|
//! C spellings are allowed to appear verbatim (see docs/coding-standards.md): a file
|
||||||
|
//! under library/posix/ *is* the foreign ABI, so it keeps the ABI's names. Everything
|
||||||
|
//! it touches on the danos side (the VFS protocol, the runtime) uses danos names,
|
||||||
|
//! which this layer translates to at the boundary.
|
||||||
|
//!
|
||||||
|
//! It is layered strictly *over* the runtime: it calls the runtime's IPC and heap,
|
||||||
|
//! never the kernel's system calls directly. danos-native applications use the
|
||||||
|
//! runtime; this exists so *POSIX* software can too.
|
||||||
|
|
||||||
|
pub const unistd = @import("unistd.zig");
|
||||||
|
pub const stdio = @import("stdio.zig");
|
||||||
@@ -0,0 +1,115 @@
|
|||||||
|
//! A small C stdio layer over the POSIX-style file API (unistd.zig). Unbuffered
|
||||||
|
//! for now — each fread/fwrite is one VFS round trip; an internal buffer (fewer
|
||||||
|
//! IPC calls) is a later optimisation. Both a Zig-callable API and `extern "C"`
|
||||||
|
//! symbols are provided, so Zig and future C programs share it.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const unistd = @import("unistd.zig");
|
||||||
|
const heap = @import("runtime").heap;
|
||||||
|
|
||||||
|
pub const SEEK_SET = unistd.SEEK_SET;
|
||||||
|
pub const SEEK_CURRENT = unistd.SEEK_CURRENT;
|
||||||
|
pub const SEEK_END = unistd.SEEK_END;
|
||||||
|
|
||||||
|
/// A C `FILE`: an fd plus sticky end-of-file / error flags. Allocated on the
|
||||||
|
/// heap; `fclose` frees it.
|
||||||
|
pub const FILE = extern struct {
|
||||||
|
fd: i32,
|
||||||
|
eof: c_int = 0,
|
||||||
|
err: c_int = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
fn flagsFor(mode: []const u8) u32 {
|
||||||
|
if (mode.len == 0) return 0;
|
||||||
|
return switch (mode[0]) {
|
||||||
|
'w', 'a' => unistd.O_CREAT,
|
||||||
|
else => 0,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Open `path` in `mode` ("r"/"w"/"a", '+' ignored for now). Returns null on error.
|
||||||
|
pub fn fopen(path: []const u8, mode: []const u8) ?*FILE {
|
||||||
|
const fd = unistd.open(path, flagsFor(mode));
|
||||||
|
if (fd < 0) return null;
|
||||||
|
const f = heap.allocator().create(FILE) catch {
|
||||||
|
unistd.close(fd);
|
||||||
|
return null;
|
||||||
|
};
|
||||||
|
f.* = .{ .fd = fd };
|
||||||
|
if (mode.len > 0 and mode[0] == 'a') _ = unistd.lseek(fd, 0, unistd.SEEK_END);
|
||||||
|
return f;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn fclose(f: *FILE) c_int {
|
||||||
|
unistd.close(f.fd);
|
||||||
|
heap.allocator().destroy(f);
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read `size*nmemb` bytes; returns the number of whole items read.
|
||||||
|
pub fn fread(buffer: []u8, size: usize, nmemb: usize, f: *FILE) usize {
|
||||||
|
const total = size * nmemb;
|
||||||
|
if (total == 0) return 0;
|
||||||
|
const n = unistd.read(f.fd, buffer[0..@min(buffer.len, total)]);
|
||||||
|
if (n <= 0) {
|
||||||
|
f.eof = 1;
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
return @as(usize, @intCast(n)) / size;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write `size*nmemb` bytes; returns the number of whole items written.
|
||||||
|
pub fn fwrite(data: []const u8, size: usize, nmemb: usize, f: *FILE) usize {
|
||||||
|
const total = @min(data.len, size * nmemb);
|
||||||
|
if (total == 0) return 0;
|
||||||
|
const n = unistd.write(f.fd, data[0..total]);
|
||||||
|
if (n <= 0) {
|
||||||
|
f.err = 1;
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
return @as(usize, @intCast(n)) / size;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn fseek(f: *FILE, off: i64, whence: u32) c_int {
|
||||||
|
f.eof = 0;
|
||||||
|
return if (unistd.lseek(f.fd, off, whence) < 0) -1 else 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn ftell(f: *FILE) i64 {
|
||||||
|
return unistd.lseek(f.fd, 0, unistd.SEEK_CURRENT);
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn rewind(f: *FILE) void {
|
||||||
|
_ = fseek(f, 0, SEEK_SET);
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn feof(f: *FILE) c_int {
|
||||||
|
return f.eof;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn ferror(f: *FILE) c_int {
|
||||||
|
return f.err;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn fputs(s: []const u8, f: *FILE) c_int {
|
||||||
|
return if (unistd.write(f.fd, s) < 0) -1 else 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn fputc(c: u8, f: *FILE) c_int {
|
||||||
|
const b = [_]u8{c};
|
||||||
|
return if (unistd.write(f.fd, &b) == 1) c else -1;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn fgetc(f: *FILE) c_int {
|
||||||
|
var b: [1]u8 = undefined;
|
||||||
|
const n = unistd.read(f.fd, &b);
|
||||||
|
if (n <= 0) {
|
||||||
|
f.eof = 1;
|
||||||
|
return -1; // EOF
|
||||||
|
}
|
||||||
|
return b[0];
|
||||||
|
}
|
||||||
|
|
||||||
|
// Real `extern "C"` symbols (fopen/fread/fseek/...) — with a C-string signature
|
||||||
|
// distinct from the Zig slice API above — land with the first C program, wired
|
||||||
|
// via @export so they don't collide with these Zig names.
|
||||||
@@ -0,0 +1,148 @@
|
|||||||
|
//! POSIX-style file API for user programs — the low level under C stdio. Files
|
||||||
|
//! are named objects served by the user-space VFS server (system/services/vfs/vfs.zig); each
|
||||||
|
//! call marshals a request, IPC_Calls the VFS, and unmarshals the reply. The
|
||||||
|
//! kernel knows nothing of files or fds — the fd table lives here, per process.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const protocol = @import("vfs-protocol");
|
||||||
|
const ipc = @import("runtime").ipc;
|
||||||
|
|
||||||
|
pub const O_CREAT = protocol.create;
|
||||||
|
pub const SEEK_SET: u32 = 0;
|
||||||
|
pub const SEEK_CURRENT: u32 = 1;
|
||||||
|
pub const SEEK_END: u32 = 2;
|
||||||
|
|
||||||
|
// Resolve (and cache) the VFS server endpoint, looked up by well-known id.
|
||||||
|
var vfs_handle: usize = 0;
|
||||||
|
var vfs_resolved = false;
|
||||||
|
fn vfs() ?usize {
|
||||||
|
if (!vfs_resolved) {
|
||||||
|
vfs_handle = ipc.lookup(.vfs) orelse return null;
|
||||||
|
vfs_resolved = true;
|
||||||
|
}
|
||||||
|
return vfs_handle;
|
||||||
|
}
|
||||||
|
|
||||||
|
const maximum_fds = 32;
|
||||||
|
const Fd = struct { used: bool = false, node: u64 = 0, offset: u64 = 0 };
|
||||||
|
var fds = [_]Fd{.{}} ** maximum_fds;
|
||||||
|
|
||||||
|
fn allocFd() ?usize {
|
||||||
|
for (&fds, 0..) |*f, i| {
|
||||||
|
if (!f.used) {
|
||||||
|
f.* = .{ .used = true };
|
||||||
|
return i;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
const Result = struct { reply: protocol.Reply, payload: []u8 };
|
||||||
|
|
||||||
|
/// One request/reply round trip: [Request header][send payload] -> VFS ->
|
||||||
|
/// [Reply header][receive payload]. The receive payload is written into `out`.
|
||||||
|
fn transact(request: protocol.Request, send: []const u8, out: []u8) ?Result {
|
||||||
|
const h = vfs() orelse return null;
|
||||||
|
var message: [protocol.message_maximum]u8 = undefined;
|
||||||
|
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||||
|
const slen = @min(send.len, protocol.maximum_payload);
|
||||||
|
@memcpy(message[protocol.request_size..][0..slen], send[0..slen]);
|
||||||
|
|
||||||
|
var rbuf: [protocol.message_maximum]u8 = undefined;
|
||||||
|
const n = ipc.call(h, message[0 .. protocol.request_size + slen], &rbuf) catch return null;
|
||||||
|
if (n < protocol.reply_size) return null;
|
||||||
|
const reply = std.mem.bytesToValue(protocol.Reply, rbuf[0..protocol.reply_size]);
|
||||||
|
const rpl = @min(n - protocol.reply_size, out.len);
|
||||||
|
@memcpy(out[0..rpl], rbuf[protocol.reply_size..][0..rpl]);
|
||||||
|
return .{ .reply = reply, .payload = out[0..rpl] };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Open (or create, with O_CREAT) `path`; returns an fd or -1.
|
||||||
|
pub fn open(path: []const u8, flags: u32) i32 {
|
||||||
|
const fd = allocFd() orelse return -1;
|
||||||
|
const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = flags };
|
||||||
|
const r = transact(request, path, &.{}) orelse {
|
||||||
|
fds[fd].used = false;
|
||||||
|
return -1;
|
||||||
|
};
|
||||||
|
if (r.reply.status != 0) {
|
||||||
|
fds[fd].used = false;
|
||||||
|
return -1;
|
||||||
|
}
|
||||||
|
fds[fd] = .{ .used = true, .node = r.reply.node, .offset = 0 };
|
||||||
|
return @intCast(fd);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn fdPtr(fd: i32) ?*Fd {
|
||||||
|
if (fd < 0 or fd >= maximum_fds) return null;
|
||||||
|
const f = &fds[@intCast(fd)];
|
||||||
|
return if (f.used) f else null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read up to `buffer.len` bytes at the current offset; returns the count or -1.
|
||||||
|
pub fn read(fd: i32, buffer: []u8) isize {
|
||||||
|
const f = fdPtr(fd) orelse return -1;
|
||||||
|
const want: u32 = @intCast(@min(buffer.len, protocol.maximum_payload));
|
||||||
|
const request = protocol.Request{ .operation = .read, .node = f.node, .offset = f.offset, .len = want, .flags = 0 };
|
||||||
|
const r = transact(request, &.{}, buffer) orelse return -1;
|
||||||
|
if (r.reply.status != 0) return -1;
|
||||||
|
f.offset += r.reply.len;
|
||||||
|
return @intCast(r.reply.len);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write `data` at the current offset; returns the count or -1.
|
||||||
|
pub fn write(fd: i32, data: []const u8) isize {
|
||||||
|
const f = fdPtr(fd) orelse return -1;
|
||||||
|
const want: u32 = @intCast(@min(data.len, protocol.maximum_payload));
|
||||||
|
const request = protocol.Request{ .operation = .write, .node = f.node, .offset = f.offset, .len = want, .flags = 0 };
|
||||||
|
const r = transact(request, data[0..want], &.{}) orelse return -1;
|
||||||
|
if (r.reply.status != 0) return -1;
|
||||||
|
f.offset += r.reply.len;
|
||||||
|
return @intCast(r.reply.len);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Reposition the fd's offset. Returns the new offset or -1. (SEEK_END needs the
|
||||||
|
/// file size, which `stat` provides; handled by fetching it here.)
|
||||||
|
pub fn lseek(fd: i32, off: i64, whence: u32) i64 {
|
||||||
|
const f = fdPtr(fd) orelse return -1;
|
||||||
|
const base: i64 = switch (whence) {
|
||||||
|
SEEK_SET => 0,
|
||||||
|
SEEK_CURRENT => @intCast(f.offset),
|
||||||
|
SEEK_END => blk: {
|
||||||
|
const request = protocol.Request{ .operation = .status, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
||||||
|
var sbuf: [@sizeOf(protocol.FileStatus)]u8 = undefined;
|
||||||
|
const r = transact(request, &.{}, &sbuf) orelse return -1;
|
||||||
|
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.FileStatus)) return -1;
|
||||||
|
const st = std.mem.bytesToValue(protocol.FileStatus, sbuf[0..@sizeOf(protocol.FileStatus)]);
|
||||||
|
break :blk @intCast(st.size);
|
||||||
|
},
|
||||||
|
else => return -1,
|
||||||
|
};
|
||||||
|
const pos = base + off;
|
||||||
|
if (pos < 0) return -1;
|
||||||
|
f.offset = @intCast(pos);
|
||||||
|
return pos;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Stat `path`. Returns 0 or -1.
|
||||||
|
pub fn stat(path: []const u8, out: *protocol.FileStatus) i32 {
|
||||||
|
// Open, stat by node, close — simple and enough for now.
|
||||||
|
const fd = open(path, 0);
|
||||||
|
if (fd < 0) return -1;
|
||||||
|
defer close(fd);
|
||||||
|
const f = fdPtr(fd).?;
|
||||||
|
const request = protocol.Request{ .operation = .status, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
||||||
|
var sbuf: [@sizeOf(protocol.FileStatus)]u8 = undefined;
|
||||||
|
const r = transact(request, &.{}, &sbuf) orelse return -1;
|
||||||
|
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.FileStatus)) return -1;
|
||||||
|
out.* = std.mem.bytesToValue(protocol.FileStatus, sbuf[0..@sizeOf(protocol.FileStatus)]);
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Close an fd (best effort — tells the VFS to release the open file).
|
||||||
|
pub fn close(fd: i32) void {
|
||||||
|
const f = fdPtr(fd) orelse return;
|
||||||
|
const request = protocol.Request{ .operation = .close, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
||||||
|
_ = transact(request, &.{}, &.{});
|
||||||
|
f.used = false;
|
||||||
|
}
|
||||||
@@ -0,0 +1,131 @@
|
|||||||
|
//! User-space device access: enumerate the kernel's device table, claim a device,
|
||||||
|
//! map its MMIO, and bind its interrupt. A driver uses these to find and take
|
||||||
|
//! ownership of its hardware; the claim is the capability the kernel checks before
|
||||||
|
//! mapping registers or routing an IRQ.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const device_abi = @import("device-abi");
|
||||||
|
const sc = @import("system-call.zig");
|
||||||
|
|
||||||
|
pub const DeviceDescriptor = device_abi.DeviceDescriptor;
|
||||||
|
pub const ResourceDescriptor = device_abi.ResourceDescriptor;
|
||||||
|
pub const DeviceClass = device_abi.DeviceClass;
|
||||||
|
pub const ResourceKind = device_abi.ResourceKind;
|
||||||
|
|
||||||
|
inline fn failed(r: usize) bool {
|
||||||
|
return r > ~@as(usize, 0) - 4095;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Copy up to `buffer.len` device descriptors into `buffer`; returns the total count.
|
||||||
|
pub fn enumerate(buffer: []DeviceDescriptor) usize {
|
||||||
|
return sc.systemCall2(.device_enumerate, @intFromPtr(buffer.ptr), buffer.len);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Take exclusive ownership of device `id`. Returns false if taken or invalid.
|
||||||
|
pub fn claim(id: u64) bool {
|
||||||
|
return !failed(sc.systemCall1(.device_claim, id));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Map resource `resource_index` (which must be an MMIO window) of claimed device
|
||||||
|
/// `device_id` into this address space; returns the register base virtual address.
|
||||||
|
pub fn mmioMap(device_id: u64, resource_index: u64) ?usize {
|
||||||
|
const r = sc.systemCall2(.mmio_map, device_id, resource_index);
|
||||||
|
return if (failed(r)) null else r;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `DeviceDescriptor.parent` for a device with no parent.
|
||||||
|
pub const no_parent = device_abi.no_parent;
|
||||||
|
|
||||||
|
/// `DeviceDescriptor.pci_class` for a device that is not a PCI function. Set this on
|
||||||
|
/// descriptors passed to `register` unless the child really is one.
|
||||||
|
pub const no_pci_class = device_abi.no_pci_class;
|
||||||
|
|
||||||
|
/// Publish `descriptor` as a child of `parent_id`, which this process must have claimed.
|
||||||
|
/// Returns the new device id. The child is left unclaimed, so whichever driver owns
|
||||||
|
/// that class of device can `claim` it — that is how a bus hands off a device.
|
||||||
|
///
|
||||||
|
/// Every resource in `descriptor` must be **contained** in a parent resource of the same
|
||||||
|
/// kind: a sub-window of the parent's MMIO, or one of its IRQs. The kernel refuses
|
||||||
|
/// anything else, because a device descriptor is a licence to map physical memory and
|
||||||
|
/// a bus driver may only subdivide what it already owns. `descriptor.id` and `descriptor.parent`
|
||||||
|
/// are ignored. A device with no resources at all is fine — a USB device is reached
|
||||||
|
/// through its controller, not by MMIO.
|
||||||
|
pub fn register(parent_id: u64, descriptor: *const DeviceDescriptor) ?u64 {
|
||||||
|
const r = sc.systemCall2(.device_register, parent_id, @intFromPtr(descriptor));
|
||||||
|
return if (failed(r)) null else r;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Bind resource `resource_index` (which must be an IRQ) of claimed device `device_id` to
|
||||||
|
/// `endpoint`. From then on the interrupt arrives as an asynchronous notification:
|
||||||
|
/// `ipc.replyWait` on that endpoint returns with the high bit set in `badge` and the
|
||||||
|
/// low bits carrying the GSI. The kernel masks the line before waking you.
|
||||||
|
pub fn irqBind(device_id: u64, resource_index: u64, endpoint: usize) bool {
|
||||||
|
return !failed(sc.systemCall3(.irq_bind, device_id, resource_index, endpoint));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Re-arm a bound IRQ. Call this **after** quieting the device (clearing whatever
|
||||||
|
/// status register holds its line asserted) — the kernel left the line masked
|
||||||
|
/// precisely because it could not do that for you. Skip it and the interrupt never
|
||||||
|
/// fires again; call it before the device is quiet and a level-triggered line storms.
|
||||||
|
pub fn irqAck(device_id: u64, resource_index: u64) bool {
|
||||||
|
return !failed(sc.systemCall2(.irq_ack, device_id, resource_index));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The Message-Signalled Interrupt address/data a driver programs into its device's
|
||||||
|
/// MSI capability. The device raises the interrupt by writing `data` to `address`.
|
||||||
|
pub const Msi = struct { address: u64, data: u32 };
|
||||||
|
|
||||||
|
/// Set up MSI for a claimed device: the kernel allocates a per-device edge-triggered
|
||||||
|
/// vector, binds it to `endpoint` (delivered like `irqBind`, but with no mask and no
|
||||||
|
/// `irqAck` cycle), and returns the (address, data) to write into the device's MSI
|
||||||
|
/// capability — found by mmio_mapping the device's ECAM config space (resource 0) and
|
||||||
|
/// walking its capability list. Returns null on failure. Two return values (address in
|
||||||
|
/// rax, data in rdx), so a hand-written stub.
|
||||||
|
pub fn msiBind(device_id: u64, endpoint: usize) ?Msi {
|
||||||
|
var rax: usize = undefined;
|
||||||
|
var rdx: usize = undefined;
|
||||||
|
asm volatile ("syscall"
|
||||||
|
: [rax] "={rax}" (rax),
|
||||||
|
[rdx] "={rdx}" (rdx),
|
||||||
|
: [n] "{rax}" (@intFromEnum(abi.SystemCall.msi_bind)),
|
||||||
|
[a0] "{rdi}" (device_id),
|
||||||
|
[a1] "{rsi}" (endpoint),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
if (failed(rax)) return null;
|
||||||
|
return .{ .address = rax, .data = @intCast(rdx) };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read `width` bytes (1, 2, or 4) from a port in a claimed device's `io_port`
|
||||||
|
/// resource, at byte `offset` within it. Ring 3 has no direct `in`/`out`, so a legacy
|
||||||
|
/// driver (PS/2, 16550 UART) reaches its ports through this claim-gated call — each
|
||||||
|
/// access is a syscall, which is fine for the low-rate hardware that needs it. Returns
|
||||||
|
/// null if the capability check fails (device not claimed, wrong resource, out of
|
||||||
|
/// range). A device that decodes no data returns all-ones, which is a valid value, not
|
||||||
|
/// a failure.
|
||||||
|
pub fn ioRead(device_id: u64, resource_index: u64, offset: u64, width: u8) ?u32 {
|
||||||
|
const r = sc.systemCall4(.io_read, device_id, resource_index, offset, width);
|
||||||
|
return if (failed(r)) null else @intCast(r);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write `value` (its low `width` bytes, 1/2/4) to a port in a claimed device's
|
||||||
|
/// `io_port` resource, at byte `offset`. Same capability gate as `ioRead`.
|
||||||
|
pub fn ioWrite(device_id: u64, resource_index: u64, offset: u64, width: u8, value: u32) bool {
|
||||||
|
return !failed(sc.systemCall5(.io_write, device_id, resource_index, offset, width, value));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Find DeviceDescription by hid
|
||||||
|
///
|
||||||
|
/// Utility function for driver development
|
||||||
|
pub fn findDeviceDescriptorByHid(buffer: []DeviceDescriptor, hid_needle: []const u8) ?DeviceDescriptor {
|
||||||
|
const total = enumerate(buffer);
|
||||||
|
const n = @min(total, buffer.len);
|
||||||
|
for (@as([]DeviceDescriptor, buffer[0..n])) |d| {
|
||||||
|
const hid_haystack = d.hid[0..@intCast(d.hid_len)];
|
||||||
|
if (std.mem.eql(u8, hid_haystack, hid_needle)) {
|
||||||
|
return d;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return null;
|
||||||
|
}
|
||||||
@@ -0,0 +1,48 @@
|
|||||||
|
//! User-space DMA memory: `dma_alloc` / `dma_free`. A driver that programs a
|
||||||
|
//! bus-mastering engine needs a descriptor ring the device can read — memory that is
|
||||||
|
//! physically contiguous, at a physical address the driver knows, uncacheable, and
|
||||||
|
//! pinned. `mmap` gives none of those; this does. Pair it with the barriers in
|
||||||
|
//! `/lib/mmio` (fill the ring, `wmb()`, ring the doorbell). See docs/driver-model.md.
|
||||||
|
|
||||||
|
const abi = @import("abi");
|
||||||
|
const sc = @import("system-call.zig");
|
||||||
|
|
||||||
|
/// Allocation flags. `coherent` (uncacheable) is the portable default; the rest are
|
||||||
|
/// opt-in for specific hardware — see `abi`.
|
||||||
|
pub const coherent: usize = abi.dma_coherent;
|
||||||
|
pub const write_combining: usize = abi.dma_write_combining;
|
||||||
|
pub const below_4g: usize = abi.dma_below_4g;
|
||||||
|
|
||||||
|
/// A DMA allocation: the `virtual` address the CPU touches, and the `physical` address
|
||||||
|
/// to program into the device's descriptor-ring / base registers.
|
||||||
|
pub const Region = struct {
|
||||||
|
virtual: usize,
|
||||||
|
physical: usize,
|
||||||
|
};
|
||||||
|
|
||||||
|
inline fn failed(r: usize) bool {
|
||||||
|
return r > ~@as(usize, 0) - 4095;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Allocate `len` bytes of DMA-capable memory with `flags` (e.g. `coherent`, or
|
||||||
|
/// `coherent | below_4g`). Returns the virtual/physical pair, or null on failure. Two
|
||||||
|
/// return values — the virtual address in rax, the physical address in rdx — so it
|
||||||
|
/// needs a hand-written stub.
|
||||||
|
pub fn alloc(len: usize, flags: usize) ?Region {
|
||||||
|
var rax: usize = undefined;
|
||||||
|
var rdx: usize = undefined; // out: physical address
|
||||||
|
asm volatile ("syscall"
|
||||||
|
: [rax] "={rax}" (rax),
|
||||||
|
[rdx] "={rdx}" (rdx),
|
||||||
|
: [n] "{rax}" (@intFromEnum(abi.SystemCall.dma_alloc)),
|
||||||
|
[a0] "{rdi}" (len),
|
||||||
|
[a1] "{rsi}" (flags),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
if (failed(rax)) return null;
|
||||||
|
return .{ .virtual = rax, .physical = rdx };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Release a region from a prior `alloc` (`virtual` and the same `len`).
|
||||||
|
pub fn free(virtual: usize, len: usize) void {
|
||||||
|
_ = sc.systemCall2(.dma_free, virtual, len);
|
||||||
|
}
|
||||||
@@ -0,0 +1,193 @@
|
|||||||
|
//! The user-space heap: C-convention dynamic allocation (`malloc`/`free`/…) plus
|
||||||
|
//! a `std.mem.Allocator` adapter over the same free list, so both C-style code
|
||||||
|
//! and Zig `std` containers share one heap.
|
||||||
|
//!
|
||||||
|
//! The algorithm is a straight port of the kernel's first-fit free list
|
||||||
|
//! (system/kernel/heap.zig): an address-ordered singly linked list of free blocks,
|
||||||
|
//! split on allocation and coalesced with neighbours on free. The only thing
|
||||||
|
//! that changes on this side of the system_call boundary is where memory comes from
|
||||||
|
//! — `grow` asks the kernel for pages via `mmap` instead of mapping frames
|
||||||
|
//! itself, and the kernel picks the base address.
|
||||||
|
//!
|
||||||
|
//! Single-threaded and 16-byte maximum alignment, exactly like the kernel heap; a
|
||||||
|
//! lock and larger alignments come when user programs gain threads.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const system_calls = @import("system.zig");
|
||||||
|
|
||||||
|
const page_size = abi.page_size;
|
||||||
|
|
||||||
|
/// A block header, at the start of every block; while free it also links the
|
||||||
|
/// free list via `next`.
|
||||||
|
const Block = extern struct {
|
||||||
|
size: usize, // total block size in bytes, including this header; a multiple of 16
|
||||||
|
next: ?*Block, // free-list link (only meaningful while free)
|
||||||
|
};
|
||||||
|
|
||||||
|
const header_size = @sizeOf(Block); // 16
|
||||||
|
const minimum_block = header_size + 16; // smallest block worth splitting off
|
||||||
|
/// Grow granularity: one `mmap` per 64 KiB amortises the system_call.
|
||||||
|
const chunk = 64 * 1024;
|
||||||
|
|
||||||
|
var free_list: ?*Block = null;
|
||||||
|
|
||||||
|
fn alignUp(value: usize, alignment: usize) usize {
|
||||||
|
return (value + alignment - 1) & ~(alignment - 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn payloadOf(block: *Block) [*]u8 {
|
||||||
|
return @ptrFromInt(@intFromPtr(block) + header_size);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Ask the kernel for more pages and add them as a free block. Because each
|
||||||
|
/// `mmap` is an independent grant, cross-grant coalescing happens only when the
|
||||||
|
/// kernel returns adjacent bases (its arena is a bump allocator, so consecutive
|
||||||
|
/// grants usually are adjacent). Returns false if the kernel is out of memory.
|
||||||
|
fn grow(minimum_bytes: usize) bool {
|
||||||
|
const bytes = alignUp(@max(minimum_bytes, chunk), page_size);
|
||||||
|
const ret = system_calls.mmap(bytes, system_calls.PROT_READ | system_calls.PROT_WRITE);
|
||||||
|
if (system_calls.mmapFailed(ret)) return false;
|
||||||
|
|
||||||
|
const block: *Block = @ptrFromInt(ret);
|
||||||
|
block.size = bytes;
|
||||||
|
insertFree(block); // coalesces if this grant is adjacent to a prior one
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Insert a block into the address-ordered free list, coalescing with the
|
||||||
|
/// physically adjacent free blocks on either side.
|
||||||
|
fn insertFree(block: *Block) void {
|
||||||
|
var previous: ?*Block = null;
|
||||||
|
var current = free_list;
|
||||||
|
while (current) |c| : (current = c.next) {
|
||||||
|
if (@intFromPtr(c) > @intFromPtr(block)) break;
|
||||||
|
previous = c;
|
||||||
|
}
|
||||||
|
|
||||||
|
block.next = current;
|
||||||
|
if (previous) |p| p.next = block else free_list = block;
|
||||||
|
|
||||||
|
// Merge forward into `current` if they're contiguous.
|
||||||
|
if (current) |c| {
|
||||||
|
if (@intFromPtr(block) + block.size == @intFromPtr(c)) {
|
||||||
|
block.size += c.size;
|
||||||
|
block.next = c.next;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Merge `previous` forward into `block` if they're contiguous.
|
||||||
|
if (previous) |p| {
|
||||||
|
if (@intFromPtr(p) + p.size == @intFromPtr(block)) {
|
||||||
|
p.size += block.size;
|
||||||
|
p.next = block.next;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Allocate `len` bytes (16-byte aligned), or null if out of memory.
|
||||||
|
fn rawAlloc(len: usize) ?[*]u8 {
|
||||||
|
const need = alignUp(header_size + len, 16);
|
||||||
|
|
||||||
|
var attempts: u32 = 0;
|
||||||
|
while (attempts < 2) : (attempts += 1) {
|
||||||
|
var previous: ?*Block = null;
|
||||||
|
var current = free_list;
|
||||||
|
while (current) |block| : ({
|
||||||
|
previous = block;
|
||||||
|
current = block.next;
|
||||||
|
}) {
|
||||||
|
if (block.size < need) continue;
|
||||||
|
|
||||||
|
if (block.size >= need + minimum_block) {
|
||||||
|
// Split: carve `need` off the front, leave the rest free.
|
||||||
|
const rest: *Block = @ptrFromInt(@intFromPtr(block) + need);
|
||||||
|
rest.size = block.size - need;
|
||||||
|
rest.next = block.next;
|
||||||
|
if (previous) |p| p.next = rest else free_list = rest;
|
||||||
|
block.size = need;
|
||||||
|
} else {
|
||||||
|
// Take the whole block.
|
||||||
|
if (previous) |p| p.next = block.next else free_list = block.next;
|
||||||
|
}
|
||||||
|
return payloadOf(block);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Nothing fit: grow and try once more.
|
||||||
|
if (!grow(need)) return null;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn rawFree(ptr: [*]u8) void {
|
||||||
|
const block: *Block = @ptrFromInt(@intFromPtr(ptr) - header_size);
|
||||||
|
insertFree(block);
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- C ABI: the global implicit heap ---------------------------------------
|
||||||
|
// `extern "C"` symbols so future C code links the same malloc/free directly.
|
||||||
|
|
||||||
|
export fn malloc(size: usize) callconv(.c) ?*anyopaque {
|
||||||
|
if (size == 0) return null;
|
||||||
|
const p = rawAlloc(size) orelse return null;
|
||||||
|
return @ptrCast(p);
|
||||||
|
}
|
||||||
|
|
||||||
|
export fn free(ptr: ?*anyopaque) callconv(.c) void {
|
||||||
|
const p = ptr orelse return;
|
||||||
|
rawFree(@ptrCast(p));
|
||||||
|
}
|
||||||
|
|
||||||
|
export fn calloc(nmemb: usize, size: usize) callconv(.c) ?*anyopaque {
|
||||||
|
const total = std.math.mul(usize, nmemb, size) catch return null; // overflow-safe
|
||||||
|
if (total == 0) return null;
|
||||||
|
const p = rawAlloc(total) orelse return null;
|
||||||
|
@memset(p[0..total], 0);
|
||||||
|
return @ptrCast(p);
|
||||||
|
}
|
||||||
|
|
||||||
|
export fn realloc(ptr: ?*anyopaque, size: usize) callconv(.c) ?*anyopaque {
|
||||||
|
const p = ptr orelse return malloc(size);
|
||||||
|
if (size == 0) {
|
||||||
|
rawFree(@ptrCast(p));
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
const block: *Block = @ptrFromInt(@intFromPtr(p) - header_size);
|
||||||
|
const old_payload = block.size - header_size;
|
||||||
|
if (size <= old_payload) return p; // shrink/same: keep the block
|
||||||
|
const np = rawAlloc(size) orelse return null; // grow: alloc + copy + free
|
||||||
|
@memcpy(np[0..old_payload], @as([*]u8, @ptrCast(p))[0..old_payload]);
|
||||||
|
rawFree(@ptrCast(p));
|
||||||
|
return @ptrCast(np);
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- std.mem.Allocator interface (same free list) --------------------------
|
||||||
|
|
||||||
|
pub fn allocator() std.mem.Allocator {
|
||||||
|
return .{ .ptr = undefined, .vtable = &vtable };
|
||||||
|
}
|
||||||
|
|
||||||
|
const vtable = std.mem.Allocator.VTable{
|
||||||
|
.alloc = allocImpl,
|
||||||
|
.resize = resizeImpl,
|
||||||
|
.remap = remapImpl,
|
||||||
|
.free = freeImpl,
|
||||||
|
};
|
||||||
|
|
||||||
|
fn allocImpl(_: *anyopaque, len: usize, alignment: std.mem.Alignment, _: usize) ?[*]u8 {
|
||||||
|
if (alignment.toByteUnits() > 16) return null; // blocks are 16-byte aligned
|
||||||
|
return rawAlloc(len);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn resizeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, new_len: usize, _: usize) bool {
|
||||||
|
// In-place iff the new payload still fits the current block.
|
||||||
|
const block: *Block = @ptrFromInt(@intFromPtr(memory.ptr) - header_size);
|
||||||
|
return new_len + header_size <= block.size;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn remapImpl(_: *anyopaque, _: []u8, _: std.mem.Alignment, _: usize, _: usize) ?[*]u8 {
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn freeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, _: usize) void {
|
||||||
|
rawFree(memory.ptr);
|
||||||
|
}
|
||||||
@@ -0,0 +1,220 @@
|
|||||||
|
//! User-space input helpers: the client and publisher sides of the input service, so a
|
||||||
|
//! program listening for input events — or a driver broadcasting them — doesn't hand-roll
|
||||||
|
//! the IPC. Layered over `ipc` (endpoints, capability passing, `send`) and the shared
|
||||||
|
//! `input-protocol` wire format, the way `device.zig` layers over the raw `device_*` calls.
|
||||||
|
//! See system/services/input/input.zig.
|
||||||
|
//!
|
||||||
|
//! The service carries several device classes (keyboard, mouse, joystick/gamepad). A
|
||||||
|
//! **source** publishes its class with the matching method:
|
||||||
|
//! var source = input.connectSource() orelse return;
|
||||||
|
//! _ = source.publishKeyboardEvent(.{ .kind = ..., .keycode = ..., ... });
|
||||||
|
//! _ = source.publishMouseEvent(.{ ... });
|
||||||
|
//! _ = source.publishJoystickEvent(.{ ... });
|
||||||
|
//!
|
||||||
|
//! A **subscriber** either takes one class with a typed helper —
|
||||||
|
//! var keys = input.subscribeKeyboard() orelse return;
|
||||||
|
//! while (true) { const key = keys.next() orelse continue; ... }
|
||||||
|
//! — or takes several at once and inspects the tagged envelope:
|
||||||
|
//! var listener = input.subscribeAll() orelse return;
|
||||||
|
//! while (true) {
|
||||||
|
//! const event = listener.next() orelse continue;
|
||||||
|
//! if (event.asKeyboard()) |k| { ... } else if (event.asMouse()) |m| { ... }
|
||||||
|
//! }
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const ipc = @import("ipc.zig");
|
||||||
|
const system = @import("system.zig");
|
||||||
|
const protocol = @import("input-protocol");
|
||||||
|
|
||||||
|
pub const DeviceKind = protocol.DeviceKind;
|
||||||
|
pub const InputEvent = protocol.InputEvent;
|
||||||
|
pub const KeyEvent = protocol.KeyEvent;
|
||||||
|
pub const MouseEvent = protocol.MouseEvent;
|
||||||
|
pub const JoystickEvent = protocol.JoystickEvent;
|
||||||
|
pub const EventKind = protocol.EventKind;
|
||||||
|
pub const MouseEventKind = protocol.MouseEventKind;
|
||||||
|
pub const JoystickEventKind = protocol.JoystickEventKind;
|
||||||
|
pub const Keycode = protocol.Keycode;
|
||||||
|
|
||||||
|
/// Interest masks re-exported so a caller can `subscribe(input.device_keyboard |
|
||||||
|
/// input.device_mouse)`.
|
||||||
|
pub const device_keyboard = protocol.device_keyboard;
|
||||||
|
pub const device_mouse = protocol.device_mouse;
|
||||||
|
pub const device_joystick = protocol.device_joystick;
|
||||||
|
pub const device_all = protocol.device_all;
|
||||||
|
|
||||||
|
/// Look up the input service, retrying while it is still coming up. Both a subscriber and
|
||||||
|
/// a source race the service's registration at boot, so both wait for it here rather than
|
||||||
|
/// failing. Returns the service endpoint handle, or null if it never appears.
|
||||||
|
fn lookupService() ?ipc.Handle {
|
||||||
|
var attempts: usize = 0;
|
||||||
|
while (attempts < 100) : (attempts += 1) {
|
||||||
|
if (ipc.lookup(.input)) |handle| return handle;
|
||||||
|
system.sleep(50);
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- subscribing ------------------------------------------------------------
|
||||||
|
|
||||||
|
/// A subscription to the input service: our own endpoint, which the service pushes events
|
||||||
|
/// to. `next` returns each event as a tagged `InputEvent`; use `asKeyboard`/`asMouse`/
|
||||||
|
/// `asJoystick` to decode. Created with `subscribe`/`subscribeAll`; for a single device
|
||||||
|
/// class prefer the typed helpers (`subscribeKeyboard`, ...), which return decoded events.
|
||||||
|
pub const Subscriber = struct {
|
||||||
|
/// The endpoint the service delivers events to (created and owned by us; its handle
|
||||||
|
/// was handed to the service as a capability at subscribe time).
|
||||||
|
endpoint: ipc.Handle,
|
||||||
|
receive: [protocol.event_size]u8 = undefined,
|
||||||
|
|
||||||
|
/// Block until the next event is pushed, and return it. Events arrive as asynchronous
|
||||||
|
/// buffered messages (`ipc_send` from the service), so nothing is owed in reply — the
|
||||||
|
/// empty reply this issues is a harmless no-op. Returns null for any non-event wake-up
|
||||||
|
/// (there should be none), so callers can loop.
|
||||||
|
pub fn next(self: *Subscriber) ?InputEvent {
|
||||||
|
const got = ipc.replyWait(self.endpoint, &.{}, &self.receive, null);
|
||||||
|
if (!got.isMessage() or got.len < protocol.event_size) return null;
|
||||||
|
return std.mem.bytesToValue(InputEvent, self.receive[0..protocol.event_size]);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Subscribe to the input classes named in `device_mask` (an OR of `device_*`, or
|
||||||
|
/// `device_all`). Creates an endpoint for the service to push to and hands it over as a
|
||||||
|
/// capability. Returns a `Subscriber` to loop `next` on, or null on failure.
|
||||||
|
pub fn subscribe(device_mask: u32) ?Subscriber {
|
||||||
|
const service = lookupService() orelse return null;
|
||||||
|
const endpoint = ipc.createIpcEndpoint() orelse return null;
|
||||||
|
|
||||||
|
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.subscribe), .device_mask = device_mask };
|
||||||
|
var reply: [protocol.reply_size]u8 = undefined;
|
||||||
|
const result = ipc.callCap(service, std.mem.asBytes(&request), &reply, endpoint) catch return null;
|
||||||
|
if (result.len < protocol.reply_size) return null;
|
||||||
|
if (std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status != 0) return null;
|
||||||
|
return .{ .endpoint = endpoint };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Subscribe to every input class (keyboard, mouse, joystick) on one stream.
|
||||||
|
pub fn subscribeAll() ?Subscriber {
|
||||||
|
return subscribe(device_all);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A subscriber filtered to keyboard events, whose `next` returns a decoded `KeyEvent`.
|
||||||
|
pub const KeyboardSubscriber = struct {
|
||||||
|
inner: Subscriber,
|
||||||
|
pub fn next(self: *KeyboardSubscriber) ?KeyEvent {
|
||||||
|
return (self.inner.next() orelse return null).asKeyboard();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A subscriber filtered to mouse events, whose `next` returns a decoded `MouseEvent`.
|
||||||
|
pub const MouseSubscriber = struct {
|
||||||
|
inner: Subscriber,
|
||||||
|
pub fn next(self: *MouseSubscriber) ?MouseEvent {
|
||||||
|
return (self.inner.next() orelse return null).asMouse();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A subscriber filtered to joystick/gamepad events, whose `next` returns a decoded
|
||||||
|
/// `JoystickEvent`.
|
||||||
|
pub const JoystickSubscriber = struct {
|
||||||
|
inner: Subscriber,
|
||||||
|
pub fn next(self: *JoystickSubscriber) ?JoystickEvent {
|
||||||
|
return (self.inner.next() orelse return null).asJoystick();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Subscribe to keyboard events only; `next` returns decoded `KeyEvent`s.
|
||||||
|
pub fn subscribeKeyboard() ?KeyboardSubscriber {
|
||||||
|
return .{ .inner = subscribe(device_keyboard) orelse return null };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Subscribe to mouse events only; `next` returns decoded `MouseEvent`s.
|
||||||
|
pub fn subscribeMouse() ?MouseSubscriber {
|
||||||
|
return .{ .inner = subscribe(device_mouse) orelse return null };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Subscribe to joystick/gamepad events only; `next` returns decoded `JoystickEvent`s.
|
||||||
|
pub fn subscribeJoystick() ?JoystickSubscriber {
|
||||||
|
return .{ .inner = subscribe(device_joystick) orelse return null };
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- publishing -------------------------------------------------------------
|
||||||
|
|
||||||
|
/// A connection to the input service for a source (a keyboard/mouse/joystick driver) that
|
||||||
|
/// publishes events. Each `publish*Event` is a short synchronous call the service answers
|
||||||
|
/// at once; its own fan-out to subscribers is asynchronous, so publishing never blocks on
|
||||||
|
/// a slow subscriber.
|
||||||
|
pub const Publisher = struct {
|
||||||
|
service: ipc.Handle,
|
||||||
|
|
||||||
|
fn publish(self: Publisher, event: InputEvent) bool {
|
||||||
|
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.publish), .event = event };
|
||||||
|
var reply: [protocol.reply_size]u8 = undefined;
|
||||||
|
const len = ipc.call(self.service, std.mem.asBytes(&request), &reply) catch return false;
|
||||||
|
if (len < protocol.reply_size) return false;
|
||||||
|
return std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Broadcast a keyboard event to every subscriber that took keyboard events.
|
||||||
|
pub fn publishKeyboardEvent(self: Publisher, event: KeyEvent) bool {
|
||||||
|
return self.publish(InputEvent.fromKeyboard(event));
|
||||||
|
}
|
||||||
|
/// Broadcast a mouse event to every subscriber that took mouse events.
|
||||||
|
pub fn publishMouseEvent(self: Publisher, event: MouseEvent) bool {
|
||||||
|
return self.publish(InputEvent.fromMouse(event));
|
||||||
|
}
|
||||||
|
/// Broadcast a joystick/gamepad event to every subscriber that took joystick events.
|
||||||
|
pub fn publishJoystickEvent(self: Publisher, event: JoystickEvent) bool {
|
||||||
|
return self.publish(InputEvent.fromJoystick(event));
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Connect to the input service as an event source, waiting for it to come up. Returns a
|
||||||
|
/// `Publisher`, or null if the service never registered.
|
||||||
|
pub fn connectSource() ?Publisher {
|
||||||
|
return .{ .service = lookupService() orelse return null };
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- synthetic scaffolding --------------------------------------------------
|
||||||
|
|
||||||
|
/// Synthetic key events, shared by the demo source and the keyboard driver's placeholder
|
||||||
|
/// stream while real scancode decoding is still a follow-up. `step` rolls through A..E,
|
||||||
|
/// emitting for each key a `key_down`, then a `key_press` carrying the character, then a
|
||||||
|
/// `key_up`. Scaffolding, not wire protocol — hence it lives with the helpers.
|
||||||
|
pub fn syntheticKeyEvent(step: usize) KeyEvent {
|
||||||
|
const Key = struct { code: Keycode, character: u32 };
|
||||||
|
const keys = [_]Key{
|
||||||
|
.{ .code = .a, .character = 'A' },
|
||||||
|
.{ .code = .b, .character = 'B' },
|
||||||
|
.{ .code = .c, .character = 'C' },
|
||||||
|
.{ .code = .d, .character = 'D' },
|
||||||
|
.{ .code = .e, .character = 'E' },
|
||||||
|
};
|
||||||
|
const key = keys[(step / 3) % keys.len];
|
||||||
|
return switch (step % 3) {
|
||||||
|
0 => .{ .kind = @intFromEnum(EventKind.key_down), .keycode = @intFromEnum(key.code), .character = 0, .modifiers = 0 },
|
||||||
|
1 => .{ .kind = @intFromEnum(EventKind.key_press), .keycode = @intFromEnum(key.code), .character = key.character, .modifiers = 0 },
|
||||||
|
else => .{ .kind = @intFromEnum(EventKind.key_up), .keycode = @intFromEnum(key.code), .character = 0, .modifiers = 0 },
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Synthetic mouse events (placeholder until real PS/2 packet decoding). `step` alternates
|
||||||
|
/// a small diagonal motion with a left-button click.
|
||||||
|
pub fn syntheticMouseEvent(step: usize) MouseEvent {
|
||||||
|
return switch (step % 3) {
|
||||||
|
0 => .{ .kind = @intFromEnum(MouseEventKind.motion), .button = 0, .dx = 1, .dy = 1, .scroll_x = 0, .scroll_y = 0, .buttons = 0 },
|
||||||
|
1 => .{ .kind = @intFromEnum(MouseEventKind.button_down), .button = protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = protocol.mouse_button_left },
|
||||||
|
else => .{ .kind = @intFromEnum(MouseEventKind.button_up), .button = protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = 0 },
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Synthetic joystick/gamepad events (placeholder until a real controller driver). `step`
|
||||||
|
/// sweeps axis 0 and toggles button 0.
|
||||||
|
pub fn syntheticJoystickEvent(step: usize) JoystickEvent {
|
||||||
|
return switch (step % 3) {
|
||||||
|
0 => .{ .kind = @intFromEnum(JoystickEventKind.axis), .control = 0, .value = 16384, .buttons = 0 },
|
||||||
|
1 => .{ .kind = @intFromEnum(JoystickEventKind.button_down), .control = 0, .value = 0, .buttons = 1 },
|
||||||
|
else => .{ .kind = @intFromEnum(JoystickEventKind.button_up), .control = 0, .value = 0, .buttons = 0 },
|
||||||
|
};
|
||||||
|
}
|
||||||
@@ -0,0 +1,194 @@
|
|||||||
|
//! User-space IPC helpers over the kernel's synchronous IPC syscalls. A client
|
||||||
|
//! `call`s an endpoint (send + block for reply); the VFS server and drivers are
|
||||||
|
//! reached this way. The server side (`replyWait`, which returns two values) is
|
||||||
|
//! added with the first server binary.
|
||||||
|
|
||||||
|
const abi = @import("abi");
|
||||||
|
const sc = @import("system-call.zig");
|
||||||
|
|
||||||
|
/// A small-int handle into the calling process's handle table.
|
||||||
|
pub const Handle = usize;
|
||||||
|
|
||||||
|
/// A fixed-size, register-friendly message payload. Server protocols (VFS, driver)
|
||||||
|
/// layer their own wire format on top of the bytes a call carries.
|
||||||
|
pub const Message = extern struct {
|
||||||
|
tag: u64 = 0,
|
||||||
|
a: u64 = 0,
|
||||||
|
b: u64 = 0,
|
||||||
|
c: u64 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Whether a system_call return value is a wrapped -errno (lands in the top page).
|
||||||
|
inline fn failed(r: usize) bool {
|
||||||
|
return r > ~@as(usize, 0) - 4095;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Create a new endpoint owned by this process; returns its handle.
|
||||||
|
pub fn createIpcEndpoint() ?Handle {
|
||||||
|
const r = sc.systemCall0(.create_ipc_endpoint);
|
||||||
|
return if (failed(r)) null else r;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Publish endpoint `h` under a well-known service id so other processes find it.
|
||||||
|
pub fn register(id: abi.ServiceId, h: Handle) bool {
|
||||||
|
return !failed(sc.systemCall2(.ipc_register, @intFromEnum(id), h));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Find the endpoint published under `id`, installing a handle to it in this
|
||||||
|
/// process.
|
||||||
|
pub fn lookup(id: abi.ServiceId) ?Handle {
|
||||||
|
const r = sc.systemCall1(.ipc_lookup, @intFromEnum(id));
|
||||||
|
return if (failed(r)) null else r;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const CallError = error{Failed};
|
||||||
|
|
||||||
|
/// The result of a capability-passing `callCap`: the reply length, and the handle of
|
||||||
|
/// an endpoint the server sent back (e.g. a per-device channel), or null.
|
||||||
|
pub const Reply = struct {
|
||||||
|
len: usize,
|
||||||
|
cap: ?Handle,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Send `message` to endpoint `h` and block until the server replies into `reply`,
|
||||||
|
/// optionally handing the server a capability (`send_cap`) and receiving one back.
|
||||||
|
/// This is the class-driver "open" primitive: call a bus with `send_cap = null`, get a
|
||||||
|
/// private per-device endpoint back in `.cap`. Two return values (reply length in rax,
|
||||||
|
/// received handle in r8) need a hand-written stub — r8 is read-write (in: reply
|
||||||
|
/// capacity, arg #4; out: the received handle).
|
||||||
|
pub fn callCap(h: Handle, message: []const u8, reply: []u8, send_cap: ?Handle) CallError!Reply {
|
||||||
|
var rax: usize = undefined;
|
||||||
|
var r8: usize = reply.len; // in: reply capacity (arg #4); out: received capability handle
|
||||||
|
asm volatile ("syscall"
|
||||||
|
: [rax] "={rax}" (rax),
|
||||||
|
[r8] "+{r8}" (r8),
|
||||||
|
: [n] "{rax}" (@intFromEnum(abi.SystemCall.ipc_call)),
|
||||||
|
[a0] "{rdi}" (h),
|
||||||
|
[a1] "{rsi}" (@intFromPtr(message.ptr)),
|
||||||
|
[a2] "{rdx}" (message.len),
|
||||||
|
[a3] "{r10}" (@intFromPtr(reply.ptr)),
|
||||||
|
[a5] "{r9}" (send_cap orelse abi.no_cap),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
if (failed(rax)) return error.Failed;
|
||||||
|
return .{ .len = rax, .cap = if (r8 == abi.no_cap) null else r8 };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Send `message` to endpoint `h` and block until the server replies into `reply`.
|
||||||
|
/// Returns the reply length. The common case: no capability passed either way.
|
||||||
|
pub fn call(h: Handle, message: []const u8, reply: []u8) CallError!usize {
|
||||||
|
return (try callCap(h, message, reply, null)).len;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Post `message` to endpoint `h`'s asynchronous queue and return immediately — no
|
||||||
|
/// rendezvous, no reply, no blocking. The receiver picks it up through `replyWait` as a
|
||||||
|
/// buffered message (`Received.isMessage`). Unlike `call`, this **cannot hang on a dead
|
||||||
|
/// or slow peer**, which is why a broadcaster (the input service) delivers events this
|
||||||
|
/// way. The payload must fit an endpoint slot (64 bytes); a full queue drops the oldest
|
||||||
|
/// message. Returns false on failure (bad handle, oversized payload, bad buffer).
|
||||||
|
pub fn send(h: Handle, message: []const u8) bool {
|
||||||
|
return !failed(sc.systemCall3(.ipc_send, h, @intFromPtr(message.ptr), message.len));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Set in `Received.badge` when what arrived is an asynchronous notification — a
|
||||||
|
/// bound device interrupt — rather than a client's message. The low bits carry the
|
||||||
|
/// GSI. See `isNotification`.
|
||||||
|
pub const notify_badge_bit: u64 = abi.notify_badge_bit;
|
||||||
|
|
||||||
|
/// Set alongside `notify_badge_bit` when the notification is a **signal** — the
|
||||||
|
/// lifecycle vocabulary of docs/process-lifecycle.md, delivered to the endpoint
|
||||||
|
/// nominated with `process.bindSignals`. Decode with `process.signalsFrom`.
|
||||||
|
pub const notify_signal_bit: u64 = abi.notify_signal_bit;
|
||||||
|
|
||||||
|
/// Set alongside `notify_badge_bit` when the notification is a **one-shot timer**
|
||||||
|
/// landing (`system.timerOnce`).
|
||||||
|
pub const notify_timer_bit: u64 = abi.notify_timer_bit;
|
||||||
|
|
||||||
|
/// Set alongside `notify_badge_bit` when the notification is a **child-exit
|
||||||
|
/// notice** — a process this one spawned (with an exit endpoint) has ended —
|
||||||
|
/// rather than a device interrupt. The low bits carry the child's process id.
|
||||||
|
pub const notify_exit_bit: u64 = abi.notify_exit_bit;
|
||||||
|
|
||||||
|
/// Set alongside `notify_badge_bit` when the wake-up is a **buffered message** — a payload
|
||||||
|
/// posted with `send` (`ipc_send`) — rather than a bare device interrupt or child-exit
|
||||||
|
/// notice. The payload is in the `replyWait` receive buffer (`Received.len` bytes); the
|
||||||
|
/// low bits of the badge carry the sender's task id. See `Received.isMessage`.
|
||||||
|
pub const notify_message_bit: u64 = abi.notify_message_bit;
|
||||||
|
|
||||||
|
/// The result of a `replyWait`: the request length, the sender's badge (a task id, or
|
||||||
|
/// an IRQ notification if the high bit is set), and any capability the request carried.
|
||||||
|
pub const Received = struct {
|
||||||
|
len: usize,
|
||||||
|
badge: u64,
|
||||||
|
cap: ?Handle,
|
||||||
|
|
||||||
|
/// True if this wake-up was an asynchronous notification (a device interrupt
|
||||||
|
/// or a child-exit notice), not a client request. An event loop branches on
|
||||||
|
/// this; there is no reply owed on the notification path.
|
||||||
|
pub fn isNotification(self: Received) bool {
|
||||||
|
return self.badge & notify_badge_bit != 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// True if this wake-up tells of a supervised child's end — the notification
|
||||||
|
/// requested by passing an exit endpoint to `system.spawnSupervised`.
|
||||||
|
pub fn isChildExit(self: Received) bool {
|
||||||
|
return self.isNotification() and self.badge & notify_exit_bit != 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// True if this wake-up is a **buffered message** posted with `send` (`ipc_send`):
|
||||||
|
/// there is a payload in the receive buffer (`self.len` bytes) and no reply is owed.
|
||||||
|
/// The subscriber side of a broadcast branches on this.
|
||||||
|
pub fn isMessage(self: Received) bool {
|
||||||
|
return self.isNotification() and self.badge & notify_message_bit != 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The task id of whoever posted a buffered message, meaningful only when
|
||||||
|
/// Whether this arrival is a signal notification — decode the set with
|
||||||
|
/// `process.signalsFrom(badge)`.
|
||||||
|
pub fn isSignal(self: Received) bool {
|
||||||
|
return self.isNotification() and self.badge & notify_signal_bit != 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether this arrival is a one-shot timer landing (`system.timerOnce`).
|
||||||
|
pub fn isTimer(self: Received) bool {
|
||||||
|
return self.isNotification() and self.badge & notify_timer_bit != 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `isMessage`. (The badge's low bits, with the three high marker bits masked off.)
|
||||||
|
pub fn senderTaskId(self: Received) u32 {
|
||||||
|
return @intCast(self.badge & ~(notify_badge_bit | notify_exit_bit | notify_message_bit));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The interrupt source (a GSI), meaningful only when `isNotification` and
|
||||||
|
/// not `isChildExit`.
|
||||||
|
pub fn source(self: Received) u64 {
|
||||||
|
return self.badge & ~notify_badge_bit;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The ended child's process id, meaningful only when `isChildExit`.
|
||||||
|
pub fn childProcessId(self: Received) u32 {
|
||||||
|
return @intCast(self.badge & ~(notify_badge_bit | notify_exit_bit));
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Server side of IPC_ReplyWait: deliver `reply` to the client last received (if any,
|
||||||
|
/// optionally handing it `send_cap`), then block until the next request arrives in
|
||||||
|
/// `receive`. Returns its length, the sender badge, and any capability the request
|
||||||
|
/// carried (in `.cap`). Three return values — length in rax, badge in rdx, received
|
||||||
|
/// handle in r8 — so it needs a hand-written stub: rdx is read-write (in: reply length,
|
||||||
|
/// arg #3; out: badge) and r8 is read-write (in: receive capacity, arg #4; out: handle).
|
||||||
|
pub fn replyWait(h: Handle, reply: []const u8, receive: []u8, send_cap: ?Handle) Received {
|
||||||
|
var rax: usize = undefined;
|
||||||
|
var rdx: usize = reply.len; // in: reply_len (arg #3); out: badge
|
||||||
|
var r8: usize = receive.len; // in: receive capacity (arg #4); out: received capability handle
|
||||||
|
asm volatile ("syscall"
|
||||||
|
: [rax] "={rax}" (rax),
|
||||||
|
[rdx] "+{rdx}" (rdx),
|
||||||
|
[r8] "+{r8}" (r8),
|
||||||
|
: [n] "{rax}" (@intFromEnum(abi.SystemCall.ipc_reply_wait)),
|
||||||
|
[a0] "{rdi}" (h),
|
||||||
|
[a1] "{rsi}" (@intFromPtr(reply.ptr)),
|
||||||
|
[a3] "{r10}" (@intFromPtr(receive.ptr)),
|
||||||
|
[a5] "{r9}" (send_cap orelse abi.no_cap),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
return .{ .len = rax, .badge = rdx, .cap = if (r8 == abi.no_cap) null else r8 };
|
||||||
|
}
|
||||||
@@ -0,0 +1,137 @@
|
|||||||
|
//! Process-level runtime types: what a user program receives at entry (`Init`,
|
||||||
|
//! the argv contract) and the process end of the lifecycle
|
||||||
|
//! (docs/process-lifecycle.md) — today the exit reason a supervisor reads to
|
||||||
|
//! decide restart; signals and the stop sequence land here with M17.4. Mirrors
|
||||||
|
//! the spirit of `std.process.Init.Minimal` in danos terms — std's `Args` holds
|
||||||
|
//! no data on freestanding targets, so the type is danos's own.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const sc = @import("system-call.zig");
|
||||||
|
const ipc = @import("ipc.zig");
|
||||||
|
const system = @import("system.zig");
|
||||||
|
|
||||||
|
/// Everything a program receives at entry. Passed to
|
||||||
|
/// `pub fn main(init: runtime.process.Init)`; programs that need nothing keep
|
||||||
|
/// `pub fn main() void`. An `environment` field is added here once the kernel
|
||||||
|
/// passes a non-empty envp (today it is always empty — see docs/sysv.md).
|
||||||
|
pub const Init = struct {
|
||||||
|
arguments: Arguments,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The process arguments (argc/argv), parsed from the kernel-built System V
|
||||||
|
/// entry block. The bytes live in the entry block at the top of the stack page,
|
||||||
|
/// NUL-terminated, valid for the process's lifetime.
|
||||||
|
pub const Arguments = struct {
|
||||||
|
/// argc — at least 1: argument 0 is the path or name this binary was
|
||||||
|
/// spawned as.
|
||||||
|
count: usize,
|
||||||
|
/// The argv pointers in the entry block (NULL-terminated after `count`
|
||||||
|
/// entries).
|
||||||
|
vector: [*]const [*:0]const u8,
|
||||||
|
|
||||||
|
/// Argument `index` (0 = the program's own path/name), or null if out of
|
||||||
|
/// range.
|
||||||
|
pub fn get(arguments: Arguments, index: usize) ?[:0]const u8 {
|
||||||
|
if (index >= arguments.count) return null;
|
||||||
|
return std.mem.span(arguments.vector[index]);
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn iterate(arguments: Arguments) Iterator {
|
||||||
|
return .{ .arguments = arguments };
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const Iterator = struct {
|
||||||
|
arguments: Arguments,
|
||||||
|
index: usize = 0,
|
||||||
|
|
||||||
|
pub fn next(iterator: *Iterator) ?[:0]const u8 {
|
||||||
|
const argument = iterator.arguments.get(iterator.index) orelse return null;
|
||||||
|
iterator.index += 1;
|
||||||
|
return argument;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
/// How a process ended — what a supervisor's restart policy reads: a clean exit
|
||||||
|
/// meant to stop, a fault wants a restart with backoff, killed means the
|
||||||
|
/// supervisor did it itself (docs/process-lifecycle.md).
|
||||||
|
pub const ExitReason = abi.ExitReason;
|
||||||
|
|
||||||
|
/// How dead child `id` ended. Ask after the exit notification arrives — the
|
||||||
|
/// kernel records the reason before it posts the notification, so this never
|
||||||
|
/// races it. Returns null for an id that never lived, is still alive, was
|
||||||
|
/// evicted from the kernel's bounded record, or is not this process's child
|
||||||
|
/// (the same authority gate as `kill`).
|
||||||
|
pub fn exitReason(id: u32) ?ExitReason {
|
||||||
|
const r = sc.systemCall1(.process_exit_reason, id);
|
||||||
|
if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno
|
||||||
|
return @enumFromInt(r);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The signal vocabulary (docs/process-lifecycle.md): POSIX's concepts, danos's
|
||||||
|
/// names, message delivery. A signal is a one-way coalescing statement — never a
|
||||||
|
/// question (liveness is the zero-length ping call) and never kill (that is
|
||||||
|
/// `system.kill`, unhandleable by definition).
|
||||||
|
pub const Signal = abi.Signal;
|
||||||
|
|
||||||
|
/// The coalesced set of signals one notification delivered: two pending
|
||||||
|
/// terminates arrive as one. Decode a received badge with `signalsFrom`.
|
||||||
|
pub const SignalSet = struct {
|
||||||
|
pending: u32,
|
||||||
|
|
||||||
|
pub fn has(set: SignalSet, signal: Signal) bool {
|
||||||
|
return set.pending & (@as(u32, 1) << @intFromEnum(signal)) != 0;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Nominate `endpoint` as this process's signal endpoint. Signals posted while
|
||||||
|
/// unbound have pended; they are delivered immediately on bind, coalesced.
|
||||||
|
pub fn bindSignals(endpoint: usize) bool {
|
||||||
|
return sc.systemCall1(.signal_bind, endpoint) == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode a received badge into the signals it delivered, or null if it is not
|
||||||
|
/// a signal notification.
|
||||||
|
pub fn signalsFrom(badge: u64) ?SignalSet {
|
||||||
|
if (badge & abi.notify_badge_bit == 0 or badge & abi.notify_signal_bit == 0) return null;
|
||||||
|
return .{ .pending = @truncate(badge & ~(abi.notify_badge_bit | abi.notify_signal_bit)) };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Post `signal` to child `id` (or to yourself). Supervisor-gated, like kill;
|
||||||
|
/// non-blocking, always — a statement, not a conversation.
|
||||||
|
pub fn sendSignal(id: u32, signal: Signal) bool {
|
||||||
|
return sc.systemCall2(.process_signal, id, @intFromEnum(signal)) == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The standard stop sequence (docs/process-lifecycle.md): terminate, wait up to
|
||||||
|
/// `deadline_ms` for the exit notification on `exit_endpoint` (the endpoint the
|
||||||
|
/// child was spawned with), then kill. Any *other* notifications arriving on
|
||||||
|
/// that endpoint while stopping are consumed and dropped — a supervisor with
|
||||||
|
/// concurrent traffic implements the same sequence inside its own event loop
|
||||||
|
/// (arm `system.timerOnce`, keep serving) instead of calling this.
|
||||||
|
pub fn stop(id: u32, deadline_ms: u64, exit_endpoint: usize) void {
|
||||||
|
_ = sendSignal(id, .terminate);
|
||||||
|
_ = system.timerOnce(exit_endpoint, deadline_ms);
|
||||||
|
var receive: [8]u8 = undefined;
|
||||||
|
while (true) {
|
||||||
|
const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null);
|
||||||
|
if (got.isChildExit() and got.childProcessId() == id) return;
|
||||||
|
if (got.isTimer()) break; // the deadline passed first — escalate
|
||||||
|
}
|
||||||
|
_ = system.kill(id);
|
||||||
|
while (true) {
|
||||||
|
const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null);
|
||||||
|
if (got.isChildExit() and got.childProcessId() == id) return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Subscribe `endpoint` to published exit events: every process death posts an
|
||||||
|
/// asynchronous notification with the same badge encoding as a supervisor's exit
|
||||||
|
/// notice (decode with `ipc.Received.isChildExit`/`childProcessId`). For stateful
|
||||||
|
/// services: release what the dead client held — file handles, subscriptions —
|
||||||
|
/// because a service must never depend on clients cleaning up after themselves
|
||||||
|
/// (docs/process-lifecycle.md). Ungated, like `system.processes`.
|
||||||
|
pub fn subscribeExits(endpoint: usize) bool {
|
||||||
|
return sc.systemCall1(.process_subscribe, endpoint) == 0;
|
||||||
|
}
|
||||||
@@ -0,0 +1,49 @@
|
|||||||
|
//! danos user-space runtime library — a nascent libc. Every user binary (init,
|
||||||
|
//! and later the VFS server + device drivers) imports this as `@import("runtime")`:
|
||||||
|
//! system_call wrappers, the C-convention heap, IPC helpers, and the process start
|
||||||
|
//! shim. It is compiled into each binary (inheriting its `.large` code model and
|
||||||
|
//! freestanding target), so all user programs share one implementation.
|
||||||
|
//!
|
||||||
|
//! A user binary needs three lines:
|
||||||
|
//! const runtime = @import("runtime");
|
||||||
|
//! pub const panic = runtime.panic;
|
||||||
|
//! comptime { _ = &runtime.start._start; } // pull the entry shim in
|
||||||
|
//! and a `pub fn main() void` or `pub fn main(init: runtime.process.Init) void`
|
||||||
|
//! (arguments arrive via `init`).
|
||||||
|
|
||||||
|
pub const system = @import("system.zig");
|
||||||
|
pub const heap = @import("heap.zig");
|
||||||
|
pub const ipc = @import("ipc.zig");
|
||||||
|
pub const start = @import("start.zig");
|
||||||
|
/// The VFS wire protocol (shared with the VFS server).
|
||||||
|
pub const vfs_protocol = @import("vfs-protocol");
|
||||||
|
|
||||||
|
/// The device-manager protocol: hello + tree reports (docs/device-manager.md).
|
||||||
|
pub const device_manager_protocol = @import("device-manager-protocol");
|
||||||
|
|
||||||
|
/// The power protocol: events (button, lid, battery) + shutdown (docs/m21-plan.md).
|
||||||
|
pub const power_protocol = @import("power-protocol");
|
||||||
|
/// Keyboard-event listening (subscribe/next) and broadcasting (publish), over the input
|
||||||
|
/// service. See library/runtime/input.zig and system/services/input/.
|
||||||
|
pub const input = @import("input.zig");
|
||||||
|
/// The input wire protocol (shared with the input service and its clients).
|
||||||
|
pub const input_protocol = @import("input-protocol");
|
||||||
|
/// POSIX-style file API: open/read/write/lseek/stat/close.
|
||||||
|
/// C stdio: fopen/fread/fwrite/fseek/ftell/fclose over unistd.
|
||||||
|
/// Device access for drivers: enumerate/claim/mmioMap.
|
||||||
|
pub const device = @import("device.zig");
|
||||||
|
/// DMA-capable memory for drivers: contiguous, pinned, uncacheable buffers.
|
||||||
|
pub const dma = @import("dma.zig");
|
||||||
|
|
||||||
|
/// Re-exported so a user binary can `pub const panic = runtime.panic;`.
|
||||||
|
pub const panic = start.panic;
|
||||||
|
|
||||||
|
/// Process entry types: the `Init` handed to `main`, and its `Arguments`.
|
||||||
|
pub const process = @import("process.zig");
|
||||||
|
|
||||||
|
/// The service harness: one replyWait loop folding requests, signals, and
|
||||||
|
/// notifications into callbacks (docs/process-lifecycle.md).
|
||||||
|
pub const service = @import("service.zig");
|
||||||
|
|
||||||
|
/// The heap as a `std.mem.Allocator`, for Zig `std` containers in user code.
|
||||||
|
pub const allocator = heap.allocator;
|
||||||
@@ -0,0 +1,83 @@
|
|||||||
|
//! The service harness (docs/process-lifecycle.md): one replyWait loop that
|
||||||
|
//! folds protocol requests, signals, and subscribed notifications into
|
||||||
|
//! callbacks — so the lifecycle contract ("answers ping, exits on terminate")
|
||||||
|
//! is satisfied by construction and a service author writes domain logic only.
|
||||||
|
//! Nothing is asynchronous inside the process: a callback runs at a point the
|
||||||
|
//! loop chose, never on a hijacked stack — the whole reason signals are
|
||||||
|
//! messages.
|
||||||
|
//!
|
||||||
|
//! The liveness probe: a **zero-length request is the universal ping**, answered
|
||||||
|
//! with a zero-length reply by the harness itself. No protocol's requests start
|
||||||
|
//! at length zero, so the encoding cannot collide, and there is nothing for a
|
||||||
|
//! service author to implement — a wedged service simply fails to answer, which
|
||||||
|
//! is the diagnosis (see docs/ipc.md).
|
||||||
|
|
||||||
|
const abi = @import("abi");
|
||||||
|
const ipc = @import("ipc.zig");
|
||||||
|
const process = @import("process.zig");
|
||||||
|
|
||||||
|
pub const Callbacks = struct {
|
||||||
|
/// Called once with the service's endpoint before the loop starts — the
|
||||||
|
/// place to subscribe to exit events, bind IRQs, or announce readiness.
|
||||||
|
/// Return false to abort startup (the process exits).
|
||||||
|
init: ?*const fn (endpoint: ipc.Handle) bool = null,
|
||||||
|
/// One protocol request from `sender` (a task id): write the reply into
|
||||||
|
/// `reply`, return its length. `capability` is the handle the request
|
||||||
|
/// carried, if any (M13 cap passing — how a subscriber hands over its
|
||||||
|
/// endpoint). The zero-length ping never reaches this.
|
||||||
|
on_message: *const fn (message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize,
|
||||||
|
/// A notification that is not a signal — a subscribed exit event, a bound
|
||||||
|
/// IRQ, a timer landing. The raw badge; decode with the ipc helpers.
|
||||||
|
on_notification: ?*const fn (badge: u64) void = null,
|
||||||
|
/// The reload signal. Default: ignored.
|
||||||
|
on_reload: ?*const fn () void = null,
|
||||||
|
/// The terminate signal, called before the loop returns. The clean exit is
|
||||||
|
/// the return itself — never put *necessary* work here (iron rule 1: a kill
|
||||||
|
/// arrives with no warning; this is for graceful extras only).
|
||||||
|
on_terminate: ?*const fn () void = null,
|
||||||
|
/// Publish the endpoint under a well-known service id at startup.
|
||||||
|
service: ?abi.ServiceId = null,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Run the service: create and (optionally) register the endpoint, bind signals
|
||||||
|
/// to it, call `init`, then serve until `terminate` arrives — at which point the
|
||||||
|
/// loop returns and main's return is the clean exit the supervisor reads as
|
||||||
|
/// `ExitReason.exited`. `maximum_message` sizes the receive and reply buffers
|
||||||
|
/// (a service passes its protocol's message maximum).
|
||||||
|
pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
|
||||||
|
const endpoint = ipc.createIpcEndpoint() orelse return;
|
||||||
|
if (callbacks.service) |id| {
|
||||||
|
if (!ipc.register(id, endpoint)) return;
|
||||||
|
}
|
||||||
|
_ = process.bindSignals(endpoint);
|
||||||
|
if (callbacks.init) |initialise| {
|
||||||
|
if (!initialise(endpoint)) return;
|
||||||
|
}
|
||||||
|
|
||||||
|
var reply_buffer: [maximum_message]u8 = undefined;
|
||||||
|
var reply_len: usize = 0;
|
||||||
|
var receive: [maximum_message]u8 = undefined;
|
||||||
|
while (true) {
|
||||||
|
const got = ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive, null);
|
||||||
|
if (got.isNotification()) {
|
||||||
|
reply_len = 0; // nothing owed for a notification
|
||||||
|
if (process.signalsFrom(got.badge)) |signals| {
|
||||||
|
if (signals.has(.reload)) {
|
||||||
|
if (callbacks.on_reload) |onReload| onReload();
|
||||||
|
}
|
||||||
|
if (signals.has(.terminate)) {
|
||||||
|
if (callbacks.on_terminate) |onTerminate| onTerminate();
|
||||||
|
return; // the loop's return IS the clean exit
|
||||||
|
}
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (callbacks.on_notification) |onNotification| onNotification(got.badge);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (got.len == 0) {
|
||||||
|
reply_len = 0; // the universal ping: a zero-length reply, from the harness
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
reply_len = callbacks.on_message(receive[0..got.len], &reply_buffer, got.senderTaskId(), got.cap);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,86 @@
|
|||||||
|
//! The user-space process entry shim. Every user binary roots `_start` here (via
|
||||||
|
//! `entry = _start` in build.zig) and forces this file to be analysed with
|
||||||
|
//! `comptime { _ = &runtime.start._start; }`, so the whole runtime is linked in.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const system = @import("system.zig");
|
||||||
|
const process = @import("process.zig");
|
||||||
|
|
||||||
|
/// The kernel enters at `_start` with rsp 16-aligned, pointing at the System V
|
||||||
|
/// process-entry block it built: argc, argv pointers, NULL, envp terminator, the
|
||||||
|
/// auxiliary vector, then the strings (see system/kernel/process.zig,
|
||||||
|
/// `buildEntryStack`). Capture that address in rdi — the first SysV argument —
|
||||||
|
/// before `call` disturbs the stack; the call's pushed return address also puts
|
||||||
|
/// rsp ≡ 8 (mod 16), satisfying the ABI before any Zig frame runs. The `ud2` is a
|
||||||
|
/// safety net if `rt_start` ever returns.
|
||||||
|
pub export fn _start() callconv(.naked) noreturn {
|
||||||
|
asm volatile (
|
||||||
|
\\mov %%rsp, %%rdi
|
||||||
|
\\call rt_start
|
||||||
|
\\ud2
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The first Zig frame, entered with `stack` pointing at the kernel-built entry
|
||||||
|
/// block. Build the `process.Init` from it and dispatch to the program's `main`,
|
||||||
|
/// whose signature is inspected at comptime. The heap is lazy (first alloc grows
|
||||||
|
/// it), so there is no other runtime init to order here.
|
||||||
|
export fn rt_start(stack: [*]const u64) callconv(.c) noreturn {
|
||||||
|
const init: process.Init = .{ .arguments = .{
|
||||||
|
.count = stack[0],
|
||||||
|
.vector = @ptrCast(stack + 1),
|
||||||
|
} };
|
||||||
|
system.exit(callMain(init));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Comptime-dispatch on root.main's signature, in the spirit of std's start.zig:
|
||||||
|
/// zero parameters or one `process.Init`; returns void, noreturn, u8, !void, or !u8.
|
||||||
|
fn callMain(init: process.Init) u8 {
|
||||||
|
const root = @import("root"); // the user binary's root source file
|
||||||
|
const main_information = @typeInfo(@TypeOf(root.main)).@"fn";
|
||||||
|
|
||||||
|
const call_arguments = switch (main_information.params.len) {
|
||||||
|
0 => .{},
|
||||||
|
1 => arguments: {
|
||||||
|
const Parameter = main_information.params[0].type orelse
|
||||||
|
@compileError("main's parameter must be runtime.process.Init (not anytype)");
|
||||||
|
if (Parameter != process.Init)
|
||||||
|
@compileError("main's parameter must be runtime.process.Init, found " ++ @typeName(Parameter));
|
||||||
|
break :arguments .{init};
|
||||||
|
},
|
||||||
|
else => @compileError("main takes no parameters or a single runtime.process.Init"),
|
||||||
|
};
|
||||||
|
|
||||||
|
const ReturnType = main_information.return_type.?;
|
||||||
|
switch (@typeInfo(ReturnType)) {
|
||||||
|
.noreturn => @call(.auto, root.main, call_arguments),
|
||||||
|
.void => {
|
||||||
|
@call(.auto, root.main, call_arguments);
|
||||||
|
return 0;
|
||||||
|
},
|
||||||
|
.int => {
|
||||||
|
if (ReturnType != u8)
|
||||||
|
@compileError("main's integer return type must be u8, found " ++ @typeName(ReturnType));
|
||||||
|
return @call(.auto, root.main, call_arguments);
|
||||||
|
},
|
||||||
|
.error_union => {
|
||||||
|
const payload = @call(.auto, root.main, call_arguments) catch |err| {
|
||||||
|
var buffer: [128]u8 = undefined;
|
||||||
|
const line = std.fmt.bufPrint(&buffer, "main returned error: {s}\n", .{@errorName(err)}) catch "main returned an error\n";
|
||||||
|
_ = system.write(line);
|
||||||
|
return 1; // distinct from panic's 127
|
||||||
|
};
|
||||||
|
if (@TypeOf(payload) == void) return 0;
|
||||||
|
if (@TypeOf(payload) == u8) return payload;
|
||||||
|
@compileError("main's error-union payload must be void or u8, found " ++ @typeName(@TypeOf(payload)));
|
||||||
|
},
|
||||||
|
else => @compileError("main must return void, noreturn, u8, !void, or !u8, found " ++ @typeName(ReturnType)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// No runtime to unwind into — report a panic as a nonzero exit code.
|
||||||
|
pub const panic = std.debug.FullPanic(struct {
|
||||||
|
fn panic(_: []const u8, _: ?usize) noreturn {
|
||||||
|
system.exit(127);
|
||||||
|
}
|
||||||
|
}.panic);
|
||||||
@@ -0,0 +1,67 @@
|
|||||||
|
//! Raw `system_call` instruction wrappers for user space — one per arity.
|
||||||
|
//!
|
||||||
|
//! ABI: number in rax, arguments in rdi, rsi, rdx, r10, r8, r9, result in rax.
|
||||||
|
//! The `system_call` instruction itself clobbers rcx (it holds the return rip) and
|
||||||
|
//! r11 (the saved rflags); the kernel entry stub preserves everything else.
|
||||||
|
//! Note argument #3 goes in **r10, not rcx** — rcx is unavailable across the
|
||||||
|
//! instruction, so the kernel reads the 4th argument from r10.
|
||||||
|
|
||||||
|
const abi = @import("abi");
|
||||||
|
const SystemCall = abi.SystemCall;
|
||||||
|
|
||||||
|
pub inline fn systemCall0(n: SystemCall) usize {
|
||||||
|
return asm volatile ("syscall"
|
||||||
|
: [ret] "={rax}" (-> usize),
|
||||||
|
: [n] "{rax}" (@intFromEnum(n)),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
}
|
||||||
|
|
||||||
|
pub inline fn systemCall1(n: SystemCall, a0: usize) usize {
|
||||||
|
return asm volatile ("syscall"
|
||||||
|
: [ret] "={rax}" (-> usize),
|
||||||
|
: [n] "{rax}" (@intFromEnum(n)),
|
||||||
|
[a0] "{rdi}" (a0),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
}
|
||||||
|
|
||||||
|
pub inline fn systemCall2(n: SystemCall, a0: usize, a1: usize) usize {
|
||||||
|
return asm volatile ("syscall"
|
||||||
|
: [ret] "={rax}" (-> usize),
|
||||||
|
: [n] "{rax}" (@intFromEnum(n)),
|
||||||
|
[a0] "{rdi}" (a0),
|
||||||
|
[a1] "{rsi}" (a1),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
}
|
||||||
|
|
||||||
|
pub inline fn systemCall3(n: SystemCall, a0: usize, a1: usize, a2: usize) usize {
|
||||||
|
return asm volatile ("syscall"
|
||||||
|
: [ret] "={rax}" (-> usize),
|
||||||
|
: [n] "{rax}" (@intFromEnum(n)),
|
||||||
|
[a0] "{rdi}" (a0),
|
||||||
|
[a1] "{rsi}" (a1),
|
||||||
|
[a2] "{rdx}" (a2),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
}
|
||||||
|
|
||||||
|
pub inline fn systemCall4(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: usize) usize {
|
||||||
|
return asm volatile ("syscall"
|
||||||
|
: [ret] "={rax}" (-> usize),
|
||||||
|
: [n] "{rax}" (@intFromEnum(n)),
|
||||||
|
[a0] "{rdi}" (a0),
|
||||||
|
[a1] "{rsi}" (a1),
|
||||||
|
[a2] "{rdx}" (a2),
|
||||||
|
[a3] "{r10}" (a3),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
}
|
||||||
|
|
||||||
|
pub inline fn systemCall5(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: usize, a4: usize) usize {
|
||||||
|
return asm volatile ("syscall"
|
||||||
|
: [ret] "={rax}" (-> usize),
|
||||||
|
: [n] "{rax}" (@intFromEnum(n)),
|
||||||
|
[a0] "{rdi}" (a0),
|
||||||
|
[a1] "{rsi}" (a1),
|
||||||
|
[a2] "{rdx}" (a2),
|
||||||
|
[a3] "{r10}" (a3),
|
||||||
|
[a4] "{r8}" (a4),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
}
|
||||||
@@ -0,0 +1,147 @@
|
|||||||
|
//! Typed system_call surface for user space — thin wrappers over the raw `system_call`
|
||||||
|
//! stubs, one per kernel call. Numbers come from `abi.SystemCall`, the single
|
||||||
|
//! source of truth shared with the kernel dispatcher.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const sc = @import("system-call.zig");
|
||||||
|
|
||||||
|
/// `mmap` protection flags (matching the usual C bit values). Grants are always
|
||||||
|
/// readable+writable today; the kernel does not yet honour finer prot.
|
||||||
|
pub const PROT_READ: usize = abi.prot_read;
|
||||||
|
pub const PROT_WRITE: usize = abi.prot_write;
|
||||||
|
pub const PROT_EXEC: usize = abi.prot_exec;
|
||||||
|
|
||||||
|
/// One `processes` entry — re-exported from the shared ABI so a user program can
|
||||||
|
/// declare its snapshot buffer without importing `abi` itself.
|
||||||
|
pub const ProcessDescriptor = abi.ProcessDescriptor;
|
||||||
|
|
||||||
|
/// Give up the rest of this quantum.
|
||||||
|
pub fn yield() void {
|
||||||
|
_ = sc.systemCall0(.yield);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write raw bytes to the kernel log (a bring-up diagnostic; real output goes
|
||||||
|
/// through the console/VFS later). Returns the byte count, or a wrapped -1.
|
||||||
|
pub fn write(message: []const u8) usize {
|
||||||
|
return sc.systemCall2(.debug_write, @intFromPtr(message.ptr), message.len);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Block the caller for `ms` milliseconds.
|
||||||
|
pub fn sleep(ms: usize) void {
|
||||||
|
_ = sc.systemCall1(.sleep, ms);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Arm a one-shot timer: after `ms` milliseconds the kernel posts a timer
|
||||||
|
/// notification (`ipc.Received.isTimer`) to `endpoint`. The timed wait of
|
||||||
|
/// docs/process-lifecycle.md — a service arms a deadline and keeps serving,
|
||||||
|
/// instead of blocking in sleep; what stop-sequence escalation, hello deadlines,
|
||||||
|
/// and restart backoff are built from.
|
||||||
|
pub fn timerOnce(endpoint: usize, ms: u64) bool {
|
||||||
|
return sc.systemCall2(.timer_bind, endpoint, ms) == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Monotonic nanoseconds since boot — a time source for timeouts and short delays. It
|
||||||
|
/// only ever moves forward. This is *not* wall-clock time (no date, no timezone — that
|
||||||
|
/// is a user-space service layered on top). Deadline pattern for a bounded poll loop:
|
||||||
|
///
|
||||||
|
/// const deadline = clock() + timeout_ns;
|
||||||
|
/// while (clock() < deadline) { ... }
|
||||||
|
pub fn clock() u64 {
|
||||||
|
return @intCast(sc.systemCall0(.clock));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// End the process. Never returns.
|
||||||
|
pub fn exit(code: usize) noreturn {
|
||||||
|
_ = sc.systemCall1(.exit, code);
|
||||||
|
unreachable; // the kernel never returns from exit
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Start the binary bundled in the initial-ramdisk under `name` as a new ring-3
|
||||||
|
/// process, returning the child's process id (or null on failure). The child's
|
||||||
|
/// argv[0] is `name`, and the caller becomes its **supervisor** — the only process
|
||||||
|
/// allowed to `kill` it. This is how a supervisor (the device manager) launches a
|
||||||
|
/// driver it matched — danos-native, not POSIX (a spawn/exec family comes with the
|
||||||
|
/// POSIX layer later).
|
||||||
|
pub fn spawn(name: []const u8) ?u32 {
|
||||||
|
return spawnSupervised(name, &.{}, null);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Like `spawn`, but hands the child command-line arguments: they arrive as
|
||||||
|
/// argv[1..] on its System V entry stack (argv[0] is still `name`).
|
||||||
|
pub fn spawnWithArguments(name: []const u8, arguments: []const []const u8) ?u32 {
|
||||||
|
return spawnSupervised(name, arguments, null);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The full spawn: command-line arguments for the child, and an optional endpoint
|
||||||
|
/// (a handle from `ipc.createIpcEndpoint`) the kernel notifies when the child ends
|
||||||
|
/// — any way it ends: clean exit, fault, or `kill`. The notification arrives via
|
||||||
|
/// `ipc.replyWait` as a badge with the child-exit bit set and the child's id in
|
||||||
|
/// the low bits (`ipc.Received.isChildExit`/`childProcessId`), so one endpoint can
|
||||||
|
/// supervise many children. Arguments are marshalled to the kernel as one
|
||||||
|
/// NUL-separated blob; the combined arguments must fit `blob` (the kernel caps the
|
||||||
|
/// blob at 256 bytes and argc at 8 anyway). Returns the child's process id, or
|
||||||
|
/// null on failure.
|
||||||
|
pub fn spawnSupervised(name: []const u8, arguments: []const []const u8, exit_endpoint: ?usize) ?u32 {
|
||||||
|
var blob: [256]u8 = undefined;
|
||||||
|
var len: usize = 0;
|
||||||
|
for (arguments, 0..) |argument, i| {
|
||||||
|
if (i != 0) {
|
||||||
|
if (len >= blob.len) return null;
|
||||||
|
blob[len] = 0;
|
||||||
|
len += 1;
|
||||||
|
}
|
||||||
|
if (len + argument.len > blob.len) return null;
|
||||||
|
@memcpy(blob[len..][0..argument.len], argument);
|
||||||
|
len += argument.len;
|
||||||
|
}
|
||||||
|
const r = sc.systemCall5(.system_spawn, @intFromPtr(name.ptr), name.len, if (len == 0) 0 else @intFromPtr(&blob), len, exit_endpoint orelse abi.no_cap);
|
||||||
|
if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno
|
||||||
|
return @intCast(r);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Snapshot the process table into `out` (up to its length) and return the total
|
||||||
|
/// number of live processes — which may exceed `out.len`; call again with a larger
|
||||||
|
/// buffer for the full listing. Kernel tasks are included, with an empty name.
|
||||||
|
/// The primitive `ps` is built on.
|
||||||
|
pub fn processes(out: []abi.ProcessDescriptor) usize {
|
||||||
|
return sc.systemCall2(.process_enumerate, @intFromPtr(out.ptr), out.len);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether a process spawned under `name` (its argv[0]) is currently alive.
|
||||||
|
pub fn isProcessRunning(name: []const u8) bool {
|
||||||
|
var table: [32]ProcessDescriptor = undefined;
|
||||||
|
const total = processes(&table);
|
||||||
|
for (table[0..@min(total, table.len)]) |descriptor| {
|
||||||
|
if (std.mem.eql(u8, descriptor.name[0..descriptor.name_length], name)) return true;
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// End process `id`. Only its supervisor — the process that spawned it — may;
|
||||||
|
/// anyone else gets false, as does a stale or unknown id (ids are never reused).
|
||||||
|
/// Delivery is prompt but asynchronous, like a signal: a target caught running on
|
||||||
|
/// another core dies at its next system call or timer tick. True means the kill
|
||||||
|
/// is accepted and irrevocable; the exit notification (if an endpoint was given
|
||||||
|
/// at spawn) confirms completion.
|
||||||
|
pub fn kill(id: u32) bool {
|
||||||
|
return sc.systemCall1(.process_kill, id) == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable
|
||||||
|
/// memory and return the base virtual address. On failure returns a value in the
|
||||||
|
/// top page (see `mmapFailed`). The user heap grows through this call.
|
||||||
|
pub fn mmap(len: usize, prot: usize) usize {
|
||||||
|
return sc.systemCall2(.mmap, len, prot);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Release a range previously handed out by `mmap`.
|
||||||
|
pub fn munmap(base: usize, len: usize) usize {
|
||||||
|
return sc.systemCall2(.munmap, base, len);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether an `mmap` return value is an error (the kernel returns a wrapped
|
||||||
|
/// -errno, which lands in the top page — no real grant base is ever that high).
|
||||||
|
pub inline fn mmapFailed(ret: usize) bool {
|
||||||
|
return ret > ~@as(usize, 0) - 4095;
|
||||||
|
}
|
||||||
@@ -0,0 +1,52 @@
|
|||||||
|
/* Shared link layout for every user binary (init, servers, drivers).
|
||||||
|
*
|
||||||
|
* Linked at a fixed user-space virtual base (set by `image_base` in build.zig,
|
||||||
|
* inside the kernel's user region). Same discipline as the kernel's script:
|
||||||
|
* one PT_LOAD per permission set, every section page-aligned, so the kernel's
|
||||||
|
* user-ELF loader can map each segment with exact W^X permissions. Note the
|
||||||
|
* linker also emits a read-only PT_LOAD covering the ELF headers at the image
|
||||||
|
* base, so the entry point comes from e_entry, not the base address.
|
||||||
|
*/
|
||||||
|
|
||||||
|
ENTRY(_start)
|
||||||
|
|
||||||
|
/* FLAGS bits: 1=X, 2=W, 4=R. */
|
||||||
|
PHDRS {
|
||||||
|
text PT_LOAD FLAGS(5); /* R + X */
|
||||||
|
rodata PT_LOAD FLAGS(4); /* R */
|
||||||
|
data PT_LOAD FLAGS(6); /* R + W */
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTIONS {
|
||||||
|
/* The `.large` code model (needed for the >4 GiB image base) emits code and
|
||||||
|
* data into .ltext/.lrodata/.ldata/.lbss; fold those into the matching
|
||||||
|
* permission segment alongside the normal names. */
|
||||||
|
.text ALIGN(4K) : {
|
||||||
|
*(.text .text.*)
|
||||||
|
*(.ltext .ltext.*)
|
||||||
|
} :text
|
||||||
|
|
||||||
|
.rodata ALIGN(4K) : {
|
||||||
|
*(.rodata .rodata.*)
|
||||||
|
*(.lrodata .lrodata.*)
|
||||||
|
} :rodata
|
||||||
|
|
||||||
|
.data ALIGN(4K) : {
|
||||||
|
*(.data .data.*)
|
||||||
|
*(.ldata .ldata.*)
|
||||||
|
} :data
|
||||||
|
|
||||||
|
/* .bss occupies memory but not file space; the loader zeroes the
|
||||||
|
* filesz..memsz gap. */
|
||||||
|
.bss ALIGN(4K) : {
|
||||||
|
*(.bss .bss.*)
|
||||||
|
*(.lbss .lbss.*)
|
||||||
|
*(COMMON)
|
||||||
|
} :data
|
||||||
|
|
||||||
|
/DISCARD/ : {
|
||||||
|
*(.comment)
|
||||||
|
*(.note .note.*)
|
||||||
|
*(.eh_frame .eh_frame_hdr)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,65 @@
|
|||||||
|
# xkeyboard-config — X11 keyboard layouts, compiled to Zig
|
||||||
|
|
||||||
|
This module turns a physical key (a **USB HID usage**, as the [input module](../../docs/input.md)
|
||||||
|
delivers in `KeyEvent.keycode`) plus a modifier state into a **keysym** and, when the key
|
||||||
|
produces one, a **character** (a Unicode scalar). It is what lets a `keycode` become a
|
||||||
|
`character` — a keymap — without danos shipping an X11 runtime.
|
||||||
|
|
||||||
|
The layout data comes from the X11 [xkeyboard-config](https://gitlab.freedesktop.org/xkeyboard-config/xkeyboard-config)
|
||||||
|
database, but it is **compiled to native Zig at build time** rather than parsed at runtime.
|
||||||
|
`tools/make-xkeyboard-config.py` reads the vendored xkb data and emits pure-data tables into
|
||||||
|
`generated/layouts.zig`; `xkeyboard-config.zig` is the hand-written API over them. This is
|
||||||
|
the same build-time-codegen pattern as `tools/make-initial-ramdisk.py`.
|
||||||
|
|
||||||
|
## Using it
|
||||||
|
|
||||||
|
```zig
|
||||||
|
const xkb = @import("xkeyboard-config");
|
||||||
|
|
||||||
|
const m = xkb.map(xkb.us, key_event.keycode, .{ .shift = shift_held, .caps_lock = caps });
|
||||||
|
if (m.character) |ch| { /* a printable Unicode scalar */ }
|
||||||
|
// m.keysym is always set (e.g. an X11 keysym for Return / F1 / a dead key).
|
||||||
|
|
||||||
|
const layout = xkb.byName("gb") orelse xkb.us; // choose a layout by name
|
||||||
|
for (xkb.all) |l| { /* enumerate available layouts */ }
|
||||||
|
```
|
||||||
|
|
||||||
|
`Modifiers` carries `shift`, `caps_lock`, `level3` (AltGr), and `control`. `map` selects the
|
||||||
|
level from the key's XKB *type* (the generated data) and those modifiers (the policy, in
|
||||||
|
`xkeyboard-config.zig`), so data and semantics stay separable.
|
||||||
|
|
||||||
|
Layouts: **us, gb, de, fr, es, dvorak**.
|
||||||
|
|
||||||
|
## Regenerating
|
||||||
|
|
||||||
|
```sh
|
||||||
|
python3 tools/make-xkeyboard-config.py fetch # network: download + vendor the data subset
|
||||||
|
python3 tools/make-xkeyboard-config.py generate # offline: emit generated/layouts.zig
|
||||||
|
# or, from the build:
|
||||||
|
zig build gen-xkeyboard-config
|
||||||
|
```
|
||||||
|
|
||||||
|
- **`fetch`** downloads the pinned xkeyboard-config release (version + sha256 in the script),
|
||||||
|
resolves the `include` graph for the configured layouts, and vendors *only* the symbols
|
||||||
|
files actually reached (plus `keysymdef.h`, `COPYING`, and `PROVENANCE.md`) into `vendor/`.
|
||||||
|
Run it when bumping the version or adding a layout.
|
||||||
|
- **`generate`** is deterministic and offline — same vendored input produces byte-identical
|
||||||
|
output. To add a layout, extend `TARGETS` (and `HID_TO_NAME` if a new physical key is
|
||||||
|
involved), then re-run `fetch` (to vendor any new includes) and `generate`.
|
||||||
|
|
||||||
|
## Scope
|
||||||
|
|
||||||
|
A pragmatic subset, enough for real Latin-script typing:
|
||||||
|
|
||||||
|
- **Group 1 only** — no multi-layout group switching.
|
||||||
|
- **No dead-key / compose composition** — a dead key returns its keysym with no `character`
|
||||||
|
(composing `´` + `e` → `é` is a higher layer's job).
|
||||||
|
- **Curated key types** — the common XKB types (one/two-level, alphabetic, four-level, …);
|
||||||
|
unmapped keys and unknown types fall back to level-by-shift.
|
||||||
|
- **6 layouts** — extend via `TARGETS` as above.
|
||||||
|
|
||||||
|
## Licensing
|
||||||
|
|
||||||
|
xkeyboard-config and `keysymdef.h` (xorgproto) are MIT/X11 licensed. The vendored data
|
||||||
|
subset carries the upstream `vendor/COPYING`, and `vendor/PROVENANCE.md` records the exact
|
||||||
|
version, source URL, and sha256. The generated tables are a derived work under the same terms.
|
||||||
File diff suppressed because it is too large
Load Diff
+190
@@ -0,0 +1,190 @@
|
|||||||
|
Copyright 1996 by Joseph Moss
|
||||||
|
Copyright (C) 2002-2007 Free Software Foundation, Inc.
|
||||||
|
Copyright (C) Dmitry Golubev <lastguru@mail.ru>, 2003-2004
|
||||||
|
Copyright (C) 2004, Gregory Mokhin <mokhin@bog.msu.ru>
|
||||||
|
Copyright (C) 2006 Erdal Ronahî
|
||||||
|
|
||||||
|
Permission to use, copy, modify, distribute, and sell this software and its
|
||||||
|
documentation for any purpose is hereby granted without fee, provided that
|
||||||
|
the above copyright notice appear in all copies and that both that
|
||||||
|
copyright notice and this permission notice appear in supporting
|
||||||
|
documentation, and that the name of the copyright holder(s) not be used in
|
||||||
|
advertising or publicity pertaining to distribution of the software without
|
||||||
|
specific, written prior permission. The copyright holder(s) makes no
|
||||||
|
representations about the suitability of this software for any purpose. It
|
||||||
|
is provided "as is" without express or implied warranty.
|
||||||
|
|
||||||
|
THE COPYRIGHT HOLDER(S) DISCLAIMS ALL WARRANTIES WITH REGARD TO THIS SOFTWARE,
|
||||||
|
INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS, IN NO
|
||||||
|
EVENT SHALL THE COPYRIGHT HOLDER(S) BE LIABLE FOR ANY SPECIAL, INDIRECT OR
|
||||||
|
CONSEQUENTIAL DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE,
|
||||||
|
DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER
|
||||||
|
TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR
|
||||||
|
PERFORMANCE OF THIS SOFTWARE.
|
||||||
|
|
||||||
|
|
||||||
|
Copyright (c) 1996 Digital Equipment Corporation
|
||||||
|
|
||||||
|
Permission is hereby granted, free of charge, to any person obtaining
|
||||||
|
a copy of this software and associated documentation files (the
|
||||||
|
"Software"), to deal in the Software without restriction, including
|
||||||
|
without limitation the rights to use, copy, modify, merge, publish,
|
||||||
|
distribute, sublicense, and sell copies of the Software, and to
|
||||||
|
permit persons to whom the Software is furnished to do so, subject to
|
||||||
|
the following conditions:
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice shall be included
|
||||||
|
in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS
|
||||||
|
OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||||
|
IN NO EVENT SHALL DIGITAL EQUIPMENT CORPORATION BE LIABLE FOR ANY CLAIM,
|
||||||
|
DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR
|
||||||
|
OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR
|
||||||
|
THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||||
|
|
||||||
|
Except as contained in this notice, the name of the Digital Equipment
|
||||||
|
Corporation shall not be used in advertising or otherwise to promote
|
||||||
|
the sale, use or other dealings in this Software without prior written
|
||||||
|
authorization from Digital Equipment Corporation.
|
||||||
|
|
||||||
|
|
||||||
|
Copyright 1996, 1998 The Open Group
|
||||||
|
|
||||||
|
Permission to use, copy, modify, distribute, and sell this software and its
|
||||||
|
documentation for any purpose is hereby granted without fee, provided that
|
||||||
|
the above copyright notice appear in all copies and that both that
|
||||||
|
copyright notice and this permission notice appear in supporting
|
||||||
|
documentation.
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice shall be
|
||||||
|
included in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||||
|
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||||
|
IN NO EVENT SHALL THE OPEN GROUP BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||||
|
OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||||
|
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||||
|
OTHER DEALINGS IN THE SOFTWARE.
|
||||||
|
|
||||||
|
Except as contained in this notice, the name of The Open Group shall
|
||||||
|
not be used in advertising or otherwise to promote the sale, use or
|
||||||
|
other dealings in this Software without prior written authorization
|
||||||
|
from The Open Group.
|
||||||
|
|
||||||
|
|
||||||
|
Copyright 2004-2005 Sun Microsystems, Inc. All rights reserved.
|
||||||
|
|
||||||
|
Permission is hereby granted, free of charge, to any person obtaining a
|
||||||
|
copy of this software and associated documentation files (the "Software"),
|
||||||
|
to deal in the Software without restriction, including without limitation
|
||||||
|
the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||||
|
and/or sell copies of the Software, and to permit persons to whom the
|
||||||
|
Software is furnished to do so, subject to the following conditions:
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice (including the next
|
||||||
|
paragraph) shall be included in all copies or substantial portions of the
|
||||||
|
Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||||
|
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||||
|
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||||
|
THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||||
|
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
|
||||||
|
FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||||
|
DEALINGS IN THE SOFTWARE.
|
||||||
|
|
||||||
|
|
||||||
|
Copyright (c) 1996 by Silicon Graphics Computer Systems, Inc.
|
||||||
|
|
||||||
|
Permission to use, copy, modify, and distribute this
|
||||||
|
software and its documentation for any purpose and without
|
||||||
|
fee is hereby granted, provided that the above copyright
|
||||||
|
notice appear in all copies and that both that copyright
|
||||||
|
notice and this permission notice appear in supporting
|
||||||
|
documentation, and that the name of Silicon Graphics not be
|
||||||
|
used in advertising or publicity pertaining to distribution
|
||||||
|
of the software without specific prior written permission.
|
||||||
|
Silicon Graphics makes no representation about the suitability
|
||||||
|
of this software for any purpose. It is provided "as is"
|
||||||
|
without any express or implied warranty.
|
||||||
|
|
||||||
|
SILICON GRAPHICS DISCLAIMS ALL WARRANTIES WITH REGARD TO THIS
|
||||||
|
SOFTWARE, INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY
|
||||||
|
AND FITNESS FOR A PARTICULAR PURPOSE. IN NO EVENT SHALL SILICON
|
||||||
|
GRAPHICS BE LIABLE FOR ANY SPECIAL, INDIRECT OR CONSEQUENTIAL
|
||||||
|
DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE,
|
||||||
|
DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE
|
||||||
|
OR OTHER TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH
|
||||||
|
THE USE OR PERFORMANCE OF THIS SOFTWARE.
|
||||||
|
|
||||||
|
|
||||||
|
Copyright (c) 1996 X Consortium
|
||||||
|
|
||||||
|
Permission is hereby granted, free of charge, to any person obtaining
|
||||||
|
a copy of this software and associated documentation files (the
|
||||||
|
"Software"), to deal in the Software without restriction, including
|
||||||
|
without limitation the rights to use, copy, modify, merge, publish,
|
||||||
|
distribute, sublicense, and/or sell copies of the Software, and to
|
||||||
|
permit persons to whom the Software is furnished to do so, subject to
|
||||||
|
the following conditions:
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice shall be
|
||||||
|
included in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||||
|
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||||
|
IN NO EVENT SHALL THE X CONSORTIUM BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||||
|
OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||||
|
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||||
|
OTHER DEALINGS IN THE SOFTWARE.
|
||||||
|
|
||||||
|
Except as contained in this notice, the name of the X Consortium shall
|
||||||
|
not be used in advertising or otherwise to promote the sale, use or
|
||||||
|
other dealings in this Software without prior written authorization
|
||||||
|
from the X Consortium.
|
||||||
|
|
||||||
|
|
||||||
|
Copyright (C) 2004, 2006 Ævar Arnfjörð Bjarmason <avarab@gmail.com>
|
||||||
|
|
||||||
|
Permission to use, copy, modify, distribute, and sell this software and its
|
||||||
|
documentation for any purpose is hereby granted without fee, provided that
|
||||||
|
the above copyright notice appear in all copies and that both that
|
||||||
|
copyright notice and this permission notice appear in supporting
|
||||||
|
documentation.
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice shall be
|
||||||
|
included in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||||
|
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||||
|
IN NO EVENT SHALL THE OPEN GROUP BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||||
|
OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||||
|
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||||
|
OTHER DEALINGS IN THE SOFTWARE.
|
||||||
|
|
||||||
|
Except as contained in this notice, the name of a copyright holder shall
|
||||||
|
not be used in advertising or otherwise to promote the sale, use or
|
||||||
|
other dealings in this Software without prior written authorization of
|
||||||
|
the copyright holder.
|
||||||
|
|
||||||
|
|
||||||
|
Copyright (C) 1999, 2000 by Anton Zinoviev <anton@lml.bas.bg>
|
||||||
|
|
||||||
|
This software may be used, modified, copied, distributed, and sold,
|
||||||
|
in both source and binary form provided that the above copyright
|
||||||
|
and these terms are retained. Under no circumstances is the author
|
||||||
|
responsible for the proper functioning of this software, nor does
|
||||||
|
the author assume any responsibility for damages incurred with its
|
||||||
|
use.
|
||||||
|
|
||||||
|
Permission is granted to anyone to use, distribute and modify
|
||||||
|
this file in any way, provided that the above copyright notice
|
||||||
|
is left intact and the author of the modification summarizes
|
||||||
|
the changes in this header.
|
||||||
|
|
||||||
|
This file is distributed without any expressed or implied warranty.
|
||||||
+21
@@ -0,0 +1,21 @@
|
|||||||
|
# Vendored xkeyboard-config subset
|
||||||
|
|
||||||
|
- **Package**: xkeyboard-config 2.44
|
||||||
|
- **Source**: https://gitlab.freedesktop.org/xkeyboard-config/xkeyboard-config/-/archive/xkeyboard-config-2.44/xkeyboard-config-2.44.tar.gz
|
||||||
|
- **sha256**: `35e34edeaf4e8da8d0696ff6b241ee11ddb1b8c6730bac7252d4d0a88ea5f05b`
|
||||||
|
- **keysymdef.h**: xorgproto, copied from `/opt/homebrew/include/X11/keysymdef.h`
|
||||||
|
- **License**: MIT/X11 (see COPYING)
|
||||||
|
|
||||||
|
Only the symbols files reachable from the generated layouts (tools/make-xkeyboard-config.py `TARGETS`) are vendored; regenerate with
|
||||||
|
`python3 tools/make-xkeyboard-config.py fetch` then `... generate`.
|
||||||
|
|
||||||
|
Vendored symbols files:
|
||||||
|
|
||||||
|
- `symbols/de`
|
||||||
|
- `symbols/es`
|
||||||
|
- `symbols/fr`
|
||||||
|
- `symbols/gb`
|
||||||
|
- `symbols/kpdl`
|
||||||
|
- `symbols/latin`
|
||||||
|
- `symbols/level3`
|
||||||
|
- `symbols/us`
|
||||||
+2584
File diff suppressed because it is too large
Load Diff
+1232
File diff suppressed because it is too large
Load Diff
+250
@@ -0,0 +1,250 @@
|
|||||||
|
// Keyboard layouts for Spain.
|
||||||
|
|
||||||
|
// Modified for a real Spanish keyboard by Jon Tombs.
|
||||||
|
default partial alphanumeric_keys
|
||||||
|
xkb_symbols "basic" {
|
||||||
|
|
||||||
|
include "latin(type4)"
|
||||||
|
|
||||||
|
name[Group1]="Spanish";
|
||||||
|
|
||||||
|
key <TLDE> { [ masculine, ordfeminine, backslash, backslash ] };
|
||||||
|
key <AE01> { [ 1, exclam, bar, exclamdown ] };
|
||||||
|
key <AE03> { [ 3, periodcentered, numbersign, sterling ] };
|
||||||
|
key <AE04> { [ 4, dollar, asciitilde, dollar ] };
|
||||||
|
key <AE11> { [apostrophe, question, backslash, questiondown ] };
|
||||||
|
key <AE12> { [exclamdown, questiondown, dead_cedilla, dead_ogonek] };
|
||||||
|
|
||||||
|
key <AD11> { [dead_grave, dead_circumflex, bracketleft, dead_abovering ] };
|
||||||
|
key <AD12> { [ plus, asterisk, bracketright, dead_macron ] };
|
||||||
|
|
||||||
|
key <AC10> { [ ntilde, Ntilde, dead_tilde, dead_doubleacute ] };
|
||||||
|
key <AC11> { [dead_acute, dead_diaeresis, braceleft, dead_caron ] };
|
||||||
|
key <BKSL> { [ ccedilla, Ccedilla, braceright, dead_breve ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "winkeys" {
|
||||||
|
|
||||||
|
include "es(basic)"
|
||||||
|
name[Group1]="Spanish (Windows)";
|
||||||
|
include "eurosign(5)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "nodeadkeys" {
|
||||||
|
|
||||||
|
include "es(basic)"
|
||||||
|
|
||||||
|
name[Group1]="Spanish (no dead keys)";
|
||||||
|
|
||||||
|
key <AE12> { [exclamdown, questiondown, cedilla, ogonek ] };
|
||||||
|
key <AD11> { [ grave, asciicircum, bracketleft, degree ] };
|
||||||
|
key <AD12> { [ plus, asterisk, bracketright, macron ] };
|
||||||
|
key <AC07> { [ j, J, ezh, EZH ] };
|
||||||
|
key <AC10> { [ ntilde, Ntilde, asciitilde, doubleacute ] };
|
||||||
|
key <AC11> { [ acute, diaeresis, braceleft, caron ] };
|
||||||
|
key <BKSL> { [ ccedilla, Ccedilla, braceright, breve ] };
|
||||||
|
key <AB10> { [ minus, underscore, ellipsis, abovedot ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Spanish Dvorak mapping (note R-H exchange)
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "dvorak" {
|
||||||
|
|
||||||
|
name[Group1]="Spanish (Dvorak)";
|
||||||
|
|
||||||
|
key <TLDE> {[ masculine, ordfeminine, backslash, degree ]};
|
||||||
|
key <AE01> {[ 1, exclam, bar, onesuperior ]};
|
||||||
|
key <AE02> {[ 2, quotedbl, at, twosuperior ]};
|
||||||
|
key <AE03> {[ 3, periodcentered, numbersign, threesuperior ]};
|
||||||
|
key <AE04> {[ 4, dollar, asciitilde, onequarter ]};
|
||||||
|
key <AE05> {[ 5, percent, brokenbar, fiveeighths ]};
|
||||||
|
key <AE06> {[ 6, ampersand, notsign, threequarters ]};
|
||||||
|
key <AE07> {[ 7, slash, onehalf, seveneighths ]};
|
||||||
|
key <AE08> {[ 8, parenleft, oneeighth, threeeighths ]};
|
||||||
|
key <AE09> {[ 9, parenright, asciicircum ]};
|
||||||
|
key <AE10> {[ 0, equal, grave, dead_doubleacute ]};
|
||||||
|
key <AE11> {[ apostrophe, question, dead_macron, dead_ogonek ]};
|
||||||
|
key <AE12> {[ exclamdown, questiondown, dead_breve, dead_abovedot ]};
|
||||||
|
|
||||||
|
key <AD01> {[ period, colon, less, guillemotleft ]};
|
||||||
|
key <AD02> {[ comma, semicolon, greater, guillemotright ]};
|
||||||
|
key <AD03> {[ ntilde, Ntilde, lstroke, Lstroke ]};
|
||||||
|
key <AD04> {[ p, P, paragraph ]};
|
||||||
|
key <AD05> {[ y, Y, yen ]};
|
||||||
|
key <AD06> {[ f, F, tslash, Tslash ]};
|
||||||
|
key <AD07> {[ g, G, dstroke, Dstroke ]};
|
||||||
|
key <AD08> {[ c, C, cent, copyright ]};
|
||||||
|
key <AD09> {[ h, H, hstroke, Hstroke ]};
|
||||||
|
key <AD10> {[ l, L, sterling ]};
|
||||||
|
key <AD11> {[ dead_grave, dead_circumflex, bracketleft, dead_caron ]};
|
||||||
|
key <AD12> {[ plus, asterisk, bracketright, plusminus ]};
|
||||||
|
|
||||||
|
key <AC01> {[ a, A, ae, AE ]};
|
||||||
|
key <AC02> {[ o, O, oslash, Oslash ]};
|
||||||
|
key <AC03> {[ e, E, EuroSign ]};
|
||||||
|
key <AC04> {[ u, U, aring, Aring ]};
|
||||||
|
key <AC05> {[ i, I, oe, OE ]};
|
||||||
|
key <AC06> {[ d, D, eth, ETH ]};
|
||||||
|
key <AC07> {[ r, R, registered, trademark ]};
|
||||||
|
key <AC08> {[ t, T, thorn, THORN ]};
|
||||||
|
key <AC09> {[ n, N, eng, ENG ]};
|
||||||
|
key <AC10> {[ s, S, ssharp, section ]};
|
||||||
|
key <AC11> {[ dead_acute, dead_diaeresis, braceleft, dead_tilde ]};
|
||||||
|
key <BKSL> {[ ccedilla, Ccedilla, braceright, dead_cedilla ]};
|
||||||
|
|
||||||
|
key <LSGT> {[ less, greater, guillemotleft, guillemotright ]};
|
||||||
|
key <AB01> {[ minus, underscore, hyphen, macron ]};
|
||||||
|
key <AB02> {[ q, Q, currency ]};
|
||||||
|
key <AB03> {[ j, J ]};
|
||||||
|
key <AB04> {[ k, K, kra ]};
|
||||||
|
key <AB05> {[ x, X, multiply, division ]};
|
||||||
|
key <AB06> {[ b, B ]};
|
||||||
|
key <AB07> {[ m, M, mu ]};
|
||||||
|
key <AB08> {[ w, W ]};
|
||||||
|
key <AB09> {[ v, V ]};
|
||||||
|
key <AB10> {[ z, Z ]};
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "cat" {
|
||||||
|
|
||||||
|
include "es(basic)"
|
||||||
|
|
||||||
|
name[Group1]="Catalan (Spain, with middle-dot L)";
|
||||||
|
|
||||||
|
key <AC09> { [ l, L, 0x1000140, 0x100013F ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "ast" {
|
||||||
|
|
||||||
|
include "es(basic)"
|
||||||
|
|
||||||
|
name[Group1]="Asturian (Spain, with bottom-dot H and L)";
|
||||||
|
|
||||||
|
key <AC06> { [ h, H, 0x1001E25, 0x1001E24 ] };
|
||||||
|
key <AC09> { [ l, L, 0x1001E37, 0x1001E36 ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "olpc" {
|
||||||
|
|
||||||
|
// #HW-SPECIFIC
|
||||||
|
|
||||||
|
// http://wiki.laptop.org/go/OLPC_Spanish_Keyboard
|
||||||
|
|
||||||
|
include "us(basic)"
|
||||||
|
name[Group1]="Spanish";
|
||||||
|
|
||||||
|
key <AE00> { [ masculine, ordfeminine ] };
|
||||||
|
key <AE01> { [ 1, exclam, bar ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, at ] };
|
||||||
|
key <AE03> { [ 3, dead_grave, numbersign, grave ] };
|
||||||
|
key <AE05> { [ 5, percent, asciicircum, dead_circumflex ] };
|
||||||
|
key <AE06> { [ 6, ampersand, notsign ] };
|
||||||
|
key <AE07> { [ 7, slash, backslash ] };
|
||||||
|
key <AE08> { [ 8, parenleft ] };
|
||||||
|
key <AE09> { [ 9, parenright ] };
|
||||||
|
key <AE10> { [ 0, equal ] };
|
||||||
|
key <AE11> { [ apostrophe, question ] };
|
||||||
|
key <AE12> { [ exclamdown, questiondown ] };
|
||||||
|
|
||||||
|
key <AD03> { [ e, E, EuroSign ] };
|
||||||
|
key <AD11> { [ dead_acute, dead_diaeresis, acute, dead_abovering ] };
|
||||||
|
key <AD12> { [ bracketleft, braceleft ] };
|
||||||
|
|
||||||
|
key <AC10> { [ ntilde, Ntilde ] };
|
||||||
|
key <AC11> { [ plus, asterisk, dead_tilde ] };
|
||||||
|
key <AC12> { [ bracketright, braceright, section ] };
|
||||||
|
|
||||||
|
key <AB08> { [ comma, semicolon ] };
|
||||||
|
key <AB09> { [ period, colon ] };
|
||||||
|
key <AB10> { [ minus, underscore ] };
|
||||||
|
|
||||||
|
key <I219> { [ less, greater, ISO_Next_Group ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "olpcm" {
|
||||||
|
|
||||||
|
// #HW-SPECIFIC
|
||||||
|
|
||||||
|
// Mechanical (non-membrane) OLPC Spanish keyboard layout.
|
||||||
|
// See: http://wiki.laptop.org/go/OLPC_Spanish_Non-membrane_Keyboard
|
||||||
|
|
||||||
|
include "us(basic)"
|
||||||
|
name[Group1]="Spanish";
|
||||||
|
|
||||||
|
key <AE00> { [ questiondown, exclamdown, backslash ] };
|
||||||
|
key <AE01> { [ 1, exclam, bar ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, at ] };
|
||||||
|
key <AE03> { [ 3, dead_grave, numbersign, grave ] };
|
||||||
|
key <AE04> { [ 4, dollar, asciitilde, dead_tilde ] };
|
||||||
|
key <AE05> { [ 5, percent, asciicircum, dead_circumflex ] };
|
||||||
|
key <AE06> { [ 6, ampersand, notsign ] };
|
||||||
|
key <AE07> { [ 7, slash, backslash ] }; // no '\' label on olpcm, leave for compatibility
|
||||||
|
key <AE08> { [ 8, parenleft, masculine ] };
|
||||||
|
key <AE09> { [ 9, parenright, ordfeminine ] };
|
||||||
|
key <AE10> { [ 0, equal ] };
|
||||||
|
key <AE11> { [ apostrophe, question ] };
|
||||||
|
|
||||||
|
key <AD03> { [ e, E, EuroSign ] };
|
||||||
|
key <AD11> { [ dead_acute, dead_diaeresis, dead_abovering, acute ] };
|
||||||
|
key <AD12> { [ plus, asterisk ] };
|
||||||
|
|
||||||
|
key <AC10> { [ ntilde, Ntilde ] };
|
||||||
|
// no AC11 or AC12 on olpcm
|
||||||
|
|
||||||
|
key <AB08> { [ comma, semicolon ] };
|
||||||
|
key <AB09> { [ period, colon ] };
|
||||||
|
key <AB10> { [ minus, underscore ] };
|
||||||
|
|
||||||
|
key <AA02> { [ less, greater ] };
|
||||||
|
key <AA06> { [ bracketleft, braceleft, ccedilla, Ccedilla ] };
|
||||||
|
key <AA07> { [ bracketright, braceright ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "deadtilde" {
|
||||||
|
|
||||||
|
include "es(basic)"
|
||||||
|
|
||||||
|
name[Group1]="Spanish (dead tilde)";
|
||||||
|
|
||||||
|
key <AE04> { [ 4, dollar, dead_tilde, dollar ] };
|
||||||
|
key <AC10> { [ ntilde, Ntilde, asciitilde, dead_doubleacute ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "olpc2" {
|
||||||
|
// #HW-SPECIFIC
|
||||||
|
|
||||||
|
// Modified variant of US International layout, specifically for Peru
|
||||||
|
// Contact: Sayamindu Dasgupta <sayamindu@laptop.org>
|
||||||
|
|
||||||
|
include "us(olpc)"
|
||||||
|
name[Group1]="Spanish";
|
||||||
|
|
||||||
|
key <AE03> { [ 3, numbersign, dead_grave, dead_grave] }; // combining grave
|
||||||
|
key <I236> { [ XF86Start ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
// EXTRAS:
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "sun_type6" {
|
||||||
|
include "sun_vndr/es(sun_type6)"
|
||||||
|
};
|
||||||
+1404
File diff suppressed because it is too large
Load Diff
+249
@@ -0,0 +1,249 @@
|
|||||||
|
// Keyboard layouts for Great Britain.
|
||||||
|
|
||||||
|
default partial alphanumeric_keys
|
||||||
|
xkb_symbols "basic" {
|
||||||
|
|
||||||
|
// The basic UK layout, also known as the IBM 166 layout,
|
||||||
|
// but with the useless brokenbar pushed two levels up.
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
name[Group1]="English (UK)";
|
||||||
|
|
||||||
|
key <TLDE> { [ grave, notsign, bar, bar ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, twosuperior, oneeighth ] };
|
||||||
|
key <AE03> { [ 3, sterling, threesuperior, sterling ] };
|
||||||
|
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||||
|
|
||||||
|
key <AC11> { [apostrophe, at, dead_circumflex, dead_caron] };
|
||||||
|
key <BKSL> { [numbersign, asciitilde, dead_grave, dead_breve ] };
|
||||||
|
|
||||||
|
key <LSGT> { [ backslash, bar, bar, brokenbar ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "intl" {
|
||||||
|
|
||||||
|
// A UK layout but with five accents made into dead keys:
|
||||||
|
// grave, diaeresis, circumflex, acute, and tilde.
|
||||||
|
// By Phil Jones <philjones1 at blueyonder.co.uk>.
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
name[Group1]="English (UK, intl., with dead keys)";
|
||||||
|
|
||||||
|
key <TLDE> { [ dead_grave, notsign, bar, bar ] };
|
||||||
|
key <AE02> { [ 2, dead_diaeresis, twosuperior, onehalf ] };
|
||||||
|
key <AE03> { [ 3, sterling, threesuperior, onethird ] };
|
||||||
|
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||||
|
key <AE06> { [ 6, dead_circumflex, threequarters, onesixth ] };
|
||||||
|
|
||||||
|
key <AC11> { [ dead_acute, at, apostrophe, bar ] };
|
||||||
|
key <BKSL> { [ numbersign, dead_tilde, bar, bar ] };
|
||||||
|
|
||||||
|
key <LSGT> { [ backslash, bar, bar, bar ] };
|
||||||
|
key <AB08> { [ comma, less, ccedilla, Ccedilla ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "extd" {
|
||||||
|
// Clone of the Microsoft "United Kingdom Extended" layout, which
|
||||||
|
// includes dead keys for: grave; diaeresis; circumflex; tilde; and
|
||||||
|
// accute. It also enables direct access to accute characters using
|
||||||
|
// the Multi_key (Alt Gr).
|
||||||
|
//
|
||||||
|
// Taken from...
|
||||||
|
// "Windows Keyboard Layouts"
|
||||||
|
// https://docs.microsoft.com/en-gb/globalization/windows-keyboard-layouts#U
|
||||||
|
//
|
||||||
|
// -- Jonathan Miles <jon@cybah.co.uk>
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
name[Group1]="English (UK, extended, Windows)";
|
||||||
|
|
||||||
|
key <TLDE> { [ dead_grave, notsign, brokenbar, NoSymbol ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, dead_diaeresis, onehalf ] };
|
||||||
|
key <AE03> { [ 3, sterling, threesuperior, onethird ] };
|
||||||
|
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||||
|
key <AE06> { [ 6, asciicircum, dead_circumflex, NoSymbol ] };
|
||||||
|
|
||||||
|
key <AD02> { [ w, W, wacute, Wacute ] };
|
||||||
|
key <AD03> { [ e, E, eacute, Eacute ] };
|
||||||
|
key <AD06> { [ y, Y, yacute, Yacute ] };
|
||||||
|
key <AD07> { [ u, U, uacute, Uacute ] };
|
||||||
|
key <AD08> { [ i, I, iacute, Iacute ] };
|
||||||
|
key <AD09> { [ o, O, oacute, Oacute ] };
|
||||||
|
key <AD12> { [ bracketright, braceright, NoSymbol, bar ] };
|
||||||
|
|
||||||
|
key <AC01> { [ a, A, aacute, Aacute ] };
|
||||||
|
key <AC11> { [ apostrophe, at, dead_acute, grave ] };
|
||||||
|
key <BKSL> { [ numbersign, asciitilde, dead_tilde, backslash ] };
|
||||||
|
|
||||||
|
key <LSGT> { [ backslash, bar, NoSymbol, NoSymbol ] };
|
||||||
|
key <AB03> { [ c, C, ccedilla, Ccedilla ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
// Describe the differences between the US Colemak layout
|
||||||
|
// and a UK variant. By Andy Buckley (andy@insectnation.org)
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "colemak" {
|
||||||
|
include "us(colemak)"
|
||||||
|
|
||||||
|
name[Group1]="English (UK, Colemak)";
|
||||||
|
|
||||||
|
key <TLDE> { [ grave, notsign, bar, asciitilde ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, twosuperior, oneeighth ] };
|
||||||
|
key <AE03> { [ 3, sterling, threesuperior, sterling ] };
|
||||||
|
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||||
|
|
||||||
|
key <AC11> { [apostrophe, at, dead_circumflex, dead_caron] };
|
||||||
|
key <BKSL> { [numbersign, asciitilde, dead_grave, dead_breve ] };
|
||||||
|
|
||||||
|
key <LSGT> { [ backslash, bar, asciitilde, brokenbar ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Colemak-DH (ISO) layout, UK Variant, https://colemakmods.github.io/mod-dh/
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "colemak_dh" {
|
||||||
|
include "us(colemak_dh)"
|
||||||
|
|
||||||
|
name[Group1]="English (UK, Colemak-DH)";
|
||||||
|
|
||||||
|
key <TLDE> { [ grave, notsign, bar, asciitilde ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, twosuperior, oneeighth ] };
|
||||||
|
key <AE03> { [ 3, sterling, threesuperior, sterling ] };
|
||||||
|
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||||
|
|
||||||
|
key <AC11> { [apostrophe, at, dead_circumflex, dead_caron] };
|
||||||
|
key <BKSL> { [numbersign, asciitilde, dead_grave, dead_breve ] };
|
||||||
|
|
||||||
|
key <AB05> { [ backslash, bar, asciitilde, brokenbar ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
// Dvorak (UK) keymap (by odaen) allowing the usage of
|
||||||
|
// the £ and ? key and swapping the @ and " keys.
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "dvorak" {
|
||||||
|
include "us(dvorak-alt-intl)"
|
||||||
|
|
||||||
|
name[Group1]="English (UK, Dvorak)";
|
||||||
|
|
||||||
|
key <TLDE> { [ grave, notsign, bar, bar ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, twosuperior, NoSymbol ] };
|
||||||
|
key <AE03> { [ 3, sterling, threesuperior, NoSymbol ] };
|
||||||
|
key <AD01> { [ apostrophe, at ] };
|
||||||
|
key <BKSL> { [ numbersign, asciitilde ] };
|
||||||
|
key <LSGT> { [ backslash, bar ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Dvorak letter positions, but punctuation all in the normal UK positions.
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "dvorakukp" {
|
||||||
|
include "gb(dvorak)"
|
||||||
|
|
||||||
|
name[Group1]="English (UK, Dvorak, with UK punctuation)";
|
||||||
|
|
||||||
|
key <AE11> { [ minus, underscore ] };
|
||||||
|
key <AE12> { [ equal, plus ] };
|
||||||
|
key <AD11> { [ bracketleft, braceleft ] };
|
||||||
|
key <AD12> { [ bracketright, braceright ] };
|
||||||
|
key <AD01> { [ slash, question ] };
|
||||||
|
key <AC11> { [apostrophe, at, dead_circumflex, dead_caron] };
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "mac" {
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
name[Group1]= "English (UK, Macintosh)";
|
||||||
|
|
||||||
|
key <TLDE> { [ section, plusminus ] };
|
||||||
|
key <AE02> { [ 2, at, EuroSign ] };
|
||||||
|
key <AE03> { [ 3, sterling, numbersign ] };
|
||||||
|
key <LSGT> { [ grave, asciitilde ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
include "level3(enter_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "mac_intl" {
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
name[Group1]="English (UK, Macintosh, intl.)";
|
||||||
|
|
||||||
|
key <TLDE> { [ section, plusminus, notsign, notsign ] }; //dead_grave
|
||||||
|
key <AE02> { [ 2, at, EuroSign, onehalf ] };
|
||||||
|
key <AE03> { [ 3, sterling, twosuperior, onethird ] };
|
||||||
|
key <AE04> { [ 4, dollar, threesuperior, onequarter ] };
|
||||||
|
key <AE06> { [ 6, dead_circumflex, NoSymbol, onesixth ] };
|
||||||
|
key <AD09> { [ o, O, oe, OE ] };
|
||||||
|
|
||||||
|
key <AC11> { [ dead_acute, dead_diaeresis, dead_diaeresis, bar ] }; //dead_doubleacute
|
||||||
|
key <BKSL> { [ backslash, bar, numbersign, bar ] };
|
||||||
|
|
||||||
|
key <LSGT> { [ dead_grave, dead_tilde, brokenbar, bar ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "pl" {
|
||||||
|
|
||||||
|
// Polish accented letters on upper levels of corresponding base letters.
|
||||||
|
// Idea from Wawrzyniec Niewodniczański, adapted by Aleksander Kowalski.
|
||||||
|
|
||||||
|
include "gb(basic)"
|
||||||
|
|
||||||
|
name[Group1]="Polish (British keyboard)";
|
||||||
|
|
||||||
|
key <AD03> { [ e, E, eogonek, Eogonek ] };
|
||||||
|
key <AD09> { [ o, O, oacute, Oacute ] };
|
||||||
|
|
||||||
|
key <AC01> { [ a, A, aogonek, Aogonek ] };
|
||||||
|
key <AC02> { [ s, S, sacute, Sacute ] };
|
||||||
|
|
||||||
|
key <AB01> { [ z, Z, zabovedot, Zabovedot ] };
|
||||||
|
key <AB02> { [ x, X, zacute, Zacute ] };
|
||||||
|
key <AB03> { [ c, C, cacute, Cacute ] };
|
||||||
|
key <AB06> { [ n, N, nacute, Nacute ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "gla" {
|
||||||
|
|
||||||
|
// Grave-accented letters on the upper levels of the relevant vowels.
|
||||||
|
|
||||||
|
include "gb(basic)"
|
||||||
|
|
||||||
|
name[Group1]="Scottish Gaelic";
|
||||||
|
|
||||||
|
key <AD03> { [ e, E, egrave, Egrave ] };
|
||||||
|
key <AD07> { [ u, U, ugrave, Ugrave ] };
|
||||||
|
key <AD08> { [ i, I, igrave, Igrave ] };
|
||||||
|
key <AD09> { [ o, O, ograve, Ograve ] };
|
||||||
|
|
||||||
|
key <AC01> { [ a, A, agrave, Agrave ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// EXTRAS:
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "sun_type6" {
|
||||||
|
include "sun_vndr/gb(sun_type6)"
|
||||||
|
};
|
||||||
+102
@@ -0,0 +1,102 @@
|
|||||||
|
// The <KPDL> key is a mess.
|
||||||
|
// It was probably originally meant to be a decimal separator.
|
||||||
|
// Except since it was declared by USA people it didn't use the original
|
||||||
|
// SI separator "," but a "." (since then the USA managed to f-up the SI
|
||||||
|
// by making "." an accepted alternative, but standards still use "," as
|
||||||
|
// default)
|
||||||
|
// As a result users of SI-abiding countries expect either a "." or a ","
|
||||||
|
// or a "decimal_separator" which may or may not be translated in one of the
|
||||||
|
// above depending on applications.
|
||||||
|
// It's not possible to define a default per-country since user expectations
|
||||||
|
// depend on the conflicting choices of their most-used applications,
|
||||||
|
// operating system, etc. Therefore it needs to be a configuration setting
|
||||||
|
// Copyright © 2007 Nicolas Mailhot <nicolas.mailhot @ laposte.net>
|
||||||
|
|
||||||
|
|
||||||
|
// Legacy <KPDL> #1
|
||||||
|
// This assumes KP_Decimal will be translated in a dot
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "dot" {
|
||||||
|
|
||||||
|
key.type[Group1]="KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ KP_Delete, KP_Decimal ] }; // <delete> <separator>
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
// Legacy <KPDL> #2
|
||||||
|
// This assumes KP_Separator will be translated in a comma
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "comma" {
|
||||||
|
|
||||||
|
key.type[Group1]="KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ KP_Delete, KP_Separator ] }; // <delete> <separator>
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
// Period <KPDL>, usual keyboard serigraphy in most countries
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "dotoss" {
|
||||||
|
|
||||||
|
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ KP_Delete, period, comma, 0x100202F ] }; // <delete> . , ⍽ (narrow no-break space)
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
// Period <KPDL>, usual keyboard serigraphy in most countries, latin-9 restriction
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "dotoss_latin9" {
|
||||||
|
|
||||||
|
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ KP_Delete, period, comma, nobreakspace ] }; // <delete> . , ⍽ (no-break space)
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
// Comma <KPDL>, what most non anglo-saxon people consider the real separator
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "commaoss" {
|
||||||
|
|
||||||
|
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ KP_Delete, comma, period, 0x100202F ] }; // <delete> , . ⍽ (narrow no-break space)
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
// Momayyez <KPDL>: Bahrain, Iran, Iraq, Kuwait, Oman, Qatar, Saudi Arabia, Syria, UAE
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "momayyezoss" {
|
||||||
|
|
||||||
|
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ KP_Delete, 0x100066B, comma, 0x100202F ] }; // <delete> ? , ⍽ (narrow no-break space)
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
// Abstracted <KPDL>, pray everything will work out (it usually does not)
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "kposs" {
|
||||||
|
|
||||||
|
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ KP_Delete, KP_Decimal, KP_Separator, 0x100202F ] }; // <delete> ? ? ⍽ (narrow no-break space)
|
||||||
|
};
|
||||||
|
|
||||||
|
// Spreadsheets may be configured to use the dot as decimal
|
||||||
|
// punctuation, comma as a thousands separator and then semi-colon as
|
||||||
|
// the list separator. Of these, dot and semi-colon is most important
|
||||||
|
// when entering data by the keyboard; the comma can then be inferred
|
||||||
|
// and added to the presentation afterwards. Using semi-colon as a
|
||||||
|
// general separator may in fact be preferred to avoid ambiguities
|
||||||
|
// in data files. Most times a decimal separator is hard-coded, it
|
||||||
|
// seems to be period, probably since this is the syntax used in
|
||||||
|
// (most) programming languages.
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "semi" {
|
||||||
|
|
||||||
|
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ NoSymbol, NoSymbol, semicolon ] };
|
||||||
|
};
|
||||||
+255
@@ -0,0 +1,255 @@
|
|||||||
|
// Common Latin alphabet layout
|
||||||
|
|
||||||
|
default partial
|
||||||
|
xkb_symbols "basic" {
|
||||||
|
|
||||||
|
key <AE01> { [ 1, exclam, onesuperior, exclamdown ] };
|
||||||
|
key <AE02> { [ 2, at, twosuperior, oneeighth ] };
|
||||||
|
key <AE03> { [ 3, numbersign, threesuperior, sterling ] };
|
||||||
|
key <AE04> { [ 4, dollar, onequarter, dollar ] };
|
||||||
|
key <AE05> { [ 5, percent, onehalf, threeeighths ] };
|
||||||
|
key <AE06> { [ 6, asciicircum, threequarters, fiveeighths ] };
|
||||||
|
key <AE07> { [ 7, ampersand, braceleft, seveneighths ] };
|
||||||
|
key <AE08> { [ 8, asterisk, bracketleft, trademark ] };
|
||||||
|
key <AE09> { [ 9, parenleft, bracketright, plusminus ] };
|
||||||
|
key <AE10> { [ 0, parenright, braceright, degree ] };
|
||||||
|
key <AE11> { [ minus, underscore, backslash, questiondown ] };
|
||||||
|
key <AE12> { [ equal, plus, dead_cedilla, dead_ogonek ] };
|
||||||
|
|
||||||
|
key <AD01> { [ q, Q, at, Greek_OMEGA ] };
|
||||||
|
key <AD02> { [ w, W, U017F, section ] };
|
||||||
|
key <AD03> { [ e, E, e, E ] };
|
||||||
|
key <AD04> { [ r, R, paragraph, registered ] };
|
||||||
|
key <AD05> { [ t, T, tslash, Tslash ] };
|
||||||
|
key <AD06> { [ y, Y, leftarrow, yen ] };
|
||||||
|
key <AD07> { [ u, U, downarrow, uparrow ] };
|
||||||
|
key <AD08> { [ i, I, rightarrow, idotless ] };
|
||||||
|
key <AD09> { [ o, O, oslash, Oslash ] };
|
||||||
|
key <AD10> { [ p, P, thorn, THORN ] };
|
||||||
|
key <AD11> { [bracketleft, braceleft, dead_diaeresis, dead_abovering ] };
|
||||||
|
key <AD12> { [bracketright, braceright, dead_tilde, dead_macron ] };
|
||||||
|
|
||||||
|
key <AC01> { [ a, A, ae, AE ] };
|
||||||
|
key <AC02> { [ s, S, ssharp, U1E9E ] };
|
||||||
|
key <AC03> { [ d, D, eth, ETH ] };
|
||||||
|
key <AC04> { [ f, F, dstroke, ordfeminine ] };
|
||||||
|
key <AC05> { [ g, G, eng, ENG ] };
|
||||||
|
key <AC06> { [ h, H, hstroke, Hstroke ] };
|
||||||
|
key <AC07> { [ j, J, dead_hook, dead_horn ] };
|
||||||
|
key <AC08> { [ k, K, kra, ampersand ] };
|
||||||
|
key <AC09> { [ l, L, lstroke, Lstroke ] };
|
||||||
|
key <AC10> { [ semicolon, colon, dead_acute, dead_doubleacute ] };
|
||||||
|
key <AC11> { [apostrophe, quotedbl, dead_circumflex, dead_caron ] };
|
||||||
|
key <TLDE> { [ grave, asciitilde, notsign, notsign ] };
|
||||||
|
|
||||||
|
key <BKSL> { [ backslash, bar, dead_grave, dead_breve ] };
|
||||||
|
key <AB01> { [ z, Z, guillemotleft, less ] };
|
||||||
|
key <AB02> { [ x, X, guillemotright, greater ] };
|
||||||
|
key <AB03> { [ c, C, cent, copyright ] };
|
||||||
|
key <AB04> { [ v, V, doublelowquotemark, singlelowquotemark ] };
|
||||||
|
key <AB05> { [ b, B, leftdoublequotemark, leftsinglequotemark ] };
|
||||||
|
key <AB06> { [ n, N, rightdoublequotemark, rightsinglequotemark ] };
|
||||||
|
key <AB07> { [ m, M, mu, masculine ] };
|
||||||
|
key <AB08> { [ comma, less, U2022, multiply ] }; // bullet
|
||||||
|
key <AB09> { [ period, greater, periodcentered, division ] };
|
||||||
|
key <AB10> { [ slash, question, dead_belowdot, dead_abovedot ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Northern Europe ( Danish, Finnish, Norwegian, Swedish) common layout
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "type2" {
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
key <AE01> { [ 1, exclam, exclamdown, onesuperior ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, at, twosuperior ] };
|
||||||
|
key <AE03> { [ 3, numbersign, sterling, threesuperior] };
|
||||||
|
key <AE04> { [ 4, currency, dollar, onequarter ] };
|
||||||
|
key <AE05> { [ 5, percent, onehalf, cent ] };
|
||||||
|
key <AE06> { [ 6, ampersand, yen, fiveeighths ] };
|
||||||
|
key <AE07> { [ 7, slash, braceleft, division ] };
|
||||||
|
key <AE08> { [ 8, parenleft, bracketleft, guillemotleft] };
|
||||||
|
key <AE09> { [ 9, parenright, bracketright, guillemotright] };
|
||||||
|
key <AE10> { [ 0, equal, braceright, degree ] };
|
||||||
|
|
||||||
|
key <AD03> { [ e, E, EuroSign, cent ] };
|
||||||
|
key <AD04> { [ r, R, registered, registered ] };
|
||||||
|
key <AD05> { [ t, T, thorn, THORN ] };
|
||||||
|
key <AD09> { [ o, O, oe, OE ] };
|
||||||
|
key <AD11> { [ aring, Aring, dead_diaeresis, dead_abovering ] };
|
||||||
|
key <AD12> { [dead_diaeresis, dead_circumflex, dead_tilde, dead_caron ] };
|
||||||
|
|
||||||
|
key <AC01> { [ a, A, ordfeminine, masculine ] };
|
||||||
|
|
||||||
|
key <AB03> { [ c, C, copyright, copyright ] };
|
||||||
|
key <AB08> { [ comma, semicolon, dead_cedilla, dead_ogonek ] };
|
||||||
|
key <AB09> { [ period, colon, periodcentered, dead_abovedot ] };
|
||||||
|
key <AB10> { [ minus, underscore, dead_belowdot, dead_abovedot ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Slavic Latin ( Albanian, Croatian, Polish, Slovene, Yugoslav)
|
||||||
|
// common layout
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "type3" {
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
key <AD01> { [ q, Q, backslash, Greek_OMEGA ] };
|
||||||
|
key <AD02> { [ w, W, bar, section ] };
|
||||||
|
key <AD06> { [ z, Z, leftarrow, yen ] };
|
||||||
|
|
||||||
|
key <AC04> { [ f, F, bracketleft, ordfeminine ] };
|
||||||
|
key <AC05> { [ g, G, bracketright, ENG ] };
|
||||||
|
key <AC08> { [ k, K, lstroke, ampersand ] };
|
||||||
|
|
||||||
|
key <AB01> { [ y, Y, guillemotleft, less ] };
|
||||||
|
key <AB04> { [ v, V, at, grave ] };
|
||||||
|
key <AB05> { [ b, B, braceleft, apostrophe ] };
|
||||||
|
key <AB06> { [ n, N, braceright, acute ] };
|
||||||
|
key <AB07> { [ m, M, section, masculine ] };
|
||||||
|
key <AB08> { [ comma, semicolon, less, multiply ] };
|
||||||
|
key <AB09> { [ period, colon, greater, division ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Another common Latin layout
|
||||||
|
// (German, Estonian, Spanish, Icelandic, Italian, Latin American, Portuguese)
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "type4" {
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
key <AE02> { [ 2, quotedbl, at, oneeighth ] };
|
||||||
|
key <AE06> { [ 6, ampersand, notsign, fiveeighths ] };
|
||||||
|
key <AE07> { [ 7, slash, braceleft, seveneighths ] };
|
||||||
|
key <AE08> { [ 8, parenleft, bracketleft, trademark ] };
|
||||||
|
key <AE09> { [ 9, parenright, bracketright, plusminus ] };
|
||||||
|
key <AE10> { [ 0, equal, braceright, degree ] };
|
||||||
|
|
||||||
|
key <AD03> { [ e, E, EuroSign, cent ] };
|
||||||
|
|
||||||
|
key <AB08> { [ comma, semicolon, U2022, multiply ] }; // bullet
|
||||||
|
key <AB09> { [ period, colon, periodcentered, division ] };
|
||||||
|
key <AB10> { [ minus, underscore, dead_belowdot, dead_abovedot ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "nodeadkeys" {
|
||||||
|
|
||||||
|
key <AE12> { [ equal, plus, cedilla, ogonek ] };
|
||||||
|
key <AD11> { [bracketleft, braceleft, diaeresis, degree ] };
|
||||||
|
key <AD12> { [bracketright, braceright, asciitilde, macron ] };
|
||||||
|
key <AC07> { [ j, J, ezh, EZH ] };
|
||||||
|
key <AC10> { [ semicolon, colon, acute, doubleacute ] };
|
||||||
|
key <AC11> { [apostrophe, quotedbl, asciicircum, caron ] };
|
||||||
|
key <BKSL> { [ backslash, bar, grave, breve ] };
|
||||||
|
key <AB10> { [ slash, question, ellipsis, abovedot ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "type2_nodeadkeys" {
|
||||||
|
|
||||||
|
include "latin(nodeadkeys)"
|
||||||
|
|
||||||
|
key <AD11> { [ aring, Aring, diaeresis, degree ] };
|
||||||
|
key <AD12> { [ diaeresis, asciicircum, asciitilde, caron ] };
|
||||||
|
key <AB08> { [ comma, semicolon, cedilla, ogonek ] };
|
||||||
|
key <AB09> { [ period, colon, periodcentered, abovedot ] };
|
||||||
|
key <AB10> { [ minus, underscore, ellipsis, abovedot ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "type3_nodeadkeys" {
|
||||||
|
|
||||||
|
include "latin(nodeadkeys)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "type4_nodeadkeys" {
|
||||||
|
|
||||||
|
include "latin(nodeadkeys)"
|
||||||
|
|
||||||
|
key <AB10> { [ minus, underscore, ellipsis, abovedot ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Added 2008.03.05 by Marcin Woliński
|
||||||
|
// See http://marcinwolinski.pl/keyboard/ for a description.
|
||||||
|
// Used by pl(intl)
|
||||||
|
//
|
||||||
|
// ┌─────┐
|
||||||
|
// │ 2 4 │ 2 = Shift, 4 = Level3 + Shift
|
||||||
|
// │ 1 3 │ 1 = Normal, 3 = Level3
|
||||||
|
// └─────┘
|
||||||
|
// ┌─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┲━━━━━━━━━┓
|
||||||
|
// │ ~ ~ │ ! ' │ @ " │ # ˝ │ $ ¸ │ % ˇ │ ^ ^ │ & ˘ │ * ̇ │ ( ̣ │ ) ° │ _ ¯ │ + ˛ ┃ ⌫ Back- ┃
|
||||||
|
// │ ` ` │ 1 ¡ │ 2 © │ 3 • │ 4 § │ 5 € │ 6 ¢ │ 7 − │ 8 × │ 9 ÷ │ 0 ° │ - – │ = — ┃ space ┃
|
||||||
|
// ┢━━━━━┷━┱───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┺━┳━━━━━━━┫
|
||||||
|
// ┃ ┃ Q │ W │ E │ R │ T │ Y │ U │ I │ O │ P │ { « │ } » ┃ Enter ┃
|
||||||
|
// ┃Tab ↹ ┃ q │ w │ e │ r │ t │ y │ u │ i │ o │ p │ [ ‹ │ ] › ┃ ⏎ ┃
|
||||||
|
// ┣━━━━━━━┻┱────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┺┓ ┃
|
||||||
|
// ┃ ┃ A │ S │ D │ F │ G │ H │ J │ K │ L │ : “ │ " ” │ | ¶ ┃ ┃
|
||||||
|
// ┃Caps ⇬ ┃ a │ s │ d │ f │ g │ h │ j │ k │ l │ ; ‘ │ ' ’ │ \ ┃ ┃
|
||||||
|
// ┣━━━━━━━━┹────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┲┷━━━━━┻━━━━━━┫
|
||||||
|
// ┃ │ Z │ X │ C │ V │ B │ N │ M │ < „ │ > · │ ? ¿ ┃ ┃
|
||||||
|
// ┃Shift ⇧ │ z │ x │ c │ v │ b │ n │ m │ , ‚ │ . … │ / ⁄ ┃Shift ⇧ ┃
|
||||||
|
// ┣━━━━━━━┳━━━━━┷━┳━━━┷━━━┱─┴─────┴─────┴─────┴─────┴─────┴───┲━┷━━━━━╈━━━━━┻━┳━━━━━━━┳━━━┛
|
||||||
|
// ┃ ┃ ┃ ┃ ␣ ⍽ ┃ ┃ ┃ ┃
|
||||||
|
// ┃Ctrl ┃Meta ┃Alt ┃ ␣ Space ⍽ ┃AltGr ⇮┃Menu ┃Ctrl ┃
|
||||||
|
// ┗━━━━━━━┻━━━━━━━┻━━━━━━━┹───────────────────────────────────┺━━━━━━━┻━━━━━━━┻━━━━━━━┛
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "intl" {
|
||||||
|
|
||||||
|
key <TLDE> { [ grave, asciitilde, dead_grave, dead_tilde ] };
|
||||||
|
key <AE01> { [ 1, exclam, exclamdown, dead_acute ] };
|
||||||
|
key <AE02> { [ 2, at, copyright, dead_diaeresis ] };
|
||||||
|
key <AE03> { [ 3, numbersign, U2022, dead_doubleacute ] }; // U+2022 is bullet (the name bullet does not work)
|
||||||
|
key <AE04> { [ 4, dollar, section, dead_cedilla ] };
|
||||||
|
key <AE05> { [ 5, percent, EuroSign, dead_caron ] };
|
||||||
|
key <AE06> { [ 6, asciicircum, cent, dead_circumflex ] };
|
||||||
|
key <AE07> { [ 7, ampersand, U2212, dead_breve ] }; // U+2212 is MINUS SIGN
|
||||||
|
key <AE08> { [ 8, asterisk, multiply, dead_abovedot ] };
|
||||||
|
key <AE09> { [ 9, parenleft, division, dead_belowdot ] };
|
||||||
|
key <AE10> { [ 0, parenright, degree, dead_abovering ] };
|
||||||
|
key <AE11> { [ minus, underscore, endash, dead_macron ] };
|
||||||
|
key <AE12> { [ equal, plus, emdash, dead_ogonek ] };
|
||||||
|
|
||||||
|
key <AD01> { [ q, Q ] };
|
||||||
|
key <AD02> { [ w, W ] };
|
||||||
|
key <AD03> { [ e, E ] };
|
||||||
|
key <AD04> { [ r, R ] };
|
||||||
|
key <AD05> { [ t, T ] };
|
||||||
|
key <AD06> { [ y, Y ] };
|
||||||
|
key <AD07> { [ u, U ] };
|
||||||
|
key <AD08> { [ i, I ] };
|
||||||
|
key <AD09> { [ o, O ] };
|
||||||
|
key <AD10> { [ p, P ] };
|
||||||
|
key <AD11> { [bracketleft, braceleft, U2039, guillemotleft ] };
|
||||||
|
key <AD12> { [bracketright, braceright, U203A, guillemotright ] };
|
||||||
|
|
||||||
|
key <AC01> { [ a, A ] };
|
||||||
|
key <AC02> { [ s, S ] };
|
||||||
|
key <AC03> { [ d, D ] };
|
||||||
|
key <AC04> { [ f, F ] };
|
||||||
|
key <AC05> { [ g, G ] };
|
||||||
|
key <AC06> { [ h, H ] };
|
||||||
|
key <AC07> { [ j, J ] };
|
||||||
|
key <AC08> { [ k, K ] };
|
||||||
|
key <AC09> { [ l, L ] };
|
||||||
|
key <AC10> { [ semicolon, colon, leftsinglequotemark, leftdoublequotemark ] };
|
||||||
|
key <AC11> { [apostrophe, quotedbl, rightsinglequotemark, rightdoublequotemark ] };
|
||||||
|
|
||||||
|
key <BKSL> { [ backslash, bar, NoSymbol, paragraph ] };
|
||||||
|
key <AB01> { [ z, Z ] };
|
||||||
|
key <AB02> { [ x, X ] };
|
||||||
|
key <AB03> { [ c, C ] };
|
||||||
|
key <AB04> { [ v, V ] };
|
||||||
|
key <AB05> { [ b, B ] };
|
||||||
|
key <AB06> { [ n, N ] };
|
||||||
|
key <AB07> { [ m, M ] };
|
||||||
|
key <AB08> { [ comma, less, singlelowquotemark, doublelowquotemark ] };
|
||||||
|
key <AB09> { [ period, greater, ellipsis, periodcentered ] };
|
||||||
|
key <AB10> { [ slash, question, U2044, questiondown ] }; // U+2044 is FRACTION SLASH
|
||||||
|
};
|
||||||
+156
@@ -0,0 +1,156 @@
|
|||||||
|
// These variants assign ISO_Level3_Shift to various keys
|
||||||
|
// so that levels 3 and 4 can be reached.
|
||||||
|
|
||||||
|
// The default behaviour:
|
||||||
|
// the right Alt key (AltGr) chooses the third symbol engraved on a key.
|
||||||
|
default partial modifier_keys
|
||||||
|
xkb_symbols "ralt_switch" {
|
||||||
|
key <RALT> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The right Alt key never chooses the third level.
|
||||||
|
// This option attempts to undo the effect of a layout's inclusion of
|
||||||
|
// 'ralt_switch'. You may want to also select another level3 option
|
||||||
|
// to map the level3 shift to some other key.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "ralt_alt" {
|
||||||
|
key <RALT> {[ Alt_R, Meta_R ], type[group1]="TWO_LEVEL" };
|
||||||
|
modifier_map Mod1 { <RALT> };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The right Alt key (while pressed) chooses the third shift level,
|
||||||
|
// and Compose is mapped to its second level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "ralt_switch_multikey" {
|
||||||
|
key <RALT> {[ ISO_Level3_Shift, Multi_key ], type[group1]="TWO_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Either Alt key (while pressed) chooses the third shift level.
|
||||||
|
// (To be used mostly to imitate Mac OS functionality.)
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "alt_switch" {
|
||||||
|
include "level3(lalt_switch)"
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
// The left Alt key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "lalt_switch" {
|
||||||
|
key <LALT> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The right Ctrl key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "switch" {
|
||||||
|
key <RCTL> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The Menu key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "menu_switch" {
|
||||||
|
key <MENU> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Either Win key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "win_switch" {
|
||||||
|
include "level3(lwin_switch)"
|
||||||
|
include "level3(rwin_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
// The left Win key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "lwin_switch" {
|
||||||
|
key <LWIN> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The right Win key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "rwin_switch" {
|
||||||
|
key <RWIN> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The Enter key on the kepypad (while pressed) chooses the third shift level.
|
||||||
|
// (This is especially useful for Mac laptops which miss the right Alt key.)
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "enter_switch" {
|
||||||
|
key <KPEN> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The CapsLock key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "caps_switch" {
|
||||||
|
key <CAPS> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The CapsLock key (while pressed) chooses the third shift level and
|
||||||
|
// Ctrl + CapsLock has the original CapsLock function.
|
||||||
|
// The 2023 DIN standard for German keyboards recommends it as an option:
|
||||||
|
// - https://de.wikipedia.org/wiki/E1_(Tastaturbelegung)#Feststelltaste/Umschaltsperre
|
||||||
|
// - https://en.wikipedia.org/wiki/Caps_Lock#Abolition
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "caps_switch_capslock_with_ctrl" {
|
||||||
|
virtual_modifiers LevelThree;
|
||||||
|
|
||||||
|
key <CAPS> {
|
||||||
|
type[Group1] = "PC_CONTROL_LEVEL2",
|
||||||
|
symbols[Group1] = [ ISO_Level3_Shift, Caps_Lock ],
|
||||||
|
// Explicit actions are preferred over modMap None/Mod5 { Caps_Lock }
|
||||||
|
// because they have no side effect
|
||||||
|
actions[Group1] = [ SetMods(modifiers = LevelThree), LockMods(modifiers = Lock) ]
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
// The Backslash key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "bksl_switch" {
|
||||||
|
key <BKSL> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The AC11 key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "ac11_switch" {
|
||||||
|
key <AC11> {[ ISO_Level3_Shift ], type[Group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The Less/Greater key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "lsgt_switch" {
|
||||||
|
key <LSGT> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The CapsLock key (while pressed) chooses the third shift level,
|
||||||
|
// and latches when pressed together with another third-level chooser.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "caps_switch_latch" {
|
||||||
|
key <CAPS> {[ ISO_Level3_Shift, ISO_Level3_Shift, ISO_Level3_Latch ],
|
||||||
|
type[group1]="THREE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The Backslash key (while pressed) chooses the third shift level,
|
||||||
|
// and latches when pressed together with another third-level chooser.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "bksl_switch_latch" {
|
||||||
|
key <BKSL> {[ ISO_Level3_Shift, ISO_Level3_Shift, ISO_Level3_Latch ],
|
||||||
|
type[group1]="THREE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The Less/Greater key (while pressed) chooses the third shift level,
|
||||||
|
// and latches when pressed together with another third-level chooser.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "lsgt_switch_latch" {
|
||||||
|
key <LSGT> {[ ISO_Level3_Shift, ISO_Level3_Shift, ISO_Level3_Latch ],
|
||||||
|
type[group1]="THREE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Top-row digit key 4 chooses third shift level when pressed alone.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "4_switch_isolated" {
|
||||||
|
override key <AE04> {[ ISO_Level3_Shift ]};
|
||||||
|
};
|
||||||
|
|
||||||
|
// Top-row digit key 9 chooses third shift level when pressed alone.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "9_switch_isolated" {
|
||||||
|
override key <AE09> {[ ISO_Level3_Shift ]};
|
||||||
|
};
|
||||||
+2238
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,153 @@
|
|||||||
|
//! xkeyboard-config — keyboard layouts, compiled from the X11 xkeyboard-config database
|
||||||
|
//! into native Zig. It turns a physical key (a USB HID usage, as the input module delivers)
|
||||||
|
//! plus a modifier state into a **keysym** and, when the key produces one, a **character**
|
||||||
|
//! (a Unicode scalar). This is the piece that lets a `KeyEvent.keycode` become a
|
||||||
|
//! `KeyEvent.character`, without shipping an X11 runtime.
|
||||||
|
//!
|
||||||
|
//! The layout tables in `generated/layouts.zig` are produced by
|
||||||
|
//! `tools/make-xkeyboard-config.py` (see ./README.md to regenerate). Those tables are
|
||||||
|
//! deliberately pure data — each key carries its up-to-four levels and an XKB *type*. The
|
||||||
|
//! type -> level selection semantics (which modifier picks which level) live here, so the
|
||||||
|
//! data and the policy are separable.
|
||||||
|
//!
|
||||||
|
//! Scope (documented in README.md): group 1 only, no dead-key/compose composition (a dead
|
||||||
|
//! key returns its keysym with no character), and a curated set of key types. Layouts:
|
||||||
|
//! us, gb, de, fr, es, dvorak.
|
||||||
|
//!
|
||||||
|
//! Upstream xkeyboard-config and keysymdef.h are MIT/X11 licensed; see vendor/COPYING and
|
||||||
|
//! vendor/PROVENANCE.md.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const generated = @import("layouts");
|
||||||
|
|
||||||
|
pub const Level = generated.Level;
|
||||||
|
pub const KeyType = generated.KeyType;
|
||||||
|
pub const Key = generated.Key;
|
||||||
|
pub const Layout = generated.Layout;
|
||||||
|
|
||||||
|
/// The generated layouts, by name — as pointers, so they share identity with `all` and
|
||||||
|
/// `byName` (and match the `*const Layout` that `map` takes).
|
||||||
|
pub const us: *const Layout = &generated.us;
|
||||||
|
pub const gb: *const Layout = &generated.gb;
|
||||||
|
pub const de: *const Layout = &generated.de;
|
||||||
|
pub const fr: *const Layout = &generated.fr;
|
||||||
|
pub const es: *const Layout = &generated.es;
|
||||||
|
pub const dvorak: *const Layout = &generated.dvorak;
|
||||||
|
|
||||||
|
/// Every generated layout, for enumeration (e.g. a settings UI).
|
||||||
|
pub const all = generated.all;
|
||||||
|
|
||||||
|
/// The modifier state that selects a key's level. `level3` is AltGr (ISO Level3 Shift);
|
||||||
|
/// `control` is accepted for completeness but does not affect level selection here.
|
||||||
|
pub const Modifiers = struct {
|
||||||
|
shift: bool = false,
|
||||||
|
caps_lock: bool = false,
|
||||||
|
level3: bool = false,
|
||||||
|
control: bool = false,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The result of a lookup: the X11 `keysym`, and the `character` it produces (a Unicode
|
||||||
|
/// scalar) when it is a printable key — null for keys that produce none (Return, F1, a
|
||||||
|
/// bare dead key, an unmapped key).
|
||||||
|
pub const Mapping = struct {
|
||||||
|
keysym: u32,
|
||||||
|
character: ?u21,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Which level (0..3) a key of `kind` selects under `mods`. XKB's canonical semantics:
|
||||||
|
/// Shift picks the odd level, AltGr (level3) adds 2, and Caps acts like Shift for the
|
||||||
|
/// alphabetic types. See the XKB "key types" — this covers the ones the vendored layouts
|
||||||
|
/// use; anything else falls back to shift-or-not.
|
||||||
|
fn selectLevel(kind: KeyType, mods: Modifiers) usize {
|
||||||
|
const shift_or_caps = mods.shift != mods.caps_lock; // XOR: Caps behaves like Shift
|
||||||
|
const low: usize = if (mods.shift) 1 else 0;
|
||||||
|
const high: usize = if (mods.level3) 2 else 0;
|
||||||
|
return switch (kind) {
|
||||||
|
.one_level => 0,
|
||||||
|
.two_level, .keypad, .other => low,
|
||||||
|
.alphabetic => if (shift_or_caps) 1 else 0,
|
||||||
|
.four_level => low + high,
|
||||||
|
.four_level_alphabetic => (if (shift_or_caps) @as(usize, 1) else 0) + high,
|
||||||
|
// Caps affects only the base pair, not the AltGr pair.
|
||||||
|
.four_level_semialphabetic => if (mods.level3) 2 + low else (if (shift_or_caps) @as(usize, 1) else 0),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Map a physical key (`hid_usage`, a USB HID keyboard-page usage) under `mods` on
|
||||||
|
/// `layout` to its keysym and character. Falls back gracefully when the selected level is
|
||||||
|
/// undefined for the key: it drops the AltGr component, then the shift component, so a key
|
||||||
|
/// with only a base/shift pair still yields something sensible under AltGr.
|
||||||
|
pub fn map(layout: *const Layout, hid_usage: u8, mods: Modifiers) Mapping {
|
||||||
|
const key = &layout.keys[hid_usage];
|
||||||
|
var level = selectLevel(key.kind, mods);
|
||||||
|
// Fall back to a defined level: full -> without AltGr -> base.
|
||||||
|
if (key.levels[level].keysym == 0 and key.levels[level].unicode == 0) {
|
||||||
|
const candidates = [_]usize{ level & 1, 0 };
|
||||||
|
for (candidates) |candidate| {
|
||||||
|
if (key.levels[candidate].keysym != 0 or key.levels[candidate].unicode != 0) {
|
||||||
|
level = candidate;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const chosen = key.levels[level];
|
||||||
|
return .{
|
||||||
|
.keysym = chosen.keysym,
|
||||||
|
.character = if (chosen.unicode != 0) @intCast(chosen.unicode) else null,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Look up a layout by its name (`"us"`, `"gb"`, ...), or null if unknown.
|
||||||
|
pub fn byName(name: []const u8) ?*const Layout {
|
||||||
|
for (all) |layout| {
|
||||||
|
if (std.mem.eql(u8, layout.name, name)) return layout;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- tests (host-run via `zig build test`) ---------------------------------
|
||||||
|
|
||||||
|
const testing = std.testing;
|
||||||
|
|
||||||
|
// USB HID usages used in the tests (keyboard page 0x07).
|
||||||
|
const hid_a: u8 = 0x04;
|
||||||
|
const hid_1: u8 = 0x1e;
|
||||||
|
const hid_3: u8 = 0x20;
|
||||||
|
|
||||||
|
test "us: letters obey shift and caps" {
|
||||||
|
try testing.expectEqual(@as(?u21, 'a'), map(us, hid_a, .{}).character);
|
||||||
|
try testing.expectEqual(@as(?u21, 'A'), map(us, hid_a, .{ .shift = true }).character);
|
||||||
|
try testing.expectEqual(@as(?u21, 'A'), map(us, hid_a, .{ .caps_lock = true }).character);
|
||||||
|
// Shift + Caps cancels for an alphabetic key.
|
||||||
|
try testing.expectEqual(@as(?u21, 'a'), map(us, hid_a, .{ .shift = true, .caps_lock = true }).character);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "us: digits and their shifted symbols" {
|
||||||
|
try testing.expectEqual(@as(?u21, '1'), map(us, hid_1, .{}).character);
|
||||||
|
try testing.expectEqual(@as(?u21, '!'), map(us, hid_1, .{ .shift = true }).character);
|
||||||
|
try testing.expectEqual(@as(?u21, '3'), map(us, hid_3, .{}).character);
|
||||||
|
try testing.expectEqual(@as(?u21, '#'), map(us, hid_3, .{ .shift = true }).character);
|
||||||
|
// A digit is not alphabetic: Caps alone must not shift it.
|
||||||
|
try testing.expectEqual(@as(?u21, '3'), map(us, hid_3, .{ .caps_lock = true }).character);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "layouts differ: GB pound vs US hash on shift+3" {
|
||||||
|
try testing.expectEqual(@as(?u21, '#'), map(us, hid_3, .{ .shift = true }).character);
|
||||||
|
try testing.expectEqual(@as(?u21, '£'), map(gb, hid_3, .{ .shift = true }).character);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "french azerty places q where us has a" {
|
||||||
|
try testing.expectEqual(@as(?u21, 'q'), map(fr, hid_a, .{}).character);
|
||||||
|
try testing.expectEqual(@as(?u21, 'Q'), map(fr, hid_a, .{ .shift = true }).character);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "byName resolves and rejects" {
|
||||||
|
try testing.expect(byName("us") == us);
|
||||||
|
try testing.expect(byName("gb") == gb);
|
||||||
|
try testing.expect(byName("nonsense") == null);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "unmapped key yields no character" {
|
||||||
|
// HID 0x00 is not a key; every level is empty.
|
||||||
|
try testing.expectEqual(@as(?u21, null), map(us, 0x00, .{}).character);
|
||||||
|
}
|
||||||
-1086
File diff suppressed because it is too large
Load Diff
@@ -1,208 +0,0 @@
|
|||||||
//! AML (ACPI Machine Language) — the bytecode in the DSDT and SSDTs that describes
|
|
||||||
//! the parts of the machine the static tables don't.
|
|
||||||
//!
|
|
||||||
//! This module has two stages. `parser.zig` walks the entire byte stream and
|
|
||||||
//! records every named object into a namespace tree (`namespace.zig`), capturing
|
|
||||||
//! method bodies and field/region layout. `interp.zig` then *evaluates* control
|
|
||||||
//! methods on demand — running operators, control flow, and OperationRegion field
|
|
||||||
//! access — so callers can resolve device status (`_STA`), current resource
|
|
||||||
//! settings (`_CRS`), sleep states (`_Sx`), and the like against the live namespace.
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
const op = @import("opcodes.zig");
|
|
||||||
const parser = @import("parser.zig");
|
|
||||||
const namespace = @import("namespace.zig");
|
|
||||||
const interp = @import("interp.zig");
|
|
||||||
|
|
||||||
pub const Namespace = namespace.Namespace;
|
|
||||||
pub const Node = namespace.Node;
|
|
||||||
pub const NodeKind = namespace.NodeKind;
|
|
||||||
|
|
||||||
/// The AML evaluator: interprets control methods (and reads Names/Fields) far
|
|
||||||
/// enough for device discovery. See `interp.zig`.
|
|
||||||
pub const Interp = interp.Interp;
|
|
||||||
pub const Object = interp.Object;
|
|
||||||
pub const EvalHal = interp.Hal;
|
|
||||||
|
|
||||||
/// The SLP_TYP values written to PM1a/PM1b control to enter a sleep state.
|
|
||||||
pub const SleepType = struct {
|
|
||||||
slp_typ_a: u8,
|
|
||||||
slp_typ_b: u8,
|
|
||||||
};
|
|
||||||
|
|
||||||
pub const ParseResult = struct {
|
|
||||||
namespace: Namespace,
|
|
||||||
/// Bytes the parser consumed across all blocks...
|
|
||||||
consumed: usize,
|
|
||||||
/// ...out of this many. A clean full traversal has `consumed == total`.
|
|
||||||
total: usize,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Parse the given AML blocks (DSDT first, then SSDTs) into one namespace. Later
|
|
||||||
/// blocks extend the namespace built by earlier ones, exactly as ACPI intends.
|
|
||||||
pub fn parse(allocator: std.mem.Allocator, blocks: []const []const u8) !ParseResult {
|
|
||||||
var ns = try Namespace.init(allocator);
|
|
||||||
var consumed: usize = 0;
|
|
||||||
var total: usize = 0;
|
|
||||||
for (blocks) |block| {
|
|
||||||
var p = parser.Parser.init(block, &ns);
|
|
||||||
consumed += p.parseAll();
|
|
||||||
total += block.len;
|
|
||||||
}
|
|
||||||
return .{ .namespace = ns, .consumed = consumed, .total = total };
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Look up the `\_S{state}` sleep package in a parsed namespace and return its
|
|
||||||
/// first two integer elements (SLP_TYP for PM1a / PM1b), or null if absent.
|
|
||||||
pub fn sleepState(ns: *Namespace, state: u8) ?SleepType {
|
|
||||||
const seg = [4]u8{ '_', 'S', '0' + state, '_' };
|
|
||||||
const node = ns.resolve(ns.root, false, 0, &.{seg}) orelse return null;
|
|
||||||
if (node.kind != .name) return null;
|
|
||||||
return parseSleepPackage(node.value);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Decode a `Package(){ SLP_TYPa, SLP_TYPb, ... }` from the raw AML of a Name's
|
|
||||||
/// value. Returns the first two elements as bytes (missing elements default to 0).
|
|
||||||
fn parseSleepPackage(value: []const u8) ?SleepType {
|
|
||||||
if (value.len == 0 or value[0] != op.package_op) return null;
|
|
||||||
var p: usize = 1;
|
|
||||||
p += pkgLengthSize(value, p) orelse return null;
|
|
||||||
if (p >= value.len) return null;
|
|
||||||
const num_elements = value[p];
|
|
||||||
p += 1;
|
|
||||||
|
|
||||||
const a: u8 = if (num_elements >= 1) @truncate(readInteger(value, &p) orelse 0) else 0;
|
|
||||||
const b: u8 = if (num_elements >= 2) @truncate(readInteger(value, &p) orelse 0) else 0;
|
|
||||||
return .{ .slp_typ_a = a, .slp_typ_b = b };
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Bytes a PkgLength field occupies at `p` (we only need to step over it here).
|
|
||||||
fn pkgLengthSize(bytes: []const u8, p: usize) ?usize {
|
|
||||||
if (p >= bytes.len) return null;
|
|
||||||
const follow: usize = bytes[p] >> 6;
|
|
||||||
if (p + 1 + follow > bytes.len) return null;
|
|
||||||
return 1 + follow;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Read one AML integer data object at `p`, advancing `p`.
|
|
||||||
fn readInteger(bytes: []const u8, p: *usize) ?u64 {
|
|
||||||
if (p.* >= bytes.len) return null;
|
|
||||||
const opcode = bytes[p.*];
|
|
||||||
p.* += 1;
|
|
||||||
return switch (opcode) {
|
|
||||||
op.zero_op => 0,
|
|
||||||
op.one_op => 1,
|
|
||||||
op.ones_op => 0xFF,
|
|
||||||
op.byte_prefix => readLittle(bytes, p, 1),
|
|
||||||
op.word_prefix => readLittle(bytes, p, 2),
|
|
||||||
op.dword_prefix => readLittle(bytes, p, 4),
|
|
||||||
op.qword_prefix => readLittle(bytes, p, 8),
|
|
||||||
else => null,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
fn readLittle(bytes: []const u8, p: *usize, n: usize) ?u64 {
|
|
||||||
if (p.* + n > bytes.len) return null;
|
|
||||||
var v: u64 = 0;
|
|
||||||
var k: usize = 0;
|
|
||||||
while (k < n) : (k += 1) v |= @as(u64, bytes[p.* + k]) << @intCast(k * 8);
|
|
||||||
p.* += n;
|
|
||||||
return v;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- tests ------------------------------------------------------------------
|
|
||||||
|
|
||||||
test "parses a nested namespace and finds the sleep package" {
|
|
||||||
// A hand-assembled AML blob (all PkgLengths computed to be single-byte):
|
|
||||||
// Name(_S5, Package(2){0x05, 0x00})
|
|
||||||
// Scope(\_SB) { Device(PCI0) {
|
|
||||||
// Name(_HID, 0x11)
|
|
||||||
// Method(MTHD, 1) {}
|
|
||||||
// Method(CALL, 0) { MTHD(Zero) } // invocation of a 1-arg method
|
|
||||||
// } }
|
|
||||||
// OperationRegion(DBG0, SystemIO, 0x0402, 1)
|
|
||||||
// Field(DBG0, ...) { DBGB, 8 }
|
|
||||||
const blob = [_]u8{
|
|
||||||
// Name(_S5, Package(2){Byte 0x05, Byte 0x00})
|
|
||||||
0x08, 0x5F, 0x53, 0x35, 0x5F, 0x12, 0x06, 0x02, 0x0A, 0x05, 0x0A, 0x00,
|
|
||||||
// Scope(\_SB) pkglen=0x27
|
|
||||||
0x10, 0x27, 0x5C, 0x5F, 0x53, 0x42, 0x5F,
|
|
||||||
// Device(PCI0) pkglen=0x1F
|
|
||||||
0x5B, 0x82, 0x1F, 0x50, 0x43, 0x49, 0x30,
|
|
||||||
// Name(_HID, 0x11)
|
|
||||||
0x08, 0x5F, 0x48, 0x49, 0x44, 0x0A, 0x11,
|
|
||||||
// Method(MTHD, flags=1) empty, pkglen=0x06
|
|
||||||
0x14, 0x06, 0x4D, 0x54, 0x48, 0x44, 0x01,
|
|
||||||
// Method(CALL, flags=0) { MTHD(Zero) }, pkglen=0x0B
|
|
||||||
0x14, 0x0B, 0x43, 0x41, 0x4C, 0x4C, 0x00, 0x4D, 0x54, 0x48, 0x44, 0x00,
|
|
||||||
// OperationRegion(DBG0, SystemIO, Word 0x0402, Byte 1)
|
|
||||||
0x5B, 0x80, 0x44, 0x42, 0x47, 0x30, 0x01, 0x0B, 0x02, 0x04, 0x0A, 0x01,
|
|
||||||
// Field(DBG0, flags=1) { DBGB, 8 }, pkglen=0x0B
|
|
||||||
0x5B, 0x81, 0x0B, 0x44, 0x42, 0x47, 0x30, 0x01, 0x44, 0x42, 0x47, 0x42, 0x08,
|
|
||||||
};
|
|
||||||
|
|
||||||
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
|
||||||
defer arena.deinit();
|
|
||||||
var result = try parse(arena.allocator(), &.{&blob});
|
|
||||||
|
|
||||||
// Integrity: the parser consumed exactly the whole blob (no desync).
|
|
||||||
try std.testing.expectEqual(blob.len, result.consumed);
|
|
||||||
try std.testing.expectEqual(blob.len, result.total);
|
|
||||||
|
|
||||||
const ns = &result.namespace;
|
|
||||||
|
|
||||||
// Expected top-level nodes.
|
|
||||||
const sb = ns.resolve(ns.root, false, 0, &.{.{ '_', 'S', 'B', '_' }}) orelse return error.NoSB;
|
|
||||||
try std.testing.expectEqual(NodeKind.scope, sb.kind);
|
|
||||||
const pci0 = ns.resolve(sb, false, 0, &.{.{ 'P', 'C', 'I', '0' }}) orelse return error.NoPCI0;
|
|
||||||
try std.testing.expectEqual(NodeKind.device, pci0.kind);
|
|
||||||
_ = ns.resolve(pci0, false, 0, &.{.{ '_', 'H', 'I', 'D' }}) orelse return error.NoHID;
|
|
||||||
|
|
||||||
// The 1-arg method's arg count was parsed from its flags byte.
|
|
||||||
const mthd = ns.resolve(pci0, false, 0, &.{.{ 'M', 'T', 'H', 'D' }}) orelse return error.NoMTHD;
|
|
||||||
try std.testing.expectEqual(NodeKind.method, mthd.kind);
|
|
||||||
try std.testing.expectEqual(@as(u8, 1), mthd.arg_count);
|
|
||||||
|
|
||||||
// OperationRegion and the Field unit made it into the namespace.
|
|
||||||
_ = ns.resolve(ns.root, false, 0, &.{.{ 'D', 'B', 'G', '0' }}) orelse return error.NoRegion;
|
|
||||||
_ = ns.resolve(ns.root, false, 0, &.{.{ 'D', 'B', 'G', 'B' }}) orelse return error.NoField;
|
|
||||||
|
|
||||||
// The sleep package decoded.
|
|
||||||
const s5 = sleepState(ns, 5) orelse return error.NoS5;
|
|
||||||
try std.testing.expectEqual(@as(u8, 5), s5.slp_typ_a);
|
|
||||||
try std.testing.expectEqual(@as(u8, 0), s5.slp_typ_b);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn noMap(_: u64, _: u64, _: bool) void {}
|
|
||||||
fn noRead(_: u8, _: u16) u32 {
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
fn noWrite(_: u8, _: u16, _: u32) void {}
|
|
||||||
|
|
||||||
test "interpreter runs a method with args, arithmetic, and control flow" {
|
|
||||||
// Method(TST_, 1) {
|
|
||||||
// Store(Arg0, Local0); Add(Local0, 5, Local0)
|
|
||||||
// If (LGreater(Local0, 10)) { Return(One) }
|
|
||||||
// Return(Zero)
|
|
||||||
// }
|
|
||||||
const blob = [_]u8{
|
|
||||||
0x14, 0x18, 0x54, 0x53, 0x54, 0x5F, 0x01, // Method TST_, 1 arg
|
|
||||||
0x70, 0x68, 0x60, // Store(Arg0, Local0)
|
|
||||||
0x72, 0x60, 0x0A, 0x05, 0x60, // Add(Local0, 5, Local0)
|
|
||||||
0xA0, 0x07, 0x94, 0x60, 0x0A, 0x0A, 0xA4, 0x01, // If(LGreater(Local0,10)) { Return(One) }
|
|
||||||
0xA4, 0x00, // Return(Zero)
|
|
||||||
};
|
|
||||||
|
|
||||||
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
|
||||||
defer arena.deinit();
|
|
||||||
var result = try parse(arena.allocator(), &.{&blob});
|
|
||||||
const ns = &result.namespace;
|
|
||||||
const tst = ns.resolve(ns.root, false, 0, &.{.{ 'T', 'S', 'T', '_' }}) orelse return error.NoMethod;
|
|
||||||
|
|
||||||
var ev = Interp.init(ns, .{ .mapMmio = noMap, .pioRead = noRead, .pioWrite = noWrite }, arena.allocator());
|
|
||||||
|
|
||||||
const hi = try ev.evaluate(tst, &.{.{ .integer = 7 }}); // 7+5=12 > 10 -> 1
|
|
||||||
try std.testing.expectEqual(@as(u64, 1), try hi.asInt());
|
|
||||||
const lo = try ev.evaluate(tst, &.{.{ .integer = 2 }}); // 2+5=7 !> 10 -> 0
|
|
||||||
try std.testing.expectEqual(@as(u64, 0), try lo.asInt());
|
|
||||||
}
|
|
||||||
@@ -1,737 +0,0 @@
|
|||||||
//! A tree-walking AML interpreter — the evaluation stage on top of the parser's
|
|
||||||
//! structural namespace. It executes control methods (their bodies captured by
|
|
||||||
//! the parser) far enough to serve device discovery: device status (`_STA`, is a
|
|
||||||
//! device present), current resource settings (`_CRS`), and the operators, control
|
|
||||||
//! flow, locals/args, and
|
|
||||||
//! OperationRegion field access those methods reach for.
|
|
||||||
//!
|
|
||||||
//! Scope: integers, buffers, strings, packages, and references; If/Else/While/
|
|
||||||
//! Return; the arithmetic/logic operators; method invocation; Name/Local/Arg
|
|
||||||
//! access; CreateField buffer patching (the common current-resource-settings
|
|
||||||
//! (`_CRS`) idiom); and field
|
|
||||||
//! reads/writes against SystemMemory and SystemIO regions. Opcodes outside this
|
|
||||||
//! set return `error.Unsupported`, which callers treat as "couldn't evaluate" and
|
|
||||||
//! fall back — never a hard failure.
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
const op = @import("opcodes.zig");
|
|
||||||
const nsp = @import("namespace.zig");
|
|
||||||
const Node = nsp.Node;
|
|
||||||
const Namespace = nsp.Namespace;
|
|
||||||
|
|
||||||
/// Injected hardware access for OperationRegion reads/writes (the arch VMM + pio).
|
|
||||||
pub const Hal = struct {
|
|
||||||
mapMmio: *const fn (virt: u64, phys: u64, writable: bool) void,
|
|
||||||
pioRead: *const fn (width: u8, port: u16) u32,
|
|
||||||
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
|
||||||
};
|
|
||||||
|
|
||||||
pub const Error = error{ Unsupported, Truncated, DivByZero } || std.mem.Allocator.Error;
|
|
||||||
|
|
||||||
/// A runtime AML value.
|
|
||||||
pub const Object = union(enum) {
|
|
||||||
uninitialized,
|
|
||||||
integer: u64,
|
|
||||||
buffer: []u8,
|
|
||||||
string: []u8,
|
|
||||||
package: []Object,
|
|
||||||
reference: *Node,
|
|
||||||
|
|
||||||
pub fn asInt(self: Object) Error!u64 {
|
|
||||||
return switch (self) {
|
|
||||||
.integer => |v| v,
|
|
||||||
.buffer => |b| blk: {
|
|
||||||
var v: u64 = 0;
|
|
||||||
for (b, 0..) |byte, i| {
|
|
||||||
if (i >= 8) break;
|
|
||||||
v |= @as(u64, byte) << @intCast(i * 8);
|
|
||||||
}
|
|
||||||
break :blk v;
|
|
||||||
},
|
|
||||||
else => error.Unsupported,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
const max_segs = 16;
|
|
||||||
const NamePath = struct {
|
|
||||||
rooted: bool = false,
|
|
||||||
parents: u8 = 0,
|
|
||||||
segs: [max_segs][4]u8 = undefined,
|
|
||||||
count: usize = 0,
|
|
||||||
fn slice(self: *const NamePath) []const [4]u8 {
|
|
||||||
return self.segs[0..self.count];
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
const Cursor = struct {
|
|
||||||
b: []const u8,
|
|
||||||
i: usize = 0,
|
|
||||||
|
|
||||||
fn eof(self: *Cursor) bool {
|
|
||||||
return self.i >= self.b.len;
|
|
||||||
}
|
|
||||||
fn peek(self: *Cursor) ?u8 {
|
|
||||||
return if (self.eof()) null else self.b[self.i];
|
|
||||||
}
|
|
||||||
fn byte(self: *Cursor) Error!u8 {
|
|
||||||
if (self.eof()) return error.Truncated;
|
|
||||||
const v = self.b[self.i];
|
|
||||||
self.i += 1;
|
|
||||||
return v;
|
|
||||||
}
|
|
||||||
fn take(self: *Cursor, n: usize) Error![]const u8 {
|
|
||||||
if (self.i + n > self.b.len) return error.Truncated;
|
|
||||||
const s = self.b[self.i .. self.i + n];
|
|
||||||
self.i += n;
|
|
||||||
return s;
|
|
||||||
}
|
|
||||||
fn pkgLen(self: *Cursor) Error!usize {
|
|
||||||
const lead = try self.byte();
|
|
||||||
const follow: usize = lead >> 6;
|
|
||||||
if (follow == 0) return lead & 0x3F;
|
|
||||||
var value: usize = lead & 0x0F;
|
|
||||||
var k: usize = 0;
|
|
||||||
while (k < follow) : (k += 1) value |= @as(usize, try self.byte()) << @intCast(4 + k * 8);
|
|
||||||
return value;
|
|
||||||
}
|
|
||||||
fn nameString(self: *Cursor) Error!NamePath {
|
|
||||||
var np = NamePath{};
|
|
||||||
if (self.peek() == op.root_char) {
|
|
||||||
np.rooted = true;
|
|
||||||
self.i += 1;
|
|
||||||
} else {
|
|
||||||
while (self.peek() == op.parent_prefix_char) : (self.i += 1) np.parents += 1;
|
|
||||||
}
|
|
||||||
const lead = self.peek() orelse return np;
|
|
||||||
switch (lead) {
|
|
||||||
0x00 => self.i += 1,
|
|
||||||
op.dual_name_prefix => {
|
|
||||||
self.i += 1;
|
|
||||||
try self.seg(&np);
|
|
||||||
try self.seg(&np);
|
|
||||||
},
|
|
||||||
op.multi_name_prefix => {
|
|
||||||
self.i += 1;
|
|
||||||
const cnt = try self.byte();
|
|
||||||
var k: usize = 0;
|
|
||||||
while (k < cnt) : (k += 1) try self.seg(&np);
|
|
||||||
},
|
|
||||||
else => try self.seg(&np),
|
|
||||||
}
|
|
||||||
return np;
|
|
||||||
}
|
|
||||||
fn seg(self: *Cursor, np: *NamePath) Error!void {
|
|
||||||
const s = try self.take(4);
|
|
||||||
if (np.count < max_segs) {
|
|
||||||
np.segs[np.count] = s[0..4].*;
|
|
||||||
np.count += 1;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
const Frame = struct {
|
|
||||||
args: [7]Object = .{.uninitialized} ** 7,
|
|
||||||
locals: [8]Object = .{.uninitialized} ** 8,
|
|
||||||
scope: *Node,
|
|
||||||
ret: Object = .uninitialized,
|
|
||||||
returned: bool = false,
|
|
||||||
broke: bool = false,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// A CreateField binding: a name that indexes into a buffer object.
|
|
||||||
const BufField = struct { buf: *Node, byte_off: usize, bit_width: u32 };
|
|
||||||
|
|
||||||
pub const Interp = struct {
|
|
||||||
ns: *Namespace,
|
|
||||||
hal: Hal,
|
|
||||||
arena: std.mem.Allocator,
|
|
||||||
/// Runtime object overrides for Name nodes (Store targets, patched buffers).
|
|
||||||
dyn: std.AutoHashMapUnmanaged(*Node, Object) = .{},
|
|
||||||
/// CreateField bindings active for the current evaluation.
|
|
||||||
fields: std.AutoHashMapUnmanaged(*Node, BufField) = .{},
|
|
||||||
|
|
||||||
pub fn init(ns: *Namespace, hal: Hal, arena: std.mem.Allocator) Interp {
|
|
||||||
return .{ .ns = ns, .hal = hal, .arena = arena };
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Evaluate a namespace object: invoke a Method, read a Name's value, or read a
|
|
||||||
/// Field. Resets per-evaluation runtime state first.
|
|
||||||
pub fn evaluate(self: *Interp, node: *Node, args: []const Object) Error!Object {
|
|
||||||
self.dyn.clearRetainingCapacity();
|
|
||||||
self.fields.clearRetainingCapacity();
|
|
||||||
return self.invoke(node, args);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn invoke(self: *Interp, node: *Node, args: []const Object) Error!Object {
|
|
||||||
switch (node.kind) {
|
|
||||||
.method => {
|
|
||||||
var frame = Frame{ .scope = node };
|
|
||||||
for (args, 0..) |a, i| {
|
|
||||||
if (i < frame.args.len) frame.args[i] = a;
|
|
||||||
}
|
|
||||||
var cur = Cursor{ .b = node.value };
|
|
||||||
try self.execList(&cur, &frame);
|
|
||||||
return frame.ret;
|
|
||||||
},
|
|
||||||
.name => {
|
|
||||||
if (self.dyn.get(node)) |o| return o;
|
|
||||||
var cur = Cursor{ .b = node.value };
|
|
||||||
var frame = Frame{ .scope = node.parent orelse self.ns.root };
|
|
||||||
return self.term(&cur, &frame);
|
|
||||||
},
|
|
||||||
.field => return .{ .integer = try self.readField(node) },
|
|
||||||
else => return .{ .reference = node },
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Execute a TermList until it ends or the frame returns/breaks.
|
|
||||||
fn execList(self: *Interp, cur: *Cursor, frame: *Frame) Error!void {
|
|
||||||
while (!cur.eof() and !frame.returned and !frame.broke) {
|
|
||||||
_ = try self.term(cur, frame);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Evaluate/execute one term, returning its value (`.uninitialized` for pure
|
|
||||||
/// statements).
|
|
||||||
fn term(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const lead = cur.peek() orelse return error.Truncated;
|
|
||||||
if (isNameStart(lead)) return self.nameRef(cur, frame);
|
|
||||||
_ = try cur.byte();
|
|
||||||
|
|
||||||
return switch (lead) {
|
|
||||||
op.zero_op => Object{ .integer = 0 },
|
|
||||||
op.one_op => Object{ .integer = 1 },
|
|
||||||
op.ones_op => Object{ .integer = ~@as(u64, 0) },
|
|
||||||
op.byte_prefix => Object{ .integer = try self.readConst(cur, 1) },
|
|
||||||
op.word_prefix => Object{ .integer = try self.readConst(cur, 2) },
|
|
||||||
op.dword_prefix => Object{ .integer = try self.readConst(cur, 4) },
|
|
||||||
op.qword_prefix => Object{ .integer = try self.readConst(cur, 8) },
|
|
||||||
op.string_prefix => try self.readString(cur),
|
|
||||||
op.buffer_op => try self.buffer(cur, frame),
|
|
||||||
op.package_op, op.var_package_op => try self.package(cur, frame, lead == op.var_package_op),
|
|
||||||
|
|
||||||
op.local0_op...op.local7_op => frame.locals[lead - op.local0_op],
|
|
||||||
op.arg0_op...op.arg6_op => frame.args[lead - op.arg0_op],
|
|
||||||
|
|
||||||
op.return_op => blk: {
|
|
||||||
frame.ret = try self.term(cur, frame);
|
|
||||||
frame.returned = true;
|
|
||||||
break :blk .uninitialized;
|
|
||||||
},
|
|
||||||
op.break_op => blk: {
|
|
||||||
frame.broke = true;
|
|
||||||
break :blk .uninitialized;
|
|
||||||
},
|
|
||||||
op.continue_op, op.noop_op => .uninitialized,
|
|
||||||
|
|
||||||
op.if_op => try self.ifElse(cur, frame),
|
|
||||||
op.while_op => try self.whileLoop(cur, frame),
|
|
||||||
op.store_op => try self.store(cur, frame),
|
|
||||||
op.increment_op => try self.incDec(cur, frame, 1),
|
|
||||||
op.decrement_op => try self.incDec(cur, frame, -1),
|
|
||||||
|
|
||||||
op.add_op => try self.binary(cur, frame, .add),
|
|
||||||
op.subtract_op => try self.binary(cur, frame, .sub),
|
|
||||||
op.multiply_op => try self.binary(cur, frame, .mul),
|
|
||||||
op.mod_op => try self.binary(cur, frame, .mod),
|
|
||||||
op.and_op => try self.binary(cur, frame, .band),
|
|
||||||
op.or_op => try self.binary(cur, frame, .bor),
|
|
||||||
op.xor_op => try self.binary(cur, frame, .bxor),
|
|
||||||
op.nand_op => try self.binary(cur, frame, .nand),
|
|
||||||
op.nor_op => try self.binary(cur, frame, .nor),
|
|
||||||
op.shift_left_op => try self.binary(cur, frame, .shl),
|
|
||||||
op.shift_right_op => try self.binary(cur, frame, .shr),
|
|
||||||
op.divide_op => try self.divide(cur, frame),
|
|
||||||
|
|
||||||
op.land_op => try self.logic2(cur, frame, .land),
|
|
||||||
op.lor_op => try self.logic2(cur, frame, .lor),
|
|
||||||
op.lequal_op => try self.logic2(cur, frame, .eq),
|
|
||||||
op.lgreater_op => try self.logic2(cur, frame, .gt),
|
|
||||||
op.lless_op => try self.logic2(cur, frame, .lt),
|
|
||||||
op.lnot_op => try self.lnot(cur, frame),
|
|
||||||
|
|
||||||
op.not_op => blk: {
|
|
||||||
const v = try self.evalInt(cur, frame);
|
|
||||||
const r = ~v;
|
|
||||||
try self.storeTarget(cur, frame, .{ .integer = r });
|
|
||||||
break :blk .{ .integer = r };
|
|
||||||
},
|
|
||||||
|
|
||||||
op.size_of_op => try self.sizeOf(cur, frame),
|
|
||||||
op.index_op => try self.index(cur, frame),
|
|
||||||
op.deref_of_op => try self.derefOf(cur, frame),
|
|
||||||
op.to_integer_op => blk: {
|
|
||||||
const v = try self.evalInt(cur, frame);
|
|
||||||
try self.storeTarget(cur, frame, .{ .integer = v });
|
|
||||||
break :blk .{ .integer = v };
|
|
||||||
},
|
|
||||||
op.to_buffer_op => try self.passThroughUnary(cur, frame),
|
|
||||||
|
|
||||||
op.ext_op_prefix => try self.ext(cur, frame),
|
|
||||||
|
|
||||||
// CreateXField: source, index, name (bit widths differ by op)
|
|
||||||
op.create_bit_field_op => try self.createField(cur, frame, 1),
|
|
||||||
op.create_byte_field_op => try self.createField(cur, frame, 8),
|
|
||||||
op.create_word_field_op => try self.createField(cur, frame, 16),
|
|
||||||
op.create_dword_field_op => try self.createField(cur, frame, 32),
|
|
||||||
op.create_qword_field_op => try self.createField(cur, frame, 64),
|
|
||||||
|
|
||||||
else => error.Unsupported,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- name references ----------------------------------------------------
|
|
||||||
|
|
||||||
fn nameRef(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const np = try cur.nameString();
|
|
||||||
const node = self.ns.resolve(frame.scope, np.rooted, np.parents, np.slice()) orelse
|
|
||||||
return .uninitialized; // unknown name -> treat as uninitialised
|
|
||||||
switch (node.kind) {
|
|
||||||
.method => {
|
|
||||||
var argbuf: [7]Object = undefined;
|
|
||||||
var i: usize = 0;
|
|
||||||
while (i < node.arg_count and i < argbuf.len) : (i += 1) argbuf[i] = try self.term(cur, frame);
|
|
||||||
return self.invoke(node, argbuf[0..@min(node.arg_count, argbuf.len)]);
|
|
||||||
},
|
|
||||||
.field => return .{ .integer = try self.readField(node) },
|
|
||||||
.name => return self.invoke(node, &.{}),
|
|
||||||
else => return .{ .reference = node },
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- data objects -------------------------------------------------------
|
|
||||||
|
|
||||||
fn readConst(self: *Interp, cur: *Cursor, n: usize) Error!u64 {
|
|
||||||
_ = self;
|
|
||||||
const bytes = try cur.take(n);
|
|
||||||
var v: u64 = 0;
|
|
||||||
for (bytes, 0..) |b, i| v |= @as(u64, b) << @intCast(i * 8);
|
|
||||||
return v;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn readString(self: *Interp, cur: *Cursor) Error!Object {
|
|
||||||
const start = cur.i;
|
|
||||||
while (cur.peek()) |c| {
|
|
||||||
cur.i += 1;
|
|
||||||
if (c == 0) break;
|
|
||||||
}
|
|
||||||
const raw = cur.b[start .. cur.i - 1];
|
|
||||||
const s = try self.arena.dupe(u8, raw);
|
|
||||||
return .{ .string = s };
|
|
||||||
}
|
|
||||||
|
|
||||||
fn buffer(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const start = cur.i;
|
|
||||||
const len = try cur.pkgLen();
|
|
||||||
const end = @min(start + len, cur.b.len);
|
|
||||||
const size = try self.evalInt(cur, frame);
|
|
||||||
const data = cur.b[@min(cur.i, end)..end];
|
|
||||||
const buf = try self.arena.alloc(u8, @intCast(size));
|
|
||||||
@memset(buf, 0);
|
|
||||||
@memcpy(buf[0..@min(buf.len, data.len)], data[0..@min(buf.len, data.len)]);
|
|
||||||
cur.i = end;
|
|
||||||
return .{ .buffer = buf };
|
|
||||||
}
|
|
||||||
|
|
||||||
fn package(self: *Interp, cur: *Cursor, frame: *Frame, variable: bool) Error!Object {
|
|
||||||
const start = cur.i;
|
|
||||||
const len = try cur.pkgLen();
|
|
||||||
const end = @min(start + len, cur.b.len);
|
|
||||||
const count: usize = if (variable) @intCast(try self.evalInt(cur, frame)) else try cur.byte();
|
|
||||||
const elems = try self.arena.alloc(Object, count);
|
|
||||||
var i: usize = 0;
|
|
||||||
while (i < count and cur.i < end) : (i += 1) elems[i] = try self.term(cur, frame);
|
|
||||||
while (i < count) : (i += 1) elems[i] = .uninitialized;
|
|
||||||
cur.i = end;
|
|
||||||
return .{ .package = elems };
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- operators ----------------------------------------------------------
|
|
||||||
|
|
||||||
const BinOp = enum { add, sub, mul, mod, band, bor, bxor, nand, nor, shl, shr };
|
|
||||||
|
|
||||||
fn binary(self: *Interp, cur: *Cursor, frame: *Frame, kind: BinOp) Error!Object {
|
|
||||||
const a = try self.evalInt(cur, frame);
|
|
||||||
const b = try self.evalInt(cur, frame);
|
|
||||||
const r: u64 = switch (kind) {
|
|
||||||
.add => a +% b,
|
|
||||||
.sub => a -% b,
|
|
||||||
.mul => a *% b,
|
|
||||||
.mod => if (b == 0) return error.DivByZero else a % b,
|
|
||||||
.band => a & b,
|
|
||||||
.bor => a | b,
|
|
||||||
.bxor => a ^ b,
|
|
||||||
.nand => ~(a & b),
|
|
||||||
.nor => ~(a | b),
|
|
||||||
.shl => if (b >= 64) 0 else a << @intCast(b),
|
|
||||||
.shr => if (b >= 64) 0 else a >> @intCast(b),
|
|
||||||
};
|
|
||||||
try self.storeTarget(cur, frame, .{ .integer = r });
|
|
||||||
return .{ .integer = r };
|
|
||||||
}
|
|
||||||
|
|
||||||
fn divide(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const a = try self.evalInt(cur, frame);
|
|
||||||
const b = try self.evalInt(cur, frame);
|
|
||||||
if (b == 0) return error.DivByZero;
|
|
||||||
try self.storeTarget(cur, frame, .{ .integer = a % b }); // remainder target
|
|
||||||
try self.storeTarget(cur, frame, .{ .integer = a / b }); // quotient target
|
|
||||||
return .{ .integer = a / b };
|
|
||||||
}
|
|
||||||
|
|
||||||
const LogicOp = enum { land, lor, eq, gt, lt };
|
|
||||||
|
|
||||||
fn logic2(self: *Interp, cur: *Cursor, frame: *Frame, kind: LogicOp) Error!Object {
|
|
||||||
const a = try self.evalInt(cur, frame);
|
|
||||||
const b = try self.evalInt(cur, frame);
|
|
||||||
const r = switch (kind) {
|
|
||||||
.land => a != 0 and b != 0,
|
|
||||||
.lor => a != 0 or b != 0,
|
|
||||||
.eq => a == b,
|
|
||||||
.gt => a > b,
|
|
||||||
.lt => a < b,
|
|
||||||
};
|
|
||||||
return .{ .integer = if (r) ~@as(u64, 0) else 0 };
|
|
||||||
}
|
|
||||||
|
|
||||||
fn lnot(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
// 0x92 0x93/94/95 are the compound comparisons.
|
|
||||||
const b = cur.peek() orelse return error.Truncated;
|
|
||||||
switch (b) {
|
|
||||||
op.lnot.not_equal => {
|
|
||||||
cur.i += 1;
|
|
||||||
const x = try self.evalInt(cur, frame);
|
|
||||||
const y = try self.evalInt(cur, frame);
|
|
||||||
return .{ .integer = if (x != y) ~@as(u64, 0) else 0 };
|
|
||||||
},
|
|
||||||
op.lnot.less_equal => {
|
|
||||||
cur.i += 1;
|
|
||||||
const x = try self.evalInt(cur, frame);
|
|
||||||
const y = try self.evalInt(cur, frame);
|
|
||||||
return .{ .integer = if (x <= y) ~@as(u64, 0) else 0 };
|
|
||||||
},
|
|
||||||
op.lnot.greater_equal => {
|
|
||||||
cur.i += 1;
|
|
||||||
const x = try self.evalInt(cur, frame);
|
|
||||||
const y = try self.evalInt(cur, frame);
|
|
||||||
return .{ .integer = if (x >= y) ~@as(u64, 0) else 0 };
|
|
||||||
},
|
|
||||||
else => {
|
|
||||||
const x = try self.evalInt(cur, frame);
|
|
||||||
return .{ .integer = if (x == 0) ~@as(u64, 0) else 0 };
|
|
||||||
},
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn incDec(self: *Interp, cur: *Cursor, frame: *Frame, delta: i64) Error!Object {
|
|
||||||
// Operand is a SuperName that is both read and written.
|
|
||||||
const save = cur.i;
|
|
||||||
const cur_val = try self.term(cur, frame);
|
|
||||||
const v = try cur_val.asInt();
|
|
||||||
const r = if (delta > 0) v +% 1 else v -% 1;
|
|
||||||
var tcur = Cursor{ .b = cur.b, .i = save };
|
|
||||||
try self.storeInto(&tcur, frame, .{ .integer = r });
|
|
||||||
return .{ .integer = r };
|
|
||||||
}
|
|
||||||
|
|
||||||
fn sizeOf(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const o = try self.term(cur, frame);
|
|
||||||
return .{ .integer = switch (o) {
|
|
||||||
.buffer => |b| b.len,
|
|
||||||
.string => |s| s.len,
|
|
||||||
.package => |p| p.len,
|
|
||||||
else => 0,
|
|
||||||
} };
|
|
||||||
}
|
|
||||||
|
|
||||||
fn passThroughUnary(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const o = try self.term(cur, frame);
|
|
||||||
try self.storeTarget(cur, frame, o);
|
|
||||||
return o;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn index(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const src = try self.term(cur, frame);
|
|
||||||
const idx: usize = @intCast(try self.evalInt(cur, frame));
|
|
||||||
// Optional target (a reference); we don't materialise references, so store
|
|
||||||
// the indexed value if a target is present.
|
|
||||||
const val: Object = switch (src) {
|
|
||||||
.buffer => |b| .{ .integer = if (idx < b.len) b[idx] else 0 },
|
|
||||||
.package => |p| if (idx < p.len) p[idx] else .uninitialized,
|
|
||||||
.string => |s| .{ .integer = if (idx < s.len) s[idx] else 0 },
|
|
||||||
else => .uninitialized,
|
|
||||||
};
|
|
||||||
try self.storeTarget(cur, frame, val);
|
|
||||||
return val;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn derefOf(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const o = try self.term(cur, frame);
|
|
||||||
return switch (o) {
|
|
||||||
.reference => |n| self.invoke(n, &.{}),
|
|
||||||
else => o,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- control flow -------------------------------------------------------
|
|
||||||
|
|
||||||
fn ifElse(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const start = cur.i;
|
|
||||||
const end = @min(start + try cur.pkgLen(), cur.b.len);
|
|
||||||
const cond = try self.evalInt(cur, frame);
|
|
||||||
if (cond != 0) {
|
|
||||||
var body = Cursor{ .b = cur.b[0..end], .i = cur.i };
|
|
||||||
try self.execList(&body, frame);
|
|
||||||
cur.i = end;
|
|
||||||
// Skip a trailing Else.
|
|
||||||
if (cur.peek() == op.else_op) {
|
|
||||||
cur.i += 1;
|
|
||||||
const es = cur.i;
|
|
||||||
const ee = @min(es + try cur.pkgLen(), cur.b.len);
|
|
||||||
cur.i = ee;
|
|
||||||
}
|
|
||||||
} else {
|
|
||||||
cur.i = end;
|
|
||||||
if (cur.peek() == op.else_op) {
|
|
||||||
cur.i += 1;
|
|
||||||
const es = cur.i;
|
|
||||||
const ee = @min(es + try cur.pkgLen(), cur.b.len);
|
|
||||||
var body = Cursor{ .b = cur.b[0..ee], .i = cur.i };
|
|
||||||
try self.execList(&body, frame);
|
|
||||||
cur.i = ee;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return .uninitialized;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn whileLoop(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const start = cur.i;
|
|
||||||
const end = @min(start + try cur.pkgLen(), cur.b.len);
|
|
||||||
const pred_at = cur.i;
|
|
||||||
var guard: usize = 0;
|
|
||||||
while (guard < 100_000) : (guard += 1) {
|
|
||||||
var pc = Cursor{ .b = cur.b[0..end], .i = pred_at };
|
|
||||||
const cond = try self.evalInt(&pc, frame);
|
|
||||||
if (cond == 0) break;
|
|
||||||
var body = Cursor{ .b = cur.b[0..end], .i = pc.i };
|
|
||||||
try self.execList(&body, frame);
|
|
||||||
if (frame.returned) break;
|
|
||||||
if (frame.broke) {
|
|
||||||
frame.broke = false;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
cur.i = end;
|
|
||||||
return .uninitialized;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- store --------------------------------------------------------------
|
|
||||||
|
|
||||||
fn store(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const value = try self.term(cur, frame);
|
|
||||||
try self.storeInto(cur, frame, value);
|
|
||||||
return value;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// A Store *target* that may be NullName (no store).
|
|
||||||
fn storeTarget(self: *Interp, cur: *Cursor, frame: *Frame, value: Object) Error!void {
|
|
||||||
if (cur.peek() == 0x00) {
|
|
||||||
cur.i += 1; // NullName
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
try self.storeInto(cur, frame, value);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn storeInto(self: *Interp, cur: *Cursor, frame: *Frame, value: Object) Error!void {
|
|
||||||
const lead = cur.peek() orelse return error.Truncated;
|
|
||||||
if (isNameStart(lead)) {
|
|
||||||
const np = try cur.nameString();
|
|
||||||
const node = self.ns.resolve(frame.scope, np.rooted, np.parents, np.slice()) orelse return;
|
|
||||||
if (self.fields.get(node)) |bf| {
|
|
||||||
try self.writeBufField(bf, try value.asInt());
|
|
||||||
} else if (node.kind == .field) {
|
|
||||||
try self.writeField(node, try value.asInt());
|
|
||||||
} else {
|
|
||||||
try self.dyn.put(self.arena, node, value);
|
|
||||||
}
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
_ = try cur.byte();
|
|
||||||
switch (lead) {
|
|
||||||
0x00 => {}, // NullName
|
|
||||||
op.local0_op...op.local7_op => frame.locals[lead - op.local0_op] = value,
|
|
||||||
op.arg0_op...op.arg6_op => frame.args[lead - op.arg0_op] = value,
|
|
||||||
op.index_op => {
|
|
||||||
const src = try self.term(cur, frame);
|
|
||||||
const idx: usize = @intCast(try self.evalInt(cur, frame));
|
|
||||||
switch (src) {
|
|
||||||
.buffer => |b| if (idx < b.len) {
|
|
||||||
b[idx] = @truncate(try value.asInt());
|
|
||||||
},
|
|
||||||
.package => |p| if (idx < p.len) {
|
|
||||||
p[idx] = value;
|
|
||||||
},
|
|
||||||
else => {},
|
|
||||||
}
|
|
||||||
},
|
|
||||||
else => return error.Unsupported,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- CreateField (buffer patching) --------------------------------------
|
|
||||||
|
|
||||||
fn createField(self: *Interp, cur: *Cursor, frame: *Frame, bit_width: u32) Error!Object {
|
|
||||||
const src = try self.term(cur, frame); // source buffer (as a reference or value)
|
|
||||||
const bit_index = try self.evalInt(cur, frame);
|
|
||||||
const np = try cur.nameString();
|
|
||||||
const node = self.ns.resolve(frame.scope, np.rooted, np.parents, np.slice()) orelse return .uninitialized;
|
|
||||||
|
|
||||||
// Bind the new name to the source buffer's node so stores land in it.
|
|
||||||
const buf_node: *Node = switch (src) {
|
|
||||||
.reference => |n| n,
|
|
||||||
else => return .uninitialized,
|
|
||||||
};
|
|
||||||
// Materialise the buffer into `dyn` so patches persist and are returned.
|
|
||||||
if (self.dyn.get(buf_node) == null) {
|
|
||||||
const val = try self.invoke(buf_node, &.{});
|
|
||||||
try self.dyn.put(self.arena, buf_node, val);
|
|
||||||
}
|
|
||||||
const byte_off: usize = @intCast(bit_index / 8);
|
|
||||||
try self.fields.put(self.arena, node, .{ .buf = buf_node, .byte_off = byte_off, .bit_width = bit_width });
|
|
||||||
return .uninitialized;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn writeBufField(self: *Interp, bf: BufField, value: u64) Error!void {
|
|
||||||
const obj = self.dyn.get(bf.buf) orelse return;
|
|
||||||
const buf = switch (obj) {
|
|
||||||
.buffer => |b| b,
|
|
||||||
else => return,
|
|
||||||
};
|
|
||||||
const nbytes = (bf.bit_width + 7) / 8;
|
|
||||||
var k: usize = 0;
|
|
||||||
while (k < nbytes and bf.byte_off + k < buf.len) : (k += 1) {
|
|
||||||
buf[bf.byte_off + k] = @truncate(value >> @intCast(k * 8));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- OperationRegion field access ---------------------------------------
|
|
||||||
|
|
||||||
fn readField(self: *Interp, field: *Node) Error!u64 {
|
|
||||||
const region = field.region orelse return error.Unsupported;
|
|
||||||
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
|
||||||
const base = try self.regionBase(region);
|
|
||||||
const start_byte = base + field.bit_offset / 8;
|
|
||||||
const shift: u7 = @intCast(field.bit_offset % 8);
|
|
||||||
const total = @as(usize, shift) + field.bit_width;
|
|
||||||
const nbytes = (total + 7) / 8;
|
|
||||||
var raw: u128 = 0;
|
|
||||||
var k: usize = 0;
|
|
||||||
while (k < nbytes) : (k += 1) {
|
|
||||||
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
|
||||||
}
|
|
||||||
const masked = (raw >> shift) & bitMask(field.bit_width);
|
|
||||||
return @truncate(masked);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn writeField(self: *Interp, field: *Node, value: u64) Error!void {
|
|
||||||
const region = field.region orelse return error.Unsupported;
|
|
||||||
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
|
||||||
const base = try self.regionBase(region);
|
|
||||||
const start_byte = base + field.bit_offset / 8;
|
|
||||||
const shift: u7 = @intCast(field.bit_offset % 8);
|
|
||||||
const total = @as(usize, shift) + field.bit_width;
|
|
||||||
const nbytes = (total + 7) / 8;
|
|
||||||
// Read-modify-write byte by byte.
|
|
||||||
var raw: u128 = 0;
|
|
||||||
var k: usize = 0;
|
|
||||||
while (k < nbytes) : (k += 1) {
|
|
||||||
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
|
||||||
}
|
|
||||||
const mask = bitMask(field.bit_width) << shift;
|
|
||||||
raw = (raw & ~mask) | ((@as(u128, value) << shift) & mask);
|
|
||||||
k = 0;
|
|
||||||
while (k < nbytes) : (k += 1) {
|
|
||||||
try self.writeRegionByte(region.region_space, start_byte + k, @truncate(raw >> @intCast(k * 8)));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn regionBase(self: *Interp, region: *Node) Error!u64 {
|
|
||||||
var cur = Cursor{ .b = region.region_offset_aml };
|
|
||||||
var frame = Frame{ .scope = region.parent orelse self.ns.root };
|
|
||||||
return (try self.term(&cur, &frame)).asInt();
|
|
||||||
}
|
|
||||||
|
|
||||||
fn readRegionByte(self: *Interp, space: u8, addr: u64) Error!u8 {
|
|
||||||
switch (space) {
|
|
||||||
0 => { // SystemMemory
|
|
||||||
self.hal.mapMmio(addr & ~@as(u64, 0xFFF), addr & ~@as(u64, 0xFFF), true);
|
|
||||||
const p: *align(1) const volatile u8 = @ptrFromInt(addr);
|
|
||||||
return p.*;
|
|
||||||
},
|
|
||||||
1 => return @truncate(self.hal.pioRead(1, @intCast(addr & 0xFFFF))), // SystemIO
|
|
||||||
else => return error.Unsupported,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn writeRegionByte(self: *Interp, space: u8, addr: u64, value: u8) Error!void {
|
|
||||||
switch (space) {
|
|
||||||
0 => {
|
|
||||||
self.hal.mapMmio(addr & ~@as(u64, 0xFFF), addr & ~@as(u64, 0xFFF), true);
|
|
||||||
const p: *align(1) volatile u8 = @ptrFromInt(addr);
|
|
||||||
p.* = value;
|
|
||||||
},
|
|
||||||
1 => self.hal.pioWrite(1, @intCast(addr & 0xFFFF), value),
|
|
||||||
else => return error.Unsupported,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- extended opcodes ---------------------------------------------------
|
|
||||||
|
|
||||||
fn ext(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const e = try cur.byte();
|
|
||||||
switch (e) {
|
|
||||||
op.ext.debug => return .uninitialized,
|
|
||||||
op.ext.revision => return .{ .integer = 2 },
|
|
||||||
op.ext.timer => return .{ .integer = 0 },
|
|
||||||
// Mutex/Event ops are no-ops in this single-threaded evaluator.
|
|
||||||
op.ext.acquire => {
|
|
||||||
_ = try self.term(cur, frame); // mutex SuperName
|
|
||||||
_ = try cur.take(2); // timeout
|
|
||||||
return .{ .integer = 0 }; // acquired
|
|
||||||
},
|
|
||||||
op.ext.release, op.ext.reset, op.ext.signal => {
|
|
||||||
_ = try self.term(cur, frame);
|
|
||||||
return .uninitialized;
|
|
||||||
},
|
|
||||||
op.ext.wait => {
|
|
||||||
_ = try self.term(cur, frame);
|
|
||||||
_ = try self.term(cur, frame);
|
|
||||||
return .{ .integer = 0 };
|
|
||||||
},
|
|
||||||
op.ext.sleep, op.ext.stall => {
|
|
||||||
_ = try self.term(cur, frame);
|
|
||||||
return .uninitialized;
|
|
||||||
},
|
|
||||||
else => return error.Unsupported,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn evalInt(self: *Interp, cur: *Cursor, frame: *Frame) Error!u64 {
|
|
||||||
return (try self.term(cur, frame)).asInt();
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
fn bitMask(width: u32) u128 {
|
|
||||||
if (width >= 128) return ~@as(u128, 0);
|
|
||||||
return (@as(u128, 1) << @intCast(width)) - 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn isNameStart(b: u8) bool {
|
|
||||||
return (b >= op.name_char_start and b <= op.name_char_end) or
|
|
||||||
b == op.name_char_underscore or
|
|
||||||
b == op.root_char or
|
|
||||||
b == op.parent_prefix_char or
|
|
||||||
b == op.dual_name_prefix or
|
|
||||||
b == op.multi_name_prefix;
|
|
||||||
}
|
|
||||||
@@ -1,137 +0,0 @@
|
|||||||
//! AML opcode constants — the full ACPI Machine Language opcode table.
|
|
||||||
//!
|
|
||||||
//! Single-byte opcodes are plain values. Extended opcodes are a two-byte sequence
|
|
||||||
//! `ext_prefix` (0x5B) followed by a byte listed under `ext`. A few comparison
|
|
||||||
//! opcodes are `lnot_op` (0x92) followed by a second byte (see `lnot`).
|
|
||||||
|
|
||||||
// --- name / path characters -------------------------------------------------
|
|
||||||
pub const zero_op = 0x00;
|
|
||||||
pub const one_op = 0x01;
|
|
||||||
pub const alias_op = 0x06;
|
|
||||||
pub const name_op = 0x08;
|
|
||||||
pub const byte_prefix = 0x0A;
|
|
||||||
pub const word_prefix = 0x0B;
|
|
||||||
pub const dword_prefix = 0x0C;
|
|
||||||
pub const string_prefix = 0x0D;
|
|
||||||
pub const qword_prefix = 0x0E;
|
|
||||||
pub const scope_op = 0x10;
|
|
||||||
pub const buffer_op = 0x11;
|
|
||||||
pub const package_op = 0x12;
|
|
||||||
pub const var_package_op = 0x13;
|
|
||||||
pub const method_op = 0x14;
|
|
||||||
pub const external_op = 0x15;
|
|
||||||
|
|
||||||
pub const dual_name_prefix = 0x2E;
|
|
||||||
pub const multi_name_prefix = 0x2F;
|
|
||||||
pub const ext_op_prefix = 0x5B;
|
|
||||||
pub const root_char = 0x5C;
|
|
||||||
pub const parent_prefix_char = 0x5E;
|
|
||||||
pub const name_char_underscore = 0x5F;
|
|
||||||
|
|
||||||
pub const digit_char_start = 0x30;
|
|
||||||
pub const digit_char_end = 0x39;
|
|
||||||
pub const name_char_start = 0x41; // 'A'
|
|
||||||
pub const name_char_end = 0x5A; // 'Z'
|
|
||||||
|
|
||||||
// --- locals / args ----------------------------------------------------------
|
|
||||||
pub const local0_op = 0x60;
|
|
||||||
pub const local7_op = 0x67;
|
|
||||||
pub const arg0_op = 0x68;
|
|
||||||
pub const arg6_op = 0x6E;
|
|
||||||
|
|
||||||
// --- store / references / arithmetic ---------------------------------------
|
|
||||||
pub const store_op = 0x70;
|
|
||||||
pub const ref_of_op = 0x71;
|
|
||||||
pub const add_op = 0x72;
|
|
||||||
pub const concat_op = 0x73;
|
|
||||||
pub const subtract_op = 0x74;
|
|
||||||
pub const increment_op = 0x75;
|
|
||||||
pub const decrement_op = 0x76;
|
|
||||||
pub const multiply_op = 0x77;
|
|
||||||
pub const divide_op = 0x78;
|
|
||||||
pub const shift_left_op = 0x79;
|
|
||||||
pub const shift_right_op = 0x7A;
|
|
||||||
pub const and_op = 0x7B;
|
|
||||||
pub const nand_op = 0x7C;
|
|
||||||
pub const or_op = 0x7D;
|
|
||||||
pub const nor_op = 0x7E;
|
|
||||||
pub const xor_op = 0x7F;
|
|
||||||
pub const not_op = 0x80;
|
|
||||||
pub const find_set_left_bit_op = 0x81;
|
|
||||||
pub const find_set_right_bit_op = 0x82;
|
|
||||||
pub const deref_of_op = 0x83;
|
|
||||||
pub const concat_res_op = 0x84;
|
|
||||||
pub const mod_op = 0x85;
|
|
||||||
pub const notify_op = 0x86;
|
|
||||||
pub const size_of_op = 0x87;
|
|
||||||
pub const index_op = 0x88;
|
|
||||||
pub const match_op = 0x89;
|
|
||||||
pub const create_dword_field_op = 0x8A;
|
|
||||||
pub const create_word_field_op = 0x8B;
|
|
||||||
pub const create_byte_field_op = 0x8C;
|
|
||||||
pub const create_bit_field_op = 0x8D;
|
|
||||||
pub const object_type_op = 0x8E;
|
|
||||||
pub const create_qword_field_op = 0x8F;
|
|
||||||
|
|
||||||
pub const land_op = 0x90;
|
|
||||||
pub const lor_op = 0x91;
|
|
||||||
pub const lnot_op = 0x92; // may be followed by a second byte (see `lnot`)
|
|
||||||
pub const lequal_op = 0x93;
|
|
||||||
pub const lgreater_op = 0x94;
|
|
||||||
pub const lless_op = 0x95;
|
|
||||||
pub const to_buffer_op = 0x96;
|
|
||||||
pub const to_decimal_string_op = 0x97;
|
|
||||||
pub const to_hex_string_op = 0x98;
|
|
||||||
pub const to_integer_op = 0x99;
|
|
||||||
pub const to_string_op = 0x9C;
|
|
||||||
pub const copy_object_op = 0x9D;
|
|
||||||
pub const mid_op = 0x9E;
|
|
||||||
pub const continue_op = 0x9F;
|
|
||||||
pub const if_op = 0xA0;
|
|
||||||
pub const else_op = 0xA1;
|
|
||||||
pub const while_op = 0xA2;
|
|
||||||
pub const noop_op = 0xA3;
|
|
||||||
pub const return_op = 0xA4;
|
|
||||||
pub const break_op = 0xA5;
|
|
||||||
pub const break_point_op = 0xCC;
|
|
||||||
pub const ones_op = 0xFF;
|
|
||||||
|
|
||||||
/// Second bytes of the `lnot_op` (0x92) compound comparison opcodes.
|
|
||||||
pub const lnot = struct {
|
|
||||||
pub const not_equal = 0x93; // LNotEqualOp: 0x92 0x93
|
|
||||||
pub const less_equal = 0x94; // LLessEqualOp: 0x92 0x94
|
|
||||||
pub const greater_equal = 0x95; // LGreaterEqualOp: 0x92 0x95
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Second bytes of extended opcodes (prefixed by `ext_op_prefix`, 0x5B).
|
|
||||||
pub const ext = struct {
|
|
||||||
pub const mutex = 0x01;
|
|
||||||
pub const event = 0x02;
|
|
||||||
pub const cond_ref_of = 0x12;
|
|
||||||
pub const create_field = 0x13;
|
|
||||||
pub const load_table = 0x1F;
|
|
||||||
pub const load = 0x20;
|
|
||||||
pub const stall = 0x21;
|
|
||||||
pub const sleep = 0x22;
|
|
||||||
pub const acquire = 0x23;
|
|
||||||
pub const signal = 0x24;
|
|
||||||
pub const wait = 0x25;
|
|
||||||
pub const reset = 0x26;
|
|
||||||
pub const release = 0x27;
|
|
||||||
pub const from_bcd = 0x28;
|
|
||||||
pub const to_bcd = 0x29;
|
|
||||||
pub const unload = 0x2A;
|
|
||||||
pub const revision = 0x30;
|
|
||||||
pub const debug = 0x31;
|
|
||||||
pub const fatal = 0x32;
|
|
||||||
pub const timer = 0x33;
|
|
||||||
pub const op_region = 0x80;
|
|
||||||
pub const field = 0x81;
|
|
||||||
pub const device = 0x82;
|
|
||||||
pub const processor = 0x83;
|
|
||||||
pub const power_res = 0x84;
|
|
||||||
pub const thermal_zone = 0x85;
|
|
||||||
pub const index_field = 0x86;
|
|
||||||
pub const bank_field = 0x87;
|
|
||||||
pub const data_region = 0x88;
|
|
||||||
};
|
|
||||||
@@ -1,517 +0,0 @@
|
|||||||
//! Recursive-descent AML parser. Walks the entire byte stream — including method
|
|
||||||
//! bodies — building the ACPI namespace as it goes. It does not *evaluate*
|
|
||||||
//! anything (no OperationRegion reads, no arithmetic); it parses structure so the
|
|
||||||
//! cursor stays aligned and every named object is recorded.
|
|
||||||
//!
|
|
||||||
//! The one genuine ambiguity in AML is method invocation: a bare NameString in an
|
|
||||||
//! operand position is a call whose argument count is only known from the method's
|
|
||||||
//! (earlier) declaration. Because we build the namespace in the same in-order pass,
|
|
||||||
//! `resolve` finds that declaration and tells us how many operands to consume.
|
|
||||||
//!
|
|
||||||
//! Safety net: every object delimited by a PkgLength (Scope/Device/Method/If/While/
|
|
||||||
//! Field/Buffer/Package/…) is parsed within its known extent, and the cursor is
|
|
||||||
//! snapped to that extent afterwards. So a mis-resolved invocation can only desync
|
|
||||||
//! *within* one such object; the enclosing walk realigns at the boundary.
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
const op = @import("opcodes.zig");
|
|
||||||
const ns = @import("namespace.zig");
|
|
||||||
const Namespace = ns.Namespace;
|
|
||||||
const Node = ns.Node;
|
|
||||||
|
|
||||||
pub const Error = error{ Truncated, Malformed } || std.mem.Allocator.Error;
|
|
||||||
|
|
||||||
const max_segs = 64;
|
|
||||||
|
|
||||||
/// A parsed NameString: an optional root anchor or some parent hops, then a list
|
|
||||||
/// of 4-byte segments.
|
|
||||||
const NamePath = struct {
|
|
||||||
rooted: bool = false,
|
|
||||||
parents: u8 = 0,
|
|
||||||
segs: [max_segs][4]u8 = undefined,
|
|
||||||
count: usize = 0,
|
|
||||||
|
|
||||||
fn slice(self: *const NamePath) []const [4]u8 {
|
|
||||||
return self.segs[0..self.count];
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
pub const Parser = struct {
|
|
||||||
aml: []const u8,
|
|
||||||
pos: usize = 0,
|
|
||||||
namespace: *Namespace,
|
|
||||||
|
|
||||||
pub fn init(aml: []const u8, namespace: *Namespace) Parser {
|
|
||||||
return .{ .aml = aml, .namespace = namespace };
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Parse the whole block as a TermList under the namespace root. Returns the
|
|
||||||
/// number of bytes consumed — equal to `aml.len` for a clean full traversal.
|
|
||||||
pub fn parseAll(self: *Parser) usize {
|
|
||||||
self.termList(self.aml.len, self.namespace.root);
|
|
||||||
return self.pos;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- cursor primitives --------------------------------------------------
|
|
||||||
|
|
||||||
fn eof(self: *Parser) bool {
|
|
||||||
return self.pos >= self.aml.len;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn peek(self: *Parser) ?u8 {
|
|
||||||
return if (self.eof()) null else self.aml[self.pos];
|
|
||||||
}
|
|
||||||
|
|
||||||
fn readByte(self: *Parser) Error!u8 {
|
|
||||||
if (self.eof()) return error.Truncated;
|
|
||||||
const b = self.aml[self.pos];
|
|
||||||
self.pos += 1;
|
|
||||||
return b;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn skip(self: *Parser, n: usize) Error!void {
|
|
||||||
if (self.pos + n > self.aml.len) return error.Truncated;
|
|
||||||
self.pos += n;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn skipCString(self: *Parser) Error!void {
|
|
||||||
while (true) {
|
|
||||||
const b = try self.readByte();
|
|
||||||
if (b == 0) return;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// AML PkgLength: the lead byte's top two bits give how many extra bytes
|
|
||||||
/// follow; the value counts from the start of the PkgLength field.
|
|
||||||
fn readPkgLength(self: *Parser) Error!usize {
|
|
||||||
const lead = try self.readByte();
|
|
||||||
const follow: usize = lead >> 6;
|
|
||||||
if (follow == 0) return lead & 0x3F;
|
|
||||||
var value: usize = lead & 0x0F;
|
|
||||||
var i: usize = 0;
|
|
||||||
while (i < follow) : (i += 1) {
|
|
||||||
const b = try self.readByte();
|
|
||||||
value |= @as(usize, b) << @intCast(4 + i * 8);
|
|
||||||
}
|
|
||||||
return value;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn readNameSeg(self: *Parser) Error![4]u8 {
|
|
||||||
if (self.pos + 4 > self.aml.len) return error.Truncated;
|
|
||||||
const seg = self.aml[self.pos..][0..4].*;
|
|
||||||
self.pos += 4;
|
|
||||||
return seg;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn readNameString(self: *Parser) Error!NamePath {
|
|
||||||
var np = NamePath{};
|
|
||||||
// A NameString is either root-anchored or parent-relative, not both.
|
|
||||||
if (self.peek() == op.root_char) {
|
|
||||||
np.rooted = true;
|
|
||||||
self.pos += 1;
|
|
||||||
} else {
|
|
||||||
while (self.peek() == op.parent_prefix_char) : (self.pos += 1) np.parents += 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
const lead = self.peek() orelse return np;
|
|
||||||
switch (lead) {
|
|
||||||
0x00 => self.pos += 1, // NullName
|
|
||||||
op.dual_name_prefix => {
|
|
||||||
self.pos += 1;
|
|
||||||
try self.appendSeg(&np);
|
|
||||||
try self.appendSeg(&np);
|
|
||||||
},
|
|
||||||
op.multi_name_prefix => {
|
|
||||||
self.pos += 1;
|
|
||||||
const cnt = try self.readByte();
|
|
||||||
var i: usize = 0;
|
|
||||||
while (i < cnt) : (i += 1) try self.appendSeg(&np);
|
|
||||||
},
|
|
||||||
else => {
|
|
||||||
if (isNameStart(lead)) try self.appendSeg(&np);
|
|
||||||
},
|
|
||||||
}
|
|
||||||
return np;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn appendSeg(self: *Parser, np: *NamePath) Error!void {
|
|
||||||
const seg = try self.readNameSeg();
|
|
||||||
if (np.count < max_segs) {
|
|
||||||
np.segs[np.count] = seg;
|
|
||||||
np.count += 1;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- term list / object -------------------------------------------------
|
|
||||||
|
|
||||||
/// Parse objects until `end`, then snap to `end`. Any parse error resyncs to
|
|
||||||
/// the boundary rather than propagating — containment for the rare desync.
|
|
||||||
fn termList(self: *Parser, end: usize, scope: *Node) void {
|
|
||||||
while (self.pos < end) {
|
|
||||||
self.object(scope) catch break;
|
|
||||||
}
|
|
||||||
self.pos = end;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Parse exactly one object/term at the cursor. Used for both TermObjs and
|
|
||||||
/// operands (TermArg / SuperName / Target all reduce to "one object" for the
|
|
||||||
/// purpose of advancing the cursor).
|
|
||||||
fn object(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const lead = self.peek() orelse return error.Truncated;
|
|
||||||
if (isNameStart(lead)) return self.nameInvocation(scope);
|
|
||||||
|
|
||||||
_ = try self.readByte();
|
|
||||||
switch (lead) {
|
|
||||||
// constants and no-operand statements
|
|
||||||
op.zero_op, op.one_op, op.ones_op => {},
|
|
||||||
op.noop_op, op.continue_op, op.break_op, op.break_point_op => {},
|
|
||||||
op.local0_op...op.local7_op => {},
|
|
||||||
op.arg0_op...op.arg6_op => {},
|
|
||||||
|
|
||||||
// literal data
|
|
||||||
op.byte_prefix => try self.skip(1),
|
|
||||||
op.word_prefix => try self.skip(2),
|
|
||||||
op.dword_prefix => try self.skip(4),
|
|
||||||
op.qword_prefix => try self.skip(8),
|
|
||||||
op.string_prefix => try self.skipCString(),
|
|
||||||
|
|
||||||
// data containers (contents skipped via their PkgLength)
|
|
||||||
op.buffer_op, op.package_op, op.var_package_op => try self.skipPkg(),
|
|
||||||
|
|
||||||
// namespace modifiers / named objects
|
|
||||||
op.name_op => try self.opName(scope),
|
|
||||||
op.alias_op => try self.opAlias(scope),
|
|
||||||
op.scope_op => try self.opScopeLike(scope, .scope),
|
|
||||||
op.method_op => try self.opMethod(scope),
|
|
||||||
op.external_op => try self.opExternal(scope),
|
|
||||||
op.ext_op_prefix => try self.opExt(scope),
|
|
||||||
|
|
||||||
// control flow
|
|
||||||
op.if_op => try self.opIf(scope),
|
|
||||||
op.else_op => try self.opElse(scope),
|
|
||||||
op.while_op => try self.opWhile(scope),
|
|
||||||
op.return_op => try self.object(scope),
|
|
||||||
op.notify_op => try self.args(scope, 2),
|
|
||||||
|
|
||||||
// stores / references / unary+target
|
|
||||||
op.store_op => try self.args(scope, 2),
|
|
||||||
op.ref_of_op, op.deref_of_op, op.size_of_op, op.object_type_op => try self.args(scope, 1),
|
|
||||||
op.increment_op, op.decrement_op => try self.args(scope, 1),
|
|
||||||
op.not_op, op.find_set_left_bit_op, op.find_set_right_bit_op => try self.args(scope, 2),
|
|
||||||
op.to_buffer_op, op.to_decimal_string_op, op.to_hex_string_op, op.to_integer_op => try self.args(scope, 2),
|
|
||||||
op.copy_object_op => try self.args(scope, 2),
|
|
||||||
|
|
||||||
// binary + target
|
|
||||||
op.add_op, op.subtract_op, op.multiply_op, op.mod_op => try self.args(scope, 3),
|
|
||||||
op.and_op, op.nand_op, op.or_op, op.nor_op, op.xor_op => try self.args(scope, 3),
|
|
||||||
op.shift_left_op, op.shift_right_op, op.concat_op, op.concat_res_op, op.index_op => try self.args(scope, 3),
|
|
||||||
op.divide_op => try self.args(scope, 4),
|
|
||||||
op.to_string_op => try self.args(scope, 3),
|
|
||||||
op.mid_op => try self.args(scope, 4),
|
|
||||||
|
|
||||||
// logical
|
|
||||||
op.land_op, op.lor_op => try self.args(scope, 2),
|
|
||||||
op.lequal_op, op.lgreater_op, op.lless_op => try self.args(scope, 2),
|
|
||||||
op.lnot_op => try self.opLnot(scope),
|
|
||||||
|
|
||||||
op.match_op => try self.opMatch(scope),
|
|
||||||
|
|
||||||
// CreateXField: <source> <index> NameString
|
|
||||||
op.create_dword_field_op,
|
|
||||||
op.create_word_field_op,
|
|
||||||
op.create_byte_field_op,
|
|
||||||
op.create_bit_field_op,
|
|
||||||
op.create_qword_field_op,
|
|
||||||
=> try self.opCreateField(scope, 2),
|
|
||||||
|
|
||||||
else => return error.Malformed,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Parse `n` operands.
|
|
||||||
fn args(self: *Parser, scope: *Node, n: usize) Error!void {
|
|
||||||
var i: usize = 0;
|
|
||||||
while (i < n) : (i += 1) try self.object(scope);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// A NameString in operand/statement position: a method invocation (consuming
|
|
||||||
/// the callee's declared argument count) or a plain name reference.
|
|
||||||
fn nameInvocation(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
if (self.namespace.resolve(scope, np.rooted, np.parents, np.slice())) |node| {
|
|
||||||
if ((node.kind == .method or node.kind == .external) and node.arg_count > 0) {
|
|
||||||
try self.args(scope, node.arg_count);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Skip a PkgLength-delimited body wholesale (Buffer / Package / VarPackage):
|
|
||||||
/// the contents are pure data, never namespace declarations.
|
|
||||||
fn skipPkg(self: *Parser) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const len = try self.readPkgLength();
|
|
||||||
const end = start + len;
|
|
||||||
if (end > self.aml.len) return error.Truncated;
|
|
||||||
self.pos = end;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- namespace objects --------------------------------------------------
|
|
||||||
|
|
||||||
fn opName(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
const val_start = self.pos;
|
|
||||||
try self.object(scope); // the DataRefObject value
|
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .name);
|
|
||||||
node.value = self.aml[val_start..self.pos];
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opAlias(self: *Parser, scope: *Node) Error!void {
|
|
||||||
_ = try self.readNameString(); // source
|
|
||||||
const np = try self.readNameString(); // the alias name
|
|
||||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .alias);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opMethod(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
const np = try self.readNameString();
|
|
||||||
const flags = try self.readByte();
|
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .method);
|
|
||||||
node.arg_count = flags & 0x7;
|
|
||||||
// Capture the body for on-demand evaluation and skip it — objects declared
|
|
||||||
// inside a method are created at *runtime*, not at load, so they must not
|
|
||||||
// become permanent namespace nodes.
|
|
||||||
node.value = self.aml[self.pos..@min(end, self.aml.len)];
|
|
||||||
self.pos = end;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opExternal(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
_ = try self.readByte(); // object type
|
|
||||||
const arg_count = try self.readByte();
|
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .external);
|
|
||||||
node.arg_count = arg_count;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Scope / Device / ThermalZone: PkgLength, NameString, then a nested TermList.
|
|
||||||
fn opScopeLike(self: *Parser, scope: *Node, kind: ns.NodeKind) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
const np = try self.readNameString();
|
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), kind);
|
|
||||||
self.termList(end, node);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opProcessor(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
const np = try self.readNameString();
|
|
||||||
try self.skip(6); // ProcID(byte) + PblkAddr(dword) + PblkLen(byte)
|
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .processor);
|
|
||||||
self.termList(end, node);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opPowerRes(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
const np = try self.readNameString();
|
|
||||||
try self.skip(3); // SystemLevel(byte) + ResourceOrder(word)
|
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .power_res);
|
|
||||||
self.termList(end, node);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// OperationRegion: NameString, RegionSpace(byte), Offset(TermArg), Len(TermArg).
|
|
||||||
/// The offset/length expressions are kept as AML for lazy evaluation.
|
|
||||||
fn opRegion(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
const space = try self.readByte();
|
|
||||||
const off_start = self.pos;
|
|
||||||
try self.object(scope);
|
|
||||||
const off_end = self.pos;
|
|
||||||
try self.object(scope);
|
|
||||||
const len_end = self.pos;
|
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .region);
|
|
||||||
node.region_space = space;
|
|
||||||
node.region_offset_aml = self.aml[off_start..off_end];
|
|
||||||
node.region_len_aml = self.aml[off_end..len_end];
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opDataRegion(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
try self.args(scope, 3); // signature, oem id, oem table id (TermArgs)
|
|
||||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .region);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opMutex(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
try self.skip(1); // sync flags
|
|
||||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .mutex);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opEvent(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .event);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// CreateXField: `count` TermArgs then the new field's NameString.
|
|
||||||
fn opCreateField(self: *Parser, scope: *Node, count: usize) Error!void {
|
|
||||||
try self.args(scope, count);
|
|
||||||
const np = try self.readNameString();
|
|
||||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .name);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Field / IndexField / BankField: a region/bank reference, flags, then a
|
|
||||||
/// FieldList whose NamedFields become nodes in the current scope. For a plain
|
|
||||||
/// Field, the first NameString is the backing region — captured so field units
|
|
||||||
/// carry a region + bit position the evaluator can read/write.
|
|
||||||
fn opField(self: *Parser, scope: *Node, name_strings: u8, bank: bool) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
var region: ?*Node = null;
|
|
||||||
var i: u8 = 0;
|
|
||||||
while (i < name_strings) : (i += 1) {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
// Only a plain Field's single NameString denotes an OperationRegion.
|
|
||||||
if (name_strings == 1) region = self.namespace.resolve(scope, np.rooted, np.parents, np.slice());
|
|
||||||
}
|
|
||||||
if (bank) try self.object(scope); // bank value TermArg
|
|
||||||
const flags = try self.readByte();
|
|
||||||
self.fieldList(end, scope, region, flags & 0x0F);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn fieldList(self: *Parser, end: usize, scope: *Node, region: ?*Node, initial_access: u8) void {
|
|
||||||
var bit_offset: u32 = 0;
|
|
||||||
var access = initial_access;
|
|
||||||
while (self.pos < end) {
|
|
||||||
const lead = self.peek() orelse break;
|
|
||||||
switch (lead) {
|
|
||||||
0x00 => { // ReservedField: advances the bit position
|
|
||||||
self.pos += 1;
|
|
||||||
const width = self.readPkgLength() catch break;
|
|
||||||
bit_offset += @intCast(width);
|
|
||||||
},
|
|
||||||
0x01 => { // AccessField: AccessType (low nibble) + AccessAttrib
|
|
||||||
self.pos += 1;
|
|
||||||
const at = self.readByte() catch break;
|
|
||||||
self.skip(1) catch break;
|
|
||||||
access = at & 0x0F;
|
|
||||||
},
|
|
||||||
0x02 => { // ConnectField: NameString | BufferData
|
|
||||||
self.pos += 1;
|
|
||||||
self.object(scope) catch break;
|
|
||||||
},
|
|
||||||
0x03 => { // ExtendedAccessField: type + attrib + length
|
|
||||||
self.pos += 1;
|
|
||||||
self.skip(3) catch break;
|
|
||||||
},
|
|
||||||
else => { // NamedField: NameSeg + PkgLength (bit width)
|
|
||||||
const seg = self.readNameSeg() catch break;
|
|
||||||
const width = self.readPkgLength() catch break;
|
|
||||||
const unit = self.namespace.newFieldUnit(scope, seg) catch break;
|
|
||||||
unit.region = region;
|
|
||||||
unit.bit_offset = bit_offset;
|
|
||||||
unit.bit_width = @intCast(width);
|
|
||||||
unit.access_type = access;
|
|
||||||
bit_offset += @intCast(width);
|
|
||||||
},
|
|
||||||
}
|
|
||||||
}
|
|
||||||
self.pos = end;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- control flow -------------------------------------------------------
|
|
||||||
|
|
||||||
fn opIf(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
try self.object(scope); // predicate
|
|
||||||
self.termList(end, scope);
|
|
||||||
if (self.peek() == op.else_op) {
|
|
||||||
self.pos += 1;
|
|
||||||
try self.opElse(scope);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opElse(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
self.termList(end, scope);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opWhile(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
try self.object(scope); // predicate
|
|
||||||
self.termList(end, scope);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opLnot(self: *Parser, scope: *Node) Error!void {
|
|
||||||
// 0x92 followed by 0x93/94/95 is a compound comparison (two operands);
|
|
||||||
// otherwise it is a plain LNot of one operand.
|
|
||||||
const b = self.peek() orelse return error.Truncated;
|
|
||||||
switch (b) {
|
|
||||||
op.lnot.not_equal, op.lnot.less_equal, op.lnot.greater_equal => {
|
|
||||||
self.pos += 1;
|
|
||||||
try self.args(scope, 2);
|
|
||||||
},
|
|
||||||
else => try self.object(scope),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opMatch(self: *Parser, scope: *Node) Error!void {
|
|
||||||
try self.object(scope); // search package
|
|
||||||
try self.skip(1); // match opcode 1
|
|
||||||
try self.object(scope); // operand 1
|
|
||||||
try self.skip(1); // match opcode 2
|
|
||||||
try self.object(scope); // operand 2
|
|
||||||
try self.object(scope); // start index
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- extended opcodes (0x5B xx) -----------------------------------------
|
|
||||||
|
|
||||||
fn opExt(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const e = try self.readByte();
|
|
||||||
switch (e) {
|
|
||||||
op.ext.mutex => try self.opMutex(scope),
|
|
||||||
op.ext.event => try self.opEvent(scope),
|
|
||||||
op.ext.op_region => try self.opRegion(scope),
|
|
||||||
op.ext.data_region => try self.opDataRegion(scope),
|
|
||||||
op.ext.field => try self.opField(scope, 1, false),
|
|
||||||
op.ext.index_field => try self.opField(scope, 2, false),
|
|
||||||
op.ext.bank_field => try self.opField(scope, 2, true),
|
|
||||||
op.ext.device => try self.opScopeLike(scope, .device),
|
|
||||||
op.ext.thermal_zone => try self.opScopeLike(scope, .thermal_zone),
|
|
||||||
op.ext.processor => try self.opProcessor(scope),
|
|
||||||
op.ext.power_res => try self.opPowerRes(scope),
|
|
||||||
|
|
||||||
op.ext.cond_ref_of => try self.args(scope, 2), // SuperName, Target
|
|
||||||
op.ext.create_field => try self.opCreateField(scope, 3),
|
|
||||||
op.ext.load_table => try self.args(scope, 6),
|
|
||||||
op.ext.load => try self.args(scope, 2), // NameString, Target
|
|
||||||
op.ext.stall, op.ext.sleep => try self.args(scope, 1),
|
|
||||||
op.ext.acquire => {
|
|
||||||
try self.object(scope); // mutex SuperName
|
|
||||||
try self.skip(2); // timeout WordData
|
|
||||||
},
|
|
||||||
op.ext.signal, op.ext.reset, op.ext.release, op.ext.unload => try self.args(scope, 1),
|
|
||||||
op.ext.wait => try self.args(scope, 2),
|
|
||||||
op.ext.from_bcd, op.ext.to_bcd => try self.args(scope, 2),
|
|
||||||
op.ext.fatal => {
|
|
||||||
try self.skip(5); // Type(byte) + Code(dword)
|
|
||||||
try self.object(scope); // Arg TermArg
|
|
||||||
},
|
|
||||||
op.ext.revision, op.ext.debug, op.ext.timer => {},
|
|
||||||
|
|
||||||
else => return error.Malformed,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
fn isNameStart(b: u8) bool {
|
|
||||||
return (b >= op.name_char_start and b <= op.name_char_end) or
|
|
||||||
b == op.name_char_underscore or
|
|
||||||
b == op.root_char or
|
|
||||||
b == op.parent_prefix_char or
|
|
||||||
b == op.dual_name_prefix or
|
|
||||||
b == op.multi_name_prefix;
|
|
||||||
}
|
|
||||||
@@ -1,76 +0,0 @@
|
|||||||
//! The firmware-agnostic discovery facade.
|
|
||||||
//!
|
|
||||||
//! The kernel calls `platform.discover()` and gets back a generic `DeviceTree`
|
|
||||||
//! without ever naming ACPI or device-tree — the same way it imports `arch`
|
|
||||||
//! without naming x86_64. Which backend runs is decided *at runtime* from what
|
|
||||||
//! the bootloader handed us (an ACPI RSDP today, a device-tree blob later),
|
|
||||||
//! because a single image — a future ARM kernel especially — may boot under
|
|
||||||
//! either firmware. That's a deliberate divergence from `arch`, which is a
|
|
||||||
//! compile-time choice.
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
const danos = @import("danos");
|
|
||||||
const device = @import("device.zig");
|
|
||||||
const acpi = @import("acpi.zig");
|
|
||||||
const power = @import("power.zig");
|
|
||||||
const devicetree = @import("devicetree.zig");
|
|
||||||
|
|
||||||
pub const DeviceTree = device.DeviceTree;
|
|
||||||
pub const Device = device.Device;
|
|
||||||
pub const DeviceClass = device.DeviceClass;
|
|
||||||
pub const Hal = device.Hal;
|
|
||||||
pub const PowerInfo = acpi.PowerInfo;
|
|
||||||
pub const AmlStats = acpi.AmlStats;
|
|
||||||
pub const PlatformInfo = acpi.PlatformInfo;
|
|
||||||
pub const RegAccess = acpi.RegAccess;
|
|
||||||
pub const IsoEntry = acpi.IsoEntry;
|
|
||||||
|
|
||||||
/// The register map + sleep types discovery extracted, for logging/diagnostics.
|
|
||||||
pub fn powerInfo() PowerInfo {
|
|
||||||
return acpi.power_info;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The scalar firmware facts the arch layer needs to avoid legacy assumptions
|
|
||||||
/// (8259 presence, LAPIC base, PM timer, SPCR UART, IRQ overrides).
|
|
||||||
pub fn platformInfo() PlatformInfo {
|
|
||||||
return acpi.platform_info;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// AML parse integrity/diagnostics (namespace node count, bytes consumed).
|
|
||||||
pub fn amlStats() AmlStats {
|
|
||||||
return acpi.aml_stats;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Enumerate hardware into a fresh device tree. `hal` supplies the hardware
|
|
||||||
/// primitives the backend needs (MMIO mapping for PCIe config space, port I/O for
|
|
||||||
/// ACPI registers); pass the arch implementation. Errors leave nothing to clean up
|
|
||||||
/// beyond the tree's own allocations.
|
|
||||||
pub fn discover(
|
|
||||||
boot_info: *const danos.BootInfo,
|
|
||||||
allocator: std.mem.Allocator,
|
|
||||||
hal: Hal,
|
|
||||||
) !DeviceTree {
|
|
||||||
var dt = try DeviceTree.init(allocator);
|
|
||||||
|
|
||||||
if (boot_info.acpi_rsdp != 0) {
|
|
||||||
try acpi.discover(boot_info.acpi_rsdp, &dt, hal);
|
|
||||||
} else {
|
|
||||||
// No ACPI RSDP. A device-tree boot would parse its blob here; today that
|
|
||||||
// path is a stub, so this reports the machine described itself no way we
|
|
||||||
// understand yet.
|
|
||||||
try devicetree.discover(&dt);
|
|
||||||
}
|
|
||||||
|
|
||||||
return dt;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Restart the machine. Never returns on success; returns only if no reset method
|
|
||||||
/// worked (extremely unlikely). Backend-agnostic entry the kernel calls.
|
|
||||||
pub fn reboot(hal: Hal) void {
|
|
||||||
power.reboot(hal);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Power the machine off (ACPI S5). Never returns on success.
|
|
||||||
pub fn shutdown(hal: Hal) void {
|
|
||||||
power.shutdown(hal);
|
|
||||||
}
|
|
||||||
@@ -1,265 +0,0 @@
|
|||||||
//! x86_64 CPU operations. This is the "arch" module: the generic kernel imports
|
|
||||||
//! it as `@import("arch")` and never names x86_64 directly, so a second
|
|
||||||
//! architecture is added by pointing that module at a different directory in
|
|
||||||
//! build.zig — no change to the generic code. Keep everything CPU-specific here
|
|
||||||
//! (halt, the descriptor tables, later paging), and nothing generic.
|
|
||||||
|
|
||||||
const danos = @import("danos");
|
|
||||||
const gdt = @import("gdt.zig");
|
|
||||||
const tss = @import("tss.zig");
|
|
||||||
const idt = @import("idt.zig");
|
|
||||||
const paging = @import("paging.zig");
|
|
||||||
const serial = @import("serial.zig");
|
|
||||||
const apic = @import("apic.zig");
|
|
||||||
const ioapic = @import("ioapic.zig");
|
|
||||||
const io = @import("io.zig");
|
|
||||||
|
|
||||||
/// The saved register/trap frame passed to a fault handler.
|
|
||||||
pub const CpuState = idt.CpuState;
|
|
||||||
|
|
||||||
/// Bring up the serial port (the kernel's machine-readable log). No dependencies,
|
|
||||||
/// so it can be the very first thing called.
|
|
||||||
pub fn serialInit() void {
|
|
||||||
serial.init();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Write bytes to the serial port.
|
|
||||||
pub fn serialWrite(bytes: []const u8) void {
|
|
||||||
serial.write(bytes);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Set up the CPU's descriptor tables: our own GDT, the TSS (with an interrupt
|
|
||||||
/// stack for double faults), then the IDT with exception handlers. After this a
|
|
||||||
/// CPU fault is reported instead of triple-faulting. Install the fault handler
|
|
||||||
/// (setFaultHandler) first so early faults are caught.
|
|
||||||
pub fn init() void {
|
|
||||||
gdt.init();
|
|
||||||
tss.init();
|
|
||||||
idt.init();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Build the kernel's own page tables (with real permissions) and switch onto
|
|
||||||
/// them. Needs the frame allocator and the boot info (for the memory map and the
|
|
||||||
/// kernel's segment layout). Call once the frame allocator is up.
|
|
||||||
pub fn enablePaging(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void {
|
|
||||||
paging.init(allocFrame, boot_info);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Map a page into the kernel address space (non-executable). For the heap, etc.
|
|
||||||
pub fn mapPage(virt: u64, phys: u64, writable: bool) void {
|
|
||||||
paging.map(virt, phys, writable);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Remove a kernel mapping.
|
|
||||||
pub fn unmapPage(virt: u64) void {
|
|
||||||
paging.unmap(virt);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// CR3 holds the physical address of the active top-level page table.
|
|
||||||
pub fn readCr3() u64 {
|
|
||||||
return asm volatile ("mov %%cr3, %[out]"
|
|
||||||
: [out] "=r" (-> u64),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Kernel tick rate: 1000 Hz (1 ms), the scheduler's time quantum.
|
|
||||||
pub const timer_hz = 1000;
|
|
||||||
|
|
||||||
/// The ACPI PM timer, as a calibration reference (re-exported for the config).
|
|
||||||
pub const PmTimer = apic.PmTimer;
|
|
||||||
/// A MADT interrupt-source override (re-exported for the config).
|
|
||||||
pub const IsoEntry = ioapic.IsoEntry;
|
|
||||||
|
|
||||||
/// Discovered platform facts the arch layer needs so it makes no legacy
|
|
||||||
/// assumptions — sourced from the device tree + ACPI, passed in by the kernel.
|
|
||||||
pub const PlatformConfig = struct {
|
|
||||||
/// Whether the legacy 8259 PIC is present (skip programming it if not).
|
|
||||||
pic_present: bool = true,
|
|
||||||
/// HPET MMIO base (0 = none) — a calibration reference for the timer.
|
|
||||||
hpet_base: u64 = 0,
|
|
||||||
/// The ACPI PM timer, another calibration reference.
|
|
||||||
pm_timer: ?PmTimer = null,
|
|
||||||
/// I/O APIC MMIO base + its first global system interrupt (0 = none).
|
|
||||||
ioapic_base: u64 = 0,
|
|
||||||
ioapic_gsi_base: u32 = 0,
|
|
||||||
/// MADT ISA-IRQ overrides, for I/O APIC routing.
|
|
||||||
overrides: []const IsoEntry = &.{},
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Apply the discovered platform config. Must run before `startTimer` (the timer
|
|
||||||
/// calibration reads `hpet_base`/`pm_timer`) and before any interrupt routing.
|
|
||||||
/// Maps + masks the I/O APIC immediately.
|
|
||||||
pub fn configurePlatform(cfg: PlatformConfig) void {
|
|
||||||
apic.configure(cfg.pic_present, cfg.hpet_base, cfg.pm_timer);
|
|
||||||
ioapic.configure(cfg.ioapic_base, cfg.ioapic_gsi_base, cfg.overrides);
|
|
||||||
ioapic.init();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Point the serial console at the UART ACPI's SPCR table named (MMIO or I/O port).
|
|
||||||
pub fn serialReconfigure(is_mmio: bool, addr: u64) void {
|
|
||||||
serial.reconfigure(is_mmio, addr);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The reference clock the timer was calibrated against ("cpuid"/"hpet"/…).
|
|
||||||
pub fn timerCalibrationSource() []const u8 {
|
|
||||||
return apic.calibrationSource();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// I/O APIC diagnostics (for boot logging / verification).
|
|
||||||
pub fn ioapicEntryCount() u32 {
|
|
||||||
return ioapic.entryCount();
|
|
||||||
}
|
|
||||||
pub fn ioapicEntryLow(n: u32) u32 {
|
|
||||||
return ioapic.entryLow(n);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Enable the Local APIC, calibrate its timer against the best available reference
|
|
||||||
/// (see apic.calibrate — no longer the PIT by default), and start it firing at
|
|
||||||
/// `timer_hz` — the kernel's real-time heartbeat. Interrupts still have to be
|
|
||||||
/// unmasked with enableInterrupts() to be delivered. Run `configurePlatform` first.
|
|
||||||
pub fn startTimer() void {
|
|
||||||
apic.init();
|
|
||||||
apic.calibrate();
|
|
||||||
idt.setHandler(apic.timer_vector, apic.timerTick);
|
|
||||||
apic.initTimer(timer_hz);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Number of timer ticks since startTimer().
|
|
||||||
pub fn ticks() u64 {
|
|
||||||
return apic.ticks();
|
|
||||||
}
|
|
||||||
|
|
||||||
// Monotonic high-resolution clock (from the TSC), one function per resolution.
|
|
||||||
pub fn nanos() u64 {
|
|
||||||
return apic.nanos();
|
|
||||||
}
|
|
||||||
pub fn micros() u64 {
|
|
||||||
return apic.micros();
|
|
||||||
}
|
|
||||||
pub fn millis() u64 {
|
|
||||||
return apic.millis();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Measured LAPIC timer / TSC frequencies in Hz (from calibration).
|
|
||||||
pub fn lapicHz() u64 {
|
|
||||||
return apic.lapicHz();
|
|
||||||
}
|
|
||||||
pub fn tscHz() u64 {
|
|
||||||
return apic.tscHz();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Unmask maskable interrupts (`sti`) so device interrupts get delivered.
|
|
||||||
pub fn enableInterrupts() void {
|
|
||||||
asm volatile ("sti");
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Mask maskable interrupts (`cli`).
|
|
||||||
pub fn disableInterrupts() void {
|
|
||||||
asm volatile ("cli");
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Disable interrupts and return the previous flags, so a nested critical section
|
|
||||||
/// can restore the caller's state rather than blindly re-enabling. Pairs with
|
|
||||||
/// restoreInterrupts.
|
|
||||||
pub fn saveInterrupts() u64 {
|
|
||||||
var flags: u64 = undefined;
|
|
||||||
asm volatile (
|
|
||||||
\\pushfq
|
|
||||||
\\pop %[f]
|
|
||||||
\\cli
|
|
||||||
: [f] "=r" (flags),
|
|
||||||
:
|
|
||||||
: .{ .memory = true }
|
|
||||||
);
|
|
||||||
return flags;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Re-enable interrupts only if they were enabled when `flags` was captured.
|
|
||||||
pub fn restoreInterrupts(flags: u64) void {
|
|
||||||
if (flags & 0x200 != 0) asm volatile ("sti" ::: .{ .memory = true }); // bit 9 = IF
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Register a callback the timer interrupt invokes each tick (e.g. the scheduler).
|
|
||||||
pub fn setTickHook(hook: *const fn () void) void {
|
|
||||||
apic.setTickHook(hook);
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- context switching (for the scheduler) -------------------------------
|
|
||||||
|
|
||||||
/// Save the current task's registers/stack and resume `new_rsp`; the old stack
|
|
||||||
/// pointer is written to `old_rsp`. Defined in isr.s.
|
|
||||||
extern fn switch_context(old_rsp: *usize, new_rsp: usize) callconv(.c) void;
|
|
||||||
|
|
||||||
pub fn switchContext(old_rsp: *usize, new_rsp: usize) void {
|
|
||||||
switch_context(old_rsp, new_rsp);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Build the initial stack for a new task so that switching to it lands in
|
|
||||||
/// `task_trampoline`, which then calls `entry`. Returns the saved stack pointer.
|
|
||||||
/// The layout must match switch_context's push order (callee-saved, then the
|
|
||||||
/// return address on top); `entry` is smuggled in via the r15 slot.
|
|
||||||
pub fn initTaskStack(stack_top: usize, entry: usize) usize {
|
|
||||||
const trampoline = @extern(*const anyopaque, .{ .name = "task_trampoline" });
|
|
||||||
var sp = stack_top;
|
|
||||||
const push = struct {
|
|
||||||
fn f(p: *usize, value: usize) void {
|
|
||||||
p.* -= @sizeOf(usize);
|
|
||||||
@as(*usize, @ptrFromInt(p.*)).* = value;
|
|
||||||
}
|
|
||||||
}.f;
|
|
||||||
push(&sp, @intFromPtr(trampoline)); // return address for switch_context's `ret`
|
|
||||||
push(&sp, 0); // rbx
|
|
||||||
push(&sp, 0); // rbp
|
|
||||||
push(&sp, 0); // r12
|
|
||||||
push(&sp, 0); // r13
|
|
||||||
push(&sp, 0); // r14
|
|
||||||
push(&sp, entry); // r15 -> task entry, read by task_trampoline
|
|
||||||
return sp;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Route CPU exceptions to `handler`, which receives the trap frame and does not
|
|
||||||
/// return. Until set, faults just halt the core.
|
|
||||||
pub fn setFaultHandler(handler: *const fn (*const CpuState) noreturn) void {
|
|
||||||
idt.on_fault = handler;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// A human-readable name for a CPU exception vector.
|
|
||||||
pub fn vectorName(vector: u64) []const u8 {
|
|
||||||
return idt.vectorName(vector);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Read `width` bytes (1/2/4) from an I/O port. The generic device layer drives
|
|
||||||
/// ACPI registers through this rather than naming x86 port instructions; on an
|
|
||||||
/// MMIO-only architecture this would be implemented differently.
|
|
||||||
pub fn pioRead(width: u8, port: u16) u32 {
|
|
||||||
return switch (width) {
|
|
||||||
1 => io.inb(port),
|
|
||||||
2 => io.inw(port),
|
|
||||||
4 => io.inl(port),
|
|
||||||
else => 0,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Write `width` bytes (1/2/4) to an I/O port.
|
|
||||||
pub fn pioWrite(width: u8, port: u16, value: u32) void {
|
|
||||||
switch (width) {
|
|
||||||
1 => io.outb(port, @truncate(value)),
|
|
||||||
2 => io.outw(port, @truncate(value)),
|
|
||||||
4 => io.outl(port, value),
|
|
||||||
else => {},
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// CR2 holds the faulting linear address after a page fault (#PF, vector 14).
|
|
||||||
pub fn readCr2() u64 {
|
|
||||||
return asm volatile ("mov %%cr2, %[out]"
|
|
||||||
: [out] "=r" (-> u64),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Park the core forever. `hlt` drops it into a low-power idle until the next
|
|
||||||
/// interrupt; the loop re-halts on every wake so the stop is permanent. See
|
|
||||||
/// docs/halting.md for the full reasoning.
|
|
||||||
pub fn halt() noreturn {
|
|
||||||
while (true) asm volatile ("hlt");
|
|
||||||
}
|
|
||||||
@@ -1,54 +0,0 @@
|
|||||||
//! Global Descriptor Table. In long mode segmentation is mostly vestigial, but
|
|
||||||
//! the CPU still needs valid code/data segment descriptors, and the IDT's gates
|
|
||||||
//! reference a code selector — so we install our own flat GDT with known
|
|
||||||
//! selectors (0x08 kernel code, 0x10 kernel data) rather than trusting whatever
|
|
||||||
//! the firmware left in place.
|
|
||||||
|
|
||||||
/// Selectors into the table below (index * 8).
|
|
||||||
pub const kernel_code = 0x08;
|
|
||||||
pub const kernel_data = 0x10;
|
|
||||||
pub const tss_selector = 0x18;
|
|
||||||
|
|
||||||
/// Flat 64-bit descriptors. Base/limit are ignored in long mode; what matters is
|
|
||||||
/// the access byte and, for code, the long-mode (L) flag.
|
|
||||||
/// code: present, ring 0, executable, readable, L=1 -> 0x00AF9A00_0000FFFF
|
|
||||||
/// data: present, ring 0, writable -> 0x00CF9200_0000FFFF
|
|
||||||
/// The last two slots hold one 16-byte TSS descriptor, filled in by setTss.
|
|
||||||
var table = [_]u64{
|
|
||||||
0, // null descriptor (required)
|
|
||||||
0x00AF9A000000FFFF, // kernel code (0x08)
|
|
||||||
0x00CF92000000FFFF, // kernel data (0x10)
|
|
||||||
0, // TSS descriptor low (0x18)
|
|
||||||
0, // TSS descriptor high
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Fill the 64-bit TSS system descriptor (two GDT slots) so the task register can
|
|
||||||
/// point at our TSS. Type 0x89 = present, ring 0, available 64-bit TSS.
|
|
||||||
pub fn setTss(base: u64, limit: u64) void {
|
|
||||||
table[3] = (limit & 0xFFFF) |
|
|
||||||
((base & 0xFFFF) << 16) |
|
|
||||||
(((base >> 16) & 0xFF) << 32) |
|
|
||||||
(@as(u64, 0x89) << 40) |
|
|
||||||
(((limit >> 16) & 0xF) << 48) |
|
|
||||||
(((base >> 24) & 0xFF) << 56);
|
|
||||||
table[4] = (base >> 32) & 0xFFFFFFFF;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The operand `lgdt` wants: table byte-length minus one, then its address.
|
|
||||||
const Descriptor = packed struct {
|
|
||||||
limit: u16,
|
|
||||||
base: u64,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Loads the GDT and reloads the segment registers (including CS). Defined in
|
|
||||||
/// isr.s — it uses the selectors 0x08 (code) and 0x10 (data) that match `table`.
|
|
||||||
extern fn gdt_flush(descriptor: *const Descriptor) callconv(.c) void;
|
|
||||||
|
|
||||||
/// Install our GDT and switch onto its segments.
|
|
||||||
pub fn init() void {
|
|
||||||
const descriptor = Descriptor{
|
|
||||||
.limit = @sizeOf(@TypeOf(table)) - 1,
|
|
||||||
.base = @intFromPtr(&table),
|
|
||||||
};
|
|
||||||
gdt_flush(&descriptor);
|
|
||||||
}
|
|
||||||
@@ -1,97 +0,0 @@
|
|||||||
//! I/O APIC — routes external device interrupts (a device's line) to a LAPIC
|
|
||||||
//! vector on a chosen CPU. Its address and the ISA-IRQ-to-GSI remappings come from
|
|
||||||
//! ACPI's MADT (via discovery), never assumed.
|
|
||||||
//!
|
|
||||||
//! Status: groundwork. The only interrupt danos handles today is the LAPIC's own
|
|
||||||
//! timer, which needs no I/O APIC — so nothing calls `routeIrq` yet. What runs now
|
|
||||||
//! is `init`, which maps the I/O APIC and **masks every input**, the correct
|
|
||||||
//! quiescent state on a legacy-free machine. `routeIrq` is ready for the first real
|
|
||||||
//! device driver (a keyboard, say).
|
|
||||||
|
|
||||||
const paging = @import("paging.zig");
|
|
||||||
|
|
||||||
/// A MADT Interrupt Source Override: an ISA IRQ that appears at a different global
|
|
||||||
/// system interrupt, with its own polarity/trigger (MPS INTI `flags`).
|
|
||||||
pub const IsoEntry = struct { source: u8, gsi: u32, flags: u16 };
|
|
||||||
|
|
||||||
var base: u64 = 0; // 0 = no I/O APIC discovered
|
|
||||||
var gsi_base: u32 = 0;
|
|
||||||
var max_entries: u32 = 0;
|
|
||||||
var overrides: [16]IsoEntry = undefined;
|
|
||||||
var override_count: usize = 0;
|
|
||||||
|
|
||||||
// The I/O APIC exposes an index register (IOREGSEL) and a data window (IOWIN).
|
|
||||||
const reg_ioregsel = 0x00;
|
|
||||||
const reg_iowin = 0x10;
|
|
||||||
const reg_version = 0x01;
|
|
||||||
const redir_base = 0x10; // redirection table: two 32-bit regs per entry
|
|
||||||
const redir_mask = 1 << 16; // mask bit in the low dword
|
|
||||||
|
|
||||||
/// Supply the discovered I/O APIC location + the MADT IRQ overrides. Call before `init`.
|
|
||||||
pub fn configure(ioapic_base: u64, ioapic_gsi_base: u32, isos: []const IsoEntry) void {
|
|
||||||
base = ioapic_base;
|
|
||||||
gsi_base = ioapic_gsi_base;
|
|
||||||
override_count = @min(isos.len, overrides.len);
|
|
||||||
for (isos[0..override_count], 0..) |iso, i| overrides[i] = iso;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn regRead(index: u32) u32 {
|
|
||||||
@as(*volatile u32, @ptrFromInt(base + reg_ioregsel)).* = index;
|
|
||||||
return @as(*volatile u32, @ptrFromInt(base + reg_iowin)).*;
|
|
||||||
}
|
|
||||||
fn regWrite(index: u32, value: u32) void {
|
|
||||||
@as(*volatile u32, @ptrFromInt(base + reg_ioregsel)).* = index;
|
|
||||||
@as(*volatile u32, @ptrFromInt(base + reg_iowin)).* = value;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn writeEntry(n: u32, low: u32, high: u32) void {
|
|
||||||
regWrite(redir_base + 2 * n, low);
|
|
||||||
regWrite(redir_base + 2 * n + 1, high);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Map the I/O APIC and mask every redirection entry — the safe quiescent state.
|
|
||||||
pub fn init() void {
|
|
||||||
if (base == 0) return;
|
|
||||||
paging.map(base & ~@as(u64, 0xFFF), base & ~@as(u64, 0xFFF), true);
|
|
||||||
max_entries = ((regRead(reg_version) >> 16) & 0xFF) + 1;
|
|
||||||
var n: u32 = 0;
|
|
||||||
while (n < max_entries) : (n += 1) writeEntry(n, redir_mask, 0);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Route ISA `irq` to `vector` on the LAPIC `apic_id`, honouring a MADT override
|
|
||||||
/// for its GSI/polarity/trigger, and unmask it. No caller yet — groundwork for the
|
|
||||||
/// first device driver.
|
|
||||||
pub fn routeIrq(irq: u8, vector: u8, apic_id: u8) void {
|
|
||||||
if (base == 0) return;
|
|
||||||
|
|
||||||
var gsi: u32 = irq;
|
|
||||||
var flags: u16 = 0;
|
|
||||||
for (overrides[0..override_count]) |o| {
|
|
||||||
if (o.source == irq) {
|
|
||||||
gsi = o.gsi;
|
|
||||||
flags = o.flags;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (gsi < gsi_base) return;
|
|
||||||
const n = gsi - gsi_base;
|
|
||||||
if (n >= max_entries) return;
|
|
||||||
|
|
||||||
// Low dword: vector + delivery mode fixed(0) + physical dest(0), unmasked.
|
|
||||||
// MPS INTI flags: bits [1:0] polarity (3 = active low), [3:2] trigger (3 = level).
|
|
||||||
var low: u32 = vector;
|
|
||||||
if (flags & 0x3 == 3) low |= (1 << 13);
|
|
||||||
if ((flags >> 2) & 0x3 == 3) low |= (1 << 15);
|
|
||||||
const high: u32 = @as(u32, apic_id) << 24; // destination APIC ID
|
|
||||||
writeEntry(n, low, high);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Number of redirection entries the I/O APIC advertises (0 until `init`).
|
|
||||||
pub fn entryCount() u32 {
|
|
||||||
return max_entries;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The low dword of redirection entry `n` — for diagnostics/read-back.
|
|
||||||
pub fn entryLow(n: u32) u32 {
|
|
||||||
if (base == 0) return 0;
|
|
||||||
return regRead(redir_base + 2 * n);
|
|
||||||
}
|
|
||||||
@@ -1,180 +0,0 @@
|
|||||||
# x86_64 low-level entry code: the CPU-exception stubs, plus the GDT/IDT load
|
|
||||||
# helpers. Kept in a dedicated assembly file rather than inline asm because these
|
|
||||||
# need real labels and cross-symbol jumps/calls (isr_common, exceptionHandler),
|
|
||||||
# and because `lgdt`/`lidt` memory operands aren't expressible in Zig inline asm.
|
|
||||||
#
|
|
||||||
# Each exception vector normalises the stack to a uniform trap frame — a dummy
|
|
||||||
# error code where the CPU pushes none, then the vector number — and jumps to the
|
|
||||||
# shared tail, which saves the general registers and calls the Zig handler with a
|
|
||||||
# pointer to the frame (matching src/arch/x86_64/idt.zig's CpuState).
|
|
||||||
|
|
||||||
.text
|
|
||||||
|
|
||||||
# gdt_flush(rdi = *GDT descriptor): load the GDT, reload the data segment
|
|
||||||
# registers to the data selector, and reload CS to the code selector. CS can't be
|
|
||||||
# set with mov, so we far-return through the caller's own return address.
|
|
||||||
.global gdt_flush
|
|
||||||
gdt_flush:
|
|
||||||
lgdt (%rdi)
|
|
||||||
mov $0x10, %ax # kernel data selector
|
|
||||||
mov %ax, %ds
|
|
||||||
mov %ax, %es
|
|
||||||
mov %ax, %ss
|
|
||||||
mov %ax, %fs
|
|
||||||
mov %ax, %gs
|
|
||||||
pop %rax # caller's return address
|
|
||||||
push $0x08 # kernel code selector (new CS)
|
|
||||||
push %rax # return address (new RIP)
|
|
||||||
lretq
|
|
||||||
|
|
||||||
# idt_flush(rdi = *IDT descriptor): load the IDT.
|
|
||||||
.global idt_flush
|
|
||||||
idt_flush:
|
|
||||||
lidt (%rdi)
|
|
||||||
ret
|
|
||||||
|
|
||||||
# load_tr(di = TSS selector): load the task register.
|
|
||||||
.global load_tr
|
|
||||||
load_tr:
|
|
||||||
ltr %di
|
|
||||||
ret
|
|
||||||
|
|
||||||
# switch_context(rdi = &old_task.rsp, rsi = new_task.rsp)
|
|
||||||
# Cooperative context switch: save the callee-saved registers on the current
|
|
||||||
# stack, stash the stack pointer in the old task, load the new task's stack
|
|
||||||
# pointer, restore its callee-saved registers, and return into it. Caller-saved
|
|
||||||
# registers are the compiler's responsibility (this looks like a normal call).
|
|
||||||
.global switch_context
|
|
||||||
switch_context:
|
|
||||||
push %rbx
|
|
||||||
push %rbp
|
|
||||||
push %r12
|
|
||||||
push %r13
|
|
||||||
push %r14
|
|
||||||
push %r15
|
|
||||||
mov %rsp, (%rdi) # save old stack pointer into old_task.rsp
|
|
||||||
mov %rsi, %rsp # switch to the new task's stack
|
|
||||||
pop %r15
|
|
||||||
pop %r14
|
|
||||||
pop %r13
|
|
||||||
pop %r12
|
|
||||||
pop %rbp
|
|
||||||
pop %rbx
|
|
||||||
ret # return into the new task's saved instruction pointer
|
|
||||||
|
|
||||||
# task_trampoline: the first thing a freshly-spawned task runs. init_task_stack
|
|
||||||
# leaves its entry function in r15. New tasks start with interrupts enabled.
|
|
||||||
.global task_trampoline
|
|
||||||
task_trampoline:
|
|
||||||
sti
|
|
||||||
call *%r15 # call the task entry (fn() void)
|
|
||||||
1: hlt # if the entry returns, idle (still preemptible)
|
|
||||||
jmp 1b
|
|
||||||
|
|
||||||
# Stub for a vector the CPU does NOT push an error code for: push a dummy 0.
|
|
||||||
.macro STUB_NOERR vec
|
|
||||||
.global isr\vec
|
|
||||||
isr\vec:
|
|
||||||
pushq $0
|
|
||||||
pushq $\vec
|
|
||||||
jmp isr_common
|
|
||||||
.endm
|
|
||||||
|
|
||||||
# Stub for a vector the CPU DOES push an error code for: leave it in place.
|
|
||||||
.macro STUB_ERR vec
|
|
||||||
.global isr\vec
|
|
||||||
isr\vec:
|
|
||||||
pushq $\vec
|
|
||||||
jmp isr_common
|
|
||||||
.endm
|
|
||||||
|
|
||||||
STUB_NOERR 0
|
|
||||||
STUB_NOERR 1
|
|
||||||
STUB_NOERR 2
|
|
||||||
STUB_NOERR 3
|
|
||||||
STUB_NOERR 4
|
|
||||||
STUB_NOERR 5
|
|
||||||
STUB_NOERR 6
|
|
||||||
STUB_NOERR 7
|
|
||||||
STUB_ERR 8
|
|
||||||
STUB_NOERR 9
|
|
||||||
STUB_ERR 10
|
|
||||||
STUB_ERR 11
|
|
||||||
STUB_ERR 12
|
|
||||||
STUB_ERR 13
|
|
||||||
STUB_ERR 14
|
|
||||||
STUB_NOERR 15
|
|
||||||
STUB_NOERR 16
|
|
||||||
STUB_ERR 17
|
|
||||||
STUB_NOERR 18
|
|
||||||
STUB_NOERR 19
|
|
||||||
STUB_NOERR 20
|
|
||||||
STUB_ERR 21
|
|
||||||
STUB_NOERR 22
|
|
||||||
STUB_NOERR 23
|
|
||||||
STUB_NOERR 24
|
|
||||||
STUB_NOERR 25
|
|
||||||
STUB_NOERR 26
|
|
||||||
STUB_NOERR 27
|
|
||||||
STUB_NOERR 28
|
|
||||||
STUB_NOERR 29
|
|
||||||
STUB_NOERR 30
|
|
||||||
STUB_NOERR 31
|
|
||||||
|
|
||||||
# Device-interrupt vectors (timer, spurious, room for more). None push an error
|
|
||||||
# code, so they all use the dummy-zero form.
|
|
||||||
STUB_NOERR 32
|
|
||||||
STUB_NOERR 33
|
|
||||||
STUB_NOERR 34
|
|
||||||
STUB_NOERR 35
|
|
||||||
STUB_NOERR 36
|
|
||||||
STUB_NOERR 37
|
|
||||||
STUB_NOERR 38
|
|
||||||
STUB_NOERR 39
|
|
||||||
STUB_NOERR 40
|
|
||||||
STUB_NOERR 41
|
|
||||||
STUB_NOERR 42
|
|
||||||
STUB_NOERR 43
|
|
||||||
STUB_NOERR 44
|
|
||||||
STUB_NOERR 45
|
|
||||||
STUB_NOERR 46
|
|
||||||
STUB_NOERR 47
|
|
||||||
|
|
||||||
.extern interruptDispatch
|
|
||||||
|
|
||||||
# Shared tail. Register push order here defines the CpuState field order.
|
|
||||||
isr_common:
|
|
||||||
push %rax
|
|
||||||
push %rbx
|
|
||||||
push %rcx
|
|
||||||
push %rdx
|
|
||||||
push %rsi
|
|
||||||
push %rdi
|
|
||||||
push %rbp
|
|
||||||
push %r8
|
|
||||||
push %r9
|
|
||||||
push %r10
|
|
||||||
push %r11
|
|
||||||
push %r12
|
|
||||||
push %r13
|
|
||||||
push %r14
|
|
||||||
push %r15
|
|
||||||
mov %rsp, %rdi # first argument: pointer to the trap frame
|
|
||||||
call interruptDispatch
|
|
||||||
pop %r15
|
|
||||||
pop %r14
|
|
||||||
pop %r13
|
|
||||||
pop %r12
|
|
||||||
pop %r11
|
|
||||||
pop %r10
|
|
||||||
pop %r9
|
|
||||||
pop %r8
|
|
||||||
pop %rbp
|
|
||||||
pop %rdi
|
|
||||||
pop %rsi
|
|
||||||
pop %rdx
|
|
||||||
pop %rcx
|
|
||||||
pop %rbx
|
|
||||||
pop %rax
|
|
||||||
add $16, %rsp # drop the vector and error code
|
|
||||||
iretq
|
|
||||||
@@ -1,47 +0,0 @@
|
|||||||
/* Kernel link layout.
|
|
||||||
*
|
|
||||||
* The kernel is linked at a fixed low physical address (set by `image_base` in
|
|
||||||
* build.zig). UEFI runs with memory identity-mapped, so the bootloader can load
|
|
||||||
* each PT_LOAD segment to the physical address matching its virtual address and
|
|
||||||
* jump straight to _start — no page tables to build yet. (Moving to a
|
|
||||||
* higher-half virtual base is a later step, once the bootloader sets up paging.)
|
|
||||||
*/
|
|
||||||
|
|
||||||
ENTRY(_start)
|
|
||||||
|
|
||||||
/* One loadable segment per permission set, so the loader can map .text as R+X,
|
|
||||||
* .rodata as R, and .data/.bss as R+W. FLAGS bits: 1=X, 2=W, 4=R. */
|
|
||||||
PHDRS {
|
|
||||||
text PT_LOAD FLAGS(5); /* R + X */
|
|
||||||
rodata PT_LOAD FLAGS(4); /* R */
|
|
||||||
data PT_LOAD FLAGS(6); /* R + W */
|
|
||||||
}
|
|
||||||
|
|
||||||
SECTIONS {
|
|
||||||
.text ALIGN(4K) : {
|
|
||||||
*(.text .text.*)
|
|
||||||
} :text
|
|
||||||
|
|
||||||
.rodata ALIGN(4K) : {
|
|
||||||
*(.rodata .rodata.*)
|
|
||||||
} :rodata
|
|
||||||
|
|
||||||
.data ALIGN(4K) : {
|
|
||||||
*(.data .data.*)
|
|
||||||
} :data
|
|
||||||
|
|
||||||
/* .bss occupies memory but not file space. The loader zeroes it via the
|
|
||||||
* gap between each PT_LOAD segment's file size and memory size, so no
|
|
||||||
* boundary symbols are needed here. (Zig's self-hosted linker also does not
|
|
||||||
* yet honour linker-script symbol assignments.) */
|
|
||||||
.bss ALIGN(4K) : {
|
|
||||||
*(.bss .bss.*)
|
|
||||||
*(COMMON)
|
|
||||||
} :data
|
|
||||||
|
|
||||||
/DISCARD/ : {
|
|
||||||
*(.comment)
|
|
||||||
*(.note .note.*)
|
|
||||||
*(.eh_frame .eh_frame_hdr)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,153 +0,0 @@
|
|||||||
//! The kernel's page tables and virtual memory manager.
|
|
||||||
//!
|
|
||||||
//! Builds our own 4-level page tables and switches CR3 onto them, replacing the
|
|
||||||
//! firmware's. Unlike the earlier bootstrap this maps with real permissions:
|
|
||||||
//! RAM is identity-mapped read-write + no-execute, the kernel's own segments get
|
|
||||||
//! their ELF permissions (code R+X, rodata R, data R+W+NX), and page 0 is left
|
|
||||||
//! unmapped as a null guard. It also exposes map/unmap for on-demand mapping,
|
|
||||||
//! which the kernel heap will build on.
|
|
||||||
//!
|
|
||||||
//! Everything is 4 KiB pages — precise and simple; the extra table memory is
|
|
||||||
//! negligible against available RAM.
|
|
||||||
|
|
||||||
const danos = @import("danos");
|
|
||||||
const io = @import("io.zig");
|
|
||||||
|
|
||||||
const page_size = danos.page_size;
|
|
||||||
|
|
||||||
// Page-table entry bits.
|
|
||||||
const present: u64 = 1 << 0;
|
|
||||||
const writable: u64 = 1 << 1;
|
|
||||||
const no_execute: u64 = 1 << 63;
|
|
||||||
const addr_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
|
||||||
|
|
||||||
// ELF segment flags (p_flags).
|
|
||||||
const pf_x: u32 = 1;
|
|
||||||
const pf_w: u32 = 2;
|
|
||||||
|
|
||||||
// State kept after init so map()/unmap() can serve later callers (e.g. the heap).
|
|
||||||
var kernel_pml4: u64 = 0;
|
|
||||||
var alloc_frame: *const fn () ?u64 = undefined;
|
|
||||||
|
|
||||||
fn tableAt(phys: u64) *[512]u64 {
|
|
||||||
return @ptrFromInt(phys);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn allocTable() u64 {
|
|
||||||
const frame = alloc_frame() orelse @panic("paging: out of memory building page tables");
|
|
||||||
@memset(tableAt(frame)[0..], 0);
|
|
||||||
return frame;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Return the table an entry points at, creating it if empty. Intermediate
|
|
||||||
/// entries are writable and executable so the leaf's bits govern (a page is
|
|
||||||
/// writable only if every level is; non-executable if any level is).
|
|
||||||
fn descend(entry: *u64) u64 {
|
|
||||||
if (entry.* & present != 0) return entry.* & addr_mask;
|
|
||||||
const frame = allocTable();
|
|
||||||
entry.* = frame | present | writable;
|
|
||||||
return frame;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Map one 4 KiB page `virt` -> `phys` with `flags` (present is added).
|
|
||||||
fn mapPage(pml4: u64, virt: u64, phys: u64, flags: u64) void {
|
|
||||||
const pml4e = &tableAt(pml4)[(virt >> 39) & 0x1FF];
|
|
||||||
const pdpt = descend(pml4e);
|
|
||||||
const pdpte = &tableAt(pdpt)[(virt >> 30) & 0x1FF];
|
|
||||||
const pd = descend(pdpte);
|
|
||||||
const pde = &tableAt(pd)[(virt >> 21) & 0x1FF];
|
|
||||||
const pt = descend(pde);
|
|
||||||
tableAt(pt)[(virt >> 12) & 0x1FF] = (phys & addr_mask) | flags | present;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Identity-map [base, base+len) with `flags`, rounded out to whole pages.
|
|
||||||
fn mapRangeIdentity(pml4: u64, base: u64, len: u64, flags: u64) void {
|
|
||||||
var addr = base & ~@as(u64, page_size - 1);
|
|
||||||
const end = base + len;
|
|
||||||
while (addr < end) : (addr += page_size) {
|
|
||||||
if (addr == 0) continue; // leave page 0 unmapped: the null guard
|
|
||||||
mapPage(pml4, addr, addr, flags);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn regions(mm: danos.MemoryMap) []const danos.MemoryRegion {
|
|
||||||
return @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Enable the NX bit in the page-table format (EFER.NXE). Must happen before we
|
|
||||||
/// load a CR3 whose entries set the NX bit, or those bits are reserved and fault.
|
|
||||||
fn enableNx() void {
|
|
||||||
const efer_msr = 0xC0000080;
|
|
||||||
io.wrmsr(efer_msr, io.rdmsr(efer_msr) | (1 << 11));
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Build the address space and switch onto it.
|
|
||||||
pub fn init(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void {
|
|
||||||
alloc_frame = allocFrame;
|
|
||||||
enableNx();
|
|
||||||
const pml4 = allocTable();
|
|
||||||
|
|
||||||
// 1. All RAM identity-mapped RW + NX. Non-RAM (MMIO) is skipped and stays
|
|
||||||
// unmapped unless mapped explicitly below.
|
|
||||||
for (regions(boot_info.memory_map)) |r| {
|
|
||||||
if (r.kind == .mmio) continue;
|
|
||||||
mapRangeIdentity(pml4, r.base, r.pages * page_size, present | writable | no_execute);
|
|
||||||
}
|
|
||||||
|
|
||||||
// 2. The framebuffer and the Local APIC (device memory we need), RW + NX.
|
|
||||||
const fb = boot_info.framebuffer;
|
|
||||||
mapRangeIdentity(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute);
|
|
||||||
mapPage(pml4, 0xFEE00000, 0xFEE00000, present | writable | no_execute);
|
|
||||||
|
|
||||||
// 3. Overlay the kernel's own segments with their real ELF permissions,
|
|
||||||
// replacing the blanket RW+NX from step 1: code becomes R+X, rodata R,
|
|
||||||
// data R+W+NX. This is the W^X guarantee.
|
|
||||||
for (boot_info.kernel_segments[0..boot_info.kernel_segment_count]) |seg| {
|
|
||||||
var flags: u64 = present;
|
|
||||||
if (seg.flags & pf_w != 0) flags |= writable;
|
|
||||||
if (seg.flags & pf_x == 0) flags |= no_execute;
|
|
||||||
var addr = seg.virt;
|
|
||||||
const end = seg.virt + seg.pages * page_size;
|
|
||||||
while (addr < end) : (addr += page_size) mapPage(pml4, addr, addr, flags);
|
|
||||||
}
|
|
||||||
|
|
||||||
kernel_pml4 = pml4;
|
|
||||||
asm volatile ("mov %[pml4], %%cr3"
|
|
||||||
:
|
|
||||||
: [pml4] "r" (pml4),
|
|
||||||
: .{ .memory = true }
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Map a page into the kernel address space on demand (for the heap, etc.).
|
|
||||||
/// `writable_page` controls W; pages are always mapped non-executable.
|
|
||||||
pub fn map(virt: u64, phys: u64, writable_page: bool) void {
|
|
||||||
var flags: u64 = present | no_execute;
|
|
||||||
if (writable_page) flags |= writable;
|
|
||||||
mapPage(kernel_pml4, virt, phys, flags);
|
|
||||||
invalidate(virt);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Remove a mapping and flush it from the TLB.
|
|
||||||
pub fn unmap(virt: u64) void {
|
|
||||||
const pml4e = tableAt(kernel_pml4)[(virt >> 39) & 0x1FF];
|
|
||||||
if (pml4e & present == 0) return;
|
|
||||||
const pdpte = tableAt(pml4e & addr_mask)[(virt >> 30) & 0x1FF];
|
|
||||||
if (pdpte & present == 0) return;
|
|
||||||
const pde = tableAt(pdpte & addr_mask)[(virt >> 21) & 0x1FF];
|
|
||||||
if (pde & present == 0) return;
|
|
||||||
tableAt(pde & addr_mask)[(virt >> 12) & 0x1FF] = 0;
|
|
||||||
invalidate(virt);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn invalidate(virt: u64) void {
|
|
||||||
// invlpg needs its operand via a register-indirect memory reference that Zig
|
|
||||||
// inline asm won't form directly, so stage the address in a register first.
|
|
||||||
asm volatile (
|
|
||||||
\\mov %[v], %%rax
|
|
||||||
\\invlpg (%%rax)
|
|
||||||
:
|
|
||||||
: [v] "r" (virt),
|
|
||||||
: .{ .rax = true, .memory = true }
|
|
||||||
);
|
|
||||||
}
|
|
||||||
@@ -1,48 +0,0 @@
|
|||||||
//! Task State Segment and its interrupt stack. In long mode the TSS's main job
|
|
||||||
//! is the Interrupt Stack Table: an IDT gate can name an IST entry, and the CPU
|
|
||||||
//! switches to that stack when the exception fires — no matter how broken the
|
|
||||||
//! interrupted stack was. We use IST1 for the double-fault handler, so a fault
|
|
||||||
//! that happens *because* the current stack is unusable still lands on solid
|
|
||||||
//! ground instead of triple-faulting.
|
|
||||||
|
|
||||||
const gdt = @import("gdt.zig");
|
|
||||||
|
|
||||||
/// x86_64 TSS. `packed` because several 64-bit fields sit at 4-byte-unaligned
|
|
||||||
/// offsets (rsp0 at byte 4), which a normal struct would pad away.
|
|
||||||
const Tss = packed struct {
|
|
||||||
reserved0: u32 = 0,
|
|
||||||
rsp0: u64 = 0,
|
|
||||||
rsp1: u64 = 0,
|
|
||||||
rsp2: u64 = 0,
|
|
||||||
reserved1: u64 = 0,
|
|
||||||
ist1: u64 = 0,
|
|
||||||
ist2: u64 = 0,
|
|
||||||
ist3: u64 = 0,
|
|
||||||
ist4: u64 = 0,
|
|
||||||
ist5: u64 = 0,
|
|
||||||
ist6: u64 = 0,
|
|
||||||
ist7: u64 = 0,
|
|
||||||
reserved2: u64 = 0,
|
|
||||||
reserved3: u16 = 0,
|
|
||||||
iomap_base: u16 = 0,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// The IST slot (1-based, as the IDT gate encodes it) used for critical faults.
|
|
||||||
pub const double_fault_ist = 1;
|
|
||||||
|
|
||||||
var tss: Tss align(16) = .{};
|
|
||||||
|
|
||||||
/// Dedicated stack for IST1. Static so it needs no allocator and is always valid.
|
|
||||||
var ist1_stack: [16 * 1024]u8 align(16) = undefined;
|
|
||||||
|
|
||||||
/// Loads the task register with the TSS selector. Defined in isr.s.
|
|
||||||
extern fn load_tr(selector: u16) callconv(.c) void;
|
|
||||||
|
|
||||||
/// Point IST1 at its stack, publish the TSS through the GDT, and load it into the
|
|
||||||
/// task register. Requires the GDT to already be loaded (gdt.init first).
|
|
||||||
pub fn init() void {
|
|
||||||
tss.ist1 = @intFromPtr(&ist1_stack) + ist1_stack.len; // stacks grow down
|
|
||||||
tss.iomap_base = @sizeOf(Tss); // == limit: no I/O permission bitmap
|
|
||||||
gdt.setTss(@intFromPtr(&tss), @sizeOf(Tss) - 1);
|
|
||||||
load_tr(gdt.tss_selector);
|
|
||||||
}
|
|
||||||
@@ -1,248 +0,0 @@
|
|||||||
const std = @import("std");
|
|
||||||
const danos = @import("danos");
|
|
||||||
const arch = @import("arch");
|
|
||||||
const console = @import("console.zig");
|
|
||||||
const pmm = @import("pmm.zig");
|
|
||||||
const heap = @import("heap.zig");
|
|
||||||
const scheduler = @import("scheduler.zig");
|
|
||||||
const platform = @import("platform");
|
|
||||||
const tests = @import("tests.zig");
|
|
||||||
const build_options = @import("build_options");
|
|
||||||
const BootInfo = danos.BootInfo;
|
|
||||||
|
|
||||||
/// The calling convention used to enter the kernel. Pinned to SysV explicitly:
|
|
||||||
/// the bootloader is built for the UEFI target, whose C convention is Microsoft
|
|
||||||
/// x64 (first argument in RCX), while the kernel is SysV (first argument in
|
|
||||||
/// RDI). Both sides reference this so the `boot_info` pointer lands in the
|
|
||||||
/// register the other expects. `danos.kernel_abi` re-exports it to the loader.
|
|
||||||
pub const kernel_abi = danos.kernel_abi;
|
|
||||||
|
|
||||||
/// The system console, valid once `kmain` has initialised it. Global so the
|
|
||||||
/// panic handler can reach it too.
|
|
||||||
var con: console.Console = undefined;
|
|
||||||
var con_ready = false;
|
|
||||||
|
|
||||||
/// Kernel entry point. The bootloader jumps here after `ExitBootServices` with a
|
|
||||||
/// pointer to the handoff data. There is no runtime, no stack unwinding, and no
|
|
||||||
/// caller to return to, so this never returns.
|
|
||||||
export fn _start(boot_info: *const BootInfo) callconv(kernel_abi) noreturn {
|
|
||||||
kmain(boot_info);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn kmain(boot_info: *const BootInfo) noreturn {
|
|
||||||
arch.serialInit(); // machine-readable log; console mirrors to it
|
|
||||||
|
|
||||||
const fb = boot_info.framebuffer;
|
|
||||||
const serial0 = console.SerialConsole;
|
|
||||||
con = console.Console.init(fb);
|
|
||||||
con.clear();
|
|
||||||
con_ready = true;
|
|
||||||
|
|
||||||
// Catch CPU exceptions before doing anything that might fault: install our
|
|
||||||
// reporter, then bring up the GDT + IDT.
|
|
||||||
arch.setFaultHandler(onException);
|
|
||||||
arch.init();
|
|
||||||
|
|
||||||
con.write("danos: initalizing kernel...");
|
|
||||||
|
|
||||||
serial0.debugWrite("danos: framebuffer console online\n");
|
|
||||||
serial0.debugWrite("danos: cpu tables online (GDT, IDT, TSS)\n");
|
|
||||||
serial0.debugPrint(" resolution : {d}x{d}\n", .{ fb.width, fb.height });
|
|
||||||
serial0.debugPrint(" pitch : {d} bytes\n", .{fb.pitch});
|
|
||||||
serial0.debugPrint(" format : {s}\n", .{@tagName(fb.format)});
|
|
||||||
serial0.debugPrint(" framebuffer: 0x{x:0>16}\n", .{fb.base});
|
|
||||||
serial0.debugPrint (" footprint : {d} MiB\n", .{(fb.pitch * fb.height) / (1024 * 1024)});
|
|
||||||
|
|
||||||
// Summarise the physical memory the loader handed us. The array is danos's
|
|
||||||
// own MemoryRegion, so this is a plain slice — no firmware layout in sight.
|
|
||||||
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(boot_info.memory_map.regions))[0..boot_info.memory_map.len];
|
|
||||||
var usable_pages: u64 = 0;
|
|
||||||
var reserved_pages: u64 = 0; // reserved RAM only — MMIO is device space, not RAM
|
|
||||||
for (regions) |r| {
|
|
||||||
switch (r.kind) {
|
|
||||||
.usable => usable_pages += r.pages,
|
|
||||||
.reserved, .acpi_tables, .acpi_nvs => reserved_pages += r.pages,
|
|
||||||
.mmio => {},
|
|
||||||
}
|
|
||||||
}
|
|
||||||
const total_pages = usable_pages + reserved_pages;
|
|
||||||
const total_bytes = total_pages * danos.page_size;
|
|
||||||
const gib = 1 << 30;
|
|
||||||
|
|
||||||
serial0.debugWrite("\ndanos: physical memory\n");
|
|
||||||
serial0.debugPrint(" total RAM : {d}.{d:0>2} GiB ({d} MiB) - RAM the firmware reported\n", .{ total_bytes / gib, (total_bytes % gib) * 100 / gib, mib(total_pages) });
|
|
||||||
serial0.debugPrint(" usable : {d} MiB - free RAM (incl. reclaimed boot-services memory)\n", .{mib(usable_pages)});
|
|
||||||
serial0.debugPrint(" reserved : {d} MiB - kernel image, boot stack, ACPI, runtime services\n", .{mib(reserved_pages)});
|
|
||||||
serial0.debugPrint(" regions : {d} - entries in the firmware memory map\n", .{regions.len});
|
|
||||||
|
|
||||||
// Bring up the physical frame allocator over that map, and prove it works:
|
|
||||||
// allocate three frames, then hand them back.
|
|
||||||
pmm.init(boot_info.memory_map);
|
|
||||||
const s1 = pmm.stats();
|
|
||||||
serial0.debugPrint("\ndanos: frame allocator online\n", .{});
|
|
||||||
serial0.debugPrint(" free frames: {d} ({d} MiB)\n", .{ s1.free_frames, mib(s1.free_frames) });
|
|
||||||
const f0 = pmm.alloc();
|
|
||||||
const f1 = pmm.alloc();
|
|
||||||
const f2 = pmm.alloc();
|
|
||||||
serial0.debugPrint(" alloc x3 : 0x{x} 0x{x} 0x{x}\n", .{ f0 orelse 0, f1 orelse 0, f2 orelse 0 });
|
|
||||||
if (f0) |p| pmm.free(p);
|
|
||||||
if (f1) |p| pmm.free(p);
|
|
||||||
if (f2) |p| pmm.free(p);
|
|
||||||
serial0.debugPrint(" after free : {d} frames free\n", .{pmm.stats().free_frames});
|
|
||||||
|
|
||||||
// Switch off the firmware's page tables onto our own (with real permissions).
|
|
||||||
arch.enablePaging(pmm.alloc, boot_info);
|
|
||||||
serial0.debugPrint("\ndanos: paging enabled\n", .{});
|
|
||||||
serial0.debugPrint(" page tables: CR3 = 0x{x:0>16}\n", .{arch.readCr3()});
|
|
||||||
serial0.debugPrint(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_info.kernel_segment_count});
|
|
||||||
|
|
||||||
// Bring up the kernel heap (dynamic allocation), built on the VMM.
|
|
||||||
heap.init();
|
|
||||||
serial0.debugWrite("\ndanos: kernel heap online\n");
|
|
||||||
// Measure the amount of resources the kernel is actually using
|
|
||||||
const s2 = pmm.stats();
|
|
||||||
serial0.debugPrint(" Kernel footprint: {d} KiB\n", .{kib(s1.free_frames - s2.free_frames)});
|
|
||||||
|
|
||||||
// Enumerate hardware from the firmware tables (ACPI here) into a generic
|
|
||||||
// device tree, then list it. Discovery walks ACPI memory directly (identity-
|
|
||||||
// mapped) and maps PCIe config space on demand via the VMM. A failure here is
|
|
||||||
// not fatal yet — log it and carry on.
|
|
||||||
const hal = platform.Hal{
|
|
||||||
.mapMmio = arch.mapPage,
|
|
||||||
.pioRead = arch.pioRead,
|
|
||||||
.pioWrite = arch.pioWrite,
|
|
||||||
};
|
|
||||||
if (platform.discover(boot_info, heap.allocator(), hal)) |devtree| {
|
|
||||||
var dt = devtree;
|
|
||||||
serial0.debugWrite("\ndanos: device discovery online\n");
|
|
||||||
dt.dump(console.SerialConsole.debugWrite);
|
|
||||||
|
|
||||||
// Power register map extracted from the FADT + AML, for confidence it parsed.
|
|
||||||
const pw = platform.powerInfo();
|
|
||||||
serial0.debugWrite("danos: power\n");
|
|
||||||
serial0.debugPrint(" pm1a_cnt : {s} 0x{x} (width {d})\n", .{ if (pw.pm1a_cnt.mmio) "mmio" else "io", pw.pm1a_cnt.address, pw.pm1a_cnt.width });
|
|
||||||
if (pw.s5) |s| {
|
|
||||||
serial0.debugPrint(" S5 slp_typ : a={d} b={d}\n", .{ s.slp_typ_a, s.slp_typ_b });
|
|
||||||
} else {
|
|
||||||
serial0.debugWrite(" S5 slp_typ : (not found)\n");
|
|
||||||
}
|
|
||||||
serial0.debugPrint(" reset : supported={} {s} 0x{x} val 0x{x}\n", .{ pw.reset_supported, if (pw.reset.mmio) "mmio" else "io", pw.reset.address, pw.reset_value });
|
|
||||||
|
|
||||||
// AML namespace parse integrity: consumed should equal total.
|
|
||||||
const am = platform.amlStats();
|
|
||||||
serial0.debugPrint(" aml : {d} namespace nodes, parsed {d}/{d} bytes\n", .{ am.nodes, am.consumed, am.total });
|
|
||||||
|
|
||||||
// Feed the arch layer the discovered addresses/facts so it makes no legacy
|
|
||||||
// assumptions — the point of all this on UEFI Class 3 firmware. MMIO bases
|
|
||||||
// (HPET, I/O APIC) come from the device tree; scalar facts from ACPI.
|
|
||||||
const pinfo = platform.platformInfo();
|
|
||||||
const hpet_base: u64 = if (dt.firstOfClass(.timer)) |t|
|
|
||||||
(if (t.firstResource(.memory)) |r| r.start else 0)
|
|
||||||
else
|
|
||||||
0;
|
|
||||||
var ioapic_base: u64 = 0;
|
|
||||||
var ioapic_gsi: u32 = 0;
|
|
||||||
if (dt.firstOfClass(.interrupt_controller)) |ic| {
|
|
||||||
if (ic.firstResource(.memory)) |r| ioapic_base = r.start;
|
|
||||||
if (ic.firstResource(.irq)) |r| ioapic_gsi = @intCast(r.start);
|
|
||||||
}
|
|
||||||
var isos: [16]arch.IsoEntry = undefined;
|
|
||||||
const iso_n = @min(pinfo.override_count, isos.len);
|
|
||||||
for (0..iso_n) |i| isos[i] = .{
|
|
||||||
.source = pinfo.overrides[i].source,
|
|
||||||
.gsi = pinfo.overrides[i].gsi,
|
|
||||||
.flags = pinfo.overrides[i].flags,
|
|
||||||
};
|
|
||||||
const pm_timer: ?arch.PmTimer = if (pinfo.pm_timer.present())
|
|
||||||
.{ .mmio = pinfo.pm_timer.mmio, .address = pinfo.pm_timer.address, .is_32bit = pinfo.pm_timer_32bit }
|
|
||||||
else
|
|
||||||
null;
|
|
||||||
arch.configurePlatform(.{
|
|
||||||
.pic_present = pinfo.pic_present,
|
|
||||||
.hpet_base = hpet_base,
|
|
||||||
.pm_timer = pm_timer,
|
|
||||||
.ioapic_base = ioapic_base,
|
|
||||||
.ioapic_gsi_base = ioapic_gsi,
|
|
||||||
.overrides = isos[0..iso_n],
|
|
||||||
});
|
|
||||||
if (pinfo.spcr_uart) |u| arch.serialReconfigure(u.mmio, u.address);
|
|
||||||
|
|
||||||
serial0.debugWrite("danos: platform\n");
|
|
||||||
serial0.debugPrint(" 8259 PIC : {s}\n", .{if (pinfo.pic_present) "present" else "absent"});
|
|
||||||
serial0.debugPrint(" lapic base : 0x{x}\n", .{pinfo.lapic_base});
|
|
||||||
serial0.debugPrint(" hpet base : 0x{x}\n", .{hpet_base});
|
|
||||||
serial0.debugPrint(" pm timer : {s} 0x{x} ({s})\n", .{ if (pinfo.pm_timer.mmio) "mmio" else "io", pinfo.pm_timer.address, if (pinfo.pm_timer_32bit) "32-bit" else "24-bit" });
|
|
||||||
if (pinfo.spcr_uart) |u| {
|
|
||||||
serial0.debugPrint(" console UART: {s} 0x{x} (SPCR type {d})\n", .{ if (u.mmio) "mmio" else "io", u.address, pinfo.spcr_kind });
|
|
||||||
} else {
|
|
||||||
serial0.debugWrite(" console UART: none in SPCR -> legacy COM1\n");
|
|
||||||
}
|
|
||||||
serial0.debugPrint(" ioapic : base 0x{x}, {d} inputs (masked); entry0 low 0x{x}\n", .{ ioapic_base, arch.ioapicEntryCount(), arch.ioapicEntryLow(0) });
|
|
||||||
} else |err| {
|
|
||||||
serial0.debugPrint("\ndanos: device discovery failed: {s}\n", .{@errorName(err)});
|
|
||||||
}
|
|
||||||
|
|
||||||
// Register the current context as the first task before enabling preemption.
|
|
||||||
scheduler.init(4);
|
|
||||||
serial0.debugWrite("\ndanos: scheduler online\n");
|
|
||||||
|
|
||||||
// Start the timer and unmask interrupts — the kernel now has a heartbeat, and
|
|
||||||
// the timer preempts among tasks.
|
|
||||||
arch.startTimer();
|
|
||||||
arch.enableInterrupts();
|
|
||||||
serial0.debugPrint("danos: timer online ({d} Hz tick; LAPIC {d} MHz, TSC {d} MHz; calibrated via {s})\n", .{ arch.timer_hz, arch.lapicHz() / 1_000_000, arch.tscHz() / 1_000_000, arch.timerCalibrationSource() });
|
|
||||||
|
|
||||||
// In a test build (`zig build -Dtest-case=<name>`), run that case and stop.
|
|
||||||
// Normal builds fall through to the idle halt.
|
|
||||||
if (build_options.test_case) |case| {
|
|
||||||
tests.run(case, boot_info);
|
|
||||||
arch.halt();
|
|
||||||
}
|
|
||||||
|
|
||||||
con.write("kernel initialised.\n");
|
|
||||||
|
|
||||||
// TODO: init process
|
|
||||||
|
|
||||||
con.write("\nnothing left to do; halting CPU.\n");
|
|
||||||
|
|
||||||
arch.halt();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Frames (4 KiB pages) to whole MiB.
|
|
||||||
fn mib(pages: u64) u64 {
|
|
||||||
return pages * danos.page_size / (1024 * 1024);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn kib(frames: u64) u64 {
|
|
||||||
return frames * danos.page_size / (1024);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Report a CPU exception in red and halt. There's no fault recovery yet, so any
|
|
||||||
/// exception is terminal — but now it debugPrints what and where instead of silently
|
|
||||||
/// resetting the machine.
|
|
||||||
fn onException(state: *const arch.CpuState) noreturn {
|
|
||||||
if (con_ready) {
|
|
||||||
con.fg = 0x00ff_5555;
|
|
||||||
con.print("\nCPU EXCEPTION: {s} (vector {d})\n", .{ arch.vectorName(state.vector), state.vector });
|
|
||||||
con.print(" error code : 0x{x}\n", .{state.error_code});
|
|
||||||
con.print(" RIP : 0x{x:0>16}\n", .{state.rip});
|
|
||||||
con.print(" RSP : 0x{x:0>16}\n", .{state.rsp});
|
|
||||||
if (state.vector == 14) con.print(" CR2 (addr) : 0x{x:0>16}\n", .{arch.readCr2()});
|
|
||||||
}
|
|
||||||
arch.halt();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Freestanding has no OS to receive a panic. debugPrint it to the console (if it is
|
|
||||||
/// up yet) in red, then halt.
|
|
||||||
pub const panic = std.debug.FullPanic(struct {
|
|
||||||
fn panic(msg: []const u8, first_trace_addr: ?usize) noreturn {
|
|
||||||
_ = first_trace_addr;
|
|
||||||
if (con_ready) {
|
|
||||||
con.fg = 0x00ff_5555;
|
|
||||||
con.write("\nKERNEL PANIC: ");
|
|
||||||
con.write(msg);
|
|
||||||
con.write("\n");
|
|
||||||
}
|
|
||||||
arch.halt();
|
|
||||||
}
|
|
||||||
}.panic);
|
|
||||||
@@ -1,251 +0,0 @@
|
|||||||
//! The scheduler: fixed-priority preemptive multitasking.
|
|
||||||
//!
|
|
||||||
//! Tasks are kernel threads (ring 0, each with its own stack). The **highest-
|
|
||||||
//! priority ready task always runs**; within a priority level, tasks round-robin.
|
|
||||||
//! Selection is O(1) — a bitmap of non-empty priority levels plus a FIFO queue per
|
|
||||||
//! level — which keeps scheduling deterministic, as a real-time kernel needs (see
|
|
||||||
//! docs/vision.md).
|
|
||||||
//!
|
|
||||||
//! Switching happens both cooperatively (`yield`) and preemptively (the timer
|
|
||||||
//! calls `tick`). See docs/scheduling.md for the interrupt-flag discipline that
|
|
||||||
//! makes those two paths coexist.
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
const arch = @import("arch");
|
|
||||||
const heap = @import("heap.zig");
|
|
||||||
|
|
||||||
/// Priority level: 0 (lowest) .. 7 (highest). 8 levels total.
|
|
||||||
pub const Priority = u3;
|
|
||||||
const num_priorities = 8;
|
|
||||||
|
|
||||||
const stack_size = 16 * 1024; // each task's kernel stack is 16 KiB
|
|
||||||
const max_tasks = 16; // the maximum number of tasks alive at once is 16 in a static sized pool
|
|
||||||
|
|
||||||
const State = enum { free, ready, running, blocked };
|
|
||||||
|
|
||||||
const Task = struct {
|
|
||||||
id: u32 = 0,
|
|
||||||
state: State = .free,
|
|
||||||
priority: Priority = 0,
|
|
||||||
rsp: usize = 0, // saved stack pointer, valid while not running
|
|
||||||
stack: []u8 = &.{},
|
|
||||||
wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping
|
|
||||||
next: ?*Task = null, // ready-queue link
|
|
||||||
};
|
|
||||||
|
|
||||||
var tasks = [_]Task{.{}} ** max_tasks;
|
|
||||||
var current: *Task = undefined;
|
|
||||||
var next_id: u32 = 1;
|
|
||||||
|
|
||||||
// Per-priority FIFO ready queues, and a bitmap of which levels are non-empty.
|
|
||||||
var ready_head: [num_priorities]?*Task = .{null} ** num_priorities;
|
|
||||||
var ready_tail: [num_priorities]?*Task = .{null} ** num_priorities;
|
|
||||||
var ready_bitmap: u8 = 0;
|
|
||||||
|
|
||||||
var preemption_enabled = true;
|
|
||||||
|
|
||||||
/// Register the currently-running kernel context as the first task, spawn the
|
|
||||||
/// idle task, and hook the timer for preemption.
|
|
||||||
pub fn init(boot_priority: Priority) void {
|
|
||||||
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
|
|
||||||
current = &tasks[0];
|
|
||||||
spawn(idle, 0); // lowest priority, always runnable — runs when nothing else is
|
|
||||||
arch.setTickHook(tick);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The idle task: run when every other task is blocked or sleeping. `hlt` waits
|
|
||||||
/// for the next interrupt at near-zero power (see docs/halting.md).
|
|
||||||
fn idle() void {
|
|
||||||
while (true) asm volatile ("hlt");
|
|
||||||
}
|
|
||||||
|
|
||||||
fn enqueue(t: *Task) void {
|
|
||||||
t.next = null;
|
|
||||||
const p: usize = t.priority;
|
|
||||||
if (ready_tail[p]) |tail| tail.next = t else ready_head[p] = t;
|
|
||||||
ready_tail[p] = t;
|
|
||||||
ready_bitmap |= levelBit(t.priority);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn dequeueHighest() ?*Task {
|
|
||||||
if (ready_bitmap == 0) return null;
|
|
||||||
const level: Priority = @intCast(num_priorities - 1 - @clz(ready_bitmap));
|
|
||||||
const t = ready_head[level].?;
|
|
||||||
ready_head[level] = t.next;
|
|
||||||
if (ready_head[level] == null) {
|
|
||||||
ready_tail[level] = null;
|
|
||||||
ready_bitmap &= ~levelBit(level);
|
|
||||||
}
|
|
||||||
t.next = null;
|
|
||||||
return t;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn levelBit(p: Priority) u8 {
|
|
||||||
return @as(u8, 1) << p;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Create a task that runs `entry` at `priority`. It becomes ready immediately.
|
|
||||||
pub fn spawn(entry: *const fn () void, priority: Priority) void {
|
|
||||||
const t = freeSlot() orelse @panic("sched: task table full");
|
|
||||||
const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack");
|
|
||||||
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack };
|
|
||||||
next_id += 1;
|
|
||||||
const top = @intFromPtr(stack.ptr) + stack.len;
|
|
||||||
t.rsp = arch.initTaskStack(top, @intFromPtr(entry));
|
|
||||||
enqueue(t);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn freeSlot() ?*Task {
|
|
||||||
for (&tasks) |*t| {
|
|
||||||
if (t.state == .free) return t;
|
|
||||||
}
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Pick the highest-priority ready task and switch to it. Interrupts must be
|
|
||||||
/// disabled by the caller.
|
|
||||||
fn schedule() void {
|
|
||||||
const prev = current;
|
|
||||||
if (prev.state == .running) {
|
|
||||||
prev.state = .ready;
|
|
||||||
enqueue(prev); // back of its level's queue (round-robin)
|
|
||||||
}
|
|
||||||
const next = dequeueHighest() orelse {
|
|
||||||
prev.state = .running; // nothing else ready — keep running
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
next.state = .running;
|
|
||||||
current = next;
|
|
||||||
if (next != prev) arch.switchContext(&prev.rsp, next.rsp);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Voluntarily give up the CPU to the next ready task.
|
|
||||||
pub fn yield() void {
|
|
||||||
const flags = arch.saveInterrupts();
|
|
||||||
schedule();
|
|
||||||
arch.restoreInterrupts(flags);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Block the current task for `ms` milliseconds, then let it become runnable
|
|
||||||
/// again. The idle task (or other work) runs in the meantime.
|
|
||||||
pub fn sleep(ms: u64) void {
|
|
||||||
const flags = arch.saveInterrupts();
|
|
||||||
current.wake_at = arch.millis() + ms;
|
|
||||||
current.state = .blocked;
|
|
||||||
schedule(); // current is blocked, so schedule() won't re-enqueue it
|
|
||||||
arch.restoreInterrupts(flags);
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- event-based blocking -------------------------------------------------
|
|
||||||
//
|
|
||||||
// A WaitQueue is a set of tasks blocked waiting for something (a resource, a
|
|
||||||
// message). Tasks link into it through the same `next` field the ready queues
|
|
||||||
// use — a task is in exactly one queue at a time. These are the primitive locks,
|
|
||||||
// semaphores and IPC channels are built on.
|
|
||||||
|
|
||||||
pub const WaitQueue = struct {
|
|
||||||
head: ?*Task = null,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Block the current task on `wq` and switch away. Precondition: interrupts are
|
|
||||||
/// disabled (the caller holds them, so a condition can be checked and the block
|
|
||||||
/// committed atomically). On return — when woken — interrupts are still disabled.
|
|
||||||
pub fn waitLocked(wq: *WaitQueue) void {
|
|
||||||
current.state = .blocked;
|
|
||||||
current.next = wq.head;
|
|
||||||
wq.head = current;
|
|
||||||
schedule();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Move the highest-priority waiter on `wq` (if any) to the ready queue.
|
|
||||||
/// Precondition: interrupts disabled. Does not preempt — the caller decides.
|
|
||||||
pub fn wakeLocked(wq: *WaitQueue) void {
|
|
||||||
// Find the highest-priority waiter (bounded scan) and unlink it.
|
|
||||||
var best_prev: ?*Task = null;
|
|
||||||
var best: ?*Task = null;
|
|
||||||
var prev: ?*Task = null;
|
|
||||||
var cur = wq.head;
|
|
||||||
while (cur) |t| : ({
|
|
||||||
prev = t;
|
|
||||||
cur = t.next;
|
|
||||||
}) {
|
|
||||||
if (best == null or t.priority > best.?.priority) {
|
|
||||||
best = t;
|
|
||||||
best_prev = prev;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
const t = best orelse return;
|
|
||||||
if (best_prev) |p| p.next = t.next else wq.head = t.next;
|
|
||||||
t.state = .ready;
|
|
||||||
enqueue(t);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Block on `wq` (a self-contained critical section).
|
|
||||||
pub fn wait(wq: *WaitQueue) void {
|
|
||||||
const flags = arch.saveInterrupts();
|
|
||||||
waitLocked(wq);
|
|
||||||
arch.restoreInterrupts(flags);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Wake the highest-priority waiter on `wq`, preempting if it outranks us.
|
|
||||||
pub fn wake(wq: *WaitQueue) void {
|
|
||||||
const flags = arch.saveInterrupts();
|
|
||||||
wakeLocked(wq);
|
|
||||||
// If a higher-priority task is now ready, run it immediately.
|
|
||||||
if (highestReadyPriority()) |p| {
|
|
||||||
if (p > current.priority) schedule();
|
|
||||||
}
|
|
||||||
arch.restoreInterrupts(flags);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn highestReadyPriority() ?Priority {
|
|
||||||
if (ready_bitmap == 0) return null;
|
|
||||||
return @intCast(num_priorities - 1 - @clz(ready_bitmap));
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Wake any sleeping task whose deadline has passed. Bounded by the task count,
|
|
||||||
/// so it stays deterministic. Called from the timer tick (interrupts disabled).
|
|
||||||
fn wakeExpired() void {
|
|
||||||
const now = arch.millis();
|
|
||||||
for (&tasks) |*t| {
|
|
||||||
if (t.state == .blocked and t.wake_at != 0 and now >= t.wake_at) {
|
|
||||||
t.wake_at = 0;
|
|
||||||
t.state = .ready;
|
|
||||||
enqueue(t);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Called from the timer interrupt (interrupts already disabled): wake due
|
|
||||||
/// sleepers, then preempt.
|
|
||||||
pub fn tick() void {
|
|
||||||
wakeExpired();
|
|
||||||
if (preemption_enabled) schedule();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Enable or disable timer-driven preemption (cooperative-only when off).
|
|
||||||
pub fn setPreemption(enabled: bool) void {
|
|
||||||
preemption_enabled = enabled;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// End the current task and switch away for good; never returns. The task's stack
|
|
||||||
/// is leaked for now (no reaper yet).
|
|
||||||
pub fn exit() noreturn {
|
|
||||||
arch.disableInterrupts();
|
|
||||||
current.state = .free;
|
|
||||||
const next = dequeueHighest() orelse @panic("sched: no task left to run");
|
|
||||||
next.state = .running;
|
|
||||||
current = next;
|
|
||||||
var discard: usize = 0;
|
|
||||||
arch.switchContext(&discard, next.rsp);
|
|
||||||
unreachable;
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn currentId() u32 {
|
|
||||||
return current.id;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Change the running task's priority (takes effect next time it's enqueued).
|
|
||||||
pub fn setPriority(p: Priority) void {
|
|
||||||
current.priority = p;
|
|
||||||
}
|
|
||||||
@@ -1,491 +0,0 @@
|
|||||||
//! In-kernel test cases, run at the end of bring-up when the kernel is built with
|
|
||||||
//! `-Dtest-case=<name>`. Each case writes structured markers to the serial port
|
|
||||||
//! that the QEMU harness (test/qemu_test.py) asserts on:
|
|
||||||
//!
|
|
||||||
//! [PASS]/[FAIL] <check> per assertion
|
|
||||||
//! DANOS-TEST-RESULT: PASS|FAIL overall, for non-faulting cases
|
|
||||||
//!
|
|
||||||
//! Faulting cases (fault-ud, fault-pf, fault-df) deliberately don't return a
|
|
||||||
//! result line — they trigger a CPU exception, and the harness asserts on the
|
|
||||||
//! exception report the handler prints (which also reaches serial).
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
const danos = @import("danos");
|
|
||||||
const arch = @import("arch");
|
|
||||||
const platform = @import("platform");
|
|
||||||
const pmm = @import("pmm.zig");
|
|
||||||
const heap = @import("heap.zig");
|
|
||||||
const sched = @import("scheduler.zig");
|
|
||||||
const ipc = @import("ipc.zig");
|
|
||||||
|
|
||||||
/// Formatted write straight to serial, independent of the framebuffer console.
|
|
||||||
fn log(comptime fmt: []const u8, args: anytype) void {
|
|
||||||
var buf: [128]u8 = undefined;
|
|
||||||
arch.serialWrite(std.fmt.bufPrint(&buf, fmt, args) catch return);
|
|
||||||
}
|
|
||||||
|
|
||||||
var passed: u32 = 0;
|
|
||||||
var failed: u32 = 0;
|
|
||||||
|
|
||||||
fn check(name: []const u8, ok: bool) void {
|
|
||||||
if (ok) {
|
|
||||||
passed += 1;
|
|
||||||
log("[PASS] {s}\n", .{name});
|
|
||||||
} else {
|
|
||||||
failed += 1;
|
|
||||||
log("[FAIL] {s}\n", .{name});
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Emit the overall result line the harness matches, then the done sentinel.
|
|
||||||
fn result() void {
|
|
||||||
log("DANOS-TEST-RESULT: {s} ({d} passed, {d} failed)\n", .{
|
|
||||||
if (failed == 0) "PASS" else "FAIL",
|
|
||||||
passed,
|
|
||||||
failed,
|
|
||||||
});
|
|
||||||
log("DANOS-TEST-DONE\n", .{});
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn run(case: []const u8, boot_info: *const BootInfo) void {
|
|
||||||
if (eql(case, "smoke")) {
|
|
||||||
smoke(boot_info);
|
|
||||||
} else if (eql(case, "timer")) {
|
|
||||||
timer();
|
|
||||||
} else if (eql(case, "clock")) {
|
|
||||||
clock();
|
|
||||||
} else if (eql(case, "vmm")) {
|
|
||||||
vmm();
|
|
||||||
} else if (eql(case, "heap")) {
|
|
||||||
heapTest();
|
|
||||||
} else if (eql(case, "sched")) {
|
|
||||||
schedTest();
|
|
||||||
} else if (eql(case, "priority")) {
|
|
||||||
priorityTest();
|
|
||||||
} else if (eql(case, "sleep")) {
|
|
||||||
sleepTest();
|
|
||||||
} else if (eql(case, "event")) {
|
|
||||||
eventTest();
|
|
||||||
} else if (eql(case, "ipc")) {
|
|
||||||
ipcTest();
|
|
||||||
} else if (eql(case, "fault-ud")) {
|
|
||||||
faultInvalidOpcode();
|
|
||||||
} else if (eql(case, "fault-pf")) {
|
|
||||||
faultPageFault();
|
|
||||||
} else if (eql(case, "fault-df")) {
|
|
||||||
faultDoubleFault();
|
|
||||||
} else if (eql(case, "fault-nx")) {
|
|
||||||
faultNoExecute();
|
|
||||||
} else if (eql(case, "fault-null")) {
|
|
||||||
faultNull();
|
|
||||||
} else if (eql(case, "poweroff")) {
|
|
||||||
powerTest(.off);
|
|
||||||
} else if (eql(case, "reboot")) {
|
|
||||||
powerTest(.reboot);
|
|
||||||
} else {
|
|
||||||
log("DANOS-TEST-RESULT: FAIL (unknown case '{s}')\n", .{case});
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn platformHal() platform.Hal {
|
|
||||||
return .{
|
|
||||||
.mapMmio = arch.mapPage,
|
|
||||||
.pioRead = arch.pioRead,
|
|
||||||
.pioWrite = arch.pioWrite,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Drive an ACPI power transition. On success the machine powers off or resets,
|
|
||||||
/// so QEMU exits — the harness observes the process exit. If control returns, the
|
|
||||||
/// transition failed and we emit a FAIL result.
|
|
||||||
fn powerTest(comptime action: enum { off, reboot }) void {
|
|
||||||
const name = if (action == .off) "poweroff" else "reboot";
|
|
||||||
log("DANOS-TEST-BEGIN: {s}\n", .{name});
|
|
||||||
const hal = platformHal();
|
|
||||||
log("DANOS-POWER: attempting {s}\n", .{name});
|
|
||||||
switch (action) {
|
|
||||||
.off => platform.shutdown(hal),
|
|
||||||
.reboot => platform.reboot(hal),
|
|
||||||
}
|
|
||||||
check("power transition took effect", false);
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
const BootInfo = danos.BootInfo;
|
|
||||||
|
|
||||||
fn eql(a: []const u8, b: []const u8) bool {
|
|
||||||
return std.mem.eql(u8, a, b);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Non-destructive checks of the memory map and frame allocator.
|
|
||||||
fn smoke(boot_info: *const BootInfo) void {
|
|
||||||
log("DANOS-TEST-BEGIN: smoke\n", .{});
|
|
||||||
|
|
||||||
// The memory map has some usable RAM.
|
|
||||||
const mm = boot_info.memory_map;
|
|
||||||
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
|
||||||
var usable: u64 = 0;
|
|
||||||
for (regions) |r| {
|
|
||||||
if (r.kind == .usable) usable += r.pages;
|
|
||||||
}
|
|
||||||
check("memory map reports usable RAM", usable > 0);
|
|
||||||
|
|
||||||
// The frame allocator hands out distinct, page-aligned frames.
|
|
||||||
const a = pmm.alloc();
|
|
||||||
const b = pmm.alloc();
|
|
||||||
check("alloc returns a frame", a != null);
|
|
||||||
check("alloc returns distinct frames", a != null and b != null and a.? != b.?);
|
|
||||||
check("frames are page-aligned", (a orelse 1) % danos.page_size == 0);
|
|
||||||
|
|
||||||
// Freeing restores the count.
|
|
||||||
const before = pmm.stats().free_frames;
|
|
||||||
if (a) |p| pmm.free(p);
|
|
||||||
if (b) |p| pmm.free(p);
|
|
||||||
check("free returns frames to the pool", pmm.stats().free_frames == before + 2);
|
|
||||||
|
|
||||||
// Paging is active on our own tables (CR3 is non-zero and page-aligned).
|
|
||||||
const cr3 = arch.readCr3();
|
|
||||||
check("paging active (CR3 set)", cr3 != 0 and cr3 % danos.page_size == 0);
|
|
||||||
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Verify device interrupts fire and return: the timer tick counter must advance
|
|
||||||
/// on its own. Interrupts are already enabled by kmain before tests run.
|
|
||||||
fn timer() void {
|
|
||||||
log("DANOS-TEST-BEGIN: timer\n", .{});
|
|
||||||
const start = arch.ticks();
|
|
||||||
// Busy-wait for the counter to advance. arch.ticks() is a volatile load, so
|
|
||||||
// the compiler re-reads it each iteration and sees the interrupt's update.
|
|
||||||
// The cap is only a safety net; the harness timeout is the real backstop.
|
|
||||||
var spins: u64 = 0;
|
|
||||||
while (arch.ticks() == start and spins < 5_000_000_000) spins +%= 1;
|
|
||||||
check("timer interrupts advance the tick count", arch.ticks() > start);
|
|
||||||
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Verify the on-demand VMM: map a fresh frame at an unused virtual address, and
|
|
||||||
/// check it's writable and reads back.
|
|
||||||
fn vmm() void {
|
|
||||||
log("DANOS-TEST-BEGIN: vmm\n", .{});
|
|
||||||
const frame = pmm.alloc();
|
|
||||||
check("frame available to map", frame != null);
|
|
||||||
if (frame) |phys| {
|
|
||||||
var virt: u64 = 0x0000_4000_0000_0000; // canonical, well clear of everything mapped
|
|
||||||
arch.mapPage(virt, phys, true);
|
|
||||||
const p: *volatile u64 = @ptrFromInt(virt);
|
|
||||||
p.* = 0xdead_c0de_cafe_babe;
|
|
||||||
check("mapped page is writable and reads back", p.* == 0xdead_c0de_cafe_babe);
|
|
||||||
arch.unmapPage(virt);
|
|
||||||
pmm.free(phys);
|
|
||||||
virt += 0;
|
|
||||||
}
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Exercise the kernel heap: basic alloc/write/free, reuse, growth beyond the
|
|
||||||
/// initial region, and a std container backed by it.
|
|
||||||
fn heapTest() void {
|
|
||||||
log("DANOS-TEST-BEGIN: heap\n", .{});
|
|
||||||
const a = heap.allocator();
|
|
||||||
|
|
||||||
// Allocate, write a pattern, read it back, free.
|
|
||||||
const buf = a.alloc(u8, 4096) catch null;
|
|
||||||
check("alloc 4096 bytes", buf != null);
|
|
||||||
if (buf) |b| {
|
|
||||||
@memset(b, 0xAB);
|
|
||||||
check("heap memory is writable and reads back", b[0] == 0xAB and b[4095] == 0xAB);
|
|
||||||
a.free(b);
|
|
||||||
}
|
|
||||||
|
|
||||||
// Freeing then re-allocating the same size should reuse the block.
|
|
||||||
const p1 = a.alloc(u64, 8) catch null;
|
|
||||||
const addr1 = if (p1) |p| @intFromPtr(p.ptr) else 0;
|
|
||||||
if (p1) |p| a.free(p);
|
|
||||||
const p2 = a.alloc(u64, 8) catch null;
|
|
||||||
const addr2 = if (p2) |p| @intFromPtr(p.ptr) else 0;
|
|
||||||
check("freed block is reused", addr1 != 0 and addr1 == addr2);
|
|
||||||
if (p2) |p| a.free(p);
|
|
||||||
|
|
||||||
// Force growth past the initial page and check every block is usable.
|
|
||||||
var blocks: [64]?[]u8 = .{null} ** 64;
|
|
||||||
var ok = true;
|
|
||||||
for (&blocks, 0..) |*slot, i| {
|
|
||||||
const b = a.alloc(u8, 4096) catch null;
|
|
||||||
slot.* = b;
|
|
||||||
if (b) |bb| @memset(bb, @intCast(i & 0xff)) else {
|
|
||||||
ok = false;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for (blocks, 0..) |slot, i| {
|
|
||||||
if (slot) |bb| {
|
|
||||||
if (bb[0] != @as(u8, @intCast(i & 0xff)) or bb[4095] != @as(u8, @intCast(i & 0xff))) ok = false;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
check("many allocations (heap growth) stay valid", ok);
|
|
||||||
for (blocks) |slot| {
|
|
||||||
if (slot) |bb| a.free(bb);
|
|
||||||
}
|
|
||||||
|
|
||||||
// A std container backed by the kernel heap.
|
|
||||||
var list: std.ArrayList(u32) = .empty;
|
|
||||||
var sum: u64 = 0;
|
|
||||||
var expected: u64 = 0;
|
|
||||||
var i: u32 = 0;
|
|
||||||
var list_ok = true;
|
|
||||||
while (i < 1000) : (i += 1) {
|
|
||||||
list.append(a, i) catch {
|
|
||||||
list_ok = false;
|
|
||||||
};
|
|
||||||
expected += i;
|
|
||||||
}
|
|
||||||
for (list.items) |v| sum += v;
|
|
||||||
list.deinit(a);
|
|
||||||
check("std.ArrayList on the kernel heap", list_ok and sum == expected);
|
|
||||||
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Verify the calibrated clocks: sane measured frequencies, monotonic uptime that
|
|
||||||
/// advances with real ticks, and — the point of the TSC clock — nanosecond
|
|
||||||
/// resolution far finer than the 1 ms tick, with the unit functions consistent.
|
|
||||||
fn clock() void {
|
|
||||||
log("DANOS-TEST-BEGIN: clock\n", .{});
|
|
||||||
|
|
||||||
const lapic = arch.lapicHz();
|
|
||||||
check("LAPIC frequency measured", lapic > 1_000_000 and lapic < 100_000_000_000);
|
|
||||||
const tsc = arch.tscHz();
|
|
||||||
check("TSC frequency measured", tsc > 100_000_000 and tsc < 100_000_000_000);
|
|
||||||
|
|
||||||
// Uptime advances over ~5 real ticks (1000 Hz => 1 tick == 1 ms).
|
|
||||||
const start_ticks = arch.ticks();
|
|
||||||
const start_ms = arch.millis();
|
|
||||||
var spins: u64 = 0;
|
|
||||||
while (arch.ticks() < start_ticks + 5 and spins < 5_000_000_000) spins +%= 1;
|
|
||||||
const elapsed_ms = arch.millis() - start_ms;
|
|
||||||
check("uptime advances with ticks", elapsed_ms >= 5 and elapsed_ms < 100);
|
|
||||||
|
|
||||||
// Sub-millisecond resolution: spin until nanos() first advances, then confirm
|
|
||||||
// that first step happened within a millisecond — so nanos() resolves finer
|
|
||||||
// than the 1 ms tick (a tick clock's smallest step *is* 1 ms). Spinning to the
|
|
||||||
// first change is robust to QEMU's coarse TSC update granularity.
|
|
||||||
const n1 = arch.nanos();
|
|
||||||
var s2: u64 = 0;
|
|
||||||
while (arch.nanos() == n1 and s2 < 10_000_000) s2 +%= 1;
|
|
||||||
const n2 = arch.nanos();
|
|
||||||
check("nanos() has sub-millisecond resolution", n2 > n1 and (n2 - n1) < 1_000_000);
|
|
||||||
|
|
||||||
// The unit functions agree (within rounding).
|
|
||||||
const ns = arch.nanos();
|
|
||||||
check("nanos/micros/millis are consistent", diffWithin(arch.micros(), ns / 1000, 1000) and diffWithin(arch.millis(), ns / 1_000_000, 2));
|
|
||||||
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
fn diffWithin(a: u64, b: u64, tol: u64) bool {
|
|
||||||
return if (a > b) a - b <= tol else b - a <= tol;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- scheduler tests ------------------------------------------------------
|
|
||||||
|
|
||||||
var counters = [_]u64{0} ** 3;
|
|
||||||
|
|
||||||
fn spin0() void {
|
|
||||||
const p: *volatile u64 = &counters[0];
|
|
||||||
while (true) p.* = p.* +% 1;
|
|
||||||
}
|
|
||||||
fn spin1() void {
|
|
||||||
const p: *volatile u64 = &counters[1];
|
|
||||||
while (true) p.* = p.* +% 1;
|
|
||||||
}
|
|
||||||
fn spin2() void {
|
|
||||||
const p: *volatile u64 = &counters[2];
|
|
||||||
while (true) p.* = p.* +% 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Preemption: spawn three tasks that busy-loop *without* yielding. If they all
|
|
||||||
/// make progress, the timer must be preempting between them (and the context
|
|
||||||
/// switch works) — because nothing yields voluntarily.
|
|
||||||
fn schedTest() void {
|
|
||||||
log("DANOS-TEST-BEGIN: sched\n", .{});
|
|
||||||
counters = .{ 0, 0, 0 };
|
|
||||||
sched.spawn(spin0, 4);
|
|
||||||
sched.spawn(spin1, 4);
|
|
||||||
sched.spawn(spin2, 4);
|
|
||||||
|
|
||||||
const c0: *volatile u64 = &counters[0];
|
|
||||||
const c1: *volatile u64 = &counters[1];
|
|
||||||
const c2: *volatile u64 = &counters[2];
|
|
||||||
var spins: u64 = 0;
|
|
||||||
while ((c0.* == 0 or c1.* == 0 or c2.* == 0) and spins < 5_000_000_000) spins +%= 1;
|
|
||||||
|
|
||||||
check("all three non-yielding tasks made progress (preemption)", c0.* > 0 and c1.* > 0 and c2.* > 0);
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
var run_order = [_]u8{0} ** 4;
|
|
||||||
var run_n: usize = 0;
|
|
||||||
|
|
||||||
fn recordExit(priority: u8) void {
|
|
||||||
run_order[run_n] = priority;
|
|
||||||
run_n += 1;
|
|
||||||
sched.exit();
|
|
||||||
}
|
|
||||||
fn taskHigh() void {
|
|
||||||
recordExit(6);
|
|
||||||
}
|
|
||||||
fn taskMid() void {
|
|
||||||
recordExit(4);
|
|
||||||
}
|
|
||||||
fn taskLow() void {
|
|
||||||
recordExit(2);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Fixed priority: with preemption off (deterministic), spawn tasks at three
|
|
||||||
/// priorities and let them run cooperatively. They must run highest-first.
|
|
||||||
fn priorityTest() void {
|
|
||||||
log("DANOS-TEST-BEGIN: priority\n", .{});
|
|
||||||
sched.setPreemption(false);
|
|
||||||
sched.setPriority(1); // above the idle task (0), below the workers — runs last
|
|
||||||
run_n = 0;
|
|
||||||
|
|
||||||
sched.spawn(taskLow, 2);
|
|
||||||
sched.spawn(taskMid, 4);
|
|
||||||
sched.spawn(taskHigh, 6);
|
|
||||||
|
|
||||||
while (run_n < 3) sched.yield(); // regain control only once the workers are done
|
|
||||||
|
|
||||||
check("tasks ran highest-priority first", run_order[0] == 6 and run_order[1] == 4 and run_order[2] == 2);
|
|
||||||
|
|
||||||
sched.setPriority(4);
|
|
||||||
sched.setPreemption(true);
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
var event_wq: sched.WaitQueue = .{};
|
|
||||||
var event_stage: u32 = 0;
|
|
||||||
|
|
||||||
fn eventWaiter() void {
|
|
||||||
event_stage = 1; // reached the wait
|
|
||||||
sched.wait(&event_wq); // block until woken
|
|
||||||
event_stage = 3; // woken and resumed
|
|
||||||
sched.exit();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Event-based blocking: a task blocks on a wait queue and is woken. The waiter is
|
|
||||||
/// higher priority, so waking it preempts us and it runs to completion at once.
|
|
||||||
fn eventTest() void {
|
|
||||||
log("DANOS-TEST-BEGIN: event\n", .{});
|
|
||||||
event_stage = 0;
|
|
||||||
sched.spawn(eventWaiter, 6); // higher priority than this task (4)
|
|
||||||
|
|
||||||
var spins: u64 = 0;
|
|
||||||
while (event_stage != 1 and spins < 1_000_000_000) : (spins += 1) sched.yield();
|
|
||||||
check("waiter reached the wait and blocked", event_stage == 1);
|
|
||||||
|
|
||||||
sched.wake(&event_wq);
|
|
||||||
check("wake resumed the blocked waiter (preempting)", event_stage == 3);
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
var channel: ipc.Channel(u64, 4) = .{};
|
|
||||||
var recv_sum: u64 = 0;
|
|
||||||
var recv_count: u64 = 0;
|
|
||||||
|
|
||||||
fn producer() void {
|
|
||||||
var i: u64 = 1;
|
|
||||||
while (i <= 100) : (i += 1) channel.send(i);
|
|
||||||
sched.exit();
|
|
||||||
}
|
|
||||||
fn consumer() void {
|
|
||||||
var n: u64 = 0;
|
|
||||||
while (n < 100) : (n += 1) {
|
|
||||||
recv_sum += channel.recv();
|
|
||||||
recv_count += 1;
|
|
||||||
}
|
|
||||||
sched.exit();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// IPC: a producer and consumer pass 100 messages through a 4-slot channel. The
|
|
||||||
/// small buffer forces the channel full and empty repeatedly, exercising both the
|
|
||||||
/// blocking-send and blocking-recv paths. The messages must arrive intact.
|
|
||||||
fn ipcTest() void {
|
|
||||||
log("DANOS-TEST-BEGIN: ipc\n", .{});
|
|
||||||
channel = .{};
|
|
||||||
recv_sum = 0;
|
|
||||||
recv_count = 0;
|
|
||||||
sched.spawn(consumer, 5); // above this task (4) so they run and we observe after
|
|
||||||
sched.spawn(producer, 5);
|
|
||||||
|
|
||||||
var spins: u64 = 0;
|
|
||||||
while (recv_count < 100 and spins < 2_000_000_000) : (spins += 1) sched.yield();
|
|
||||||
|
|
||||||
check("all 100 messages received", recv_count == 100);
|
|
||||||
check("messages arrived intact (sum 1..100 == 5050)", recv_sum == 5050);
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Blocking: sleep(50) should block this task for about 50 ms (measured on the
|
|
||||||
/// calibrated clock) — not busy-wait — while the idle task runs.
|
|
||||||
fn sleepTest() void {
|
|
||||||
log("DANOS-TEST-BEGIN: sleep\n", .{});
|
|
||||||
const t0 = arch.millis();
|
|
||||||
sched.sleep(50);
|
|
||||||
const elapsed = arch.millis() - t0;
|
|
||||||
check("sleep(50) blocked for ~50 ms", elapsed >= 50 and elapsed <= 70);
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
fn faultInvalidOpcode() void {
|
|
||||||
log("DANOS-TEST-BEGIN: fault-ud\n", .{});
|
|
||||||
asm volatile ("ud2");
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Verify NX: fetching an instruction from a data page (mapped no-execute) faults.
|
|
||||||
fn faultNoExecute() void {
|
|
||||||
log("DANOS-TEST-BEGIN: fault-nx\n", .{});
|
|
||||||
var scratch: u64 = 0xC3; // a lone `ret` — harmless if NX somehow let it run
|
|
||||||
const f: *const fn () void = @ptrFromInt(@intFromPtr(&scratch));
|
|
||||||
f(); // instruction fetch from an NX page -> #PF before it executes
|
|
||||||
log("DANOS-TEST-RESULT: FAIL (NX not enforced)\n", .{});
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Verify the null guard: dereferencing address 0 (page 0 left unmapped) faults.
|
|
||||||
fn faultNull() void {
|
|
||||||
log("DANOS-TEST-BEGIN: fault-null\n", .{});
|
|
||||||
// Launder the address through empty asm so the compiler no longer knows it's
|
|
||||||
// 0 (otherwise it folds a null-pointer safety panic instead of doing the real
|
|
||||||
// access). `allowzero` skips the same null check on the cast. The write then
|
|
||||||
// hits the unmapped page 0 and takes a real hardware #PF.
|
|
||||||
var addr: u64 = 0;
|
|
||||||
addr = asm ("" : [ret] "=r" (-> u64) : [in] "0" (addr));
|
|
||||||
const p: *allowzero volatile u64 = @ptrFromInt(addr);
|
|
||||||
p.* = 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn faultPageFault() void {
|
|
||||||
log("DANOS-TEST-BEGIN: fault-pf\n", .{});
|
|
||||||
// Runtime address so the backend emits a register store (not a `mov moffs`,
|
|
||||||
// which the self-hosted x86_64 backend can't encode).
|
|
||||||
var addr: u64 = 0xdeadbeef000; // well above all mapped RAM
|
|
||||||
const p: *volatile u64 = @ptrFromInt(addr);
|
|
||||||
p.* = 1;
|
|
||||||
addr += 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn faultDoubleFault() void {
|
|
||||||
log("DANOS-TEST-BEGIN: fault-df\n", .{});
|
|
||||||
arch.disableInterrupts(); // so only the ud2 delivery (not a timer tick) triggers the #DF
|
|
||||||
// Point RSP at unmapped memory, then fault: the CPU can't push the fault
|
|
||||||
// frame, which escalates to #DF — survivable only because #DF runs on IST1.
|
|
||||||
var bad_sp: u64 = 0x5000000000;
|
|
||||||
asm volatile (
|
|
||||||
\\mov %[sp], %%rsp
|
|
||||||
\\ud2
|
|
||||||
:
|
|
||||||
: [sp] "r" (bad_sp),
|
|
||||||
: .{ .memory = true }
|
|
||||||
);
|
|
||||||
bad_sp += 0;
|
|
||||||
}
|
|
||||||
-103
@@ -1,103 +0,0 @@
|
|||||||
//! Shared definitions that form the contract between a bootloader
|
|
||||||
//! (src/boot/, e.g. efi.zig built as BOOTX64.efi) and the kernel (src/kernel/main.zig).
|
|
||||||
//!
|
|
||||||
//! Both binaries import this as the "danos" module, so the handoff layout is
|
|
||||||
//! defined in exactly one place.
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
|
|
||||||
/// Calling convention for the bootloader→kernel jump. Pinned to SysV so it does
|
|
||||||
/// not depend on each binary's target default: the UEFI bootloader's C
|
|
||||||
/// convention is Microsoft x64 (first arg in RCX), the freestanding kernel's is
|
|
||||||
/// SysV (first arg in RDI). Both reference this to agree on where `*BootInfo`
|
|
||||||
/// is passed.
|
|
||||||
pub const kernel_abi: std.builtin.CallingConvention = .{ .x86_64_sysv = .{} };
|
|
||||||
|
|
||||||
/// Pixel byte order of the linear framebuffer the firmware handed us.
|
|
||||||
pub const PixelFormat = enum(u32) {
|
|
||||||
/// Byte 0 = Red, 1 = Green, 2 = Blue, 3 = reserved.
|
|
||||||
rgbx,
|
|
||||||
/// Byte 0 = Blue, 1 = Green, 2 = Red, 3 = reserved.
|
|
||||||
bgrx,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// A linear framebuffer: `width`x`height` pixels, each a 32-bit value, with
|
|
||||||
/// `pitch` bytes between the start of one row and the next (which may be larger
|
|
||||||
/// than `width * 4` due to hardware padding).
|
|
||||||
pub const Framebuffer = extern struct {
|
|
||||||
base: usize, // the memory address where pixel data starts
|
|
||||||
width: u32, // visible pixels per row (e.g. 1920)
|
|
||||||
height: u32, // visible rows (e.g. 1080)
|
|
||||||
pitch: u32, // bytes from the start of one row to the start of the next
|
|
||||||
format: PixelFormat,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Page size the memory map is measured in. 4 KiB on every architecture danos
|
|
||||||
/// targets so far.
|
|
||||||
pub const page_size = 4096;
|
|
||||||
|
|
||||||
/// danos's own classification of a span of physical memory — deliberately not
|
|
||||||
/// UEFI's vocabulary. Each boot path (UEFI now, device tree later) translates its
|
|
||||||
/// native memory description into these kinds, so the kernel never learns what
|
|
||||||
/// booted it. [[arch]] keeps the same discipline for CPU code.
|
|
||||||
pub const MemoryKind = enum(u32) {
|
|
||||||
/// Free RAM the kernel may allocate. Each boot path folds its own transient
|
|
||||||
/// memory into this once it's genuinely free (e.g. the UEFI loader classifies
|
|
||||||
/// boot-services memory as usable after ExitBootServices), so the kernel never
|
|
||||||
/// has to know about boot-protocol-specific "reclaimable" states.
|
|
||||||
usable,
|
|
||||||
/// Firmware, MMIO, the kernel image, our own boot buffers, the boot stack —
|
|
||||||
/// never hand out.
|
|
||||||
reserved,
|
|
||||||
/// ACPI tables: parse, then reclaim.
|
|
||||||
acpi_tables,
|
|
||||||
/// ACPI non-volatile storage: preserve across sleep, do not allocate.
|
|
||||||
acpi_nvs,
|
|
||||||
/// Not backed by RAM: memory-mapped device registers or a reserved
|
|
||||||
/// address-space window (e.g. PCIe config space). Kept distinct from
|
|
||||||
/// `reserved` so RAM accounting doesn't count device address space.
|
|
||||||
mmio,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// One contiguous span of physical memory. Because danos defines this layout
|
|
||||||
/// itself (unlike the UEFI descriptor it's built from), `@sizeOf` is
|
|
||||||
/// authoritative — the kernel walks a plain `[]MemoryRegion`, with none of the
|
|
||||||
/// firmware's variable descriptor-stride to worry about.
|
|
||||||
pub const MemoryRegion = extern struct {
|
|
||||||
base: u64, // physical start address
|
|
||||||
pages: u64, // length in `page_size` units
|
|
||||||
kind: MemoryKind,
|
|
||||||
_pad: u32 = 0,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// The physical memory layout handed to the kernel: a pointer to an array of
|
|
||||||
/// `len` `MemoryRegion`s, in a buffer that outlives the loader.
|
|
||||||
pub const MemoryMap = extern struct {
|
|
||||||
regions: usize, // address of a `[len]MemoryRegion`
|
|
||||||
len: usize,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// One PT_LOAD segment of the kernel image, so the kernel can re-map itself with
|
|
||||||
/// correct permissions (code R+X, rodata R, data R+W+NX). `flags` are raw ELF
|
|
||||||
/// segment flags: PF_X=1, PF_W=2, PF_R=4.
|
|
||||||
pub const KernelSegment = extern struct {
|
|
||||||
virt: u64,
|
|
||||||
pages: u64,
|
|
||||||
flags: u32,
|
|
||||||
_pad: u32 = 0,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Handoff structure the bootloader fills in and passes to the kernel's
|
|
||||||
/// `_start` in RDI (the first argument under the SysV AMD64 C ABI).
|
|
||||||
pub const BootInfo = extern struct {
|
|
||||||
framebuffer: Framebuffer,
|
|
||||||
memory_map: MemoryMap,
|
|
||||||
/// The kernel's own PT_LOAD segments (it has three: text, rodata, data).
|
|
||||||
kernel_segments: [8]KernelSegment,
|
|
||||||
kernel_segment_count: u32,
|
|
||||||
/// Physical address of the ACPI RSDP the firmware exposed, or 0 if none. The
|
|
||||||
/// kernel's device layer parses the ACPI tables from here to discover hardware.
|
|
||||||
/// A device-tree boot path leaves this 0 and (later) fills a `device_tree_blob`
|
|
||||||
/// field instead, so the kernel discovers devices without knowing what booted it.
|
|
||||||
acpi_rsdp: u64 = 0,
|
|
||||||
};
|
|
||||||
+192
@@ -0,0 +1,192 @@
|
|||||||
|
//! The **private kernel ↔ runtime** ABI: the raw system_call contract — the call
|
||||||
|
//! numbers, `mmap` protection flags, the page size those calls work in, and the IPC
|
||||||
|
//! name-registry ids and notification bit. Shared by the kernel dispatcher
|
||||||
|
//! (system/kernel/process.zig) and the user-space runtime library (library/runtime/),
|
||||||
|
//! so the two can never drift.
|
||||||
|
//!
|
||||||
|
//! **Application code does not speak this.** danos programs call the `runtime` library —
|
||||||
|
//! the stable, danos-native ABI — and the runtime is the one thing that issues the
|
||||||
|
//! actual system calls (POSIX code layers over the runtime, never on this directly). It
|
||||||
|
//! is the same split as libSystem on macOS or win32 over the NT syscalls: the numbers
|
||||||
|
//! here are an implementation detail the runtime hides and may renumber, not a public
|
||||||
|
//! interface. See docs/coding-standards.md and library/runtime/.
|
||||||
|
//!
|
||||||
|
//! This is the *core* contract; the device half — `DeviceDescriptor` and friends, which
|
||||||
|
//! also cross this boundary — lives with the device sub-project as [[device-abi]]
|
||||||
|
//! (system/devices/device-abi.zig). The loader↔kernel handoff is [[boot-handoff]].
|
||||||
|
|
||||||
|
/// Page size every `mmap`/`munmap` grant and the boot memory map are measured in.
|
||||||
|
/// 4 KiB on every architecture danos targets so far. Part of the ABI because the
|
||||||
|
/// runtime aligns to it (grants are page-granular) and the kernel guarantees it.
|
||||||
|
pub const page_size = 4096;
|
||||||
|
|
||||||
|
/// The kernel system_call numbers — the single source of truth shared by the kernel
|
||||||
|
/// dispatcher (system/kernel/process.zig) and the user runtime library, so the two
|
||||||
|
/// can never drift. The set is deliberately microkernel-minimal: file/device I/O
|
||||||
|
/// is not here — it lives in user-space servers reached through the IPC calls.
|
||||||
|
/// The table grows one milestone at a time; see docs/syscall.md.
|
||||||
|
pub const SystemCall = enum(u64) {
|
||||||
|
exit = 0, // exit(code): end the calling process
|
||||||
|
yield = 1, // yield(): give up the rest of this quantum
|
||||||
|
debug_write = 2, // debug_write(ptr, len): raw bytes to the kernel log (bring-up only)
|
||||||
|
sleep = 3, // sleep(ms): block the caller for ms milliseconds
|
||||||
|
mmap = 4, // mmap(len, prot) -> base: grant zeroed, page-aligned user pages
|
||||||
|
munmap = 5, // munmap(base, len): release pages from a prior mmap
|
||||||
|
create_ipc_endpoint = 6, // create_ipc_endpoint() -> handle: a new IPC endpoint
|
||||||
|
ipc_register = 7, // ipc_register(service_id, handle): publish an endpoint by well-known id
|
||||||
|
ipc_lookup = 8, // ipc_lookup(service_id) -> handle: find a published endpoint
|
||||||
|
ipc_call = 9, // ipc_call(h, message, len, reply, cap) -> reply_len: send + block for reply
|
||||||
|
ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, receive, cap) -> receive_len (+badge in rdx)
|
||||||
|
device_enumerate = 11, // device_enumerate(buffer, maximum) -> count: snapshot the device table
|
||||||
|
device_claim = 12, // device_claim(id) -> ok: take exclusive ownership of a device
|
||||||
|
mmio_map = 13, // mmio_map(id, resource_index) -> vaddr: map a claimed device's MMIO into this AS
|
||||||
|
irq_bind = 14, // irq_bind(id, resource_index, endpoint): deliver a device IRQ as an IPC notification
|
||||||
|
irq_ack = 15, // irq_ack(id, resource_index): re-arm a bound IRQ after servicing it
|
||||||
|
device_register = 16, // device_register(parent_id, descriptor) -> id: publish a child of a device you claimed
|
||||||
|
system_spawn = 17, // system_spawn(name_ptr, name_len, arguments_ptr, arguments_len, exit_endpoint) -> child process id: start a named initial-ramdisk binary as a new ring-3 process
|
||||||
|
dma_alloc = 18, // dma_alloc(len, flags) -> vaddr (rax), paddr (rdx): contiguous, pinned, uncacheable DMA memory
|
||||||
|
dma_free = 19, // dma_free(vaddr, len) -> 0: release a prior dma_alloc
|
||||||
|
msi_bind = 20, // msi_bind(device_id, endpoint) -> address (rax), data (rdx): a per-device MSI vector for a claimed device
|
||||||
|
io_read = 21, // io_read(device_id, resource_index, offset, width) -> value: read a port in a claimed device's io_port resource
|
||||||
|
io_write = 22, // io_write(device_id, resource_index, offset, width, value) -> 0: write a port in a claimed device's io_port resource
|
||||||
|
clock = 23, // clock() -> nanoseconds since boot: a monotonic time source (for timeouts/delays)
|
||||||
|
process_enumerate = 24, // process_enumerate(buffer, maximum) -> total: snapshot the task table
|
||||||
|
process_kill = 25, // process_kill(id) -> 0/-errno: end a process this process spawned
|
||||||
|
ipc_send = 26, // ipc_send(handle, message_ptr, message_len) -> 0/-errno: post a payload to an endpoint's async queue without blocking
|
||||||
|
process_exit_reason = 27, // process_exit_reason(id) -> ExitReason/-errno: how a dead child ended (its supervisor only)
|
||||||
|
process_subscribe = 28, // process_subscribe(endpoint) -> 0/-errno: subscribe to published exit events — every death posts a notification
|
||||||
|
signal_bind = 29, // signal_bind(endpoint) -> 0/-errno: nominate the endpoint this process's signals arrive on
|
||||||
|
process_signal = 30, // process_signal(id, signal) -> 0/-errno: post a signal to a child (or to yourself)
|
||||||
|
timer_bind = 31, // timer_bind(endpoint, ms) -> 0/-errno: one-shot timer — posts a notification when ms elapse
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// How a process ended — recorded by the kernel at death, queried by the
|
||||||
|
/// supervisor with `process_exit_reason`, and the input to its restart decision
|
||||||
|
/// (docs/process-lifecycle.md): a clean exit meant to stop, a fault wants a
|
||||||
|
/// restart with backoff, killed means the supervisor did it itself. The faults
|
||||||
|
/// mirror the CPU exceptions a ring-3 process can die of; they are exit reasons,
|
||||||
|
/// never delivered to the faulting process (recovery is restart, not a handler).
|
||||||
|
pub const ExitReason = enum(u8) {
|
||||||
|
exited = 0, // returned from main / called exit
|
||||||
|
aborted = 1, // deliberate self-termination (reserved: no abort path yet)
|
||||||
|
segmentation_fault = 2, // page fault
|
||||||
|
illegal_instruction = 3, // invalid opcode
|
||||||
|
arithmetic_fault = 4, // divide error, x87 or SIMD fault
|
||||||
|
protection_fault = 5, // general protection fault
|
||||||
|
fault = 6, // any other CPU exception
|
||||||
|
killed = 7, // process_kill
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The x86 MSI message address base (`0xFEE0_0000`): a device raises an MSI by writing
|
||||||
|
/// `data` to this address, which the Local APIC turns into an interrupt at the vector
|
||||||
|
/// in `data`. The kernel returns the concrete (address, data) from `msi_bind`; this is
|
||||||
|
/// the fixed prefix, exposed so a driver's config-space programming reads clearly.
|
||||||
|
pub const msi_address_base: u64 = 0xFEE0_0000;
|
||||||
|
|
||||||
|
/// `dma_alloc` flags. `coherent` (uncacheable) is the portable default; the others are
|
||||||
|
/// opt-in for specific hardware. `write_combining` needs PAT programming (not yet — it
|
||||||
|
/// currently falls back to coherent); see docs/driver-model.md (M14).
|
||||||
|
pub const dma_coherent: u64 = 1; // strong-uncacheable — the default, the only portable one
|
||||||
|
pub const dma_write_combining: u64 = 2; // write-combining (framebuffers); needs PAT
|
||||||
|
pub const dma_below_4g: u64 = 4; // physical address must fit 32 bits (legacy DMA engines)
|
||||||
|
|
||||||
|
/// Set in the badge returned by `ipc_reply_wait` when what arrived is an
|
||||||
|
/// **asynchronous notification** (a device interrupt bound with `irq_bind`, or a
|
||||||
|
/// child-exit notice — see `notify_exit_bit`) rather than a message from a client.
|
||||||
|
/// There is no payload and no reply owed; the low bits carry the source. Shared so
|
||||||
|
/// the kernel's ISR and the driver's event loop can't disagree about which bit
|
||||||
|
/// means "the hardware spoke".
|
||||||
|
pub const notify_badge_bit: u64 = 1 << 63;
|
||||||
|
|
||||||
|
/// Set (alongside `notify_badge_bit`) in the badge of a **child-exit notification**:
|
||||||
|
/// posted to the endpoint a supervisor passed to `system_spawn` when that child ends
|
||||||
|
/// — by clean exit, by a fault, or by `process_kill`. The low bits carry the child's
|
||||||
|
/// process id, so one endpoint can supervise many children (and even share with IRQ
|
||||||
|
/// notifications, which never set this bit). The microkernel's SIGCHLD.
|
||||||
|
pub const notify_exit_bit: u64 = 1 << 62;
|
||||||
|
|
||||||
|
/// Set (alongside `notify_badge_bit`) in the badge of a **buffered message** — a payload
|
||||||
|
/// posted to an endpoint's async queue by `ipc_send`, delivered through `ipc_reply_wait`
|
||||||
|
/// like a notification (no reply owed) but carrying bytes in the receive buffer, not just
|
||||||
|
/// a badge. This is what distinguishes a payload-bearing async message from a bare IRQ /
|
||||||
|
/// child-exit notification (which sets neither this nor `notify_exit_bit`). The low bits
|
||||||
|
/// carry the sender's task id. The async counterpart of the synchronous `ipc_call`, for
|
||||||
|
/// broadcasts where a rendezvous is the wrong shape (the input service is the first user).
|
||||||
|
pub const notify_message_bit: u64 = 1 << 61;
|
||||||
|
|
||||||
|
/// Set (alongside `notify_badge_bit`) in the badge of a **signal notification** —
|
||||||
|
/// the process-lifecycle vocabulary of docs/process-lifecycle.md, delivered to the
|
||||||
|
/// endpoint the process nominated with `signal_bind`. The low bits carry the
|
||||||
|
/// coalesced pending mask (bit positions = `Signal` values): signals are
|
||||||
|
/// statements, not questions, and two pending terminates are one terminate.
|
||||||
|
pub const notify_signal_bit: u64 = 1 << 60;
|
||||||
|
|
||||||
|
/// Set (alongside `notify_badge_bit`) in the badge of a **timer notification** —
|
||||||
|
/// a one-shot `timer_bind` deadline landing. No payload bits: what to do when the
|
||||||
|
/// deadline fires is whatever the receiver armed it for (a stop-sequence
|
||||||
|
/// escalation, a restart backoff, an alarm).
|
||||||
|
pub const notify_timer_bit: u64 = 1 << 59;
|
||||||
|
|
||||||
|
/// The signal vocabulary (docs/process-lifecycle.md): POSIX's concepts, danos's
|
||||||
|
/// names, message delivery. The value is the bit position in the pending mask — a
|
||||||
|
/// private kernel/runtime detail, free to change while they ship together. Kill
|
||||||
|
/// is not here (it is `process_kill`, unhandleable by definition); faults are not
|
||||||
|
/// here (they are `ExitReason`s — recovery is restart, not a handler); liveness is
|
||||||
|
/// not here (a question, asked as the zero-length ping call, not a statement).
|
||||||
|
pub const Signal = enum(u5) {
|
||||||
|
terminate = 0, // finish up and exit (the polite half of the stop sequence)
|
||||||
|
reload = 1, // re-read configuration / re-scan
|
||||||
|
interrupt = 2, // interactive interrupt (no sender until a console exists)
|
||||||
|
quit = 3, // as interrupt, by convention more final
|
||||||
|
alarm = 4, // a timer the process armed for itself (unbuilt: no consumer yet)
|
||||||
|
user_1 = 5, // service-defined
|
||||||
|
user_2 = 6, // service-defined
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Capacity of `ProcessDescriptor.name` — matches the longest name `system_spawn`
|
||||||
|
/// accepts, so a process's recorded name (its argv[0]) is never truncated.
|
||||||
|
pub const maximum_process_name = 64;
|
||||||
|
|
||||||
|
/// What a process is doing right now, as reported by `process_enumerate`. Crosses
|
||||||
|
/// the system_call boundary as `ProcessDescriptor.state`.
|
||||||
|
pub const ProcessState = enum(u32) {
|
||||||
|
ready = 0, // runnable, waiting for a core
|
||||||
|
running = 1, // executing on a core right now
|
||||||
|
blocked = 2, // waiting (sleeping, or blocked in IPC)
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One `process_enumerate` entry — the kernel's view of a live task, kernel tasks
|
||||||
|
/// included (they carry an empty name and id 0 is the boot task). Fixed layout
|
||||||
|
/// (extern) because it crosses the kernel↔user boundary by memory copy, like
|
||||||
|
/// `DeviceDescriptor` in the device ABI.
|
||||||
|
pub const ProcessDescriptor = extern struct {
|
||||||
|
id: u32, // kernel-assigned process id; never reused (monotonic)
|
||||||
|
supervisor: u32, // id of the process that spawned it (0 = the kernel)
|
||||||
|
state: u32, // a ProcessState value
|
||||||
|
priority: u32,
|
||||||
|
name_length: u32,
|
||||||
|
name: [maximum_process_name]u8, // argv[0] at spawn; empty for kernel tasks
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Well-known IPC service ids for the bootstrap name registry (create_ipc_endpoint +
|
||||||
|
/// ipc_register/ipc_lookup). Small integers, so no string interning is needed
|
||||||
|
/// during bring-up. The VFS server registers under `vfs`; clients look it up.
|
||||||
|
pub const ServiceId = enum(u32) {
|
||||||
|
vfs = 1,
|
||||||
|
input = 2,
|
||||||
|
ps2_bus = 3, // the 8042 owner; child device drivers attach here for raw bytes
|
||||||
|
device_manager = 4, // the tree, the matcher, the supervisor (docs/device-manager.md)
|
||||||
|
power = 5, // system power: events (button, lid, battery) + shutdown (docs/m21-plan.md; domain-named per decision 7 — the acpi service registers it on x86, a PSCI service will on ARM)
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Protection flags for `mmap` (matching the usual C bit values).
|
||||||
|
pub const prot_read: u64 = 1;
|
||||||
|
pub const prot_write: u64 = 2;
|
||||||
|
pub const prot_exec: u64 = 4;
|
||||||
|
|
||||||
|
/// `send_cap` / `received_cap` sentinel meaning "no capability" on the `ipc_call` /
|
||||||
|
/// `ipc_reply_wait` cap-passing path (M13). `~0`, like `no_parent` — a real handle is
|
||||||
|
/// a small index, so it can never collide.
|
||||||
|
pub const no_cap: u64 = ~@as(u64, 0);
|
||||||
@@ -0,0 +1,156 @@
|
|||||||
|
//! The **loader ↔ kernel** contract: everything a bootloader (boot/, e.g. efi.zig
|
||||||
|
//! built as BOOTX64.efi) and the kernel (system/kernel/kernel.zig) must agree on to
|
||||||
|
//! hand control over — the handoff structures the loader fills in, plus the kernel's
|
||||||
|
//! virtual-memory layout and the physical↔virtual addressing both sides use.
|
||||||
|
//!
|
||||||
|
//! Both binaries import this as the `boot-handoff` module, so the layout is defined
|
||||||
|
//! in exactly one place. **User space never sees this** — the kernel↔user contract is
|
||||||
|
//! [[abi]] (system/abi.zig); device types are [[device-abi]] (system/devices/device-abi.zig).
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
/// Calling convention for the bootloader→kernel jump. Pinned to SystemV so it does
|
||||||
|
/// not depend on each binary's target default: the UEFI bootloader's C
|
||||||
|
/// convention is Microsoft x64 (first arg in RCX), the freestanding kernel's is
|
||||||
|
/// SystemV (first arg in RDI). Both reference this to agree on where `*BootInformation`
|
||||||
|
/// is passed.
|
||||||
|
pub const kernel_abi: std.builtin.CallingConvention = .{ .x86_64_sysv = .{} };
|
||||||
|
|
||||||
|
/// Pixel byte order of the linear framebuffer the firmware handed us.
|
||||||
|
pub const PixelFormat = enum(u32) {
|
||||||
|
/// Byte 0 = Red, 1 = Green, 2 = Blue, 3 = reserved.
|
||||||
|
rgbx,
|
||||||
|
/// Byte 0 = Blue, 1 = Green, 2 = Red, 3 = reserved.
|
||||||
|
bgrx,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A linear framebuffer: `width`x`height` pixels, each a 32-bit value, with
|
||||||
|
/// `pitch` bytes between the start of one row and the next (which may be larger
|
||||||
|
/// than `width * 4` due to hardware padding).
|
||||||
|
///
|
||||||
|
/// A `base` of 0 means **no framebuffer** — the firmware exposed no Graphics
|
||||||
|
/// Output Protocol (a headless server, say). The kernel must treat on-screen
|
||||||
|
/// output as optional and never assume a framebuffer exists.
|
||||||
|
pub const Framebuffer = extern struct {
|
||||||
|
base: usize, // the memory address where pixel data starts (0 = none)
|
||||||
|
width: u32, // visible pixels per row (e.g. 1920)
|
||||||
|
height: u32, // visible rows (e.g. 1080)
|
||||||
|
pitch: u32, // bytes from the start of one row to the start of the next
|
||||||
|
format: PixelFormat,
|
||||||
|
|
||||||
|
/// Whether a usable framebuffer was handed over.
|
||||||
|
pub fn present(self: Framebuffer) bool {
|
||||||
|
return self.base != 0 and self.width != 0 and self.height != 0;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The kernel's virtual-memory layout (higher-half). The kernel is linked at
|
||||||
|
/// `kernel_virt_base` but loaded at a low physical address; all of RAM (and the
|
||||||
|
/// device MMIO windows) is also mapped at `physmap_base + physical`, so the kernel
|
||||||
|
/// can reach any physical address by adding a constant. The low half is left
|
||||||
|
/// entirely to user space.
|
||||||
|
///
|
||||||
|
/// user image + stack : 0x0000_7000_0000_0000 (PML4[224], low half)
|
||||||
|
/// kernel heap : 0xFFFF_8000_0000_0000 (PML4[256])
|
||||||
|
/// physmap : 0xFFFF_8800_0000_0000 (PML4[272]) + physical
|
||||||
|
/// kernel image : 0xFFFF_FFFF_8000_0000 (PML4[511])
|
||||||
|
pub const physmap_base: u64 = 0xFFFF_8800_0000_0000;
|
||||||
|
pub const kernel_virt_base: u64 = 0xFFFF_FFFF_8000_0000;
|
||||||
|
|
||||||
|
/// Physical address -> its virtual address in the physmap. The single way the
|
||||||
|
/// kernel dereferences a physical address once paging is up.
|
||||||
|
///
|
||||||
|
/// **Hazard:** valid only once the (bootstrap or final) page tables are live.
|
||||||
|
/// The bootloader may use the *constant* `physmap_base` to build those tables,
|
||||||
|
/// but must not call this to dereference memory before its own CR3 is loaded —
|
||||||
|
/// it runs under the firmware's identity map, where these addresses are unmapped.
|
||||||
|
pub inline fn physicalToVirtual(physical: u64) u64 {
|
||||||
|
return physical + physmap_base;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Physmap virtual address -> physical. Inverse of `physicalToVirtual`; for producing
|
||||||
|
/// the physical address of something the kernel holds a physmap pointer to
|
||||||
|
/// (e.g. a page-table frame for CR3, a post-mortem breadcrumb's RAM location).
|
||||||
|
pub inline fn virtualToPhysical(virtual: u64) u64 {
|
||||||
|
return virtual - physmap_base;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// danos's own classification of a span of physical memory — deliberately not
|
||||||
|
/// UEFI's vocabulary. Each boot path (UEFI now, device tree later) translates its
|
||||||
|
/// native memory description into these kinds, so the kernel never learns what
|
||||||
|
/// booted it. [[architecture]] keeps the same discipline for CPU code.
|
||||||
|
pub const MemoryKind = enum(u32) {
|
||||||
|
/// Free RAM the kernel may allocate. Each boot path folds its own transient
|
||||||
|
/// memory into this once it's genuinely free (e.g. the UEFI loader classifies
|
||||||
|
/// boot-services memory as usable after ExitBootServices), so the kernel never
|
||||||
|
/// has to know about boot-protocol-specific "reclaimable" states.
|
||||||
|
usable,
|
||||||
|
/// Firmware, MMIO, the kernel image, our own boot buffers, the boot stack —
|
||||||
|
/// never hand out.
|
||||||
|
reserved,
|
||||||
|
/// ACPI tables: parse, then reclaim.
|
||||||
|
acpi_tables,
|
||||||
|
/// ACPI non-volatile storage: preserve across sleep, do not allocate.
|
||||||
|
acpi_nvs,
|
||||||
|
/// Not backed by RAM: memory-mapped device registers or a reserved
|
||||||
|
/// address-space window (e.g. PCIe configuration space). Kept distinct from
|
||||||
|
/// `reserved` so RAM accounting doesn't count device address space.
|
||||||
|
mmio,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One contiguous span of physical memory. Because danos defines this layout
|
||||||
|
/// itself (unlike the UEFI descriptor it's built from), `@sizeOf` is
|
||||||
|
/// authoritative — the kernel walks a plain `[]MemoryRegion`, with none of the
|
||||||
|
/// firmware's variable descriptor-stride to worry about.
|
||||||
|
pub const MemoryRegion = extern struct {
|
||||||
|
base: u64, // physical start address
|
||||||
|
pages: u64, // length in 4 KiB pages (the [[abi]] `page_size` unit)
|
||||||
|
kind: MemoryKind,
|
||||||
|
_pad: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The physical memory layout handed to the kernel: a pointer to an array of
|
||||||
|
/// `len` `MemoryRegion`s, in a buffer that outlives the loader.
|
||||||
|
pub const MemoryMap = extern struct {
|
||||||
|
regions: usize, // address of a `[len]MemoryRegion`
|
||||||
|
len: usize,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One PT_LOAD segment of the kernel image, so the kernel can re-map itself with
|
||||||
|
/// correct permissions (code R+X, rodata R, data R+W+NX). `flags` are raw ELF
|
||||||
|
/// segment flags: PF_X=1, PF_W=2, PF_R=4. `virtual` is the higher-half link address;
|
||||||
|
/// `physical` is where the loader actually placed the segment (they differ once the
|
||||||
|
/// kernel links high — the loader records the real load address here).
|
||||||
|
pub const KernelSegment = extern struct {
|
||||||
|
virtual: u64,
|
||||||
|
physical: u64,
|
||||||
|
pages: u64,
|
||||||
|
flags: u32,
|
||||||
|
_pad: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Handoff structure the bootloader fills in and passes to the kernel's
|
||||||
|
/// `_start` in RDI (the first argument under the SystemV AMD64 C ABI).
|
||||||
|
pub const BootInformation = extern struct {
|
||||||
|
framebuffer: Framebuffer,
|
||||||
|
memory_map: MemoryMap,
|
||||||
|
/// The kernel's own PT_LOAD segments (it has three: text, rodata, data).
|
||||||
|
kernel_segments: [8]KernelSegment,
|
||||||
|
kernel_segment_count: u32,
|
||||||
|
/// Physical address of the ACPI RSDP the firmware exposed, or 0 if none. The
|
||||||
|
/// kernel's device layer parses the ACPI tables from here to discover hardware.
|
||||||
|
/// A device-tree boot path leaves this 0 and (later) fills a `device_tree_blob`
|
||||||
|
/// field instead, so the kernel discovers devices without knowing what booted it.
|
||||||
|
acpi_rsdp: u64 = 0,
|
||||||
|
/// The raw `/system/services/init` ELF image, read off the boot volume by the loader
|
||||||
|
/// into memory that survives the handoff (classified reserved, so the kernel
|
||||||
|
/// identity-maps it and never allocates over it). 0/0 = no init found — the
|
||||||
|
/// kernel boots without user space. Grows into a full initial_ramdisk handoff later.
|
||||||
|
init_base: u64 = 0,
|
||||||
|
init_len: u64 = 0,
|
||||||
|
/// The initial_ramdisk image (a bundle of extra user binaries — the VFS server and
|
||||||
|
/// device drivers), read off the boot volume into memory that survives the
|
||||||
|
/// handoff, same as `init` above. 0/0 = no initial_ramdisk. See system/initial-ramdisk.zig.
|
||||||
|
initial_ramdisk_base: u64 = 0,
|
||||||
|
initial_ramdisk_len: u64 = 0,
|
||||||
|
};
|
||||||
@@ -0,0 +1,138 @@
|
|||||||
|
//! ACPI / PnP hardware-ID (`_HID`) names: the flat analog of pci-class.zig for
|
||||||
|
//! `acpi_device` nodes. Unlike PCI, ACPI has no class/subclass/prog-IF taxonomy — a
|
||||||
|
//! device's identity *is* its `_HID` string (`PNP0303` simply means "PS/2 keyboard"),
|
||||||
|
//! so this is a plain id <-> name registry rather than a hierarchical decoder.
|
||||||
|
//! The well-known PnP/ACPI IDs; vendor-specific ids (e.g. `QEMU0002`, `INTC1234`) have
|
||||||
|
//! no standard name and decode to nothing. Pure reference data, so it is shared by
|
||||||
|
//! kernel discovery (the device-tree dump) and any user-space driver or tool.
|
||||||
|
//!
|
||||||
|
//! Code that means a specific device names the `HardwareId` variant instead of its
|
||||||
|
//! `_HID` string — `HardwareId.ps2_keyboard.hid()` reads without a registry lookup,
|
||||||
|
//! where a bare `"PNP0303"` does not.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
/// The common standard PnP/ACPI hardware IDs, as named values. Prefix ranges hint at
|
||||||
|
/// the grouping (PNP03xx keyboards, PNP0Fxx pointing devices, PNP0Cxx ACPI
|
||||||
|
/// power/thermal, PNP0Axx buses), but there is no formal hierarchy — hence a flat
|
||||||
|
/// enum over a flat registry.
|
||||||
|
pub const HardwareId = enum {
|
||||||
|
programmable_interrupt_controller,
|
||||||
|
system_timer,
|
||||||
|
high_precision_event_timer,
|
||||||
|
dma_controller,
|
||||||
|
ps2_keyboard,
|
||||||
|
parallel_port,
|
||||||
|
ecp_parallel_port,
|
||||||
|
serial_port,
|
||||||
|
floppy_disk_controller,
|
||||||
|
system_speaker,
|
||||||
|
pci_bus,
|
||||||
|
generic_container,
|
||||||
|
/// The second id the ACPI spec assigns the same "Generic Container Device" name.
|
||||||
|
generic_container_extended,
|
||||||
|
pci_express_root_bridge,
|
||||||
|
real_time_clock,
|
||||||
|
system_board,
|
||||||
|
motherboard_reserved_resources,
|
||||||
|
math_coprocessor,
|
||||||
|
acpi_system_board,
|
||||||
|
embedded_controller,
|
||||||
|
control_method_battery,
|
||||||
|
fan,
|
||||||
|
power_button,
|
||||||
|
lid,
|
||||||
|
sleep_button,
|
||||||
|
pci_interrupt_link,
|
||||||
|
microsoft_ps2_mouse,
|
||||||
|
ps2_mouse,
|
||||||
|
ac_adapter,
|
||||||
|
processor_device,
|
||||||
|
processor_aggregator,
|
||||||
|
processor_container,
|
||||||
|
|
||||||
|
const Entry = struct { hid: []const u8, name: []const u8 };
|
||||||
|
|
||||||
|
/// The registry row for this id: its `_HID` string and human-readable name.
|
||||||
|
fn entry(self: HardwareId) Entry {
|
||||||
|
return switch (self) {
|
||||||
|
.programmable_interrupt_controller => .{ .hid = "PNP0000", .name = "Programmable Interrupt Controller (PIC)" },
|
||||||
|
.system_timer => .{ .hid = "PNP0100", .name = "System Timer (PIT)" },
|
||||||
|
.high_precision_event_timer => .{ .hid = "PNP0103", .name = "High Precision Event Timer (HPET)" },
|
||||||
|
.dma_controller => .{ .hid = "PNP0200", .name = "DMA Controller" },
|
||||||
|
.ps2_keyboard => .{ .hid = "PNP0303", .name = "PS/2 Keyboard" },
|
||||||
|
.parallel_port => .{ .hid = "PNP0400", .name = "Standard LPT Parallel Port" },
|
||||||
|
.ecp_parallel_port => .{ .hid = "PNP0401", .name = "ECP Parallel Port" },
|
||||||
|
.serial_port => .{ .hid = "PNP0501", .name = "16550A-compatible Serial Port" },
|
||||||
|
.floppy_disk_controller => .{ .hid = "PNP0700", .name = "PC Floppy Disk Controller" },
|
||||||
|
.system_speaker => .{ .hid = "PNP0800", .name = "System Speaker" },
|
||||||
|
.pci_bus => .{ .hid = "PNP0A03", .name = "PCI Bus" },
|
||||||
|
.generic_container => .{ .hid = "PNP0A05", .name = "Generic Container Device" },
|
||||||
|
.generic_container_extended => .{ .hid = "PNP0A06", .name = "Generic Container Device" },
|
||||||
|
.pci_express_root_bridge => .{ .hid = "PNP0A08", .name = "PCI Express Root Bridge" },
|
||||||
|
.real_time_clock => .{ .hid = "PNP0B00", .name = "Real-Time Clock (RTC)" },
|
||||||
|
.system_board => .{ .hid = "PNP0C01", .name = "System Board" },
|
||||||
|
.motherboard_reserved_resources => .{ .hid = "PNP0C02", .name = "Motherboard Reserved Resources" },
|
||||||
|
.math_coprocessor => .{ .hid = "PNP0C04", .name = "Math Coprocessor" },
|
||||||
|
.acpi_system_board => .{ .hid = "PNP0C08", .name = "ACPI System Board" },
|
||||||
|
.embedded_controller => .{ .hid = "PNP0C09", .name = "ACPI Embedded Controller" },
|
||||||
|
.control_method_battery => .{ .hid = "PNP0C0A", .name = "ACPI Control Method Battery" },
|
||||||
|
.fan => .{ .hid = "PNP0C0B", .name = "ACPI Fan" },
|
||||||
|
.power_button => .{ .hid = "PNP0C0C", .name = "ACPI Power Button" },
|
||||||
|
.lid => .{ .hid = "PNP0C0D", .name = "ACPI Lid" },
|
||||||
|
.sleep_button => .{ .hid = "PNP0C0E", .name = "ACPI Sleep Button" },
|
||||||
|
.pci_interrupt_link => .{ .hid = "PNP0C0F", .name = "PCI Interrupt Link Device" },
|
||||||
|
.microsoft_ps2_mouse => .{ .hid = "PNP0F03", .name = "Microsoft PS/2 Mouse" },
|
||||||
|
.ps2_mouse => .{ .hid = "PNP0F13", .name = "PS/2 Mouse" },
|
||||||
|
.ac_adapter => .{ .hid = "ACPI0003", .name = "AC Adapter" },
|
||||||
|
.processor_device => .{ .hid = "ACPI0007", .name = "Processor Device" },
|
||||||
|
.processor_aggregator => .{ .hid = "ACPI000C", .name = "Processor Aggregator" },
|
||||||
|
.processor_container => .{ .hid = "ACPI0010", .name = "Processor Container" },
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// This id's `_HID` string (e.g. `.ps2_keyboard` -> "PNP0303").
|
||||||
|
pub fn hid(self: HardwareId) []const u8 {
|
||||||
|
return self.entry().hid;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// This id's human-readable name (e.g. `.ps2_keyboard` -> "PS/2 Keyboard").
|
||||||
|
pub fn description(self: HardwareId) []const u8 {
|
||||||
|
return self.entry().name;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The named value for a `_HID` string, or null if it is not a known standard
|
||||||
|
/// id (vendor-specific ids are not in the registry).
|
||||||
|
pub fn fromHid(hid_string: []const u8) ?HardwareId {
|
||||||
|
for (std.enums.values(HardwareId)) |id| {
|
||||||
|
if (std.mem.eql(u8, id.hid(), hid_string)) return id;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The human-readable name for a `_HID` string, or "" if it is not a known standard
|
||||||
|
/// id (vendor-specific ids have no registry name — callers just print the raw HID).
|
||||||
|
pub fn description(hid: []const u8) []const u8 {
|
||||||
|
return (HardwareId.fromHid(hid) orelse return "").description();
|
||||||
|
}
|
||||||
|
|
||||||
|
test "decodes standard PnP/ACPI ids and leaves the rest alone" {
|
||||||
|
const eq = std.testing.expectEqualStrings;
|
||||||
|
try eq("PS/2 Keyboard", description("PNP0303"));
|
||||||
|
try eq("PS/2 Mouse", description("PNP0F13"));
|
||||||
|
try eq("PCI Express Root Bridge", description("PNP0A08"));
|
||||||
|
try eq("Real-Time Clock (RTC)", description("PNP0B00"));
|
||||||
|
try eq("", description("QEMU0002")); // vendor-specific: no standard name
|
||||||
|
try eq("", description("")); // no HID at all
|
||||||
|
}
|
||||||
|
|
||||||
|
test "named values round-trip through their _HID strings" {
|
||||||
|
const testing = std.testing;
|
||||||
|
try testing.expectEqualStrings("PNP0303", HardwareId.ps2_keyboard.hid());
|
||||||
|
try testing.expectEqual(@as(?HardwareId, .ps2_mouse), HardwareId.fromHid("PNP0F13"));
|
||||||
|
try testing.expectEqual(@as(?HardwareId, null), HardwareId.fromHid("QEMU0002"));
|
||||||
|
for (std.enums.values(HardwareId)) |id| {
|
||||||
|
try testing.expectEqual(@as(?HardwareId, id), HardwareId.fromHid(id.hid()));
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,875 @@
|
|||||||
|
//! ACPI discovery backend.
|
||||||
|
//!
|
||||||
|
//! Walks the ACPI tables the firmware left in memory (starting from the RSDP the
|
||||||
|
//! bootloader handed us) and translates the static tables into the generic
|
||||||
|
//! `device` model, so the kernel enumerates hardware without knowing ACPI is the
|
||||||
|
//! source. This is deliberately the *static-table* path: MADT (CPUs / interrupt
|
||||||
|
//! controllers), MCFG (PCIe ECAM -> PCI enumeration), HPET (timer), and FADT
|
||||||
|
//! (power register map). The DSDT/SSDT bytecode is handed to the `aml` submodule
|
||||||
|
//! only to extract the sleep-state (`_Sx`) values for power management; full AML namespace
|
||||||
|
//! interpretation is a separate, larger subproject.
|
||||||
|
//!
|
||||||
|
//! ACPI tables live in `.acpi_tables` / `.acpi_nvs` memory, which the kernel
|
||||||
|
//! identity-maps, so table addresses are dereferenced directly. PCIe ECAM is MMIO
|
||||||
|
//! and is *not* mapped up front, so configuration-space pages are mapped on demand via
|
||||||
|
//! the `Hal.mapMmio` callback the caller supplies (the architecture VMM's map primitive).
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const boot_handoff = @import("boot-handoff");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const parameters = @import("parameters");
|
||||||
|
const device_model = @import("device-model.zig");
|
||||||
|
const aml = @import("aml/aml.zig");
|
||||||
|
const DeviceTree = device_model.DeviceTree;
|
||||||
|
const Hal = device_model.Hal;
|
||||||
|
|
||||||
|
/// A hardware register located either in MMIO or I/O-port space, as ACPI's
|
||||||
|
/// Generic Address Structure describes. `address == 0` means "not present".
|
||||||
|
pub const RegisterAccess = struct {
|
||||||
|
/// true = system memory (MMIO), false = system I/O port space.
|
||||||
|
mmio: bool = false,
|
||||||
|
address: u64 = 0,
|
||||||
|
/// Access width in bytes.
|
||||||
|
width: u8 = 0,
|
||||||
|
|
||||||
|
pub fn present(self: RegisterAccess) bool {
|
||||||
|
return self.address != 0;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Everything the power subsystem needs, extracted from the FADT and the AML
|
||||||
|
/// sleep packages during discovery. Populated by `discover`, read by `power`.
|
||||||
|
pub const PowerInformation = struct {
|
||||||
|
/// The System Control Interrupt's GSI (FADT SCI_INT) — the line ACPI events
|
||||||
|
/// (power button, GPEs) arrive on. Published to the acpi service for M21.
|
||||||
|
sci_interrupt: u16 = 0,
|
||||||
|
/// The SMM command port and the value that switches the platform into ACPI mode.
|
||||||
|
smi_cmd: u16 = 0,
|
||||||
|
acpi_enable: u8 = 0,
|
||||||
|
acpi_disable: u8 = 0,
|
||||||
|
/// PM1 control registers — writing SLP_TYP|SLP_EN here enters a sleep state.
|
||||||
|
pm1a_cnt: RegisterAccess = .{},
|
||||||
|
pm1b_cnt: RegisterAccess = .{},
|
||||||
|
/// The FADT reset register and the value to write to it.
|
||||||
|
reset: RegisterAccess = .{},
|
||||||
|
reset_value: u8 = 0,
|
||||||
|
reset_supported: bool = false,
|
||||||
|
/// SLP_TYP values for S5 (soft off) and S3 (suspend), from the AML sleep-state (`_Sx`)
|
||||||
|
/// packages.
|
||||||
|
s5: ?aml.SleepType = null,
|
||||||
|
s3: ?aml.SleepType = null,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Filled in by `discover`; the power service reads it to reboot/shutdown.
|
||||||
|
pub var power_information: PowerInformation = .{};
|
||||||
|
|
||||||
|
/// A legacy ISA IRQ remapped to a different global system interrupt (GSI), from a
|
||||||
|
/// MADT Interrupt Source Override. `flags` are the MPS INTI polarity/trigger bits.
|
||||||
|
pub const IsoEntry = struct {
|
||||||
|
source: u8,
|
||||||
|
gsi: u32,
|
||||||
|
flags: u16,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Firmware facts the architecture layer needs to avoid legacy assumptions (so danos boots
|
||||||
|
/// on legacy-free UEFI Class 3 machines). MMIO device *addresses* (HPET, IOAPIC)
|
||||||
|
/// come from the device tree instead; this holds the scalar facts that have no
|
||||||
|
/// natural device node.
|
||||||
|
pub const PlatformInformation = struct {
|
||||||
|
/// Whether the legacy 8259 PIC is present (MADT flags bit 0, PCAT_COMPAT). When
|
||||||
|
/// false, the PIC must not be programmed (it may not exist).
|
||||||
|
pic_present: bool = false,
|
||||||
|
/// Local APIC MMIO base (MADT, honouring a type-5 address override).
|
||||||
|
lapic_base: u64 = 0xFEE00000,
|
||||||
|
/// The ACPI power-management timer — a fixed 3.579545 MHz counter usable as a
|
||||||
|
/// calibration reference when no HPET is present.
|
||||||
|
pm_timer: RegisterAccess = .{},
|
||||||
|
/// true = 32-bit PM timer counter, false = 24-bit (FADT flag TMR_VALUE_EXT).
|
||||||
|
pm_timer_32bit: bool = false,
|
||||||
|
/// The console UART the firmware points at (SPCR), if any — MMIO or I/O port.
|
||||||
|
spcr_uart: ?RegisterAccess = null,
|
||||||
|
/// SPCR interface type (0/1 = 16550/16450, …).
|
||||||
|
spcr_kind: u8 = 0,
|
||||||
|
/// ISA-IRQ-to-GSI remappings from the MADT (for future IOAPIC routing).
|
||||||
|
overrides: [16]IsoEntry = undefined,
|
||||||
|
override_count: usize = 0,
|
||||||
|
/// Whether an IOMMU (VT-d DMA-remapping unit) was found in the ACPI DMAR table.
|
||||||
|
/// When false, `device_claim` on a DMA-capable device is equivalent to granting
|
||||||
|
/// ring 0 — a device can DMA to any physical address (docs/driver-model.md M16).
|
||||||
|
/// Detection is the first step; per-device domain enforcement lands with the first
|
||||||
|
/// DMA driver.
|
||||||
|
iommu_present: bool = false,
|
||||||
|
/// MMIO base of the first DMA-remapping hardware unit (DMAR DRHD), when present.
|
||||||
|
iommu_base: u64 = 0,
|
||||||
|
/// The unit's Version register (offset 0x00) — its low byte is major.minor;
|
||||||
|
/// reading it back nonzero confirms a real, mappable VT-d unit.
|
||||||
|
iommu_version: u32 = 0,
|
||||||
|
/// The unit's Capability register (offset 0x08): supported address widths, number
|
||||||
|
/// of domains, etc. Recorded now; consumed when enforcement is built.
|
||||||
|
iommu_capabilities: u64 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Filled in by `discover`; the architecture layer reads it during bring-up.
|
||||||
|
pub var platform_information: PlatformInformation = .{};
|
||||||
|
|
||||||
|
/// One usable logical processor, from a MADT type-0 (Local APIC) record. The
|
||||||
|
/// `apic_id` is the Local APIC ID that SMP bring-up targets to wake this core
|
||||||
|
/// (INIT–SIPI–SIPI); `processor_id` is the ACPI namespace handle. Only processors
|
||||||
|
/// the firmware marks *enabled* are recorded — a disabled one can't be started.
|
||||||
|
pub const Cpu = struct {
|
||||||
|
processor_id: u8,
|
||||||
|
apic_id: u8,
|
||||||
|
/// MADT flags bit 1: usable but firmware-started offline (hot-plug / deferred
|
||||||
|
/// bring-up), as opposed to already available. Informational for now.
|
||||||
|
online_capable: bool,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The set of usable logical processors the MADT listed — the hardware's degree of
|
||||||
|
/// parallelism. Includes the bootstrap processor danos already runs on; the rest
|
||||||
|
/// are the application processors SMP bring-up would start (see docs/smp.md).
|
||||||
|
pub const CpuInformation = struct {
|
||||||
|
/// A static pool sized well above any danos target (a desktop, two 4-core Pis).
|
||||||
|
/// If the MADT ever lists more, the surplus is dropped and counted in `dropped`
|
||||||
|
/// so the truncation is never silent.
|
||||||
|
cpus: [maximum_cpus]Cpu = undefined,
|
||||||
|
count: usize = 0,
|
||||||
|
dropped: usize = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
const maximum_cpus = parameters.maximum_cpus;
|
||||||
|
|
||||||
|
/// Filled in by `discover` (from the MADT); SMP bring-up reads it to wake the APs.
|
||||||
|
pub var cpu_information: CpuInformation = .{};
|
||||||
|
|
||||||
|
/// Integrity/diagnostics for the AML parse. `consumed == total` means the parser
|
||||||
|
/// walked every byte of the DSDT/SSDTs without desyncing.
|
||||||
|
pub const AmlStats = struct {
|
||||||
|
nodes: usize = 0,
|
||||||
|
consumed: usize = 0,
|
||||||
|
total: usize = 0,
|
||||||
|
};
|
||||||
|
pub var aml_stats: AmlStats = .{};
|
||||||
|
|
||||||
|
/// The ACPI namespace built from the DSDT/SSDTs, kept for sleep-state (`_Sx`) lookup now and
|
||||||
|
/// device enumeration later. Null until `discover` runs successfully.
|
||||||
|
pub var namespace: ?aml.Namespace = null;
|
||||||
|
|
||||||
|
/// Physical address of the DSDT the FADT points at, or 0.
|
||||||
|
pub var dsdt_physical: u64 = 0;
|
||||||
|
|
||||||
|
/// The FADT itself (physical + length), published on the acpi-tables node so
|
||||||
|
/// the ring-3 acpi service can read the PM1 event and GPE blocks it needs for
|
||||||
|
/// the event side (docs/m21-plan.md decision 3). Distinguished from the AML
|
||||||
|
/// blob resources by its intact "FACP" header — the blobs are header-stripped.
|
||||||
|
var fadt_physical: u64 = 0;
|
||||||
|
var fadt_length: u64 = 0;
|
||||||
|
|
||||||
|
// AML blocks (DSDT + any SSDTs) collected during the table walk, as physical
|
||||||
|
// address + length of each table's post-header bytecode. Scanned after the walk
|
||||||
|
// for the sleep-state (`_Sx`) packages.
|
||||||
|
var aml_block_physical: [32]u64 = undefined;
|
||||||
|
var aml_block_len: [32]usize = undefined;
|
||||||
|
var aml_block_count: usize = 0;
|
||||||
|
|
||||||
|
fn addAmlBlock(sdt_physical: u64) void {
|
||||||
|
if (aml_block_count >= aml_block_physical.len or sdt_physical == 0) return;
|
||||||
|
const h: *const SystemDescriptorTableHeader = @ptrFromInt(boot_handoff.physicalToVirtual(sdt_physical));
|
||||||
|
if (h.length <= @sizeOf(SystemDescriptorTableHeader)) return;
|
||||||
|
aml_block_physical[aml_block_count] = sdt_physical + @sizeOf(SystemDescriptorTableHeader);
|
||||||
|
aml_block_len[aml_block_count] = h.length - @sizeOf(SystemDescriptorTableHeader);
|
||||||
|
aml_block_count += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// RSDP structure for revision 0 (version 1.0)
|
||||||
|
const RootSystemDescriptionPointer = extern struct {
|
||||||
|
/// An 8 byte magic number used for locating the RSDP, containing RSD PTR.
|
||||||
|
signature: [8]u8,
|
||||||
|
/// A byte used to verify the first 20 bytes of the RSDP
|
||||||
|
checksum: u8,
|
||||||
|
/// An OEM-supplied string that identified the OEM.
|
||||||
|
oem_id: [6]u8,
|
||||||
|
/// The RSDP revision, used for determining which fields are available.
|
||||||
|
revision: u8,
|
||||||
|
/// A 32-bit physical address pointing to the RSDT.
|
||||||
|
root_system_description_table_address: u32 align(1),
|
||||||
|
};
|
||||||
|
|
||||||
|
/// XSDP structure for revision 2 (version 2.0+)
|
||||||
|
const ExtendedSystemDescriptorPointer = extern struct {
|
||||||
|
/// An 8 byte magic number used for locating the RSDP, containing RSD PTR.
|
||||||
|
signature: [8]u8,
|
||||||
|
/// A byte used to verify the first 20 bytes of the RSDP
|
||||||
|
checksum: u8,
|
||||||
|
/// An OEM-supplied string that identified the OEM.
|
||||||
|
oem_id: [6]u8,
|
||||||
|
/// The RSDP revision, used for determining which fields are available.
|
||||||
|
revision: u8,
|
||||||
|
/// deprecated since version 2.0. A 32-bit physical address pointing to the RSDT.
|
||||||
|
root_system_description_table_address: u32 align(1),
|
||||||
|
/// The size of the RSDP.
|
||||||
|
length: u32 align(1),
|
||||||
|
/// A 64-bit physical address pointing to the XSDT. If the revision is at least 2, the XSDT
|
||||||
|
/// should be used regardless of architecture, as the RSDT was deprecated.
|
||||||
|
extended_system_descriptor_table_address: u64 align(1),
|
||||||
|
/// A checksum used for the entire table.
|
||||||
|
extended_checksum: u8,
|
||||||
|
reserved: [3]u8,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Multiple APIC Description Table (MADT)
|
||||||
|
const APIC: [4]u8 = "APIC".*;
|
||||||
|
/// Boot Error Record Table (BERT)
|
||||||
|
const BERT: [4]u8 = "BERT".*;
|
||||||
|
/// Corrected Platform Error Polling Table (CPEP)
|
||||||
|
const CPEP: [4]u8 = "CPEP".*;
|
||||||
|
/// Differentiated System Description Table (DSDT)
|
||||||
|
const DSDT: [4]u8 = "DSDT".*;
|
||||||
|
/// Embedded Controller Boot Resources Table (ECDT)
|
||||||
|
const ECDT: [4]u8 = "ECDT".*;
|
||||||
|
/// Error Injection Table (EINJ)
|
||||||
|
const EINJ: [4]u8 = "EINJ".*;
|
||||||
|
/// Error Record Serialization Table (ERST)
|
||||||
|
const ERST: [4]u8 = "ERST".*;
|
||||||
|
/// Fixed ACPI Description Table (FADT)
|
||||||
|
const FACP: [4]u8 = "FACP".*;
|
||||||
|
/// Firmware ACPI Control Structure (FACS)
|
||||||
|
const FACS: [4]u8 = "FACS".*;
|
||||||
|
/// Hardware Error Source Table (HEST)
|
||||||
|
const HEST: [4]u8 = "HEST".*;
|
||||||
|
/// High Precision Event Timer table (HPET)
|
||||||
|
const HPET: [4]u8 = "HPET".*;
|
||||||
|
/// PCI Express memory-mapped configuration space table (MCFG)
|
||||||
|
const MCFG: [4]u8 = "MCFG".*;
|
||||||
|
/// Maximum System Characteristics Table (MSCT)
|
||||||
|
const MSCT: [4]u8 = "MSCT".*;
|
||||||
|
/// Memory Power State Table (MPST)
|
||||||
|
const MPST: [4]u8 = "MPST".*;
|
||||||
|
// Platform Memory Topology Table (PMTT)
|
||||||
|
const PMTT: [4]u8 = "PMTT".*;
|
||||||
|
/// Persistent System Description Table (PSDT)
|
||||||
|
const PSDT: [4]u8 = "PSDT".*;
|
||||||
|
/// ACPI RAS Feature Table (RASF)
|
||||||
|
const RASF: [4]u8 = "RASF".*;
|
||||||
|
/// Root System Description Table
|
||||||
|
const RSDT: [4]u8 = "RSDT".*;
|
||||||
|
/// Smart Battery Specification Table (SBST)
|
||||||
|
const SBST: [4]u8 = "SBST".*;
|
||||||
|
/// System Locality System Information Table (SLIT)
|
||||||
|
const SLIT: [4]u8 = "SLIT".*;
|
||||||
|
/// System Resource Affinity Table (SRAT)
|
||||||
|
const SRAT: [4]u8 = "SRAT".*;
|
||||||
|
/// Secondary System Description Table (SSDT)
|
||||||
|
const DMAR: [4]u8 = "DMAR".*;
|
||||||
|
const SSDT: [4]u8 = "SSDT".*;
|
||||||
|
/// Serial Port Console Redirection table (SPCR) — the firmware's console UART.
|
||||||
|
const SPCR: [4]u8 = "SPCR".*;
|
||||||
|
/// Extended System Description Table (XSDT; 64-bit version of the RSDT)
|
||||||
|
const XSDT: [4]u8 = "XSDT".*;
|
||||||
|
|
||||||
|
/// The header every system descriptor table (RSDT/XSDT and each SDT) begins with.
|
||||||
|
const SystemDescriptorTableHeader = extern struct {
|
||||||
|
/// A 4 byte signature used for identification (e.g. "RSDT", "APIC").
|
||||||
|
signature: [4]u8,
|
||||||
|
/// The length of the entire table, including the header.
|
||||||
|
length: u32 align(1),
|
||||||
|
/// The revision of the ACPI spec this table conforms to.
|
||||||
|
revision: u8,
|
||||||
|
/// An 8-bit checksum field for the whole table, inclusive of the header.
|
||||||
|
checksum: u8,
|
||||||
|
/// An OEM-supplied string that identified the OEM.
|
||||||
|
oem_id: [6]u8,
|
||||||
|
oem_table_id: [8]u8,
|
||||||
|
oem_revision: u32 align(1),
|
||||||
|
creator_id: u32 align(1),
|
||||||
|
creator_revision: u32 align(1),
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- MADT: Multiple APIC Description Table (signature "APIC") ---------------
|
||||||
|
|
||||||
|
const Madt = extern struct {
|
||||||
|
header: SystemDescriptorTableHeader,
|
||||||
|
local_apic_address: u32 align(1),
|
||||||
|
flags: u32 align(1),
|
||||||
|
// Followed by a variable-length run of interrupt-controller records, each a
|
||||||
|
// MadtRecordHeader plus a type-specific body.
|
||||||
|
};
|
||||||
|
|
||||||
|
const MadtRecordHeader = extern struct {
|
||||||
|
type: u8,
|
||||||
|
length: u8,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// MADT record type 0: a processor's Local APIC.
|
||||||
|
const MadtLocalApic = extern struct {
|
||||||
|
record: MadtRecordHeader,
|
||||||
|
processor_id: u8,
|
||||||
|
apic_id: u8,
|
||||||
|
/// bit 0 = enabled, bit 1 = online-capable.
|
||||||
|
flags: u32 align(1),
|
||||||
|
};
|
||||||
|
|
||||||
|
/// MADT record type 1: an I/O APIC.
|
||||||
|
const MadtIoApic = extern struct {
|
||||||
|
record: MadtRecordHeader,
|
||||||
|
io_apic_id: u8,
|
||||||
|
reserved: u8,
|
||||||
|
address: u32 align(1),
|
||||||
|
/// First global system interrupt this I/O APIC handles.
|
||||||
|
gsi_base: u32 align(1),
|
||||||
|
};
|
||||||
|
|
||||||
|
/// MADT record type 2: an Interrupt Source Override (ISA IRQ -> GSI remap).
|
||||||
|
const MadtIso = extern struct {
|
||||||
|
record: MadtRecordHeader,
|
||||||
|
bus: u8,
|
||||||
|
source: u8,
|
||||||
|
gsi: u32 align(1),
|
||||||
|
flags: u16 align(1),
|
||||||
|
};
|
||||||
|
|
||||||
|
/// MADT record type 5: Local APIC Address Override (64-bit MMIO base).
|
||||||
|
const MadtLapicOverride = extern struct {
|
||||||
|
record: MadtRecordHeader,
|
||||||
|
reserved: u16 align(1),
|
||||||
|
address: u64 align(1),
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- MCFG: PCIe ECAM configuration space (signature "MCFG") -----------------
|
||||||
|
|
||||||
|
const Mcfg = extern struct {
|
||||||
|
header: SystemDescriptorTableHeader,
|
||||||
|
reserved: u64 align(1),
|
||||||
|
// Followed by one or more McfgAllocation entries.
|
||||||
|
};
|
||||||
|
|
||||||
|
const McfgAllocation = extern struct {
|
||||||
|
/// Physical base of this segment group's ECAM window.
|
||||||
|
base_address: u64 align(1),
|
||||||
|
segment_group: u16 align(1),
|
||||||
|
start_bus: u8,
|
||||||
|
end_bus: u8,
|
||||||
|
reserved: u32 align(1),
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- HPET (signature "HPET") ------------------------------------------------
|
||||||
|
|
||||||
|
const Hpet = extern struct {
|
||||||
|
header: SystemDescriptorTableHeader,
|
||||||
|
hardware_rev_id: u8,
|
||||||
|
flags: u8,
|
||||||
|
pci_vendor_id: u16 align(1),
|
||||||
|
// Generic Address Structure describing the register block.
|
||||||
|
address_space_id: u8,
|
||||||
|
register_bit_width: u8,
|
||||||
|
register_bit_offset: u8,
|
||||||
|
gas_reserved: u8,
|
||||||
|
address: u64 align(1),
|
||||||
|
hpet_number: u8,
|
||||||
|
minimum_tick: u16 align(1),
|
||||||
|
page_protection: u8,
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- Entry point ------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Discover hardware from the ACPI tables rooted at `rsdp_physical` and populate
|
||||||
|
/// `device_tree`. `hal` provides MMIO mapping (for PCIe ECAM) and port I/O. Also parses the
|
||||||
|
/// FADT and the AML sleep-state (`_Sx`) packages into `power_information` for the power service.
|
||||||
|
pub fn discover(rsdp_physical: u64, memory_regions: []const boot_handoff.MemoryRegion, device_tree: *DeviceTree, hal: Hal) !void {
|
||||||
|
if (rsdp_physical == 0) return error.NoRsdp;
|
||||||
|
boot_memory_regions = memory_regions;
|
||||||
|
|
||||||
|
// Start clean so a re-run doesn't accumulate stale state.
|
||||||
|
power_information = .{};
|
||||||
|
fadt_physical = 0;
|
||||||
|
fadt_length = 0;
|
||||||
|
platform_information = .{};
|
||||||
|
aml_stats = .{};
|
||||||
|
namespace = null;
|
||||||
|
dsdt_physical = 0;
|
||||||
|
aml_block_count = 0;
|
||||||
|
|
||||||
|
const rsdp: *const RootSystemDescriptionPointer = @ptrFromInt(boot_handoff.physicalToVirtual(rsdp_physical));
|
||||||
|
if (!std.mem.eql(u8, &rsdp.signature, "RSD PTR ")) return error.BadRsdpSignature;
|
||||||
|
// Revision 0 checksums only the first 20 bytes (the v1.0 RSDP).
|
||||||
|
if (!checksumOk(@ptrFromInt(boot_handoff.physicalToVirtual(rsdp_physical)), 20)) return error.BadRsdpChecksum;
|
||||||
|
|
||||||
|
if (rsdp.revision >= 2) {
|
||||||
|
const xsdp: *const ExtendedSystemDescriptorPointer = @ptrFromInt(boot_handoff.physicalToVirtual(rsdp_physical));
|
||||||
|
if (!checksumOk(@ptrFromInt(boot_handoff.physicalToVirtual(rsdp_physical)), xsdp.length)) return error.BadXsdpChecksum;
|
||||||
|
try walkRoot(u64, xsdp.extended_system_descriptor_table_address, device_tree, hal);
|
||||||
|
} else {
|
||||||
|
try walkRoot(u32, rsdp.root_system_description_table_address, device_tree, hal);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Now that the DSDT and any SSDTs are collected, build the AML namespace and
|
||||||
|
// read the sleep types from it.
|
||||||
|
var blocks: [aml_block_physical.len][]const u8 = undefined;
|
||||||
|
for (0..aml_block_count) |i| {
|
||||||
|
blocks[i] = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(aml_block_physical[i])))[0..aml_block_len[i]];
|
||||||
|
}
|
||||||
|
const active = blocks[0..aml_block_count];
|
||||||
|
if (aml.parse(device_tree.allocator, active)) |pr| {
|
||||||
|
namespace = pr.namespace;
|
||||||
|
aml_stats = .{ .nodes = namespace.?.nodeCount(), .consumed = pr.consumed, .total = pr.total };
|
||||||
|
power_information.s5 = aml.sleepState(&namespace.?, 5);
|
||||||
|
power_information.s3 = aml.sleepState(&namespace.?, 3);
|
||||||
|
// The namespace's Device objects are no longer folded into the kernel
|
||||||
|
// tree (M20.3): the ring-3 acpi service claims the acpi-tables node
|
||||||
|
// (published below), re-parses the same blobs, and registers + reports
|
||||||
|
// the _HID devices itself. The kernel keeps the namespace only for the
|
||||||
|
// \_S5 sleep type above.
|
||||||
|
} else |_| {
|
||||||
|
// AML parse failed (e.g. out of memory); power stays best-effort with
|
||||||
|
// whatever the FADT alone provided.
|
||||||
|
}
|
||||||
|
|
||||||
|
// Publish the acpi-tables node (docs/m19-m20-plan.md M20): the AML blobs as
|
||||||
|
// memory resources for the acpi service to map and parse in ring 3, a broad
|
||||||
|
// io_port grant for the OperationRegion access its interpreter needs, and
|
||||||
|
// the SCI for the events track (M21). Exactly one node, one trusted
|
||||||
|
// claimant. Kept even when the kernel-side device building (above) retires
|
||||||
|
// in M20.3 — the kernel still owns the *static* tables and \_S5.
|
||||||
|
publishAcpiTablesNode(device_tree) catch {};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Build the acpi-tables node (see the call site in discover). Best-effort: a
|
||||||
|
/// failure here leaves the kernel-seeded tree working, only the ring-3 service
|
||||||
|
/// finds nothing to claim.
|
||||||
|
fn publishAcpiTablesNode(device_tree: *DeviceTree) !void {
|
||||||
|
const node = try device_tree.addChild(device_tree.root, .acpi_tables, "acpi-tables");
|
||||||
|
// One memory resource per AML block — page-aligned base down, length padded
|
||||||
|
// up to cover the bytecode, so mmio_map hands the service a pointer into it.
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < aml_block_count and i < device_model.maximum_resources - 2) : (i += 1) {
|
||||||
|
// mmio_map preserves the sub-page offset, so the service maps this and
|
||||||
|
// gets a pointer straight to the bytecode.
|
||||||
|
_ = node.addResource(.memory, aml_block_physical[i], aml_block_len[i]);
|
||||||
|
}
|
||||||
|
// The broad I/O grant: OperationRegions name whatever ports the firmware
|
||||||
|
// chose (EC, PM1, GPE, SMBus); which ports cannot be known before the AML
|
||||||
|
// that names them is parsed, so the grant is the whole space — the honest
|
||||||
|
// trust boundary of docs/m19-m20-plan.md decision 5.
|
||||||
|
_ = node.addResource(.io_port, 0, 1 << 16);
|
||||||
|
// A broad interrupt window: ACPI _CRS names legacy ISA IRQs (the PS/2 lines
|
||||||
|
// 1 and 12, the RTC, …), and the service registers those devices under this
|
||||||
|
// node, so it must own a superset. The range [0, 256) covers every GSI; the
|
||||||
|
// SCI (recorded first, len 1) stays distinct so M21 can pick it out.
|
||||||
|
if (power_information.sci_interrupt != 0) _ = node.addResource(.irq, power_information.sci_interrupt, 1);
|
||||||
|
_ = node.addResource(.irq, 0, 256);
|
||||||
|
// The FADT rides along (M21): the service reads the PM1 event / GPE blocks
|
||||||
|
// from its own copy, telling it apart from the AML blobs by signature.
|
||||||
|
if (fadt_physical != 0) _ = node.addResource(.memory, fadt_physical, fadt_length);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The number of Device objects in the namespace built during discovery, or 0.
|
||||||
|
pub fn amlDeviceCount() usize {
|
||||||
|
if (namespace) |*ns| return aml.deviceCount(ns);
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Walk the RSDT (Entry = u32) or XSDT (Entry = u64): validate it, then dispatch
|
||||||
|
/// each SDT it points at. A bad individual table is skipped, not fatal.
|
||||||
|
fn walkRoot(comptime Entry: type, root_physical: u64, device_tree: *DeviceTree, hal: Hal) !void {
|
||||||
|
const header: *const SystemDescriptorTableHeader = @ptrFromInt(boot_handoff.physicalToVirtual(root_physical));
|
||||||
|
if (!checksumOk(@ptrFromInt(boot_handoff.physicalToVirtual(root_physical)), header.length)) return error.BadRootChecksum;
|
||||||
|
|
||||||
|
const count = (header.length - @sizeOf(SystemDescriptorTableHeader)) / @sizeOf(Entry);
|
||||||
|
const base: [*]const u8 = @ptrFromInt(boot_handoff.physicalToVirtual(root_physical));
|
||||||
|
const entries: [*]align(1) const Entry = @ptrCast(base + @sizeOf(SystemDescriptorTableHeader));
|
||||||
|
|
||||||
|
for (entries[0..count]) |ent| {
|
||||||
|
const sdt_physical: u64 = ent; // u32 entries widen; u64 pass through
|
||||||
|
handleTable(device_tree, hal, sdt_physical) catch continue;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Dispatch a single SDT on its signature.
|
||||||
|
fn handleTable(device_tree: *DeviceTree, hal: Hal, sdt_physical: u64) !void {
|
||||||
|
const header: *const SystemDescriptorTableHeader = @ptrFromInt(boot_handoff.physicalToVirtual(sdt_physical));
|
||||||
|
const sig = header.signature;
|
||||||
|
if (std.mem.eql(u8, &sig, &APIC)) {
|
||||||
|
try parseMadt(device_tree, header);
|
||||||
|
} else if (std.mem.eql(u8, &sig, &MCFG)) {
|
||||||
|
try parseMcfg(device_tree, header);
|
||||||
|
} else if (std.mem.eql(u8, &sig, &HPET)) {
|
||||||
|
try parseHpet(device_tree, hal, header);
|
||||||
|
} else if (std.mem.eql(u8, &sig, &FACP)) {
|
||||||
|
fadt_physical = sdt_physical;
|
||||||
|
fadt_length = header.length;
|
||||||
|
parseFadt(header);
|
||||||
|
} else if (std.mem.eql(u8, &sig, &SPCR)) {
|
||||||
|
parseSpcr(header);
|
||||||
|
} else if (std.mem.eql(u8, &sig, &DMAR)) {
|
||||||
|
parseDmar(hal, header);
|
||||||
|
} else if (std.mem.eql(u8, &sig, &SSDT)) {
|
||||||
|
// Secondary namespace bytecode — collect for the sleep-state (`_Sx`) scan.
|
||||||
|
addAmlBlock(sdt_physical);
|
||||||
|
}
|
||||||
|
// Any other signature is recognised but left opaque for now.
|
||||||
|
}
|
||||||
|
|
||||||
|
/// MADT -> one processor node per Local APIC, one interrupt_controller per I/O APIC.
|
||||||
|
fn parseMadt(device_tree: *DeviceTree, header: *const SystemDescriptorTableHeader) !void {
|
||||||
|
const madt: *const Madt = @ptrCast(header);
|
||||||
|
const total: usize = header.length;
|
||||||
|
const base: [*]const u8 = @ptrCast(header);
|
||||||
|
var ioapic_index: usize = 0;
|
||||||
|
|
||||||
|
// MADT header: local APIC base + flags (bit 0 = 8259 PIC present).
|
||||||
|
platform_information.lapic_base = madt.local_apic_address;
|
||||||
|
platform_information.pic_present = madt.flags & 1 != 0;
|
||||||
|
|
||||||
|
var off: usize = @sizeOf(Madt);
|
||||||
|
while (off + @sizeOf(MadtRecordHeader) <= total) {
|
||||||
|
const rec: *const MadtRecordHeader = @ptrCast(base + off);
|
||||||
|
if (rec.length < @sizeOf(MadtRecordHeader)) break; // malformed; avoid a spin
|
||||||
|
switch (rec.type) {
|
||||||
|
0 => {
|
||||||
|
const la: *const MadtLocalApic = @ptrCast(base + off);
|
||||||
|
// bit 0 = enabled: skip processors the firmware marks unusable.
|
||||||
|
if (la.flags & 1 != 0) {
|
||||||
|
var nb: [24]u8 = undefined;
|
||||||
|
const nm = std.fmt.bufPrint(&nb, "cpu{d}", .{la.processor_id}) catch "cpu";
|
||||||
|
_ = try device_tree.addChild(device_tree.root, .processor, nm);
|
||||||
|
// Also record it as a schedulable core (with the APIC ID an AP
|
||||||
|
// wake needs, which the device node name doesn't preserve).
|
||||||
|
if (cpu_information.count < cpu_information.cpus.len) {
|
||||||
|
cpu_information.cpus[cpu_information.count] = .{
|
||||||
|
.processor_id = la.processor_id,
|
||||||
|
.apic_id = la.apic_id,
|
||||||
|
.online_capable = la.flags & 2 != 0,
|
||||||
|
};
|
||||||
|
cpu_information.count += 1;
|
||||||
|
} else {
|
||||||
|
cpu_information.dropped += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
1 => {
|
||||||
|
const io: *const MadtIoApic = @ptrCast(base + off);
|
||||||
|
var nb: [24]u8 = undefined;
|
||||||
|
const nm = std.fmt.bufPrint(&nb, "ioapic{d}", .{ioapic_index}) catch "ioapic";
|
||||||
|
ioapic_index += 1;
|
||||||
|
const d = try device_tree.addChild(device_tree.root, .interrupt_controller, nm);
|
||||||
|
_ = d.addResource(.memory, io.address, 0x20);
|
||||||
|
// The GSI range this I/O APIC handles, starting at gsi_base.
|
||||||
|
_ = d.addResource(.irq, io.gsi_base, 0);
|
||||||
|
},
|
||||||
|
2 => {
|
||||||
|
const iso: *const MadtIso = @ptrCast(base + off);
|
||||||
|
if (platform_information.override_count < platform_information.overrides.len) {
|
||||||
|
platform_information.overrides[platform_information.override_count] = .{
|
||||||
|
.source = iso.source,
|
||||||
|
.gsi = iso.gsi,
|
||||||
|
.flags = iso.flags,
|
||||||
|
};
|
||||||
|
platform_information.override_count += 1;
|
||||||
|
}
|
||||||
|
},
|
||||||
|
5 => {
|
||||||
|
const ovr: *const MadtLapicOverride = @ptrCast(base + off);
|
||||||
|
platform_information.lapic_base = ovr.address;
|
||||||
|
},
|
||||||
|
else => {},
|
||||||
|
}
|
||||||
|
off += rec.length;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// MCFG -> a pci_host_bridge per ECAM segment, then a PCI enumeration underneath.
|
||||||
|
fn parseMcfg(device_tree: *DeviceTree, header: *const SystemDescriptorTableHeader) !void {
|
||||||
|
const total: usize = header.length;
|
||||||
|
const base: [*]const u8 = @ptrCast(header);
|
||||||
|
|
||||||
|
var off: usize = @sizeOf(Mcfg);
|
||||||
|
while (off + @sizeOf(McfgAllocation) <= total) : (off += @sizeOf(McfgAllocation)) {
|
||||||
|
const alloc: *const McfgAllocation = @ptrCast(base + off);
|
||||||
|
const bus_count: u64 = @as(u64, alloc.end_bus - alloc.start_bus) + 1;
|
||||||
|
|
||||||
|
var nb: [24]u8 = undefined;
|
||||||
|
const nm = std.fmt.bufPrint(&nb, "pci{d}", .{alloc.segment_group}) catch "pci";
|
||||||
|
const bridge = try device_tree.addChild(device_tree.root, .pci_host_bridge, nm);
|
||||||
|
// ECAM window: 1 MiB of configuration space per bus.
|
||||||
|
_ = bridge.addResource(.memory, alloc.base_address, bus_count << 20);
|
||||||
|
_ = bridge.addResource(.bus_range, alloc.start_bus, bus_count);
|
||||||
|
addBridgeApertures(bridge);
|
||||||
|
// The bridge decodes the whole 16-bit I/O space toward its bus — the
|
||||||
|
// window functions' I/O BARs must register-contain within (M19.2).
|
||||||
|
_ = bridge.addResource(.io_port, 0, 1 << 16);
|
||||||
|
|
||||||
|
// The function walk itself retired to ring 3 (M19.3): the pci-bus
|
||||||
|
// driver claims this bridge, repeats the scan through its ECAM grant,
|
||||||
|
// and device_registers what it finds — the kernel seeds only the
|
||||||
|
// bridge. The scan's equivalence was proven before the hand-off
|
||||||
|
// (pci-scan), and the walk's history is in git if archaeology calls.
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The boot memory map, stored at discover() entry for the aperture derivation
|
||||||
|
/// below (and, in M20, for the acpi-tables node's containment windows).
|
||||||
|
var boot_memory_regions: []const boot_handoff.MemoryRegion = &.{};
|
||||||
|
|
||||||
|
/// The bridge's MMIO apertures, derived from the boot memory map's holes
|
||||||
|
/// (docs/m19-m20-plan.md decision 2): registered PCI functions carry BAR
|
||||||
|
/// resources, and `device_register` containment demands the bridge own windows
|
||||||
|
/// that cover them. Everything the firmware described is "not hole"; the low
|
||||||
|
/// aperture runs from the end of the described space below 4 GiB up to the
|
||||||
|
/// I/O-APIC region, the high one from 4 GiB (or the end of RAM above it) to
|
||||||
|
/// the 46-bit line. Coarse, mechanical, and AML-free — available at boot no
|
||||||
|
/// matter what later moved to user space.
|
||||||
|
fn addBridgeApertures(bridge: *device_model.Device) void {
|
||||||
|
// Below 4 GiB the described regions are sparse (RAM low, firmware flash
|
||||||
|
// and tables high), so the holes are the *gaps between* them — a single
|
||||||
|
// "after the last region" rule dies on OVMF's flash at the very top.
|
||||||
|
// Sort-merge the described ranges, then keep the three largest gaps
|
||||||
|
// (resource slots are bounded at 8 per device; ECAM + bus range + 3 + the
|
||||||
|
// high aperture fits). Above 4 GiB one aperture runs from the end of the
|
||||||
|
// described space to the 46-bit line.
|
||||||
|
const Range = struct { base: u64, end: u64 };
|
||||||
|
var below: [64]Range = undefined;
|
||||||
|
var below_count: usize = 0;
|
||||||
|
var high_end: u64 = 1 << 32;
|
||||||
|
for (boot_memory_regions) |region| {
|
||||||
|
const end = region.base + region.pages * 4096;
|
||||||
|
// Above 4 GiB only *usable RAM* blocks the aperture: OVMF describes
|
||||||
|
// its own 64-bit PCI window as a reserved region and then programs
|
||||||
|
// BARs inside it — honoring reserved there would exclude the very
|
||||||
|
// space BARs live in. Below 4 GiB every described region blocks (the
|
||||||
|
// kernel image, the tables, the ramdisk all live there). Bring-up
|
||||||
|
// trust: only the bridge's claimant can register into the aperture.
|
||||||
|
if (region.kind == .usable and end > high_end) high_end = end;
|
||||||
|
if (region.base >= (1 << 32) or below_count == below.len) continue;
|
||||||
|
below[below_count] = .{ .base = region.base, .end = @min(end, 1 << 32) };
|
||||||
|
below_count += 1;
|
||||||
|
}
|
||||||
|
// Insertion sort by base (the map is small and this runs once at boot).
|
||||||
|
for (1..below_count) |i| {
|
||||||
|
const key = below[i];
|
||||||
|
var j = i;
|
||||||
|
while (j > 0 and below[j - 1].base > key.base) : (j -= 1) below[j] = below[j - 1];
|
||||||
|
below[j] = key;
|
||||||
|
}
|
||||||
|
// Walk the sorted ranges, collecting inter-region gaps of at least 1 MiB.
|
||||||
|
var gaps: [3]Range = .{Range{ .base = 0, .end = 0 }} ** 3;
|
||||||
|
var cursor: u64 = 0;
|
||||||
|
var index: usize = 0;
|
||||||
|
while (index <= below_count) : (index += 1) {
|
||||||
|
const gap_end = if (index == below_count) (1 << 32) else below[index].base;
|
||||||
|
if (gap_end > cursor and gap_end - cursor >= (1 << 20)) {
|
||||||
|
// Keep the three largest, replacing the smallest kept so far.
|
||||||
|
var smallest: usize = 0;
|
||||||
|
for (gaps, 0..) |gap, gi| {
|
||||||
|
if (gap.end - gap.base < gaps[smallest].end - gaps[smallest].base) smallest = gi;
|
||||||
|
}
|
||||||
|
if (gap_end - cursor > gaps[smallest].end - gaps[smallest].base) {
|
||||||
|
gaps[smallest] = .{ .base = cursor, .end = gap_end };
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (index < below_count and below[index].end > cursor) cursor = below[index].end;
|
||||||
|
}
|
||||||
|
for (gaps) |gap| {
|
||||||
|
if (gap.end > gap.base) _ = bridge.addResource(.memory, gap.base, gap.end - gap.base);
|
||||||
|
}
|
||||||
|
_ = bridge.addResource(.memory, high_end, (@as(u64, 1) << 46) - high_end);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// HPET -> a timer node with its register block as an MMIO resource, plus the GSI
|
||||||
|
/// its comparators can raise.
|
||||||
|
///
|
||||||
|
/// Unlike a PCI device or an ACPI `_CRS` node, the HPET table carries **no interrupt
|
||||||
|
/// number**: which I/O APIC inputs a comparator may drive is advertised at runtime,
|
||||||
|
/// as a bitmask in `Tn_INT_ROUTE_CAP` (bits 63:32 of the Timer 0 configuration register).
|
||||||
|
/// So discovery maps the register block, reads the mask, and records one concrete
|
||||||
|
/// `irq` resource — the GSI a driver is entitled to bind. The driver commits to it
|
||||||
|
/// by writing `Tn_INT_ROUTE_CNF`; the kernel checks the binding against this
|
||||||
|
/// resource (see process.ownedGsi), which is what keeps `irq_bind` a capability
|
||||||
|
/// rather than a request for an arbitrary interrupt line.
|
||||||
|
fn parseHpet(device_tree: *DeviceTree, hal: Hal, header: *const SystemDescriptorTableHeader) !void {
|
||||||
|
const hpet: *const Hpet = @ptrCast(header);
|
||||||
|
const d = try device_tree.addChild(device_tree.root, .timer, "hpet");
|
||||||
|
|
||||||
|
// The GAS tag must say System Memory (0) before we treat `address` as a physical
|
||||||
|
// address. The HPET spec mandates it, but firmware is not a thing to trust: a
|
||||||
|
// System I/O (1) tag here would have us map an arbitrary page and read a bogus
|
||||||
|
// route-capability mask out of it.
|
||||||
|
if (hpet.address_space_id != gas_system_memory) return;
|
||||||
|
|
||||||
|
_ = d.addResource(.memory, hpet.address, 0x400);
|
||||||
|
|
||||||
|
const regs = hal.mapMmio(hpet.address, 0x400, true);
|
||||||
|
const t0_configuration: *const volatile u64 = @ptrFromInt(regs + 0x100);
|
||||||
|
const route_cap: u32 = @truncate(t0_configuration.* >> 32);
|
||||||
|
if (hpetGsi(route_cap)) |gsi| _ = d.addResource(.irq, gsi, 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// ACPI Generic Address Structure address-space ids we care about.
|
||||||
|
const gas_system_memory: u8 = 0;
|
||||||
|
|
||||||
|
/// Pick a GSI for the HPET out of its route-capability mask. Prefer an input at or
|
||||||
|
/// above 16: the low ones overlap the legacy ISA lines (2 = cascaded PIT, 8 = RTC),
|
||||||
|
/// which the MADT may separately override, whereas 16+ are the free upper inputs on
|
||||||
|
/// every I/O APIC we care about. Falls back to the lowest bit set if there are none.
|
||||||
|
fn hpetGsi(route_cap: u32) ?u32 {
|
||||||
|
if (route_cap == 0) return null;
|
||||||
|
var gsi: u32 = 16;
|
||||||
|
while (gsi < 32) : (gsi += 1) {
|
||||||
|
if (route_cap & (@as(u32, 1) << @intCast(gsi)) != 0) return gsi;
|
||||||
|
}
|
||||||
|
return @ctz(route_cap);
|
||||||
|
}
|
||||||
|
|
||||||
|
// FADT field offsets (bytes from the table start). The FADT grew across ACPI
|
||||||
|
// revisions, so every field is read through `fadt()` with a length guard rather
|
||||||
|
// than a fixed struct — an older/shorter FADT simply lacks the later (X_) fields.
|
||||||
|
const fadt_dsdt = 40; // u32
|
||||||
|
const fadt_smi_cmd = 48; // u32 (an I/O port)
|
||||||
|
const fadt_acpi_enable = 52; // u8
|
||||||
|
const fadt_acpi_disable = 53; // u8
|
||||||
|
const fadt_pm1a_cnt_blk = 64; // u32 (I/O port)
|
||||||
|
const fadt_pm1b_cnt_blk = 68; // u32 (I/O port)
|
||||||
|
const fadt_pm_tmr_blk = 76; // u32 (I/O port) — the PM timer counter
|
||||||
|
const fadt_pm1_cnt_len = 89; // u8 (bytes)
|
||||||
|
const fadt_sci_int = 46; // u16 (the SCI's GSI)
|
||||||
|
const fadt_flags = 112; // u32
|
||||||
|
const fadt_reset_register = 116; // GAS (12 bytes)
|
||||||
|
const fadt_reset_value = 128; // u8
|
||||||
|
const fadt_x_dsdt = 140; // u64
|
||||||
|
const fadt_x_pm1a_cnt_blk = 172; // GAS
|
||||||
|
const fadt_x_pm1b_cnt_blk = 184; // GAS
|
||||||
|
const fadt_x_pm_tmr_blk = 208; // GAS
|
||||||
|
const flag_reset_register_supported = 1 << 10;
|
||||||
|
const flag_tmr_value_ext = 1 << 8; // PM timer counter is 32-bit (else 24-bit)
|
||||||
|
|
||||||
|
/// FADT -> the power register map (into `power_information`) and the DSDT address, which
|
||||||
|
/// is queued for the AML sleep-state (`_Sx`) scan. No AML interpretation happens here.
|
||||||
|
fn parseFadt(header: *const SystemDescriptorTableHeader) void {
|
||||||
|
const base: [*]align(1) const u8 = @ptrCast(header);
|
||||||
|
const len: usize = header.length;
|
||||||
|
const pi = &power_information;
|
||||||
|
|
||||||
|
pi.sci_interrupt = @truncate(fadt(u16, base, len, fadt_sci_int) orelse 0);
|
||||||
|
pi.smi_cmd = @truncate(fadt(u32, base, len, fadt_smi_cmd) orelse 0);
|
||||||
|
pi.acpi_enable = fadt(u8, base, len, fadt_acpi_enable) orelse 0;
|
||||||
|
pi.acpi_disable = fadt(u8, base, len, fadt_acpi_disable) orelse 0;
|
||||||
|
|
||||||
|
const cnt_width = fadt(u8, base, len, fadt_pm1_cnt_len) orelse 2;
|
||||||
|
pi.pm1a_cnt = readCntRegister(base, len, fadt_x_pm1a_cnt_blk, fadt_pm1a_cnt_blk, cnt_width);
|
||||||
|
pi.pm1b_cnt = readCntRegister(base, len, fadt_x_pm1b_cnt_blk, fadt_pm1b_cnt_blk, cnt_width);
|
||||||
|
|
||||||
|
const flags = fadt(u32, base, len, fadt_flags) orelse 0;
|
||||||
|
pi.reset_supported = flags & flag_reset_register_supported != 0;
|
||||||
|
pi.reset = readGas(base, len, fadt_reset_register) orelse .{};
|
||||||
|
pi.reset_value = fadt(u8, base, len, fadt_reset_value) orelse 0;
|
||||||
|
|
||||||
|
// The PM timer — a fixed-rate counter used as a calibration reference when no
|
||||||
|
// HPET is present. Prefer the 64-bit-capable X_ GAS, fall back to the port.
|
||||||
|
platform_information.pm_timer = readCntRegister(base, len, fadt_x_pm_tmr_blk, fadt_pm_tmr_blk, 4);
|
||||||
|
platform_information.pm_timer_32bit = flags & flag_tmr_value_ext != 0;
|
||||||
|
|
||||||
|
var dsdt: u64 = fadt(u32, base, len, fadt_dsdt) orelse 0;
|
||||||
|
if (fadt(u64, base, len, fadt_x_dsdt)) |x| {
|
||||||
|
if (x != 0) dsdt = x;
|
||||||
|
}
|
||||||
|
dsdt_physical = dsdt;
|
||||||
|
addAmlBlock(dsdt);
|
||||||
|
}
|
||||||
|
|
||||||
|
// SPCR field offsets (bytes from the table start).
|
||||||
|
const spcr_interface_type = 36; // u8
|
||||||
|
const spcr_base_address = 40; // GAS (12 bytes)
|
||||||
|
|
||||||
|
/// SPCR -> the console UART's address + interface type, so serial can target the
|
||||||
|
/// firmware's actual debug port instead of assuming legacy COM1.
|
||||||
|
fn parseSpcr(header: *const SystemDescriptorTableHeader) void {
|
||||||
|
const base: [*]align(1) const u8 = @ptrCast(header);
|
||||||
|
const len: usize = header.length;
|
||||||
|
const gas = readGas(base, len, spcr_base_address) orelse return;
|
||||||
|
if (gas.address == 0) return;
|
||||||
|
platform_information.spcr_uart = gas;
|
||||||
|
platform_information.spcr_kind = fadt(u8, base, len, spcr_interface_type) orelse 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
// DMAR remapping-structure layout (Intel VT-d spec §8): the DMAR-specific header is 12
|
||||||
|
// bytes (host-address-width, flags, 10 reserved), then a list of {type u16, length u16}
|
||||||
|
// structures. Type 0 is a DRHD (DMA Remapping Hardware Unit Definition), whose 64-bit
|
||||||
|
// register base sits at offset 8 within it.
|
||||||
|
const dmar_structures_offset = 48; // 36-byte ACPI header + 12-byte DMAR header
|
||||||
|
const dmar_type_drhd: u16 = 0;
|
||||||
|
const drhd_register_base_offset = 8;
|
||||||
|
|
||||||
|
/// DMAR -> detect the IOMMU. Find the first DMA-remapping hardware unit, map its
|
||||||
|
/// register block, and record its version and capabilities. This is *detection only*:
|
||||||
|
/// it tells the system an IOMMU exists (so `device_claim` on a DMA device could one day
|
||||||
|
/// be gated by a per-device translation domain), but no domains are programmed yet —
|
||||||
|
/// enforcement is built with the first DMA driver, which is what there is to protect and
|
||||||
|
/// test against. See docs/driver-model.md (M16), the honest caveat.
|
||||||
|
fn parseDmar(hal: Hal, header: *const SystemDescriptorTableHeader) void {
|
||||||
|
const base: [*]align(1) const u8 = @ptrCast(header);
|
||||||
|
const total: usize = header.length;
|
||||||
|
|
||||||
|
var off: usize = dmar_structures_offset;
|
||||||
|
while (off + 4 <= total) {
|
||||||
|
const kind = fadt(u16, base, total, off) orelse break;
|
||||||
|
const length = fadt(u16, base, total, off + 2) orelse break;
|
||||||
|
if (length < 4 or off + length > total) break; // malformed; stop rather than loop
|
||||||
|
if (kind == dmar_type_drhd) {
|
||||||
|
const register_base = fadt(u64, base, total, off + drhd_register_base_offset) orelse 0;
|
||||||
|
if (register_base != 0) {
|
||||||
|
const regs = hal.mapMmio(register_base, abi.page_size, true);
|
||||||
|
platform_information.iommu_present = true;
|
||||||
|
platform_information.iommu_base = register_base;
|
||||||
|
platform_information.iommu_version = @as(*const volatile u32, @ptrFromInt(regs + 0x00)).*;
|
||||||
|
platform_information.iommu_capabilities = @as(*const volatile u64, @ptrFromInt(regs + 0x08)).*;
|
||||||
|
return; // first unit is enough for detection; multi-unit is future
|
||||||
|
}
|
||||||
|
}
|
||||||
|
off += length;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- helpers ----------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Sum `len` bytes; an ACPI table/pointer is valid when the low 8 bits are zero.
|
||||||
|
fn checksumOk(bytes: [*]const u8, len: usize) bool {
|
||||||
|
var sum: u8 = 0;
|
||||||
|
for (0..len) |i| sum +%= bytes[i];
|
||||||
|
return sum == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read a FADT field of type `T` at `off`, or null if the table is too short to
|
||||||
|
/// contain it (a legal state for older FADT revisions).
|
||||||
|
fn fadt(comptime T: type, base: [*]align(1) const u8, len: usize, off: usize) ?T {
|
||||||
|
if (off + @sizeOf(T) > len) return null;
|
||||||
|
return rd(T, base, off);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode a Generic Address Structure at `off` into a `RegisterAccess`. GAS layout:
|
||||||
|
/// address_space(u8), bit_width(u8), bit_offset(u8), access_size(u8), address(u64).
|
||||||
|
fn readGas(base: [*]align(1) const u8, len: usize, off: usize) ?RegisterAccess {
|
||||||
|
if (off + 12 > len) return null;
|
||||||
|
const address_space = rd(u8, base, off);
|
||||||
|
const bit_width = rd(u8, base, off + 1);
|
||||||
|
const address = rd(u64, base, off + 4);
|
||||||
|
return .{
|
||||||
|
.mmio = address_space == 0, // 0 = system memory, 1 = system I/O
|
||||||
|
.address = address,
|
||||||
|
.width = bit_width / 8,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A PM1 control register: prefer the 64-bit-capable X_ GAS form; fall back to the
|
||||||
|
/// legacy 32-bit I/O-port field. Width comes from PM1_CNT_LEN either way.
|
||||||
|
fn readCntRegister(base: [*]align(1) const u8, len: usize, xoff: usize, legacy_off: usize, width: u8) RegisterAccess {
|
||||||
|
if (readGas(base, len, xoff)) |g| {
|
||||||
|
if (g.address != 0) return .{ .mmio = g.mmio, .address = g.address, .width = width };
|
||||||
|
}
|
||||||
|
const port = fadt(u32, base, len, legacy_off) orelse 0;
|
||||||
|
return .{ .mmio = false, .address = port, .width = width };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read a little-endian integer at `off` from a (possibly unaligned) byte pointer.
|
||||||
|
/// x86 is little-endian and native, so an unaligned load suffices.
|
||||||
|
fn rd(comptime T: type, bytes: [*]align(1) const u8, off: usize) T {
|
||||||
|
const p: *align(1) const T = @ptrCast(bytes + off);
|
||||||
|
return p.*;
|
||||||
|
}
|
||||||
@@ -0,0 +1,260 @@
|
|||||||
|
//! AML (ACPI Machine Language) — the bytecode in the DSDT and SSDTs that describes
|
||||||
|
//! the parts of the machine the static tables don't.
|
||||||
|
//!
|
||||||
|
//! This module has two stages. `parser.zig` walks the entire byte stream and
|
||||||
|
//! records every named object into a namespace tree (`namespace.zig`), capturing
|
||||||
|
//! method bodies and field/region layout. `interpreter.zig` then *evaluates* control
|
||||||
|
//! methods on demand — running operators, control flow, and OperationRegion field
|
||||||
|
//! access — so callers can resolve device status (`_STA`), current resource
|
||||||
|
//! settings (`_CRS`), sleep states (`_Sx`), and the like against the live namespace.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const opcode = @import("opcodes.zig");
|
||||||
|
const parser = @import("parser.zig");
|
||||||
|
|
||||||
|
/// The named AML opcode/prefix bytes (`zero_opcode`, `byte_prefix`, …). Re-exported so
|
||||||
|
/// callers that decode raw AML bytes — e.g. the acpi service reading a `_HID` integer —
|
||||||
|
/// name the opcodes instead of writing bare 0x0A/0x0B/… literals (docs/coding-standards.md).
|
||||||
|
pub const opcodes = @import("opcodes.zig");
|
||||||
|
|
||||||
|
pub const Namespace = @import("namespace.zig").Namespace;
|
||||||
|
pub const Node = @import("namespace.zig").Node;
|
||||||
|
pub const NodeKind = @import("namespace.zig").NodeKind;
|
||||||
|
|
||||||
|
/// The AML evaluator: interprets control methods (and reads Names/Fields) far
|
||||||
|
/// enough for device discovery. See `interpreter.zig`.
|
||||||
|
pub const Interpreter = @import("interpreter.zig").Interpreter;
|
||||||
|
pub const Object = @import("interpreter.zig").Object;
|
||||||
|
pub const EvaluateHal = @import("interpreter.zig").Hal;
|
||||||
|
|
||||||
|
/// The SLP_TYP values written to PM1a/PM1b control to enter a sleep state.
|
||||||
|
pub const SleepType = struct {
|
||||||
|
slp_typ_a: u8,
|
||||||
|
slp_typ_b: u8,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const ParseResult = struct {
|
||||||
|
namespace: Namespace,
|
||||||
|
/// Bytes the parser consumed across all blocks...
|
||||||
|
consumed: usize,
|
||||||
|
/// ...out of this many. A clean full traversal has `consumed == total`.
|
||||||
|
total: usize,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Parse the given AML blocks (DSDT first, then SSDTs) into one namespace. Later
|
||||||
|
/// blocks extend the namespace built by earlier ones, exactly as ACPI intends.
|
||||||
|
pub fn parse(allocator: std.mem.Allocator, blocks: []const []const u8) !ParseResult {
|
||||||
|
var namespace = try Namespace.init(allocator);
|
||||||
|
var consumed: usize = 0;
|
||||||
|
var total: usize = 0;
|
||||||
|
for (blocks) |block| {
|
||||||
|
var p = parser.Parser.init(block, &namespace);
|
||||||
|
consumed += p.parseAll();
|
||||||
|
total += block.len;
|
||||||
|
}
|
||||||
|
return .{ .namespace = namespace, .consumed = consumed, .total = total };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Count the Device objects in a parsed namespace — what the acpi service
|
||||||
|
/// (docs/m19-m20-plan.md M20) reports, and what the kernel's own parse counts
|
||||||
|
/// so the two can be checked equal across the ring-3 move.
|
||||||
|
pub fn deviceCount(namespace: *const Namespace) usize {
|
||||||
|
return countKind(namespace.root, .device);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn countKind(node: *const Node, kind: NodeKind) usize {
|
||||||
|
var n: usize = if (node.kind == kind) 1 else 0;
|
||||||
|
var c = node.first_child;
|
||||||
|
while (c) |child| : (c = child.next_sibling) n += countKind(child, kind);
|
||||||
|
return n;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Look up the `\_S{state}` sleep package in a parsed namespace and return its
|
||||||
|
/// first two integer elements (SLP_TYP for PM1a / PM1b), or null if absent.
|
||||||
|
pub fn sleepState(namespace: *Namespace, state: u8) ?SleepType {
|
||||||
|
const segment = [4]u8{ '_', 'S', '0' + state, '_' };
|
||||||
|
const node = namespace.resolve(namespace.root, false, 0, &.{segment}) orelse return null;
|
||||||
|
if (node.kind != .name) return null;
|
||||||
|
return parseSleepPackage(node.value);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode a `Package(){ SLP_TYPa, SLP_TYPb, ... }` from the raw AML of a Name's
|
||||||
|
/// value. Returns the first two elements as bytes (missing elements default to 0).
|
||||||
|
fn parseSleepPackage(value: []const u8) ?SleepType {
|
||||||
|
if (value.len == 0 or value[0] != opcode.package_opcode) return null;
|
||||||
|
var p: usize = 1;
|
||||||
|
p += packageLengthSize(value, p) orelse return null;
|
||||||
|
if (p >= value.len) return null;
|
||||||
|
const number_elements = value[p];
|
||||||
|
p += 1;
|
||||||
|
|
||||||
|
const a: u8 = if (number_elements >= 1) @truncate(readInteger(value, &p) orelse 0) else 0;
|
||||||
|
const b: u8 = if (number_elements >= 2) @truncate(readInteger(value, &p) orelse 0) else 0;
|
||||||
|
return .{ .slp_typ_a = a, .slp_typ_b = b };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Bytes a PkgLength field occupies at `p` (we only need to step over it here).
|
||||||
|
fn packageLengthSize(bytes: []const u8, p: usize) ?usize {
|
||||||
|
if (p >= bytes.len) return null;
|
||||||
|
const follow: usize = bytes[p] >> 6;
|
||||||
|
if (p + 1 + follow > bytes.len) return null;
|
||||||
|
return 1 + follow;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read one AML integer data object at `p`, advancing `p`.
|
||||||
|
fn readInteger(bytes: []const u8, p: *usize) ?u64 {
|
||||||
|
if (p.* >= bytes.len) return null;
|
||||||
|
const opcode_byte = bytes[p.*];
|
||||||
|
p.* += 1;
|
||||||
|
return switch (opcode_byte) {
|
||||||
|
opcode.zero_opcode => 0,
|
||||||
|
opcode.one_opcode => 1,
|
||||||
|
opcode.ones_opcode => 0xFF,
|
||||||
|
opcode.byte_prefix => readLittle(bytes, p, 1),
|
||||||
|
opcode.word_prefix => readLittle(bytes, p, 2),
|
||||||
|
opcode.dword_prefix => readLittle(bytes, p, 4),
|
||||||
|
opcode.qword_prefix => readLittle(bytes, p, 8),
|
||||||
|
else => null,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
fn readLittle(bytes: []const u8, p: *usize, n: usize) ?u64 {
|
||||||
|
if (p.* + n > bytes.len) return null;
|
||||||
|
var v: u64 = 0;
|
||||||
|
var k: usize = 0;
|
||||||
|
while (k < n) : (k += 1) v |= @as(u64, bytes[p.* + k]) << @intCast(k * 8);
|
||||||
|
p.* += n;
|
||||||
|
return v;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- tests ------------------------------------------------------------------
|
||||||
|
|
||||||
|
test "parses a nested namespace and finds the sleep package" {
|
||||||
|
// A hand-assembled AML blob (all PkgLengths computed to be single-byte):
|
||||||
|
// Name(_S5, Package(2){0x05, 0x00})
|
||||||
|
// Scope(\_SB) { Device(PCI0) {
|
||||||
|
// Name(_HID, 0x11)
|
||||||
|
// Method(MTHD, 1) {}
|
||||||
|
// Method(CALL, 0) { MTHD(Zero) } // invocation of a 1-arg method
|
||||||
|
// } }
|
||||||
|
// OperationRegion(DBG0, SystemIO, 0x0402, 1)
|
||||||
|
// Field(DBG0, ...) { DBGB, 8 }
|
||||||
|
const blob = [_]u8{
|
||||||
|
// Name(_S5, Package(2){Byte 0x05, Byte 0x00})
|
||||||
|
0x08, 0x5F, 0x53, 0x35, 0x5F, 0x12, 0x06, 0x02, 0x0A, 0x05, 0x0A, 0x00,
|
||||||
|
// Scope(\_SB) packagelen=0x27
|
||||||
|
0x10, 0x27, 0x5C, 0x5F, 0x53, 0x42, 0x5F,
|
||||||
|
// Device(PCI0) packagelen=0x1F
|
||||||
|
0x5B, 0x82, 0x1F, 0x50, 0x43,
|
||||||
|
0x49, 0x30,
|
||||||
|
// Name(_HID, 0x11)
|
||||||
|
0x08, 0x5F, 0x48, 0x49, 0x44, 0x0A, 0x11,
|
||||||
|
// Method(MTHD, flags=1) empty, packagelen=0x06
|
||||||
|
0x14, 0x06, 0x4D,
|
||||||
|
0x54, 0x48, 0x44, 0x01,
|
||||||
|
// Method(CALL, flags=0) { MTHD(Zero) }, packagelen=0x0B
|
||||||
|
0x14, 0x0B, 0x43, 0x41, 0x4C, 0x4C, 0x00, 0x4D,
|
||||||
|
0x54, 0x48, 0x44, 0x00,
|
||||||
|
// OperationRegion(DBG0, SystemIO, Word 0x0402, Byte 1)
|
||||||
|
0x5B, 0x80, 0x44, 0x42, 0x47, 0x30, 0x01, 0x0B,
|
||||||
|
0x02, 0x04, 0x0A, 0x01,
|
||||||
|
// Field(DBG0, flags=1) { DBGB, 8 }, packagelen=0x0B
|
||||||
|
0x5B, 0x81, 0x0B, 0x44, 0x42, 0x47, 0x30, 0x01,
|
||||||
|
0x44, 0x42, 0x47, 0x42, 0x08,
|
||||||
|
};
|
||||||
|
|
||||||
|
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
||||||
|
defer arena.deinit();
|
||||||
|
var result = try parse(arena.allocator(), &.{&blob});
|
||||||
|
|
||||||
|
// Integrity: the parser consumed exactly the whole blob (no desync).
|
||||||
|
try std.testing.expectEqual(blob.len, result.consumed);
|
||||||
|
try std.testing.expectEqual(blob.len, result.total);
|
||||||
|
|
||||||
|
const namespace = &result.namespace;
|
||||||
|
|
||||||
|
// Expected top-level nodes.
|
||||||
|
const sb = namespace.resolve(namespace.root, false, 0, &.{.{ '_', 'S', 'B', '_' }}) orelse return error.NoSB;
|
||||||
|
try std.testing.expectEqual(NodeKind.scope, sb.kind);
|
||||||
|
const pci0 = namespace.resolve(sb, false, 0, &.{.{ 'P', 'C', 'I', '0' }}) orelse return error.NoPCI0;
|
||||||
|
try std.testing.expectEqual(NodeKind.device, pci0.kind);
|
||||||
|
_ = namespace.resolve(pci0, false, 0, &.{.{ '_', 'H', 'I', 'D' }}) orelse return error.NoHID;
|
||||||
|
|
||||||
|
// The 1-arg method's arg count was parsed from its flags byte.
|
||||||
|
const mthd = namespace.resolve(pci0, false, 0, &.{.{ 'M', 'T', 'H', 'D' }}) orelse return error.NoMTHD;
|
||||||
|
try std.testing.expectEqual(NodeKind.method, mthd.kind);
|
||||||
|
try std.testing.expectEqual(@as(u8, 1), mthd.arg_count);
|
||||||
|
|
||||||
|
// OperationRegion and the Field unit made it into the namespace.
|
||||||
|
_ = namespace.resolve(namespace.root, false, 0, &.{.{ 'D', 'B', 'G', '0' }}) orelse return error.NoRegion;
|
||||||
|
_ = namespace.resolve(namespace.root, false, 0, &.{.{ 'D', 'B', 'G', 'B' }}) orelse return error.NoField;
|
||||||
|
|
||||||
|
// The sleep package decoded.
|
||||||
|
const s5 = sleepState(namespace, 5) orelse return error.NoS5;
|
||||||
|
try std.testing.expectEqual(@as(u8, 5), s5.slp_typ_a);
|
||||||
|
try std.testing.expectEqual(@as(u8, 0), s5.slp_typ_b);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn noMap(physical: u64, _: u64, _: bool) u64 {
|
||||||
|
return physical;
|
||||||
|
}
|
||||||
|
fn noRead(_: u8, _: u16) u32 {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
fn noWrite(_: u8, _: u16, _: u32) void {}
|
||||||
|
|
||||||
|
test "interpreter runs a method with args, arithmetic, and control flow" {
|
||||||
|
// Method(TST_, 1) {
|
||||||
|
// Store(Arg0, Local0); Add(Local0, 5, Local0)
|
||||||
|
// If (LGreater(Local0, 10)) { Return(One) }
|
||||||
|
// Return(Zero)
|
||||||
|
// }
|
||||||
|
const blob = [_]u8{
|
||||||
|
0x14, 0x18, 0x54, 0x53, 0x54, 0x5F, 0x01, // Method TST_, 1 arg
|
||||||
|
0x70, 0x68, 0x60, // Store(Arg0, Local0)
|
||||||
|
0x72, 0x60, 0x0A, 0x05, 0x60, // Add(Local0, 5, Local0)
|
||||||
|
0xA0, 0x07, 0x94, 0x60, 0x0A, 0x0A, 0xA4, 0x01, // If(LGreater(Local0,10)) { Return(One) }
|
||||||
|
0xA4, 0x00, // Return(Zero)
|
||||||
|
};
|
||||||
|
|
||||||
|
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
||||||
|
defer arena.deinit();
|
||||||
|
var result = try parse(arena.allocator(), &.{&blob});
|
||||||
|
const namespace = &result.namespace;
|
||||||
|
const tst = namespace.resolve(namespace.root, false, 0, &.{.{ 'T', 'S', 'T', '_' }}) orelse return error.NoMethod;
|
||||||
|
|
||||||
|
var interpreter = Interpreter.init(namespace, .{ .mapMmio = noMap, .pioRead = noRead, .pioWrite = noWrite }, arena.allocator());
|
||||||
|
|
||||||
|
const hi = try interpreter.evaluate(tst, &.{.{ .integer = 7 }}); // 7+5=12 > 10 -> 1
|
||||||
|
try std.testing.expectEqual(@as(u64, 1), try hi.asInteger());
|
||||||
|
const lo = try interpreter.evaluate(tst, &.{.{ .integer = 2 }}); // 2+5=7 !> 10 -> 0
|
||||||
|
try std.testing.expectEqual(@as(u64, 0), try lo.asInteger());
|
||||||
|
}
|
||||||
|
|
||||||
|
test "interpreter records Notify(device, code)" {
|
||||||
|
// Device(DEV_) { Name(_HID, 0x030AD041) } // PNP0A03-ish placeholder
|
||||||
|
// Method(TST_, 0) { Notify(DEV_, 0x80); Return(Zero) }
|
||||||
|
// Encoded: a Device holding a Name, then a Method issuing Notify on it.
|
||||||
|
const blob = [_]u8{
|
||||||
|
0x5B, 0x82, 0x0F, 0x44, 0x45, 0x56, 0x5F, // Device(DEV_) len=0x0F (pkglen + DEV_ + Name)
|
||||||
|
0x08, 0x5F, 0x48, 0x49, 0x44, 0x0C, 0x41, 0xD0, 0x0A, 0x03, // Name(_HID, DWord 0x030AD041)
|
||||||
|
0x14, 0x0F, 0x54, 0x53, 0x54, 0x5F, 0x00, // Method(TST_, 0) len=0x0F (pkglen + TST_ + flags + body)
|
||||||
|
0x86, 0x44, 0x45, 0x56, 0x5F, 0x0A, 0x80, // Notify(DEV_, 0x80)
|
||||||
|
0xA4, 0x00, // Return(Zero)
|
||||||
|
};
|
||||||
|
|
||||||
|
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
||||||
|
defer arena.deinit();
|
||||||
|
var result = try parse(arena.allocator(), &.{&blob});
|
||||||
|
const namespace = &result.namespace;
|
||||||
|
const tst = namespace.resolve(namespace.root, false, 0, &.{.{ 'T', 'S', 'T', '_' }}) orelse return error.NoMethod;
|
||||||
|
const dev = namespace.resolve(namespace.root, false, 0, &.{.{ 'D', 'E', 'V', '_' }}) orelse return error.NoDevice;
|
||||||
|
|
||||||
|
var interpreter = Interpreter.init(namespace, .{ .mapMmio = noMap, .pioRead = noRead, .pioWrite = noWrite }, arena.allocator());
|
||||||
|
_ = try interpreter.evaluate(tst, &.{});
|
||||||
|
|
||||||
|
const events = interpreter.takeNotifications();
|
||||||
|
try std.testing.expectEqual(@as(usize, 1), events.len);
|
||||||
|
try std.testing.expectEqual(dev, events[0].node);
|
||||||
|
try std.testing.expectEqual(@as(u64, 0x80), events[0].code);
|
||||||
|
}
|
||||||
@@ -0,0 +1,777 @@
|
|||||||
|
//! A tree-walking AML interpreter — the evaluation stage on top of the parser's
|
||||||
|
//! structural namespace. It executes control methods (their bodies captured by
|
||||||
|
//! the parser) far enough to serve device discovery: device status (`_STA`, is a
|
||||||
|
//! device present), current resource settings (`_CRS`), and the operators, control
|
||||||
|
//! flow, locals/args, and
|
||||||
|
//! OperationRegion field access those methods reach for.
|
||||||
|
//!
|
||||||
|
//! Scope: integers, buffers, strings, packages, and references; If/Else/While/
|
||||||
|
//! Return; the arithmetic/logic operators; method invocation; Name/Local/Arg
|
||||||
|
//! access; CreateField buffer patching (the common current-resource-settings
|
||||||
|
//! (`_CRS`) idiom); and field
|
||||||
|
//! reads/writes against SystemMemory and SystemIO regions. Opcodes outside this
|
||||||
|
//! set return `error.Unsupported`, which callers treat as "couldn't evaluate" and
|
||||||
|
//! fall back — never a hard failure.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const opcode = @import("opcodes.zig");
|
||||||
|
const Node = @import("namespace.zig").Node;
|
||||||
|
const Namespace = @import("namespace.zig").Namespace;
|
||||||
|
|
||||||
|
/// Injected hardware access for OperationRegion reads/writes (the architecture VMM + pio).
|
||||||
|
pub const Hal = struct {
|
||||||
|
mapMmio: *const fn (physical: u64, len: u64, writable: bool) u64,
|
||||||
|
pioRead: *const fn (width: u8, port: u16) u32,
|
||||||
|
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const Error = error{ Unsupported, Truncated, DivByZero } || std.mem.Allocator.Error;
|
||||||
|
|
||||||
|
/// A runtime AML value.
|
||||||
|
pub const Object = union(enum) {
|
||||||
|
uninitialized,
|
||||||
|
integer: u64,
|
||||||
|
buffer: []u8,
|
||||||
|
string: []u8,
|
||||||
|
package: []Object,
|
||||||
|
reference: *Node,
|
||||||
|
|
||||||
|
pub fn asInteger(self: Object) Error!u64 {
|
||||||
|
return switch (self) {
|
||||||
|
.integer => |v| v,
|
||||||
|
.buffer => |b| blk: {
|
||||||
|
var v: u64 = 0;
|
||||||
|
for (b, 0..) |byte, i| {
|
||||||
|
if (i >= 8) break;
|
||||||
|
v |= @as(u64, byte) << @intCast(i * 8);
|
||||||
|
}
|
||||||
|
break :blk v;
|
||||||
|
},
|
||||||
|
else => error.Unsupported,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
const maximum_segments = 16;
|
||||||
|
const NamePath = struct {
|
||||||
|
rooted: bool = false,
|
||||||
|
parents: u8 = 0,
|
||||||
|
segments: [maximum_segments][4]u8 = undefined,
|
||||||
|
count: usize = 0,
|
||||||
|
fn slice(self: *const NamePath) []const [4]u8 {
|
||||||
|
return self.segments[0..self.count];
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
const Cursor = struct {
|
||||||
|
b: []const u8,
|
||||||
|
i: usize = 0,
|
||||||
|
|
||||||
|
fn eof(self: *Cursor) bool {
|
||||||
|
return self.i >= self.b.len;
|
||||||
|
}
|
||||||
|
fn peek(self: *Cursor) ?u8 {
|
||||||
|
return if (self.eof()) null else self.b[self.i];
|
||||||
|
}
|
||||||
|
fn byte(self: *Cursor) Error!u8 {
|
||||||
|
if (self.eof()) return error.Truncated;
|
||||||
|
const v = self.b[self.i];
|
||||||
|
self.i += 1;
|
||||||
|
return v;
|
||||||
|
}
|
||||||
|
fn take(self: *Cursor, n: usize) Error![]const u8 {
|
||||||
|
if (self.i + n > self.b.len) return error.Truncated;
|
||||||
|
const s = self.b[self.i .. self.i + n];
|
||||||
|
self.i += n;
|
||||||
|
return s;
|
||||||
|
}
|
||||||
|
fn packageLength(self: *Cursor) Error!usize {
|
||||||
|
const lead = try self.byte();
|
||||||
|
const follow: usize = lead >> 6;
|
||||||
|
if (follow == 0) return lead & 0x3F;
|
||||||
|
var value: usize = lead & 0x0F;
|
||||||
|
var k: usize = 0;
|
||||||
|
while (k < follow) : (k += 1) value |= @as(usize, try self.byte()) << @intCast(4 + k * 8);
|
||||||
|
return value;
|
||||||
|
}
|
||||||
|
fn nameString(self: *Cursor) Error!NamePath {
|
||||||
|
var name_path = NamePath{};
|
||||||
|
if (self.peek() == opcode.root_char) {
|
||||||
|
name_path.rooted = true;
|
||||||
|
self.i += 1;
|
||||||
|
} else {
|
||||||
|
while (self.peek() == opcode.parent_prefix_char) : (self.i += 1) name_path.parents += 1;
|
||||||
|
}
|
||||||
|
const lead = self.peek() orelse return name_path;
|
||||||
|
switch (lead) {
|
||||||
|
0x00 => self.i += 1,
|
||||||
|
opcode.dual_name_prefix => {
|
||||||
|
self.i += 1;
|
||||||
|
try self.segment(&name_path);
|
||||||
|
try self.segment(&name_path);
|
||||||
|
},
|
||||||
|
opcode.multi_name_prefix => {
|
||||||
|
self.i += 1;
|
||||||
|
const count = try self.byte();
|
||||||
|
var k: usize = 0;
|
||||||
|
while (k < count) : (k += 1) try self.segment(&name_path);
|
||||||
|
},
|
||||||
|
else => try self.segment(&name_path),
|
||||||
|
}
|
||||||
|
return name_path;
|
||||||
|
}
|
||||||
|
fn segment(self: *Cursor, name_path: *NamePath) Error!void {
|
||||||
|
const s = try self.take(4);
|
||||||
|
if (name_path.count < maximum_segments) {
|
||||||
|
name_path.segments[name_path.count] = s[0..4].*;
|
||||||
|
name_path.count += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
const Frame = struct {
|
||||||
|
args: [7]Object = .{.uninitialized} ** 7,
|
||||||
|
locals: [8]Object = .{.uninitialized} ** 8,
|
||||||
|
scope: *Node,
|
||||||
|
ret: Object = .uninitialized,
|
||||||
|
returned: bool = false,
|
||||||
|
broke: bool = false,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A CreateField binding: a name that indexes into a buffer object.
|
||||||
|
const BufferField = struct { buffer: *Node, byte_off: usize, bit_width: u32 };
|
||||||
|
|
||||||
|
/// One Notify(device, code) the interpreter executed.
|
||||||
|
pub const NotifyEvent = struct { node: *Node, code: u64 };
|
||||||
|
|
||||||
|
pub const Interpreter = struct {
|
||||||
|
namespace: *Namespace,
|
||||||
|
hal: Hal,
|
||||||
|
arena: std.mem.Allocator,
|
||||||
|
/// Runtime object overrides for Name nodes (Store targets, patched buffers).
|
||||||
|
dynamic_overrides: std.AutoHashMapUnmanaged(*Node, Object) = .{},
|
||||||
|
/// CreateField bindings active for the current evaluation.
|
||||||
|
fields: std.AutoHashMapUnmanaged(*Node, BufferField) = .{},
|
||||||
|
/// Notify(device, code) operations the last evaluation executed — a GPE or
|
||||||
|
/// EC handler tells the OS "look at this device" this way. Bounded; the
|
||||||
|
/// caller drains it with `takeNotifications` after `evaluate` (M21).
|
||||||
|
notify_queue: [16]NotifyEvent = undefined,
|
||||||
|
notify_count: usize = 0,
|
||||||
|
|
||||||
|
pub fn init(namespace: *Namespace, hal: Hal, arena: std.mem.Allocator) Interpreter {
|
||||||
|
return .{ .namespace = namespace, .hal = hal, .arena = arena };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Evaluate a namespace object: invoke a Method, read a Name's value, or read a
|
||||||
|
/// Field. Resets per-evaluation runtime state first.
|
||||||
|
pub fn evaluate(self: *Interpreter, node: *Node, args: []const Object) Error!Object {
|
||||||
|
self.notify_count = 0;
|
||||||
|
self.dynamic_overrides.clearRetainingCapacity();
|
||||||
|
self.fields.clearRetainingCapacity();
|
||||||
|
return self.invoke(node, args);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn invoke(self: *Interpreter, node: *Node, args: []const Object) Error!Object {
|
||||||
|
switch (node.kind) {
|
||||||
|
.method => {
|
||||||
|
var frame = Frame{ .scope = node };
|
||||||
|
for (args, 0..) |a, i| {
|
||||||
|
if (i < frame.args.len) frame.args[i] = a;
|
||||||
|
}
|
||||||
|
var current = Cursor{ .b = node.value };
|
||||||
|
try self.executeList(¤t, &frame);
|
||||||
|
return frame.ret;
|
||||||
|
},
|
||||||
|
.name => {
|
||||||
|
if (self.dynamic_overrides.get(node)) |o| return o;
|
||||||
|
var current = Cursor{ .b = node.value };
|
||||||
|
var frame = Frame{ .scope = node.parent orelse self.namespace.root };
|
||||||
|
return self.term(¤t, &frame);
|
||||||
|
},
|
||||||
|
.field => return .{ .integer = try self.readField(node) },
|
||||||
|
else => return .{ .reference = node },
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Execute a TermList until it ends or the frame returns/breaks.
|
||||||
|
fn executeList(self: *Interpreter, current: *Cursor, frame: *Frame) Error!void {
|
||||||
|
while (!current.eof() and !frame.returned and !frame.broke) {
|
||||||
|
_ = try self.term(current, frame);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Evaluate/execute one term, returning its value (`.uninitialized` for pure
|
||||||
|
/// statements).
|
||||||
|
fn term(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const lead = current.peek() orelse return error.Truncated;
|
||||||
|
if (isNameStart(lead)) return self.nameReference(current, frame);
|
||||||
|
_ = try current.byte();
|
||||||
|
|
||||||
|
return switch (lead) {
|
||||||
|
opcode.zero_opcode => Object{ .integer = 0 },
|
||||||
|
opcode.one_opcode => Object{ .integer = 1 },
|
||||||
|
opcode.ones_opcode => Object{ .integer = ~@as(u64, 0) },
|
||||||
|
opcode.byte_prefix => Object{ .integer = try self.readConstant(current, 1) },
|
||||||
|
opcode.word_prefix => Object{ .integer = try self.readConstant(current, 2) },
|
||||||
|
opcode.dword_prefix => Object{ .integer = try self.readConstant(current, 4) },
|
||||||
|
opcode.qword_prefix => Object{ .integer = try self.readConstant(current, 8) },
|
||||||
|
opcode.string_prefix => try self.readString(current),
|
||||||
|
opcode.buffer_opcode => try self.buffer(current, frame),
|
||||||
|
opcode.package_opcode, opcode.var_package_opcode => try self.package(current, frame, lead == opcode.var_package_opcode),
|
||||||
|
|
||||||
|
opcode.local0_opcode...opcode.local7_opcode => frame.locals[lead - opcode.local0_opcode],
|
||||||
|
opcode.arg0_opcode...opcode.arg6_opcode => frame.args[lead - opcode.arg0_opcode],
|
||||||
|
|
||||||
|
opcode.return_opcode => blk: {
|
||||||
|
frame.ret = try self.term(current, frame);
|
||||||
|
frame.returned = true;
|
||||||
|
break :blk .uninitialized;
|
||||||
|
},
|
||||||
|
opcode.break_opcode => blk: {
|
||||||
|
frame.broke = true;
|
||||||
|
break :blk .uninitialized;
|
||||||
|
},
|
||||||
|
opcode.continue_opcode, opcode.noop_opcode => .uninitialized,
|
||||||
|
|
||||||
|
opcode.if_opcode => try self.ifElse(current, frame),
|
||||||
|
opcode.while_opcode => try self.whileLoop(current, frame),
|
||||||
|
opcode.store_opcode => try self.store(current, frame),
|
||||||
|
opcode.increment_opcode => try self.incDec(current, frame, 1),
|
||||||
|
opcode.decrement_opcode => try self.incDec(current, frame, -1),
|
||||||
|
|
||||||
|
opcode.add_opcode => try self.binary(current, frame, .add),
|
||||||
|
opcode.subtract_opcode => try self.binary(current, frame, .sub),
|
||||||
|
opcode.multiply_opcode => try self.binary(current, frame, .mul),
|
||||||
|
opcode.mod_opcode => try self.binary(current, frame, .mod),
|
||||||
|
opcode.and_opcode => try self.binary(current, frame, .band),
|
||||||
|
opcode.or_opcode => try self.binary(current, frame, .bor),
|
||||||
|
opcode.xor_opcode => try self.binary(current, frame, .bxor),
|
||||||
|
opcode.nand_opcode => try self.binary(current, frame, .nand),
|
||||||
|
opcode.nor_opcode => try self.binary(current, frame, .nor),
|
||||||
|
opcode.shift_left_opcode => try self.binary(current, frame, .shl),
|
||||||
|
opcode.shift_right_opcode => try self.binary(current, frame, .shr),
|
||||||
|
opcode.divide_opcode => try self.divide(current, frame),
|
||||||
|
|
||||||
|
opcode.land_opcode => try self.logic2(current, frame, .land),
|
||||||
|
opcode.lor_opcode => try self.logic2(current, frame, .lor),
|
||||||
|
opcode.lequal_opcode => try self.logic2(current, frame, .eq),
|
||||||
|
opcode.lgreater_opcode => try self.logic2(current, frame, .gt),
|
||||||
|
opcode.lless_opcode => try self.logic2(current, frame, .lt),
|
||||||
|
opcode.lnot_opcode => try self.lnot(current, frame),
|
||||||
|
|
||||||
|
opcode.not_opcode => blk: {
|
||||||
|
const v = try self.evaluateInteger(current, frame);
|
||||||
|
const r = ~v;
|
||||||
|
try self.storeTarget(current, frame, .{ .integer = r });
|
||||||
|
break :blk .{ .integer = r };
|
||||||
|
},
|
||||||
|
|
||||||
|
opcode.size_of_opcode => try self.sizeOf(current, frame),
|
||||||
|
opcode.index_opcode => try self.index(current, frame),
|
||||||
|
opcode.dereference_of_opcode => try self.dereferenceOf(current, frame),
|
||||||
|
opcode.to_integer_opcode => blk: {
|
||||||
|
const v = try self.evaluateInteger(current, frame);
|
||||||
|
try self.storeTarget(current, frame, .{ .integer = v });
|
||||||
|
break :blk .{ .integer = v };
|
||||||
|
},
|
||||||
|
opcode.to_buffer_opcode => try self.passThroughUnary(current, frame),
|
||||||
|
|
||||||
|
opcode.notify_opcode => try self.notify(current, frame),
|
||||||
|
|
||||||
|
opcode.extended_opcode_prefix => try self.ext(current, frame),
|
||||||
|
|
||||||
|
// CreateXField: source, index, name (bit widths differ by op)
|
||||||
|
opcode.create_bit_field_opcode => try self.createField(current, frame, 1),
|
||||||
|
opcode.create_byte_field_opcode => try self.createField(current, frame, 8),
|
||||||
|
opcode.create_word_field_opcode => try self.createField(current, frame, 16),
|
||||||
|
opcode.create_dword_field_opcode => try self.createField(current, frame, 32),
|
||||||
|
opcode.create_qword_field_opcode => try self.createField(current, frame, 64),
|
||||||
|
|
||||||
|
else => error.Unsupported,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- name references ----------------------------------------------------
|
||||||
|
|
||||||
|
fn nameReference(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const name_path = try current.nameString();
|
||||||
|
const node = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice()) orelse
|
||||||
|
return .uninitialized; // unknown name -> treat as uninitialised
|
||||||
|
switch (node.kind) {
|
||||||
|
.method => {
|
||||||
|
var argbuf: [7]Object = undefined;
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < node.arg_count and i < argbuf.len) : (i += 1) argbuf[i] = try self.term(current, frame);
|
||||||
|
return self.invoke(node, argbuf[0..@min(node.arg_count, argbuf.len)]);
|
||||||
|
},
|
||||||
|
.field => return .{ .integer = try self.readField(node) },
|
||||||
|
.name => return self.invoke(node, &.{}),
|
||||||
|
else => return .{ .reference = node },
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- data objects -------------------------------------------------------
|
||||||
|
|
||||||
|
fn readConstant(self: *Interpreter, current: *Cursor, n: usize) Error!u64 {
|
||||||
|
_ = self;
|
||||||
|
const bytes = try current.take(n);
|
||||||
|
var v: u64 = 0;
|
||||||
|
for (bytes, 0..) |b, i| v |= @as(u64, b) << @intCast(i * 8);
|
||||||
|
return v;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn readString(self: *Interpreter, current: *Cursor) Error!Object {
|
||||||
|
const start = current.i;
|
||||||
|
while (current.peek()) |c| {
|
||||||
|
current.i += 1;
|
||||||
|
if (c == 0) break;
|
||||||
|
}
|
||||||
|
const raw = current.b[start .. current.i - 1];
|
||||||
|
const s = try self.arena.dupe(u8, raw);
|
||||||
|
return .{ .string = s };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn buffer(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const start = current.i;
|
||||||
|
const len = try current.packageLength();
|
||||||
|
const end = @min(start + len, current.b.len);
|
||||||
|
const size = try self.evaluateInteger(current, frame);
|
||||||
|
const data = current.b[@min(current.i, end)..end];
|
||||||
|
const bytes = try self.arena.alloc(u8, @intCast(size));
|
||||||
|
@memset(bytes, 0);
|
||||||
|
@memcpy(bytes[0..@min(bytes.len, data.len)], data[0..@min(bytes.len, data.len)]);
|
||||||
|
current.i = end;
|
||||||
|
return .{ .buffer = bytes };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn package(self: *Interpreter, current: *Cursor, frame: *Frame, variable: bool) Error!Object {
|
||||||
|
const start = current.i;
|
||||||
|
const len = try current.packageLength();
|
||||||
|
const end = @min(start + len, current.b.len);
|
||||||
|
const count: usize = if (variable) @intCast(try self.evaluateInteger(current, frame)) else try current.byte();
|
||||||
|
const elems = try self.arena.alloc(Object, count);
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < count and current.i < end) : (i += 1) elems[i] = try self.term(current, frame);
|
||||||
|
while (i < count) : (i += 1) elems[i] = .uninitialized;
|
||||||
|
current.i = end;
|
||||||
|
return .{ .package = elems };
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- operators ----------------------------------------------------------
|
||||||
|
|
||||||
|
const BinaryOperation = enum { add, sub, mul, mod, band, bor, bxor, nand, nor, shl, shr };
|
||||||
|
|
||||||
|
fn binary(self: *Interpreter, current: *Cursor, frame: *Frame, kind: BinaryOperation) Error!Object {
|
||||||
|
const a = try self.evaluateInteger(current, frame);
|
||||||
|
const b = try self.evaluateInteger(current, frame);
|
||||||
|
const r: u64 = switch (kind) {
|
||||||
|
.add => a +% b,
|
||||||
|
.sub => a -% b,
|
||||||
|
.mul => a *% b,
|
||||||
|
.mod => if (b == 0) return error.DivByZero else a % b,
|
||||||
|
.band => a & b,
|
||||||
|
.bor => a | b,
|
||||||
|
.bxor => a ^ b,
|
||||||
|
.nand => ~(a & b),
|
||||||
|
.nor => ~(a | b),
|
||||||
|
.shl => if (b >= 64) 0 else a << @intCast(b),
|
||||||
|
.shr => if (b >= 64) 0 else a >> @intCast(b),
|
||||||
|
};
|
||||||
|
try self.storeTarget(current, frame, .{ .integer = r });
|
||||||
|
return .{ .integer = r };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn divide(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const a = try self.evaluateInteger(current, frame);
|
||||||
|
const b = try self.evaluateInteger(current, frame);
|
||||||
|
if (b == 0) return error.DivByZero;
|
||||||
|
try self.storeTarget(current, frame, .{ .integer = a % b }); // remainder target
|
||||||
|
try self.storeTarget(current, frame, .{ .integer = a / b }); // quotient target
|
||||||
|
return .{ .integer = a / b };
|
||||||
|
}
|
||||||
|
|
||||||
|
const LogicOperation = enum { land, lor, eq, gt, lt };
|
||||||
|
|
||||||
|
fn logic2(self: *Interpreter, current: *Cursor, frame: *Frame, kind: LogicOperation) Error!Object {
|
||||||
|
const a = try self.evaluateInteger(current, frame);
|
||||||
|
const b = try self.evaluateInteger(current, frame);
|
||||||
|
const r = switch (kind) {
|
||||||
|
.land => a != 0 and b != 0,
|
||||||
|
.lor => a != 0 or b != 0,
|
||||||
|
.eq => a == b,
|
||||||
|
.gt => a > b,
|
||||||
|
.lt => a < b,
|
||||||
|
};
|
||||||
|
return .{ .integer = if (r) ~@as(u64, 0) else 0 };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn lnot(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
// 0x92 0x93/94/95 are the compound comparisons.
|
||||||
|
const b = current.peek() orelse return error.Truncated;
|
||||||
|
switch (b) {
|
||||||
|
opcode.lnot.not_equal => {
|
||||||
|
current.i += 1;
|
||||||
|
const x = try self.evaluateInteger(current, frame);
|
||||||
|
const y = try self.evaluateInteger(current, frame);
|
||||||
|
return .{ .integer = if (x != y) ~@as(u64, 0) else 0 };
|
||||||
|
},
|
||||||
|
opcode.lnot.less_equal => {
|
||||||
|
current.i += 1;
|
||||||
|
const x = try self.evaluateInteger(current, frame);
|
||||||
|
const y = try self.evaluateInteger(current, frame);
|
||||||
|
return .{ .integer = if (x <= y) ~@as(u64, 0) else 0 };
|
||||||
|
},
|
||||||
|
opcode.lnot.greater_equal => {
|
||||||
|
current.i += 1;
|
||||||
|
const x = try self.evaluateInteger(current, frame);
|
||||||
|
const y = try self.evaluateInteger(current, frame);
|
||||||
|
return .{ .integer = if (x >= y) ~@as(u64, 0) else 0 };
|
||||||
|
},
|
||||||
|
else => {
|
||||||
|
const x = try self.evaluateInteger(current, frame);
|
||||||
|
return .{ .integer = if (x == 0) ~@as(u64, 0) else 0 };
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn incDec(self: *Interpreter, current: *Cursor, frame: *Frame, delta: i64) Error!Object {
|
||||||
|
// Operand is a SuperName that is both read and written.
|
||||||
|
const save = current.i;
|
||||||
|
const current_value = try self.term(current, frame);
|
||||||
|
const v = try current_value.asInteger();
|
||||||
|
const r = if (delta > 0) v +% 1 else v -% 1;
|
||||||
|
var tcur = Cursor{ .b = current.b, .i = save };
|
||||||
|
try self.storeInto(&tcur, frame, .{ .integer = r });
|
||||||
|
return .{ .integer = r };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn sizeOf(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const o = try self.term(current, frame);
|
||||||
|
return .{ .integer = switch (o) {
|
||||||
|
.buffer => |b| b.len,
|
||||||
|
.string => |s| s.len,
|
||||||
|
.package => |p| p.len,
|
||||||
|
else => 0,
|
||||||
|
} };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn passThroughUnary(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const o = try self.term(current, frame);
|
||||||
|
try self.storeTarget(current, frame, o);
|
||||||
|
return o;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn index(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const source = try self.term(current, frame);
|
||||||
|
const element_index: usize = @intCast(try self.evaluateInteger(current, frame));
|
||||||
|
// Optional target (a reference); we don't materialise references, so store
|
||||||
|
// the indexed value if a target is present.
|
||||||
|
const value: Object = switch (source) {
|
||||||
|
.buffer => |b| .{ .integer = if (element_index < b.len) b[element_index] else 0 },
|
||||||
|
.package => |p| if (element_index < p.len) p[element_index] else .uninitialized,
|
||||||
|
.string => |s| .{ .integer = if (element_index < s.len) s[element_index] else 0 },
|
||||||
|
else => .uninitialized,
|
||||||
|
};
|
||||||
|
try self.storeTarget(current, frame, value);
|
||||||
|
return value;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn dereferenceOf(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const o = try self.term(current, frame);
|
||||||
|
return switch (o) {
|
||||||
|
.reference => |n| self.invoke(n, &.{}),
|
||||||
|
else => o,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- control flow -------------------------------------------------------
|
||||||
|
|
||||||
|
fn ifElse(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const start = current.i;
|
||||||
|
const end = @min(start + try current.packageLength(), current.b.len);
|
||||||
|
const cond = try self.evaluateInteger(current, frame);
|
||||||
|
if (cond != 0) {
|
||||||
|
var body = Cursor{ .b = current.b[0..end], .i = current.i };
|
||||||
|
try self.executeList(&body, frame);
|
||||||
|
current.i = end;
|
||||||
|
// Skip a trailing Else.
|
||||||
|
if (current.peek() == opcode.else_opcode) {
|
||||||
|
current.i += 1;
|
||||||
|
const es = current.i;
|
||||||
|
const ee = @min(es + try current.packageLength(), current.b.len);
|
||||||
|
current.i = ee;
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
current.i = end;
|
||||||
|
if (current.peek() == opcode.else_opcode) {
|
||||||
|
current.i += 1;
|
||||||
|
const es = current.i;
|
||||||
|
const ee = @min(es + try current.packageLength(), current.b.len);
|
||||||
|
var body = Cursor{ .b = current.b[0..ee], .i = current.i };
|
||||||
|
try self.executeList(&body, frame);
|
||||||
|
current.i = ee;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return .uninitialized;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn whileLoop(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const start = current.i;
|
||||||
|
const end = @min(start + try current.packageLength(), current.b.len);
|
||||||
|
const pred_at = current.i;
|
||||||
|
var guard: usize = 0;
|
||||||
|
while (guard < 100_000) : (guard += 1) {
|
||||||
|
var pc = Cursor{ .b = current.b[0..end], .i = pred_at };
|
||||||
|
const cond = try self.evaluateInteger(&pc, frame);
|
||||||
|
if (cond == 0) break;
|
||||||
|
var body = Cursor{ .b = current.b[0..end], .i = pc.i };
|
||||||
|
try self.executeList(&body, frame);
|
||||||
|
if (frame.returned) break;
|
||||||
|
if (frame.broke) {
|
||||||
|
frame.broke = false;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
current.i = end;
|
||||||
|
return .uninitialized;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- store --------------------------------------------------------------
|
||||||
|
|
||||||
|
fn store(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const value = try self.term(current, frame);
|
||||||
|
try self.storeInto(current, frame, value);
|
||||||
|
return value;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A Store *target* that may be NullName (no store).
|
||||||
|
fn storeTarget(self: *Interpreter, current: *Cursor, frame: *Frame, value: Object) Error!void {
|
||||||
|
if (current.peek() == 0x00) {
|
||||||
|
current.i += 1; // NullName
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
try self.storeInto(current, frame, value);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Notify(SuperName, NotifyValue): resolve the named device, evaluate the
|
||||||
|
/// code, and record the pair for the caller to dispatch. AML control flow
|
||||||
|
/// continues (Notify returns nothing).
|
||||||
|
fn notify(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const lead = current.peek() orelse return error.Truncated;
|
||||||
|
var target: ?*Node = null;
|
||||||
|
if (isNameStart(lead)) {
|
||||||
|
const name_path = try current.nameString();
|
||||||
|
target = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice());
|
||||||
|
} else {
|
||||||
|
// A non-name SuperName (Local/Arg holding a reference).
|
||||||
|
const obj = try self.term(current, frame);
|
||||||
|
if (obj == .reference) target = obj.reference;
|
||||||
|
}
|
||||||
|
const code = try self.evaluateInteger(current, frame);
|
||||||
|
if (target) |node| {
|
||||||
|
if (self.notify_count < self.notify_queue.len) {
|
||||||
|
self.notify_queue[self.notify_count] = .{ .node = node, .code = code };
|
||||||
|
self.notify_count += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return .uninitialized;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The Notify events the last `evaluate` produced. Valid until the next
|
||||||
|
/// `evaluate` clears the queue.
|
||||||
|
pub fn takeNotifications(self: *Interpreter) []const NotifyEvent {
|
||||||
|
return self.notify_queue[0..self.notify_count];
|
||||||
|
}
|
||||||
|
|
||||||
|
fn storeInto(self: *Interpreter, current: *Cursor, frame: *Frame, value: Object) Error!void {
|
||||||
|
const lead = current.peek() orelse return error.Truncated;
|
||||||
|
if (isNameStart(lead)) {
|
||||||
|
const name_path = try current.nameString();
|
||||||
|
const node = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice()) orelse return;
|
||||||
|
if (self.fields.get(node)) |buffer_field| {
|
||||||
|
try self.writeBufferField(buffer_field, try value.asInteger());
|
||||||
|
} else if (node.kind == .field) {
|
||||||
|
try self.writeField(node, try value.asInteger());
|
||||||
|
} else {
|
||||||
|
try self.dynamic_overrides.put(self.arena, node, value);
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
_ = try current.byte();
|
||||||
|
switch (lead) {
|
||||||
|
0x00 => {}, // NullName
|
||||||
|
opcode.local0_opcode...opcode.local7_opcode => frame.locals[lead - opcode.local0_opcode] = value,
|
||||||
|
opcode.arg0_opcode...opcode.arg6_opcode => frame.args[lead - opcode.arg0_opcode] = value,
|
||||||
|
opcode.index_opcode => {
|
||||||
|
const source = try self.term(current, frame);
|
||||||
|
const element_index: usize = @intCast(try self.evaluateInteger(current, frame));
|
||||||
|
switch (source) {
|
||||||
|
.buffer => |b| if (element_index < b.len) {
|
||||||
|
b[element_index] = @truncate(try value.asInteger());
|
||||||
|
},
|
||||||
|
.package => |p| if (element_index < p.len) {
|
||||||
|
p[element_index] = value;
|
||||||
|
},
|
||||||
|
else => {},
|
||||||
|
}
|
||||||
|
},
|
||||||
|
else => return error.Unsupported,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- CreateField (buffer patching) --------------------------------------
|
||||||
|
|
||||||
|
fn createField(self: *Interpreter, current: *Cursor, frame: *Frame, bit_width: u32) Error!Object {
|
||||||
|
const source = try self.term(current, frame); // source buffer (as a reference or value)
|
||||||
|
const bit_index = try self.evaluateInteger(current, frame);
|
||||||
|
const name_path = try current.nameString();
|
||||||
|
const node = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice()) orelse return .uninitialized;
|
||||||
|
|
||||||
|
// Bind the new name to the source buffer's node so stores land in it.
|
||||||
|
const buffer_node: *Node = switch (source) {
|
||||||
|
.reference => |n| n,
|
||||||
|
else => return .uninitialized,
|
||||||
|
};
|
||||||
|
// Materialise the buffer into `dynamic_overrides` so patches persist and are returned.
|
||||||
|
if (self.dynamic_overrides.get(buffer_node) == null) {
|
||||||
|
const value = try self.invoke(buffer_node, &.{});
|
||||||
|
try self.dynamic_overrides.put(self.arena, buffer_node, value);
|
||||||
|
}
|
||||||
|
const byte_off: usize = @intCast(bit_index / 8);
|
||||||
|
try self.fields.put(self.arena, node, .{ .buffer = buffer_node, .byte_off = byte_off, .bit_width = bit_width });
|
||||||
|
return .uninitialized;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn writeBufferField(self: *Interpreter, buffer_field: BufferField, value: u64) Error!void {
|
||||||
|
const obj = self.dynamic_overrides.get(buffer_field.buffer) orelse return;
|
||||||
|
const bytes = switch (obj) {
|
||||||
|
.buffer => |b| b,
|
||||||
|
else => return,
|
||||||
|
};
|
||||||
|
const byte_count = (buffer_field.bit_width + 7) / 8;
|
||||||
|
var k: usize = 0;
|
||||||
|
while (k < byte_count and buffer_field.byte_off + k < bytes.len) : (k += 1) {
|
||||||
|
bytes[buffer_field.byte_off + k] = @truncate(value >> @intCast(k * 8));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- OperationRegion field access ---------------------------------------
|
||||||
|
|
||||||
|
fn readField(self: *Interpreter, field: *Node) Error!u64 {
|
||||||
|
const region = field.region orelse return error.Unsupported;
|
||||||
|
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
||||||
|
const base = try self.regionBase(region);
|
||||||
|
const start_byte = base + field.bit_offset / 8;
|
||||||
|
const shift: u7 = @intCast(field.bit_offset % 8);
|
||||||
|
const total = @as(usize, shift) + field.bit_width;
|
||||||
|
const byte_count = (total + 7) / 8;
|
||||||
|
var raw: u128 = 0;
|
||||||
|
var k: usize = 0;
|
||||||
|
while (k < byte_count) : (k += 1) {
|
||||||
|
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
||||||
|
}
|
||||||
|
const masked = (raw >> shift) & bitMask(field.bit_width);
|
||||||
|
return @truncate(masked);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn writeField(self: *Interpreter, field: *Node, value: u64) Error!void {
|
||||||
|
const region = field.region orelse return error.Unsupported;
|
||||||
|
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
||||||
|
const base = try self.regionBase(region);
|
||||||
|
const start_byte = base + field.bit_offset / 8;
|
||||||
|
const shift: u7 = @intCast(field.bit_offset % 8);
|
||||||
|
const total = @as(usize, shift) + field.bit_width;
|
||||||
|
const byte_count = (total + 7) / 8;
|
||||||
|
// Read-modify-write byte by byte.
|
||||||
|
var raw: u128 = 0;
|
||||||
|
var k: usize = 0;
|
||||||
|
while (k < byte_count) : (k += 1) {
|
||||||
|
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
||||||
|
}
|
||||||
|
const mask = bitMask(field.bit_width) << shift;
|
||||||
|
raw = (raw & ~mask) | ((@as(u128, value) << shift) & mask);
|
||||||
|
k = 0;
|
||||||
|
while (k < byte_count) : (k += 1) {
|
||||||
|
try self.writeRegionByte(region.region_space, start_byte + k, @truncate(raw >> @intCast(k * 8)));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn regionBase(self: *Interpreter, region: *Node) Error!u64 {
|
||||||
|
var current = Cursor{ .b = region.region_offset_aml };
|
||||||
|
var frame = Frame{ .scope = region.parent orelse self.namespace.root };
|
||||||
|
return (try self.term(¤t, &frame)).asInteger();
|
||||||
|
}
|
||||||
|
|
||||||
|
fn readRegionByte(self: *Interpreter, space: u8, address: u64) Error!u8 {
|
||||||
|
switch (space) {
|
||||||
|
0 => { // SystemMemory
|
||||||
|
const virtual = self.hal.mapMmio(address & ~@as(u64, 0xFFF), 0x1000, true);
|
||||||
|
const p: *align(1) const volatile u8 = @ptrFromInt(virtual + (address & 0xFFF));
|
||||||
|
return p.*;
|
||||||
|
},
|
||||||
|
1 => return @truncate(self.hal.pioRead(1, @intCast(address & 0xFFFF))), // SystemIO
|
||||||
|
else => return error.Unsupported,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn writeRegionByte(self: *Interpreter, space: u8, address: u64, value: u8) Error!void {
|
||||||
|
switch (space) {
|
||||||
|
0 => {
|
||||||
|
const virtual = self.hal.mapMmio(address & ~@as(u64, 0xFFF), 0x1000, true);
|
||||||
|
const p: *align(1) volatile u8 = @ptrFromInt(virtual + (address & 0xFFF));
|
||||||
|
p.* = value;
|
||||||
|
},
|
||||||
|
1 => self.hal.pioWrite(1, @intCast(address & 0xFFFF), value),
|
||||||
|
else => return error.Unsupported,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- extended opcodes ---------------------------------------------------
|
||||||
|
|
||||||
|
fn ext(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const e = try current.byte();
|
||||||
|
switch (e) {
|
||||||
|
opcode.extended.debug => return .uninitialized,
|
||||||
|
opcode.extended.revision => return .{ .integer = 2 },
|
||||||
|
opcode.extended.timer => return .{ .integer = 0 },
|
||||||
|
// Mutex/Event ops are no-ops in this single-threaded evaluator.
|
||||||
|
opcode.extended.acquire => {
|
||||||
|
_ = try self.term(current, frame); // mutex SuperName
|
||||||
|
_ = try current.take(2); // timeout
|
||||||
|
return .{ .integer = 0 }; // acquired
|
||||||
|
},
|
||||||
|
opcode.extended.release, opcode.extended.reset, opcode.extended.signal => {
|
||||||
|
_ = try self.term(current, frame);
|
||||||
|
return .uninitialized;
|
||||||
|
},
|
||||||
|
opcode.extended.wait => {
|
||||||
|
_ = try self.term(current, frame);
|
||||||
|
_ = try self.term(current, frame);
|
||||||
|
return .{ .integer = 0 };
|
||||||
|
},
|
||||||
|
opcode.extended.sleep, opcode.extended.stall => {
|
||||||
|
_ = try self.term(current, frame);
|
||||||
|
return .uninitialized;
|
||||||
|
},
|
||||||
|
else => return error.Unsupported,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn evaluateInteger(self: *Interpreter, current: *Cursor, frame: *Frame) Error!u64 {
|
||||||
|
return (try self.term(current, frame)).asInteger();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
fn bitMask(width: u32) u128 {
|
||||||
|
if (width >= 128) return ~@as(u128, 0);
|
||||||
|
return (@as(u128, 1) << @intCast(width)) - 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn isNameStart(b: u8) bool {
|
||||||
|
return (b >= opcode.name_char_start and b <= opcode.name_char_end) or
|
||||||
|
b == opcode.name_char_underscore or
|
||||||
|
b == opcode.root_char or
|
||||||
|
b == opcode.parent_prefix_char or
|
||||||
|
b == opcode.dual_name_prefix or
|
||||||
|
b == opcode.multi_name_prefix;
|
||||||
|
}
|
||||||
@@ -18,7 +18,7 @@ pub const NodeKind = enum {
|
|||||||
mutex,
|
mutex,
|
||||||
event,
|
event,
|
||||||
processor,
|
processor,
|
||||||
power_res,
|
power_resource,
|
||||||
thermal_zone,
|
thermal_zone,
|
||||||
alias,
|
alias,
|
||||||
external,
|
external,
|
||||||
@@ -28,12 +28,12 @@ pub const NodeKind = enum {
|
|||||||
pub const Node = struct {
|
pub const Node = struct {
|
||||||
/// The 4-byte NameSeg identifying this node within its parent. The root uses
|
/// The 4-byte NameSeg identifying this node within its parent. The root uses
|
||||||
/// all-zero.
|
/// all-zero.
|
||||||
seg: [4]u8 = .{ 0, 0, 0, 0 },
|
segment: [4]u8 = .{ 0, 0, 0, 0 },
|
||||||
kind: NodeKind = .other,
|
kind: NodeKind = .other,
|
||||||
/// For Method / External: the declared argument count (0..7). Used to resolve
|
/// For Method / External: the declared argument count (0..7). Used to resolve
|
||||||
/// how many TermArgs a method invocation consumes.
|
/// how many TermArgs a method invocation consumes.
|
||||||
arg_count: u8 = 0,
|
arg_count: u8 = 0,
|
||||||
/// For Name: the AML bytes of its DataRefObject (so a value like a sleep
|
/// For Name: the AML bytes of its DataReferenceObject (so a value like a sleep
|
||||||
/// state's (`_Sx`) Package can be parsed on demand). For Method: the AML bytes of the body,
|
/// state's (`_Sx`) Package can be parsed on demand). For Method: the AML bytes of the body,
|
||||||
/// interpreted on demand by the evaluator. Empty otherwise.
|
/// interpreted on demand by the evaluator. Empty otherwise.
|
||||||
value: []const u8 = &.{},
|
value: []const u8 = &.{},
|
||||||
@@ -78,49 +78,49 @@ pub const Namespace = struct {
|
|||||||
return self.root.subtreeCount();
|
return self.root.subtreeCount();
|
||||||
}
|
}
|
||||||
|
|
||||||
fn findChild(parent: *Node, seg: [4]u8) ?*Node {
|
fn findChild(parent: *Node, segment: [4]u8) ?*Node {
|
||||||
var c = parent.first_child;
|
var c = parent.first_child;
|
||||||
while (c) |child| : (c = child.next_sibling) {
|
while (c) |child| : (c = child.next_sibling) {
|
||||||
if (std.mem.eql(u8, &child.seg, &seg)) return child;
|
if (std.mem.eql(u8, &child.segment, &segment)) return child;
|
||||||
}
|
}
|
||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The direct child of `node` named `seg`, or null. Unlike `resolve`, this does
|
/// The direct child of `node` named `segment`, or null. Unlike `resolve`, this does
|
||||||
/// not apply the search-rule walk-up — it looks only at immediate children (for
|
/// not apply the search-rule walk-up — it looks only at immediate children (for
|
||||||
/// reading a device's own hardware ID (`_HID`) / current resource settings (`_CRS`)).
|
/// reading a device's own hardware ID (`_HID`) / current resource settings (`_CRS`)).
|
||||||
pub fn childOf(node: *Node, seg: [4]u8) ?*Node {
|
pub fn childOf(node: *Node, segment: [4]u8) ?*Node {
|
||||||
return findChild(node, seg);
|
return findChild(node, segment);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn newChild(self: *Namespace, parent: *Node, seg: [4]u8, kind: NodeKind) !*Node {
|
fn newChild(self: *Namespace, parent: *Node, segment: [4]u8, kind: NodeKind) !*Node {
|
||||||
const n = try self.allocator.create(Node);
|
const n = try self.allocator.create(Node);
|
||||||
n.* = .{ .seg = seg, .kind = kind, .parent = parent };
|
n.* = .{ .segment = segment, .kind = kind, .parent = parent };
|
||||||
// Append at the tail so a dump reads in declaration order.
|
// Append at the tail so a dump reads in declaration order.
|
||||||
if (parent.first_child == null) {
|
if (parent.first_child == null) {
|
||||||
parent.first_child = n;
|
parent.first_child = n;
|
||||||
} else {
|
} else {
|
||||||
var cur = parent.first_child.?;
|
var current = parent.first_child.?;
|
||||||
while (cur.next_sibling) |sib| cur = sib;
|
while (current.next_sibling) |sib| current = sib;
|
||||||
cur.next_sibling = n;
|
current.next_sibling = n;
|
||||||
}
|
}
|
||||||
return n;
|
return n;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Create a Field unit node directly under `scope` (field units live in the
|
/// Create a Field unit node directly under `scope` (field units live in the
|
||||||
/// scope of the Field/IndexField/BankField, not under the region).
|
/// scope of the Field/IndexField/BankField, not under the region).
|
||||||
pub fn newFieldUnit(self: *Namespace, scope: *Node, seg: [4]u8) !*Node {
|
pub fn newFieldUnit(self: *Namespace, scope: *Node, segment: [4]u8) !*Node {
|
||||||
return self.findOrCreate(scope, seg, .field);
|
return self.findOrCreate(scope, segment, .field);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn findOrCreate(self: *Namespace, parent: *Node, seg: [4]u8, kind: NodeKind) !*Node {
|
fn findOrCreate(self: *Namespace, parent: *Node, segment: [4]u8, kind: NodeKind) !*Node {
|
||||||
if (findChild(parent, seg)) |existing| {
|
if (findChild(parent, segment)) |existing| {
|
||||||
// Reopening a scope (e.g. Scope(\_SB) after Device \_SB) keeps the more
|
// Reopening a scope (e.g. Scope(\_SB) after Device \_SB) keeps the more
|
||||||
// specific kind rather than downgrading to a plain scope.
|
// specific kind rather than downgrading to a plain scope.
|
||||||
if (existing.kind == .scope and kind != .scope) existing.kind = kind;
|
if (existing.kind == .scope and kind != .scope) existing.kind = kind;
|
||||||
return existing;
|
return existing;
|
||||||
}
|
}
|
||||||
return self.newChild(parent, seg, kind);
|
return self.newChild(parent, segment, kind);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The node a definition's NameString names, creating any intermediate scopes.
|
/// The node a definition's NameString names, creating any intermediate scopes.
|
||||||
@@ -131,16 +131,16 @@ pub const Namespace = struct {
|
|||||||
current: *Node,
|
current: *Node,
|
||||||
rooted: bool,
|
rooted: bool,
|
||||||
parents: u8,
|
parents: u8,
|
||||||
segs: []const [4]u8,
|
segments: []const [4]u8,
|
||||||
kind: NodeKind,
|
kind: NodeKind,
|
||||||
) !*Node {
|
) !*Node {
|
||||||
var base = startNode(self, current, rooted, parents);
|
var base = startNode(self, current, rooted, parents);
|
||||||
if (segs.len == 0) return base;
|
if (segments.len == 0) return base;
|
||||||
var i: usize = 0;
|
var i: usize = 0;
|
||||||
while (i + 1 < segs.len) : (i += 1) {
|
while (i + 1 < segments.len) : (i += 1) {
|
||||||
base = try self.findOrCreate(base, segs[i], .scope);
|
base = try self.findOrCreate(base, segments[i], .scope);
|
||||||
}
|
}
|
||||||
return self.findOrCreate(base, segs[segs.len - 1], kind);
|
return self.findOrCreate(base, segments[segments.len - 1], kind);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Resolve a NameString *reference* to an existing node, or null. A single
|
/// Resolve a NameString *reference* to an existing node, or null. A single
|
||||||
@@ -151,22 +151,22 @@ pub const Namespace = struct {
|
|||||||
current: *Node,
|
current: *Node,
|
||||||
rooted: bool,
|
rooted: bool,
|
||||||
parents: u8,
|
parents: u8,
|
||||||
segs: []const [4]u8,
|
segments: []const [4]u8,
|
||||||
) ?*Node {
|
) ?*Node {
|
||||||
if (segs.len == 0) return null;
|
if (segments.len == 0) return null;
|
||||||
|
|
||||||
if (!rooted and parents == 0 and segs.len == 1) {
|
if (!rooted and parents == 0 and segments.len == 1) {
|
||||||
// Search rule: this scope, then each ancestor up to the root.
|
// Search rule: this scope, then each ancestor up to the root.
|
||||||
var scope: ?*Node = current;
|
var scope: ?*Node = current;
|
||||||
while (scope) |s| : (scope = s.parent) {
|
while (scope) |s| : (scope = s.parent) {
|
||||||
if (findChild(s, segs[0])) |n| return n;
|
if (findChild(s, segments[0])) |n| return n;
|
||||||
}
|
}
|
||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
|
||||||
var base = startNode(self, current, rooted, parents);
|
var base = startNode(self, current, rooted, parents);
|
||||||
for (segs) |seg| {
|
for (segments) |segment| {
|
||||||
base = findChild(base, seg) orelse return null;
|
base = findChild(base, segment) orelse return null;
|
||||||
}
|
}
|
||||||
return base;
|
return base;
|
||||||
}
|
}
|
||||||
@@ -0,0 +1,137 @@
|
|||||||
|
//! AML opcode constants — the full ACPI Machine Language opcode table.
|
||||||
|
//!
|
||||||
|
//! Single-byte opcodes are plain values. Extended opcodes are a two-byte sequence
|
||||||
|
//! `ext_prefix` (0x5B) followed by a byte listed under `ext`. A few comparison
|
||||||
|
//! opcodes are `lnot_opcode` (0x92) followed by a second byte (see `lnot`).
|
||||||
|
|
||||||
|
// --- name / path characters -------------------------------------------------
|
||||||
|
pub const zero_opcode = 0x00;
|
||||||
|
pub const one_opcode = 0x01;
|
||||||
|
pub const alias_opcode = 0x06;
|
||||||
|
pub const name_opcode = 0x08;
|
||||||
|
pub const byte_prefix = 0x0A;
|
||||||
|
pub const word_prefix = 0x0B;
|
||||||
|
pub const dword_prefix = 0x0C;
|
||||||
|
pub const string_prefix = 0x0D;
|
||||||
|
pub const qword_prefix = 0x0E;
|
||||||
|
pub const scope_opcode = 0x10;
|
||||||
|
pub const buffer_opcode = 0x11;
|
||||||
|
pub const package_opcode = 0x12;
|
||||||
|
pub const var_package_opcode = 0x13;
|
||||||
|
pub const method_opcode = 0x14;
|
||||||
|
pub const external_opcode = 0x15;
|
||||||
|
|
||||||
|
pub const dual_name_prefix = 0x2E;
|
||||||
|
pub const multi_name_prefix = 0x2F;
|
||||||
|
pub const extended_opcode_prefix = 0x5B;
|
||||||
|
pub const root_char = 0x5C;
|
||||||
|
pub const parent_prefix_char = 0x5E;
|
||||||
|
pub const name_char_underscore = 0x5F;
|
||||||
|
|
||||||
|
pub const digit_char_start = 0x30;
|
||||||
|
pub const digit_char_end = 0x39;
|
||||||
|
pub const name_char_start = 0x41; // 'A'
|
||||||
|
pub const name_char_end = 0x5A; // 'Z'
|
||||||
|
|
||||||
|
// --- locals / args ----------------------------------------------------------
|
||||||
|
pub const local0_opcode = 0x60;
|
||||||
|
pub const local7_opcode = 0x67;
|
||||||
|
pub const arg0_opcode = 0x68;
|
||||||
|
pub const arg6_opcode = 0x6E;
|
||||||
|
|
||||||
|
// --- store / references / arithmetic ---------------------------------------
|
||||||
|
pub const store_opcode = 0x70;
|
||||||
|
pub const ref_of_opcode = 0x71;
|
||||||
|
pub const add_opcode = 0x72;
|
||||||
|
pub const concat_opcode = 0x73;
|
||||||
|
pub const subtract_opcode = 0x74;
|
||||||
|
pub const increment_opcode = 0x75;
|
||||||
|
pub const decrement_opcode = 0x76;
|
||||||
|
pub const multiply_opcode = 0x77;
|
||||||
|
pub const divide_opcode = 0x78;
|
||||||
|
pub const shift_left_opcode = 0x79;
|
||||||
|
pub const shift_right_opcode = 0x7A;
|
||||||
|
pub const and_opcode = 0x7B;
|
||||||
|
pub const nand_opcode = 0x7C;
|
||||||
|
pub const or_opcode = 0x7D;
|
||||||
|
pub const nor_opcode = 0x7E;
|
||||||
|
pub const xor_opcode = 0x7F;
|
||||||
|
pub const not_opcode = 0x80;
|
||||||
|
pub const find_set_left_bit_opcode = 0x81;
|
||||||
|
pub const find_set_right_bit_opcode = 0x82;
|
||||||
|
pub const dereference_of_opcode = 0x83;
|
||||||
|
pub const concat_resource_opcode = 0x84;
|
||||||
|
pub const mod_opcode = 0x85;
|
||||||
|
pub const notify_opcode = 0x86;
|
||||||
|
pub const size_of_opcode = 0x87;
|
||||||
|
pub const index_opcode = 0x88;
|
||||||
|
pub const match_opcode = 0x89;
|
||||||
|
pub const create_dword_field_opcode = 0x8A;
|
||||||
|
pub const create_word_field_opcode = 0x8B;
|
||||||
|
pub const create_byte_field_opcode = 0x8C;
|
||||||
|
pub const create_bit_field_opcode = 0x8D;
|
||||||
|
pub const object_type_opcode = 0x8E;
|
||||||
|
pub const create_qword_field_opcode = 0x8F;
|
||||||
|
|
||||||
|
pub const land_opcode = 0x90;
|
||||||
|
pub const lor_opcode = 0x91;
|
||||||
|
pub const lnot_opcode = 0x92; // may be followed by a second byte (see `lnot`)
|
||||||
|
pub const lequal_opcode = 0x93;
|
||||||
|
pub const lgreater_opcode = 0x94;
|
||||||
|
pub const lless_opcode = 0x95;
|
||||||
|
pub const to_buffer_opcode = 0x96;
|
||||||
|
pub const to_decimal_string_opcode = 0x97;
|
||||||
|
pub const to_hex_string_opcode = 0x98;
|
||||||
|
pub const to_integer_opcode = 0x99;
|
||||||
|
pub const to_string_opcode = 0x9C;
|
||||||
|
pub const copy_object_opcode = 0x9D;
|
||||||
|
pub const mid_opcode = 0x9E;
|
||||||
|
pub const continue_opcode = 0x9F;
|
||||||
|
pub const if_opcode = 0xA0;
|
||||||
|
pub const else_opcode = 0xA1;
|
||||||
|
pub const while_opcode = 0xA2;
|
||||||
|
pub const noop_opcode = 0xA3;
|
||||||
|
pub const return_opcode = 0xA4;
|
||||||
|
pub const break_opcode = 0xA5;
|
||||||
|
pub const break_point_opcode = 0xCC;
|
||||||
|
pub const ones_opcode = 0xFF;
|
||||||
|
|
||||||
|
/// Second bytes of the `lnot_opcode` (0x92) compound comparison opcodes.
|
||||||
|
pub const lnot = struct {
|
||||||
|
pub const not_equal = 0x93; // LNotEqualOp: 0x92 0x93
|
||||||
|
pub const less_equal = 0x94; // LLessEqualOp: 0x92 0x94
|
||||||
|
pub const greater_equal = 0x95; // LGreaterEqualOp: 0x92 0x95
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Second bytes of extended opcodes (prefixed by `extended_opcode_prefix`, 0x5B).
|
||||||
|
pub const extended = struct {
|
||||||
|
pub const mutex = 0x01;
|
||||||
|
pub const event = 0x02;
|
||||||
|
pub const conditional_reference_of = 0x12;
|
||||||
|
pub const create_field = 0x13;
|
||||||
|
pub const load_table = 0x1F;
|
||||||
|
pub const load = 0x20;
|
||||||
|
pub const stall = 0x21;
|
||||||
|
pub const sleep = 0x22;
|
||||||
|
pub const acquire = 0x23;
|
||||||
|
pub const signal = 0x24;
|
||||||
|
pub const wait = 0x25;
|
||||||
|
pub const reset = 0x26;
|
||||||
|
pub const release = 0x27;
|
||||||
|
pub const from_bcd = 0x28;
|
||||||
|
pub const to_bcd = 0x29;
|
||||||
|
pub const unload = 0x2A;
|
||||||
|
pub const revision = 0x30;
|
||||||
|
pub const debug = 0x31;
|
||||||
|
pub const fatal = 0x32;
|
||||||
|
pub const timer = 0x33;
|
||||||
|
pub const operation_region = 0x80;
|
||||||
|
pub const field = 0x81;
|
||||||
|
pub const device = 0x82;
|
||||||
|
pub const processor = 0x83;
|
||||||
|
pub const power_resource = 0x84;
|
||||||
|
pub const thermal_zone = 0x85;
|
||||||
|
pub const index_field = 0x86;
|
||||||
|
pub const bank_field = 0x87;
|
||||||
|
pub const data_region = 0x88;
|
||||||
|
};
|
||||||
@@ -0,0 +1,517 @@
|
|||||||
|
//! Recursive-descent AML parser. Walks the entire byte stream — including method
|
||||||
|
//! bodies — building the ACPI namespace as it goes. It does not *evaluate*
|
||||||
|
//! anything (no OperationRegion reads, no arithmetic); it parses structure so the
|
||||||
|
//! cursor stays aligned and every named object is recorded.
|
||||||
|
//!
|
||||||
|
//! The one genuine ambiguity in AML is method invocation: a bare NameString in an
|
||||||
|
//! operand position is a call whose argument count is only known from the method's
|
||||||
|
//! (earlier) declaration. Because we build the namespace in the same in-order pass,
|
||||||
|
//! `resolve` finds that declaration and tells us how many operands to consume.
|
||||||
|
//!
|
||||||
|
//! Safety net: every object delimited by a PkgLength (Scope/Device/Method/If/While/
|
||||||
|
//! Field/Buffer/Package/…) is parsed within its known extent, and the cursor is
|
||||||
|
//! snapped to that extent afterwards. So a mis-resolved invocation can only desync
|
||||||
|
//! *within* one such object; the enclosing walk realigns at the boundary.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const opcode = @import("opcodes.zig");
|
||||||
|
const Namespace = @import("namespace.zig").Namespace;
|
||||||
|
const Node = @import("namespace.zig").Node;
|
||||||
|
const NodeKind = @import("namespace.zig").NodeKind;
|
||||||
|
|
||||||
|
pub const Error = error{ Truncated, Malformed } || std.mem.Allocator.Error;
|
||||||
|
|
||||||
|
const maximum_segments = 64;
|
||||||
|
|
||||||
|
/// A parsed NameString: an optional root anchor or some parent hops, then a list
|
||||||
|
/// of 4-byte segments.
|
||||||
|
const NamePath = struct {
|
||||||
|
rooted: bool = false,
|
||||||
|
parents: u8 = 0,
|
||||||
|
segments: [maximum_segments][4]u8 = undefined,
|
||||||
|
count: usize = 0,
|
||||||
|
|
||||||
|
fn slice(self: *const NamePath) []const [4]u8 {
|
||||||
|
return self.segments[0..self.count];
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const Parser = struct {
|
||||||
|
aml: []const u8,
|
||||||
|
position: usize = 0,
|
||||||
|
namespace: *Namespace,
|
||||||
|
|
||||||
|
pub fn init(aml: []const u8, namespace: *Namespace) Parser {
|
||||||
|
return .{ .aml = aml, .namespace = namespace };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Parse the whole block as a TermList under the namespace root. Returns the
|
||||||
|
/// number of bytes consumed — equal to `aml.len` for a clean full traversal.
|
||||||
|
pub fn parseAll(self: *Parser) usize {
|
||||||
|
self.termList(self.aml.len, self.namespace.root);
|
||||||
|
return self.position;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- cursor primitives --------------------------------------------------
|
||||||
|
|
||||||
|
fn eof(self: *Parser) bool {
|
||||||
|
return self.position >= self.aml.len;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn peek(self: *Parser) ?u8 {
|
||||||
|
return if (self.eof()) null else self.aml[self.position];
|
||||||
|
}
|
||||||
|
|
||||||
|
fn readByte(self: *Parser) Error!u8 {
|
||||||
|
if (self.eof()) return error.Truncated;
|
||||||
|
const b = self.aml[self.position];
|
||||||
|
self.position += 1;
|
||||||
|
return b;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn skip(self: *Parser, n: usize) Error!void {
|
||||||
|
if (self.position + n > self.aml.len) return error.Truncated;
|
||||||
|
self.position += n;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn skipCString(self: *Parser) Error!void {
|
||||||
|
while (true) {
|
||||||
|
const b = try self.readByte();
|
||||||
|
if (b == 0) return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// AML PkgLength: the lead byte's top two bits give how many extra bytes
|
||||||
|
/// follow; the value counts from the start of the PkgLength field.
|
||||||
|
fn readPackageLength(self: *Parser) Error!usize {
|
||||||
|
const lead = try self.readByte();
|
||||||
|
const follow: usize = lead >> 6;
|
||||||
|
if (follow == 0) return lead & 0x3F;
|
||||||
|
var value: usize = lead & 0x0F;
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < follow) : (i += 1) {
|
||||||
|
const b = try self.readByte();
|
||||||
|
value |= @as(usize, b) << @intCast(4 + i * 8);
|
||||||
|
}
|
||||||
|
return value;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn readNameSegment(self: *Parser) Error![4]u8 {
|
||||||
|
if (self.position + 4 > self.aml.len) return error.Truncated;
|
||||||
|
const segment = self.aml[self.position..][0..4].*;
|
||||||
|
self.position += 4;
|
||||||
|
return segment;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn readNameString(self: *Parser) Error!NamePath {
|
||||||
|
var name_path = NamePath{};
|
||||||
|
// A NameString is either root-anchored or parent-relative, not both.
|
||||||
|
if (self.peek() == opcode.root_char) {
|
||||||
|
name_path.rooted = true;
|
||||||
|
self.position += 1;
|
||||||
|
} else {
|
||||||
|
while (self.peek() == opcode.parent_prefix_char) : (self.position += 1) name_path.parents += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
const lead = self.peek() orelse return name_path;
|
||||||
|
switch (lead) {
|
||||||
|
0x00 => self.position += 1, // NullName
|
||||||
|
opcode.dual_name_prefix => {
|
||||||
|
self.position += 1;
|
||||||
|
try self.appendSegment(&name_path);
|
||||||
|
try self.appendSegment(&name_path);
|
||||||
|
},
|
||||||
|
opcode.multi_name_prefix => {
|
||||||
|
self.position += 1;
|
||||||
|
const count = try self.readByte();
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < count) : (i += 1) try self.appendSegment(&name_path);
|
||||||
|
},
|
||||||
|
else => {
|
||||||
|
if (isNameStart(lead)) try self.appendSegment(&name_path);
|
||||||
|
},
|
||||||
|
}
|
||||||
|
return name_path;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn appendSegment(self: *Parser, name_path: *NamePath) Error!void {
|
||||||
|
const segment = try self.readNameSegment();
|
||||||
|
if (name_path.count < maximum_segments) {
|
||||||
|
name_path.segments[name_path.count] = segment;
|
||||||
|
name_path.count += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- term list / object -------------------------------------------------
|
||||||
|
|
||||||
|
/// Parse objects until `end`, then snap to `end`. Any parse error resyncs to
|
||||||
|
/// the boundary rather than propagating — containment for the rare desync.
|
||||||
|
fn termList(self: *Parser, end: usize, scope: *Node) void {
|
||||||
|
while (self.position < end) {
|
||||||
|
self.object(scope) catch break;
|
||||||
|
}
|
||||||
|
self.position = end;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Parse exactly one object/term at the cursor. Used for both TermObjs and
|
||||||
|
/// operands (TermArg / SuperName / Target all reduce to "one object" for the
|
||||||
|
/// purpose of advancing the cursor).
|
||||||
|
fn object(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const lead = self.peek() orelse return error.Truncated;
|
||||||
|
if (isNameStart(lead)) return self.nameInvocation(scope);
|
||||||
|
|
||||||
|
_ = try self.readByte();
|
||||||
|
switch (lead) {
|
||||||
|
// constants and no-operand statements
|
||||||
|
opcode.zero_opcode, opcode.one_opcode, opcode.ones_opcode => {},
|
||||||
|
opcode.noop_opcode, opcode.continue_opcode, opcode.break_opcode, opcode.break_point_opcode => {},
|
||||||
|
opcode.local0_opcode...opcode.local7_opcode => {},
|
||||||
|
opcode.arg0_opcode...opcode.arg6_opcode => {},
|
||||||
|
|
||||||
|
// literal data
|
||||||
|
opcode.byte_prefix => try self.skip(1),
|
||||||
|
opcode.word_prefix => try self.skip(2),
|
||||||
|
opcode.dword_prefix => try self.skip(4),
|
||||||
|
opcode.qword_prefix => try self.skip(8),
|
||||||
|
opcode.string_prefix => try self.skipCString(),
|
||||||
|
|
||||||
|
// data containers (contents skipped via their PkgLength)
|
||||||
|
opcode.buffer_opcode, opcode.package_opcode, opcode.var_package_opcode => try self.skipPackage(),
|
||||||
|
|
||||||
|
// namespace modifiers / named objects
|
||||||
|
opcode.name_opcode => try self.parseName(scope),
|
||||||
|
opcode.alias_opcode => try self.parseAlias(scope),
|
||||||
|
opcode.scope_opcode => try self.parseScopeLike(scope, .scope),
|
||||||
|
opcode.method_opcode => try self.parseMethod(scope),
|
||||||
|
opcode.external_opcode => try self.parseExternal(scope),
|
||||||
|
opcode.extended_opcode_prefix => try self.parseExtended(scope),
|
||||||
|
|
||||||
|
// control flow
|
||||||
|
opcode.if_opcode => try self.parseIf(scope),
|
||||||
|
opcode.else_opcode => try self.parseElse(scope),
|
||||||
|
opcode.while_opcode => try self.parseWhile(scope),
|
||||||
|
opcode.return_opcode => try self.object(scope),
|
||||||
|
opcode.notify_opcode => try self.args(scope, 2),
|
||||||
|
|
||||||
|
// stores / references / unary+target
|
||||||
|
opcode.store_opcode => try self.args(scope, 2),
|
||||||
|
opcode.ref_of_opcode, opcode.dereference_of_opcode, opcode.size_of_opcode, opcode.object_type_opcode => try self.args(scope, 1),
|
||||||
|
opcode.increment_opcode, opcode.decrement_opcode => try self.args(scope, 1),
|
||||||
|
opcode.not_opcode, opcode.find_set_left_bit_opcode, opcode.find_set_right_bit_opcode => try self.args(scope, 2),
|
||||||
|
opcode.to_buffer_opcode, opcode.to_decimal_string_opcode, opcode.to_hex_string_opcode, opcode.to_integer_opcode => try self.args(scope, 2),
|
||||||
|
opcode.copy_object_opcode => try self.args(scope, 2),
|
||||||
|
|
||||||
|
// binary + target
|
||||||
|
opcode.add_opcode, opcode.subtract_opcode, opcode.multiply_opcode, opcode.mod_opcode => try self.args(scope, 3),
|
||||||
|
opcode.and_opcode, opcode.nand_opcode, opcode.or_opcode, opcode.nor_opcode, opcode.xor_opcode => try self.args(scope, 3),
|
||||||
|
opcode.shift_left_opcode, opcode.shift_right_opcode, opcode.concat_opcode, opcode.concat_resource_opcode, opcode.index_opcode => try self.args(scope, 3),
|
||||||
|
opcode.divide_opcode => try self.args(scope, 4),
|
||||||
|
opcode.to_string_opcode => try self.args(scope, 3),
|
||||||
|
opcode.mid_opcode => try self.args(scope, 4),
|
||||||
|
|
||||||
|
// logical
|
||||||
|
opcode.land_opcode, opcode.lor_opcode => try self.args(scope, 2),
|
||||||
|
opcode.lequal_opcode, opcode.lgreater_opcode, opcode.lless_opcode => try self.args(scope, 2),
|
||||||
|
opcode.lnot_opcode => try self.parseLnot(scope),
|
||||||
|
|
||||||
|
opcode.match_opcode => try self.parseMatch(scope),
|
||||||
|
|
||||||
|
// CreateXField: <source> <index> NameString
|
||||||
|
opcode.create_dword_field_opcode,
|
||||||
|
opcode.create_word_field_opcode,
|
||||||
|
opcode.create_byte_field_opcode,
|
||||||
|
opcode.create_bit_field_opcode,
|
||||||
|
opcode.create_qword_field_opcode,
|
||||||
|
=> try self.parseCreateField(scope, 2),
|
||||||
|
|
||||||
|
else => return error.Malformed,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Parse `n` operands.
|
||||||
|
fn args(self: *Parser, scope: *Node, n: usize) Error!void {
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < n) : (i += 1) try self.object(scope);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A NameString in operand/statement position: a method invocation (consuming
|
||||||
|
/// the callee's declared argument count) or a plain name reference.
|
||||||
|
fn nameInvocation(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
if (self.namespace.resolve(scope, name_path.rooted, name_path.parents, name_path.slice())) |node| {
|
||||||
|
if ((node.kind == .method or node.kind == .external) and node.arg_count > 0) {
|
||||||
|
try self.args(scope, node.arg_count);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Skip a PkgLength-delimited body wholesale (Buffer / Package / VarPackage):
|
||||||
|
/// the contents are pure data, never namespace declarations.
|
||||||
|
fn skipPackage(self: *Parser) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const len = try self.readPackageLength();
|
||||||
|
const end = start + len;
|
||||||
|
if (end > self.aml.len) return error.Truncated;
|
||||||
|
self.position = end;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- namespace objects --------------------------------------------------
|
||||||
|
|
||||||
|
fn parseName(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
const value_start = self.position;
|
||||||
|
try self.object(scope); // the DataReferenceObject value
|
||||||
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .name);
|
||||||
|
node.value = self.aml[value_start..self.position];
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseAlias(self: *Parser, scope: *Node) Error!void {
|
||||||
|
_ = try self.readNameString(); // source
|
||||||
|
const name_path = try self.readNameString(); // the alias name
|
||||||
|
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .alias);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseMethod(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
const flags = try self.readByte();
|
||||||
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .method);
|
||||||
|
node.arg_count = flags & 0x7;
|
||||||
|
// Capture the body for on-demand evaluation and skip it — objects declared
|
||||||
|
// inside a method are created at *runtime*, not at load, so they must not
|
||||||
|
// become permanent namespace nodes.
|
||||||
|
node.value = self.aml[self.position..@min(end, self.aml.len)];
|
||||||
|
self.position = end;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseExternal(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
_ = try self.readByte(); // object type
|
||||||
|
const arg_count = try self.readByte();
|
||||||
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .external);
|
||||||
|
node.arg_count = arg_count;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Scope / Device / ThermalZone: PkgLength, NameString, then a nested TermList.
|
||||||
|
fn parseScopeLike(self: *Parser, scope: *Node, kind: NodeKind) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), kind);
|
||||||
|
self.termList(end, node);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseProcessor(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
try self.skip(6); // ProcID(byte) + PblkAddress(dword) + PblkLen(byte)
|
||||||
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .processor);
|
||||||
|
self.termList(end, node);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parsePowerResource(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
try self.skip(3); // SystemLevel(byte) + ResourceOrder(word)
|
||||||
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .power_resource);
|
||||||
|
self.termList(end, node);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// OperationRegion: NameString, RegionSpace(byte), Offset(TermArg), Len(TermArg).
|
||||||
|
/// The offset/length expressions are kept as AML for lazy evaluation.
|
||||||
|
fn parseRegion(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
const space = try self.readByte();
|
||||||
|
const off_start = self.position;
|
||||||
|
try self.object(scope);
|
||||||
|
const off_end = self.position;
|
||||||
|
try self.object(scope);
|
||||||
|
const len_end = self.position;
|
||||||
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .region);
|
||||||
|
node.region_space = space;
|
||||||
|
node.region_offset_aml = self.aml[off_start..off_end];
|
||||||
|
node.region_len_aml = self.aml[off_end..len_end];
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseDataRegion(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
try self.args(scope, 3); // signature, oem id, oem table id (TermArgs)
|
||||||
|
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .region);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseMutex(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
try self.skip(1); // sync flags
|
||||||
|
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .mutex);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseEvent(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .event);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// CreateXField: `count` TermArgs then the new field's NameString.
|
||||||
|
fn parseCreateField(self: *Parser, scope: *Node, count: usize) Error!void {
|
||||||
|
try self.args(scope, count);
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .name);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Field / IndexField / BankField: a region/bank reference, flags, then a
|
||||||
|
/// FieldList whose NamedFields become nodes in the current scope. For a plain
|
||||||
|
/// Field, the first NameString is the backing region — captured so field units
|
||||||
|
/// carry a region + bit position the evaluator can read/write.
|
||||||
|
fn parseField(self: *Parser, scope: *Node, name_strings: u8, bank: bool) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
var region: ?*Node = null;
|
||||||
|
var i: u8 = 0;
|
||||||
|
while (i < name_strings) : (i += 1) {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
// Only a plain Field's single NameString denotes an OperationRegion.
|
||||||
|
if (name_strings == 1) region = self.namespace.resolve(scope, name_path.rooted, name_path.parents, name_path.slice());
|
||||||
|
}
|
||||||
|
if (bank) try self.object(scope); // bank value TermArg
|
||||||
|
const flags = try self.readByte();
|
||||||
|
self.fieldList(end, scope, region, flags & 0x0F);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn fieldList(self: *Parser, end: usize, scope: *Node, region: ?*Node, initial_access: u8) void {
|
||||||
|
var bit_offset: u32 = 0;
|
||||||
|
var access = initial_access;
|
||||||
|
while (self.position < end) {
|
||||||
|
const lead = self.peek() orelse break;
|
||||||
|
switch (lead) {
|
||||||
|
0x00 => { // ReservedField: advances the bit position
|
||||||
|
self.position += 1;
|
||||||
|
const width = self.readPackageLength() catch break;
|
||||||
|
bit_offset += @intCast(width);
|
||||||
|
},
|
||||||
|
0x01 => { // AccessField: AccessType (low nibble) + AccessAttrib
|
||||||
|
self.position += 1;
|
||||||
|
const at = self.readByte() catch break;
|
||||||
|
self.skip(1) catch break;
|
||||||
|
access = at & 0x0F;
|
||||||
|
},
|
||||||
|
0x02 => { // ConnectField: NameString | BufferData
|
||||||
|
self.position += 1;
|
||||||
|
self.object(scope) catch break;
|
||||||
|
},
|
||||||
|
0x03 => { // ExtendedAccessField: type + attrib + length
|
||||||
|
self.position += 1;
|
||||||
|
self.skip(3) catch break;
|
||||||
|
},
|
||||||
|
else => { // NamedField: NameSegment + PkgLength (bit width)
|
||||||
|
const segment = self.readNameSegment() catch break;
|
||||||
|
const width = self.readPackageLength() catch break;
|
||||||
|
const unit = self.namespace.newFieldUnit(scope, segment) catch break;
|
||||||
|
unit.region = region;
|
||||||
|
unit.bit_offset = bit_offset;
|
||||||
|
unit.bit_width = @intCast(width);
|
||||||
|
unit.access_type = access;
|
||||||
|
bit_offset += @intCast(width);
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
self.position = end;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- control flow -------------------------------------------------------
|
||||||
|
|
||||||
|
fn parseIf(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
try self.object(scope); // predicate
|
||||||
|
self.termList(end, scope);
|
||||||
|
if (self.peek() == opcode.else_opcode) {
|
||||||
|
self.position += 1;
|
||||||
|
try self.parseElse(scope);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseElse(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
self.termList(end, scope);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseWhile(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
try self.object(scope); // predicate
|
||||||
|
self.termList(end, scope);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseLnot(self: *Parser, scope: *Node) Error!void {
|
||||||
|
// 0x92 followed by 0x93/94/95 is a compound comparison (two operands);
|
||||||
|
// otherwise it is a plain LNot of one operand.
|
||||||
|
const b = self.peek() orelse return error.Truncated;
|
||||||
|
switch (b) {
|
||||||
|
opcode.lnot.not_equal, opcode.lnot.less_equal, opcode.lnot.greater_equal => {
|
||||||
|
self.position += 1;
|
||||||
|
try self.args(scope, 2);
|
||||||
|
},
|
||||||
|
else => try self.object(scope),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseMatch(self: *Parser, scope: *Node) Error!void {
|
||||||
|
try self.object(scope); // search package
|
||||||
|
try self.skip(1); // match opcode 1
|
||||||
|
try self.object(scope); // operand 1
|
||||||
|
try self.skip(1); // match opcode 2
|
||||||
|
try self.object(scope); // operand 2
|
||||||
|
try self.object(scope); // start index
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- extended opcodes (0x5B xx) -----------------------------------------
|
||||||
|
|
||||||
|
fn parseExtended(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const e = try self.readByte();
|
||||||
|
switch (e) {
|
||||||
|
opcode.extended.mutex => try self.parseMutex(scope),
|
||||||
|
opcode.extended.event => try self.parseEvent(scope),
|
||||||
|
opcode.extended.operation_region => try self.parseRegion(scope),
|
||||||
|
opcode.extended.data_region => try self.parseDataRegion(scope),
|
||||||
|
opcode.extended.field => try self.parseField(scope, 1, false),
|
||||||
|
opcode.extended.index_field => try self.parseField(scope, 2, false),
|
||||||
|
opcode.extended.bank_field => try self.parseField(scope, 2, true),
|
||||||
|
opcode.extended.device => try self.parseScopeLike(scope, .device),
|
||||||
|
opcode.extended.thermal_zone => try self.parseScopeLike(scope, .thermal_zone),
|
||||||
|
opcode.extended.processor => try self.parseProcessor(scope),
|
||||||
|
opcode.extended.power_resource => try self.parsePowerResource(scope),
|
||||||
|
|
||||||
|
opcode.extended.conditional_reference_of => try self.args(scope, 2), // SuperName, Target
|
||||||
|
opcode.extended.create_field => try self.parseCreateField(scope, 3),
|
||||||
|
opcode.extended.load_table => try self.args(scope, 6),
|
||||||
|
opcode.extended.load => try self.args(scope, 2), // NameString, Target
|
||||||
|
opcode.extended.stall, opcode.extended.sleep => try self.args(scope, 1),
|
||||||
|
opcode.extended.acquire => {
|
||||||
|
try self.object(scope); // mutex SuperName
|
||||||
|
try self.skip(2); // timeout WordData
|
||||||
|
},
|
||||||
|
opcode.extended.signal, opcode.extended.reset, opcode.extended.release, opcode.extended.unload => try self.args(scope, 1),
|
||||||
|
opcode.extended.wait => try self.args(scope, 2),
|
||||||
|
opcode.extended.from_bcd, opcode.extended.to_bcd => try self.args(scope, 2),
|
||||||
|
opcode.extended.fatal => {
|
||||||
|
try self.skip(5); // Type(byte) + Code(dword)
|
||||||
|
try self.object(scope); // Arg TermArg
|
||||||
|
},
|
||||||
|
opcode.extended.revision, opcode.extended.debug, opcode.extended.timer => {},
|
||||||
|
|
||||||
|
else => return error.Malformed,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
fn isNameStart(b: u8) bool {
|
||||||
|
return (b >= opcode.name_char_start and b <= opcode.name_char_end) or
|
||||||
|
b == opcode.name_char_underscore or
|
||||||
|
b == opcode.root_char or
|
||||||
|
b == opcode.parent_prefix_char or
|
||||||
|
b == opcode.dual_name_prefix or
|
||||||
|
b == opcode.multi_name_prefix;
|
||||||
|
}
|
||||||
@@ -0,0 +1,90 @@
|
|||||||
|
//! The **device ABI**: the flat, `extern` device types that cross the system_call
|
||||||
|
//! boundary — what `device_enumerate` hands a user-space driver, what
|
||||||
|
//! `device_register` takes back. This is the devices sub-project's *public
|
||||||
|
//! interface*, exposed as its own `device-abi` module the same way the VFS server
|
||||||
|
//! exposes `vfs-protocol` — so both the kernel and user space depend on the contract
|
||||||
|
//! by name, and neither reaches into the other's files.
|
||||||
|
//!
|
||||||
|
//! It is also the **single source of truth** for `DeviceClass` and `ResourceKind`:
|
||||||
|
//! the kernel's rich, pointer-based device tree (system/devices/device-model.zig,
|
||||||
|
//! which user space must never import) re-exports these, so the enum that a driver
|
||||||
|
//! matches on and the enum the kernel classifies with are the *same* type — no
|
||||||
|
//! hand-kept "mirror in order" to drift. The core kernel↔user ABI is [[abi]]; the
|
||||||
|
//! loader↔kernel handoff is [[boot-handoff]].
|
||||||
|
|
||||||
|
/// A coarse classification of a device, independent of the describing firmware.
|
||||||
|
/// Kept small on purpose; refine as real drivers arrive. `enum(u32)` because the
|
||||||
|
/// `@intFromEnum` value crosses the system_call boundary in `DeviceDescriptor.class`.
|
||||||
|
pub const DeviceClass = enum(u32) {
|
||||||
|
/// The synthetic root every discovered device hangs beneath.
|
||||||
|
root,
|
||||||
|
processor,
|
||||||
|
interrupt_controller,
|
||||||
|
timer,
|
||||||
|
/// A PCI(e) host bridge — the root of a PCI segment (owns an ECAM window).
|
||||||
|
pci_host_bridge,
|
||||||
|
/// A single PCI function.
|
||||||
|
pci_device,
|
||||||
|
/// A device named in the ACPI namespace (from the DSDT/SSDT), carrying a
|
||||||
|
/// hardware ID (`_HID`) and, where static, current resource settings (`_CRS`).
|
||||||
|
acpi_device,
|
||||||
|
/// The ACPI tables themselves, published as one node for the user-space acpi
|
||||||
|
/// service (docs/m19-m20-plan.md M20): memory resources over the AML blobs,
|
||||||
|
/// a broad io_port grant for OperationRegion access, and the SCI interrupt.
|
||||||
|
/// The one node whose claimant is trusted to run firmware bytecode.
|
||||||
|
acpi_tables,
|
||||||
|
unknown,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The kind of hardware resource a device occupies. `enum(u32)` for the same
|
||||||
|
/// boundary-crossing reason as `DeviceClass` (see `ResourceDescriptor.kind`).
|
||||||
|
pub const ResourceKind = enum(u32) {
|
||||||
|
/// A memory-mapped I/O window: `start` is the physical base, `len` its size.
|
||||||
|
memory,
|
||||||
|
/// A legacy I/O-port range: `start` is the first port, `len` the count.
|
||||||
|
io_port,
|
||||||
|
/// An interrupt: `start` is the global system interrupt (GSI), `len` is 1.
|
||||||
|
irq,
|
||||||
|
/// A range of bus numbers owned by a bridge: `start`..`start+len`.
|
||||||
|
bus_range,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One device resource, as handed to a user-space driver (flat, extern).
|
||||||
|
pub const ResourceDescriptor = extern struct {
|
||||||
|
kind: u64, // a ResourceKind value
|
||||||
|
start: u64,
|
||||||
|
len: u64,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const maximum_device_resources = 8;
|
||||||
|
|
||||||
|
/// `DeviceDescriptor.parent` for a device with no parent — a root of the device tree.
|
||||||
|
pub const no_parent: u64 = ~@as(u64, 0);
|
||||||
|
|
||||||
|
/// `DeviceDescriptor.pci_class` for a device that is not a PCI function. (Zero would be
|
||||||
|
/// ambiguous: 0x000000 is a real class code, "unclassified device".)
|
||||||
|
pub const no_pci_class: u64 = ~@as(u64, 0);
|
||||||
|
|
||||||
|
/// A device, as snapshotted for user space by `device_enumerate`. A driver scans
|
||||||
|
/// these to find the hardware it owns, claims it, and maps its MMIO.
|
||||||
|
///
|
||||||
|
/// `parent` makes the table a tree rather than a list, which is what a **bus driver**
|
||||||
|
/// needs: it claims the bus, finds the devices below it, and publishes any it
|
||||||
|
/// discovers itself with `device_register`. A registered child's resources must lie
|
||||||
|
/// within its parent's (the kernel enforces this) — that containment is what makes
|
||||||
|
/// delegation safe, since a device descriptor is otherwise a licence to map physical
|
||||||
|
/// memory.
|
||||||
|
pub const DeviceDescriptor = extern struct {
|
||||||
|
id: u64,
|
||||||
|
parent: u64, // a device id, or `no_parent`
|
||||||
|
class: u64, // a DeviceClass value
|
||||||
|
// The PCI class/subclass/prog-IF triple packed as 0xCCSSPP when this device is a PCI
|
||||||
|
// function, or `no_pci_class` otherwise. This is how a manager tells *what* a
|
||||||
|
// `pci_device` is (an xHCI controller, an AHCI controller) — decode the triple into
|
||||||
|
// names with the pci-class module.
|
||||||
|
pci_class: u64,
|
||||||
|
hid_len: u64,
|
||||||
|
resource_count: u64,
|
||||||
|
hid: [8]u8,
|
||||||
|
resources: [maximum_device_resources]ResourceDescriptor,
|
||||||
|
};
|
||||||
@@ -3,7 +3,7 @@
|
|||||||
//! Discovery backends (ACPI today, device-tree later) translate their native
|
//! Discovery backends (ACPI today, device-tree later) translate their native
|
||||||
//! hardware description into this one shape, so the rest of the kernel walks a
|
//! hardware description into this one shape, so the rest of the kernel walks a
|
||||||
//! plain `Device` tree without knowing which firmware described the machine —
|
//! plain `Device` tree without knowing which firmware described the machine —
|
||||||
//! the same discipline `root.zig`'s `MemoryKind` applies to memory and `arch`
|
//! the same discipline `ps2-library.zig`'s `MemoryKind` applies to memory and `architecture`
|
||||||
//! applies to the CPU.
|
//! applies to the CPU.
|
||||||
//!
|
//!
|
||||||
//! This is deliberately minimal: enough to *describe* what was discovered (a
|
//! This is deliberately minimal: enough to *describe* what was discovered (a
|
||||||
@@ -12,28 +12,27 @@
|
|||||||
//! on top of this — nothing here presumes them.
|
//! on top of this — nothing here presumes them.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
|
const device_abi = @import("device-abi");
|
||||||
|
const pci_class = @import("pci-class");
|
||||||
|
const acpi_ids = @import("acpi-ids");
|
||||||
|
|
||||||
/// The hardware primitives a discovery backend needs but can't express portably.
|
/// The hardware primitives a discovery backend needs but can't express portably.
|
||||||
/// The kernel injects an implementation (the arch VMM + port I/O), so the device
|
/// The kernel injects an implementation (the architecture VMM + port I/O), so the device
|
||||||
/// layer touches hardware without importing `arch` — the same discipline that lets
|
/// layer touches hardware without importing `architecture` — the same discipline that lets
|
||||||
/// it stay firmware-agnostic. `pioRead`/`pioWrite` take a width in bytes (1/2/4).
|
/// it stay firmware-agnostic. `pioRead`/`pioWrite` take a width in bytes (1/2/4).
|
||||||
pub const Hal = struct {
|
pub const Hal = struct {
|
||||||
mapMmio: *const fn (virt: u64, phys: u64, writable: bool) void,
|
/// Map a physical MMIO range and return the virtual address to reach it at.
|
||||||
|
/// The device layer dereferences the returned address and never learns how
|
||||||
|
/// the kernel places it (identity, physmap, a window — the kernel's choice).
|
||||||
|
mapMmio: *const fn (physical: u64, len: u64, writable: bool) u64,
|
||||||
pioRead: *const fn (width: u8, port: u16) u32,
|
pioRead: *const fn (width: u8, port: u16) u32,
|
||||||
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
||||||
};
|
};
|
||||||
|
|
||||||
/// The kind of hardware resource a device occupies.
|
/// The kind of hardware resource a device occupies. Canonically defined by the
|
||||||
pub const ResourceKind = enum {
|
/// device ABI (system/devices/device-abi.zig) and re-exported here, so the kernel's
|
||||||
/// A memory-mapped I/O window: `start` is the physical base, `len` its size.
|
/// internal tree and the descriptors it hands user space share one enum.
|
||||||
memory,
|
pub const ResourceKind = device_abi.ResourceKind;
|
||||||
/// A legacy I/O-port range: `start` is the first port, `len` the count.
|
|
||||||
io_port,
|
|
||||||
/// An interrupt: `start` is the global system interrupt (GSI), `len` is 1.
|
|
||||||
irq,
|
|
||||||
/// A range of bus numbers owned by a bridge: `start`..`start+len`.
|
|
||||||
bus_range,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// One hardware resource claimed by a device.
|
/// One hardware resource claimed by a device.
|
||||||
pub const Resource = struct {
|
pub const Resource = struct {
|
||||||
@@ -43,22 +42,10 @@ pub const Resource = struct {
|
|||||||
};
|
};
|
||||||
|
|
||||||
/// A coarse classification of a device, independent of the describing firmware.
|
/// A coarse classification of a device, independent of the describing firmware.
|
||||||
/// Kept small on purpose; refine as real drivers arrive.
|
/// Canonically defined by the device ABI (system/devices/device-abi.zig) and
|
||||||
pub const DeviceClass = enum {
|
/// re-exported here — the enum a driver matches on and the one the kernel classifies
|
||||||
/// The synthetic root every discovered device hangs beneath.
|
/// with are the same type. Kept small on purpose; refine as real drivers arrive.
|
||||||
root,
|
pub const DeviceClass = device_abi.DeviceClass;
|
||||||
processor,
|
|
||||||
interrupt_controller,
|
|
||||||
timer,
|
|
||||||
/// A PCI(e) host bridge — the root of a PCI segment (owns an ECAM window).
|
|
||||||
pci_host_bridge,
|
|
||||||
/// A single PCI function.
|
|
||||||
pci_device,
|
|
||||||
/// A device named in the ACPI namespace (from the DSDT/SSDT), carrying a
|
|
||||||
/// hardware ID (`_HID`) and, where static, current resource settings (`_CRS`).
|
|
||||||
acpi_device,
|
|
||||||
unknown,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Firmware-independent identity. Each backend fills only the fields it knows;
|
/// Firmware-independent identity. Each backend fills only the fields it knows;
|
||||||
/// the rest stay null. The generic layer never branches on *how* an id was
|
/// the rest stay null. The generic layer never branches on *how* an id was
|
||||||
@@ -71,29 +58,29 @@ pub const Ids = struct {
|
|||||||
pci_device: ?u16 = null,
|
pci_device: ?u16 = null,
|
||||||
/// PCI class/subclass/prog-if packed as 0xCCSSPP.
|
/// PCI class/subclass/prog-if packed as 0xCCSSPP.
|
||||||
pci_class: ?u24 = null,
|
pci_class: ?u24 = null,
|
||||||
/// PCI bus/device/function packed as (bus << 8) | (dev << 3) | func — the key
|
/// PCI bus/device/function packed as (bus << 8) | (device << 3) | function — the key
|
||||||
/// the ACPI address (`_ADR`) merge uses to match a namespace device to this node.
|
/// the ACPI address (`_ADR`) merge uses to match a namespace device to this node.
|
||||||
pci_bdf: ?u16 = null,
|
pci_bdf: ?u16 = null,
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Upper bound on resources tracked per device (6 PCI BARs + a couple of IRQs is
|
/// Upper bound on resources tracked per device (6 PCI BARs + a couple of IRQs is
|
||||||
/// the busy case). Stored inline so a device is a single allocation.
|
/// the busy case). Stored inline so a device is a single allocation.
|
||||||
pub const max_resources = 8;
|
pub const maximum_resources = 8;
|
||||||
|
|
||||||
/// One node in the device tree. Nodes are individually heap-allocated and linked
|
/// One node in the device tree. Nodes are individually heap-allocated and linked
|
||||||
/// intrusively (first-child / next-sibling), the classic device-tree layout —
|
/// intrusively (first-child / next-sibling), the classic device-tree layout —
|
||||||
/// no per-node dynamic arrays to manage.
|
/// no per-node dynamic arrays to manage.
|
||||||
pub const Device = struct {
|
pub const Device = struct {
|
||||||
name_buf: [24]u8 = undefined,
|
name_buffer: [24]u8 = undefined,
|
||||||
name_len: u8 = 0,
|
name_len: u8 = 0,
|
||||||
class: DeviceClass = .unknown,
|
class: DeviceClass = .unknown,
|
||||||
ids: Ids = .{},
|
ids: Ids = .{},
|
||||||
/// Human-readable hardware id (e.g. "PNP0A03"), when known. Backed inline like
|
/// Human-readable hardware id (e.g. "PNP0A03"), when known. Backed inline like
|
||||||
/// `name`; empty when unset. The generic layer stores/prints it without knowing
|
/// `name`; empty when unset. The generic layer stores/prints it without knowing
|
||||||
/// how a backend encoded it.
|
/// how a backend encoded it.
|
||||||
hid_buf: [8]u8 = undefined,
|
hid_buffer: [8]u8 = undefined,
|
||||||
hid_len: u8 = 0,
|
hid_len: u8 = 0,
|
||||||
resources: [max_resources]Resource = undefined,
|
resources: [maximum_resources]Resource = undefined,
|
||||||
resource_count: u8 = 0,
|
resource_count: u8 = 0,
|
||||||
|
|
||||||
parent: ?*Device = null,
|
parent: ?*Device = null,
|
||||||
@@ -103,30 +90,30 @@ pub const Device = struct {
|
|||||||
/// The device's short name (e.g. "cpu0", "pci0:00:1f.0"). Backed by an inline
|
/// The device's short name (e.g. "cpu0", "pci0:00:1f.0"). Backed by an inline
|
||||||
/// buffer, so it stays valid for the life of the node with no extra allocation.
|
/// buffer, so it stays valid for the life of the node with no extra allocation.
|
||||||
pub fn name(self: *const Device) []const u8 {
|
pub fn name(self: *const Device) []const u8 {
|
||||||
return self.name_buf[0..self.name_len];
|
return self.name_buffer[0..self.name_len];
|
||||||
}
|
}
|
||||||
|
|
||||||
fn setName(self: *Device, s: []const u8) void {
|
fn setName(self: *Device, s: []const u8) void {
|
||||||
const n: u8 = @intCast(@min(s.len, self.name_buf.len));
|
const n: u8 = @intCast(@min(s.len, self.name_buffer.len));
|
||||||
@memcpy(self.name_buf[0..n], s[0..n]);
|
@memcpy(self.name_buffer[0..n], s[0..n]);
|
||||||
self.name_len = n;
|
self.name_len = n;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The device's hardware id string, or empty if none is set.
|
/// The device's hardware id string, or empty if none is set.
|
||||||
pub fn hid(self: *const Device) []const u8 {
|
pub fn hid(self: *const Device) []const u8 {
|
||||||
return self.hid_buf[0..self.hid_len];
|
return self.hid_buffer[0..self.hid_len];
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn setHid(self: *Device, s: []const u8) void {
|
pub fn setHid(self: *Device, s: []const u8) void {
|
||||||
const n: u8 = @intCast(@min(s.len, self.hid_buf.len));
|
const n: u8 = @intCast(@min(s.len, self.hid_buffer.len));
|
||||||
@memcpy(self.hid_buf[0..n], s[0..n]);
|
@memcpy(self.hid_buffer[0..n], s[0..n]);
|
||||||
self.hid_len = n;
|
self.hid_len = n;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Record a resource. Silently drops beyond `max_resources` — discovery logs
|
/// Record a resource. Silently drops beyond `maximum_resources` — discovery logs
|
||||||
/// the truncation rather than failing the whole tree.
|
/// the truncation rather than failing the whole tree.
|
||||||
pub fn addResource(self: *Device, kind: ResourceKind, start: u64, len: u64) bool {
|
pub fn addResource(self: *Device, kind: ResourceKind, start: u64, len: u64) bool {
|
||||||
if (self.resource_count >= max_resources) return false;
|
if (self.resource_count >= maximum_resources) return false;
|
||||||
self.resources[self.resource_count] = .{ .kind = kind, .start = start, .len = len };
|
self.resources[self.resource_count] = .{ .kind = kind, .start = start, .len = len };
|
||||||
self.resource_count += 1;
|
self.resource_count += 1;
|
||||||
return true;
|
return true;
|
||||||
@@ -167,17 +154,17 @@ pub const DeviceTree = struct {
|
|||||||
self: *DeviceTree,
|
self: *DeviceTree,
|
||||||
parent: *Device,
|
parent: *Device,
|
||||||
class: DeviceClass,
|
class: DeviceClass,
|
||||||
dev_name: []const u8,
|
device_name: []const u8,
|
||||||
) !*Device {
|
) !*Device {
|
||||||
const d = try self.allocator.create(Device);
|
const d = try self.allocator.create(Device);
|
||||||
d.* = .{ .class = class, .parent = parent };
|
d.* = .{ .class = class, .parent = parent };
|
||||||
d.setName(dev_name);
|
d.setName(device_name);
|
||||||
if (parent.first_child == null) {
|
if (parent.first_child == null) {
|
||||||
parent.first_child = d;
|
parent.first_child = d;
|
||||||
} else {
|
} else {
|
||||||
var cur = parent.first_child.?;
|
var current = parent.first_child.?;
|
||||||
while (cur.next_sibling) |sib| cur = sib;
|
while (current.next_sibling) |sib| current = sib;
|
||||||
cur.next_sibling = d;
|
current.next_sibling = d;
|
||||||
}
|
}
|
||||||
return d;
|
return d;
|
||||||
}
|
}
|
||||||
@@ -199,18 +186,37 @@ fn firstOfClassIn(node: *Device, class: DeviceClass) ?*Device {
|
|||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn dumpNode(dev: *const Device, depth: usize, emit: *const fn ([]const u8) void) void {
|
fn dumpNode(device: *const Device, depth: usize, emit: *const fn ([]const u8) void) void {
|
||||||
const indent = @min(depth * 2, 40);
|
const indent = @min(depth * 2, 40);
|
||||||
|
|
||||||
var buf: [200]u8 = undefined;
|
var buffer: [200]u8 = undefined;
|
||||||
@memset(buf[0..indent], ' ');
|
@memset(buffer[0..indent], ' ');
|
||||||
const body = if (dev.hid_len != 0)
|
const body = if (device.hid_len != 0) blk: {
|
||||||
std.fmt.bufPrint(buf[indent..], "{s} [{s}] hid={s}\n", .{ dev.name(), @tagName(dev.class), dev.hid() }) catch return
|
// Decode the _HID to a human name when it's a known standard PnP/ACPI id.
|
||||||
else
|
const desc = acpi_ids.description(device.hid());
|
||||||
std.fmt.bufPrint(buf[indent..], "{s} [{s}]\n", .{ dev.name(), @tagName(dev.class) }) catch return;
|
break :blk if (desc.len != 0)
|
||||||
emit(buf[0 .. indent + body.len]);
|
std.fmt.bufPrint(buffer[indent..], "{s} [{s}] hid={s} ({s})\n", .{ device.name(), @tagName(device.class), device.hid(), desc }) catch return
|
||||||
|
else
|
||||||
|
std.fmt.bufPrint(buffer[indent..], "{s} [{s}] hid={s}\n", .{ device.name(), @tagName(device.class), device.hid() }) catch return;
|
||||||
|
} else std.fmt.bufPrint(buffer[indent..], "{s} [{s}]\n", .{ device.name(), @tagName(device.class) }) catch return;
|
||||||
|
emit(buffer[0 .. indent + body.len]);
|
||||||
|
|
||||||
for (dev.resources[0..dev.resource_count]) |r| {
|
// For a PCI function, decode its class code — the (class / subclass / prog-IF)
|
||||||
|
// triple that says what it actually is, which the coarse `DeviceClass` can't.
|
||||||
|
if (device.ids.pci_class) |packed_code| {
|
||||||
|
const cc = pci_class.ClassCode.unpack(packed_code);
|
||||||
|
var cbuf: [200]u8 = undefined;
|
||||||
|
const pad = @min(indent + 2, 42);
|
||||||
|
@memset(cbuf[0..pad], ' ');
|
||||||
|
const pif = pci_class.progIfName(cc.base, cc.subclass, cc.prog_if);
|
||||||
|
const cline = if (pif.len != 0)
|
||||||
|
std.fmt.bufPrint(cbuf[pad..], "class 0x{x:0>2} ({s}) subclass 0x{x:0>2} ({s}) progif 0x{x:0>2} ({s})\n", .{ cc.base, pci_class.className(cc.base), cc.subclass, pci_class.subclassName(cc.base, cc.subclass), cc.prog_if, pif }) catch return
|
||||||
|
else
|
||||||
|
std.fmt.bufPrint(cbuf[pad..], "class 0x{x:0>2} ({s}) subclass 0x{x:0>2} ({s}) progif 0x{x:0>2}\n", .{ cc.base, pci_class.className(cc.base), cc.subclass, pci_class.subclassName(cc.base, cc.subclass), cc.prog_if }) catch return;
|
||||||
|
emit(cbuf[0 .. pad + cline.len]);
|
||||||
|
}
|
||||||
|
|
||||||
|
for (device.resources[0..device.resource_count]) |r| {
|
||||||
var rbuf: [200]u8 = undefined;
|
var rbuf: [200]u8 = undefined;
|
||||||
const pad = @min(indent + 2, 42);
|
const pad = @min(indent + 2, 42);
|
||||||
@memset(rbuf[0..pad], ' ');
|
@memset(rbuf[0..pad], ' ');
|
||||||
@@ -222,6 +228,6 @@ fn dumpNode(dev: *const Device, depth: usize, emit: *const fn ([]const u8) void)
|
|||||||
emit(rbuf[0 .. pad + rline.len]);
|
emit(rbuf[0 .. pad + rline.len]);
|
||||||
}
|
}
|
||||||
|
|
||||||
var child = dev.first_child;
|
var child = device.first_child;
|
||||||
while (child) |c| : (child = c.next_sibling) dumpNode(c, depth + 1, emit);
|
while (child) |c| : (child = c.next_sibling) dumpNode(c, depth + 1, emit);
|
||||||
}
|
}
|
||||||
@@ -7,10 +7,10 @@
|
|||||||
//! already routes to a backend rather than hard-coding ACPI — wiring the FDT
|
//! already routes to a backend rather than hard-coding ACPI — wiring the FDT
|
||||||
//! parser in later is a local change here, not an architectural one.
|
//! parser in later is a local change here, not an architectural one.
|
||||||
|
|
||||||
const device = @import("device.zig");
|
const device_model = @import("device-model.zig");
|
||||||
|
|
||||||
/// Populate `dt` from a device-tree blob. Not implemented yet.
|
/// Populate `device_tree` from a device-tree blob. Not implemented yet.
|
||||||
pub fn discover(dt: *device.DeviceTree) !void {
|
pub fn discover(device_tree: *device_model.DeviceTree) !void {
|
||||||
_ = dt;
|
_ = device_tree;
|
||||||
return error.Unsupported;
|
return error.Unsupported;
|
||||||
}
|
}
|
||||||
@@ -0,0 +1,568 @@
|
|||||||
|
//! PCI class-code decoding: turn the (class, subclass, prog-IF) triple a PCI function
|
||||||
|
//! reports in its configuration header into human-readable names. Every PCI function
|
||||||
|
//! carries a 24-bit class code — base class (config byte 0x0B), subclass (0x0A), and
|
||||||
|
//! programming interface (0x09) — that says *what it is* far more precisely than
|
||||||
|
//! danos's coarse `DeviceClass`: an ISA bridge, a SATA/AHCI controller, and an xHCI USB
|
||||||
|
//! controller are all just `pci_device` by class, and only this triple tells them
|
||||||
|
//! apart. Pure reference data (from the PCI spec; see https://wiki.osdev.org/PCI) — no
|
||||||
|
//! hardware access — so it is shared by kernel discovery (the device-tree dump) and any
|
||||||
|
//! user-space tool (a future lspci, driver matching).
|
||||||
|
//!
|
||||||
|
//! The taxonomy is named, not numbered (docs/coding-standards.md, "Named values"): the
|
||||||
|
//! base class is a `BaseClass` enum, and each class with defined subclasses gets a
|
||||||
|
//! namespace holding its `SubClass` enum (and, where the spec defines them, per-subclass
|
||||||
|
//! `ProgIf` enums) — the same shape as `usb-ids.zig`. Code that *means* a specific class
|
||||||
|
//! names it (`BaseClass.serial_bus`, `serial_bus.usb.ProgIf.xhci`) rather than writing a
|
||||||
|
//! bare 0x0C/0x03/0x30. The `className`/`subclassName`/`progIfName` functions still take
|
||||||
|
//! the raw bytes a function reports in its header, because that is what hardware hands us.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
/// The three bytes of a PCI class code, unpacked from the `0xCCSSPP` value discovery
|
||||||
|
/// records in `Device.ids.pci_class` (CC = base class, SS = subclass, PP = prog-IF).
|
||||||
|
pub const ClassCode = struct {
|
||||||
|
base: u8, // class code (config offset 0x0B)
|
||||||
|
subclass: u8, // subclass (0x0A)
|
||||||
|
prog_if: u8, // programming interface (0x09)
|
||||||
|
|
||||||
|
pub fn unpack(packed_code: u24) ClassCode {
|
||||||
|
return .{
|
||||||
|
.base = @intCast((packed_code >> 16) & 0xFF),
|
||||||
|
.subclass = @intCast((packed_code >> 8) & 0xFF),
|
||||||
|
.prog_if = @intCast(packed_code & 0xFF),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Re-pack the triple into the `0xCCSSPP` form. Lets code name a whole class code
|
||||||
|
/// from its parts — `pack(.{ .base = @intFromEnum(BaseClass.serial_bus), … })` —
|
||||||
|
/// instead of writing the literal 0x0C0330.
|
||||||
|
pub fn pack(self: ClassCode) u24 {
|
||||||
|
return (@as(u24, self.base) << 16) | (@as(u24, self.subclass) << 8) | self.prog_if;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Base class (config byte 0x0B). Non-exhaustive: an unlisted code is a real but
|
||||||
|
/// unnamed class, decoded as "Unknown" rather than rejected.
|
||||||
|
pub const BaseClass = enum(u8) {
|
||||||
|
unclassified = 0x00,
|
||||||
|
mass_storage = 0x01,
|
||||||
|
network = 0x02,
|
||||||
|
display = 0x03,
|
||||||
|
multimedia = 0x04,
|
||||||
|
memory = 0x05,
|
||||||
|
bridge = 0x06,
|
||||||
|
simple_communication = 0x07,
|
||||||
|
base_system_peripheral = 0x08,
|
||||||
|
input_device = 0x09,
|
||||||
|
docking_station = 0x0A,
|
||||||
|
processor = 0x0B,
|
||||||
|
serial_bus = 0x0C,
|
||||||
|
wireless = 0x0D,
|
||||||
|
intelligent = 0x0E,
|
||||||
|
satellite_communication = 0x0F,
|
||||||
|
encryption = 0x10,
|
||||||
|
signal_processing = 0x11,
|
||||||
|
processing_accelerator = 0x12,
|
||||||
|
non_essential_instrumentation = 0x13,
|
||||||
|
co_processor = 0x40,
|
||||||
|
unassigned = 0xFF,
|
||||||
|
_,
|
||||||
|
|
||||||
|
pub fn name(self: BaseClass) []const u8 {
|
||||||
|
return switch (self) {
|
||||||
|
.unclassified => "Unclassified",
|
||||||
|
.mass_storage => "Mass Storage Controller",
|
||||||
|
.network => "Network Controller",
|
||||||
|
.display => "Display Controller",
|
||||||
|
.multimedia => "Multimedia Controller",
|
||||||
|
.memory => "Memory Controller",
|
||||||
|
.bridge => "Bridge",
|
||||||
|
.simple_communication => "Simple Communication Controller",
|
||||||
|
.base_system_peripheral => "Base System Peripheral",
|
||||||
|
.input_device => "Input Device Controller",
|
||||||
|
.docking_station => "Docking Station",
|
||||||
|
.processor => "Processor",
|
||||||
|
.serial_bus => "Serial Bus Controller",
|
||||||
|
.wireless => "Wireless Controller",
|
||||||
|
.intelligent => "Intelligent Controller",
|
||||||
|
.satellite_communication => "Satellite Communication Controller",
|
||||||
|
.encryption => "Encryption Controller",
|
||||||
|
.signal_processing => "Signal Processing Controller",
|
||||||
|
.processing_accelerator => "Processing Accelerator",
|
||||||
|
.non_essential_instrumentation => "Non-Essential Instrumentation",
|
||||||
|
.co_processor => "Co-Processor",
|
||||||
|
.unassigned => "Unassigned Class (Vendor specific)",
|
||||||
|
_ => "Unknown",
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- Per-class subclass (and prog-IF) taxonomies --------------------------------------
|
||||||
|
// One namespace per base class that has defined subclasses, named after the class. Each
|
||||||
|
// holds an exhaustive `SubClass` enum (so an unlisted code decodes to the class default,
|
||||||
|
// not a wrong name), and, where the spec assigns them, per-subclass `ProgIf` enums.
|
||||||
|
|
||||||
|
pub const mass_storage = struct {
|
||||||
|
pub const SubClass = enum(u8) {
|
||||||
|
scsi_bus = 0x00,
|
||||||
|
ide = 0x01,
|
||||||
|
floppy = 0x02,
|
||||||
|
ipi_bus = 0x03,
|
||||||
|
raid = 0x04,
|
||||||
|
ata = 0x05,
|
||||||
|
serial_ata = 0x06,
|
||||||
|
serial_attached_scsi = 0x07,
|
||||||
|
non_volatile_memory = 0x08,
|
||||||
|
|
||||||
|
pub fn name(self: SubClass) []const u8 {
|
||||||
|
return switch (self) {
|
||||||
|
.scsi_bus => "SCSI Bus Controller",
|
||||||
|
.ide => "IDE Controller",
|
||||||
|
.floppy => "Floppy Disk Controller",
|
||||||
|
.ipi_bus => "IPI Bus Controller",
|
||||||
|
.raid => "RAID Controller",
|
||||||
|
.ata => "ATA Controller",
|
||||||
|
.serial_ata => "Serial ATA Controller",
|
||||||
|
.serial_attached_scsi => "Serial Attached SCSI Controller",
|
||||||
|
.non_volatile_memory => "Non-Volatile Memory Controller",
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const serial_ata = struct {
|
||||||
|
pub const ProgIf = enum(u8) {
|
||||||
|
vendor_specific = 0x00,
|
||||||
|
ahci = 0x01,
|
||||||
|
serial_storage_bus = 0x02,
|
||||||
|
|
||||||
|
pub fn name(self: ProgIf) []const u8 {
|
||||||
|
return switch (self) {
|
||||||
|
.vendor_specific => "Vendor Specific Interface",
|
||||||
|
.ahci => "AHCI 1.0",
|
||||||
|
.serial_storage_bus => "Serial Storage Bus",
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const non_volatile_memory = struct {
|
||||||
|
pub const ProgIf = enum(u8) {
|
||||||
|
nvmhci = 0x01,
|
||||||
|
nvm_express = 0x02,
|
||||||
|
|
||||||
|
pub fn name(self: ProgIf) []const u8 {
|
||||||
|
return switch (self) {
|
||||||
|
.nvmhci => "NVMHCI",
|
||||||
|
.nvm_express => "NVM Express",
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const network = struct {
|
||||||
|
pub const SubClass = enum(u8) {
|
||||||
|
ethernet = 0x00,
|
||||||
|
token_ring = 0x01,
|
||||||
|
fddi = 0x02,
|
||||||
|
atm = 0x03,
|
||||||
|
isdn = 0x04,
|
||||||
|
picmg_multi_computing = 0x06,
|
||||||
|
infiniband = 0x07,
|
||||||
|
fabric = 0x08,
|
||||||
|
|
||||||
|
pub fn name(self: SubClass) []const u8 {
|
||||||
|
return switch (self) {
|
||||||
|
.ethernet => "Ethernet Controller",
|
||||||
|
.token_ring => "Token Ring Controller",
|
||||||
|
.fddi => "FDDI Controller",
|
||||||
|
.atm => "ATM Controller",
|
||||||
|
.isdn => "ISDN Controller",
|
||||||
|
.picmg_multi_computing => "PICMG 2.14 Multi Computing Controller",
|
||||||
|
.infiniband => "Infiniband Controller",
|
||||||
|
.fabric => "Fabric Controller",
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const display = struct {
|
||||||
|
pub const SubClass = enum(u8) {
|
||||||
|
vga_compatible = 0x00,
|
||||||
|
xga = 0x01,
|
||||||
|
three_dimensional = 0x02,
|
||||||
|
|
||||||
|
pub fn name(self: SubClass) []const u8 {
|
||||||
|
return switch (self) {
|
||||||
|
.vga_compatible => "VGA Compatible Controller",
|
||||||
|
.xga => "XGA Controller",
|
||||||
|
.three_dimensional => "3D Controller (Not VGA-Compatible)",
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const vga_compatible = struct {
|
||||||
|
pub const ProgIf = enum(u8) {
|
||||||
|
vga = 0x00,
|
||||||
|
compatible_8514 = 0x01,
|
||||||
|
|
||||||
|
pub fn name(self: ProgIf) []const u8 {
|
||||||
|
return switch (self) {
|
||||||
|
.vga => "VGA Controller",
|
||||||
|
.compatible_8514 => "8514-Compatible Controller",
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const multimedia = struct {
|
||||||
|
pub const SubClass = enum(u8) {
|
||||||
|
video = 0x00,
|
||||||
|
audio = 0x01,
|
||||||
|
telephony = 0x02,
|
||||||
|
audio_device = 0x03,
|
||||||
|
|
||||||
|
pub fn name(self: SubClass) []const u8 {
|
||||||
|
return switch (self) {
|
||||||
|
.video => "Multimedia Video Controller",
|
||||||
|
.audio => "Multimedia Audio Controller",
|
||||||
|
.telephony => "Computer Telephony Device",
|
||||||
|
.audio_device => "Audio Device",
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const memory = struct {
|
||||||
|
pub const SubClass = enum(u8) {
|
||||||
|
ram = 0x00,
|
||||||
|
flash = 0x01,
|
||||||
|
|
||||||
|
pub fn name(self: SubClass) []const u8 {
|
||||||
|
return switch (self) {
|
||||||
|
.ram => "RAM Controller",
|
||||||
|
.flash => "Flash Controller",
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const bridge = struct {
|
||||||
|
pub const SubClass = enum(u8) {
|
||||||
|
host = 0x00,
|
||||||
|
isa = 0x01,
|
||||||
|
eisa = 0x02,
|
||||||
|
mca = 0x03,
|
||||||
|
pci_to_pci = 0x04,
|
||||||
|
pcmcia = 0x05,
|
||||||
|
nubus = 0x06,
|
||||||
|
cardbus = 0x07,
|
||||||
|
raceway = 0x08,
|
||||||
|
pci_to_pci_semi_transparent = 0x09,
|
||||||
|
infiniband_to_pci = 0x0A,
|
||||||
|
|
||||||
|
pub fn name(self: SubClass) []const u8 {
|
||||||
|
return switch (self) {
|
||||||
|
.host => "Host Bridge",
|
||||||
|
.isa => "ISA Bridge",
|
||||||
|
.eisa => "EISA Bridge",
|
||||||
|
.mca => "MCA Bridge",
|
||||||
|
.pci_to_pci => "PCI-to-PCI Bridge",
|
||||||
|
.pcmcia => "PCMCIA Bridge",
|
||||||
|
.nubus => "NuBus Bridge",
|
||||||
|
.cardbus => "CardBus Bridge",
|
||||||
|
.raceway => "RACEway Bridge",
|
||||||
|
.pci_to_pci_semi_transparent => "PCI-to-PCI Bridge (Semi-Transparent)",
|
||||||
|
.infiniband_to_pci => "InfiniBand-to-PCI Host Bridge",
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const pci_to_pci = struct {
|
||||||
|
pub const ProgIf = enum(u8) {
|
||||||
|
normal_decode = 0x00,
|
||||||
|
subtractive_decode = 0x01,
|
||||||
|
|
||||||
|
pub fn name(self: ProgIf) []const u8 {
|
||||||
|
return switch (self) {
|
||||||
|
.normal_decode => "Normal Decode",
|
||||||
|
.subtractive_decode => "Subtractive Decode",
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const simple_communication = struct {
|
||||||
|
pub const SubClass = enum(u8) {
|
||||||
|
serial = 0x00,
|
||||||
|
parallel = 0x01,
|
||||||
|
multiport_serial = 0x02,
|
||||||
|
modem = 0x03,
|
||||||
|
gpib = 0x04,
|
||||||
|
smart_card = 0x05,
|
||||||
|
|
||||||
|
pub fn name(self: SubClass) []const u8 {
|
||||||
|
return switch (self) {
|
||||||
|
.serial => "Serial Controller",
|
||||||
|
.parallel => "Parallel Controller",
|
||||||
|
.multiport_serial => "Multiport Serial Controller",
|
||||||
|
.modem => "Modem",
|
||||||
|
.gpib => "IEEE 488.1/2 (GPIB) Controller",
|
||||||
|
.smart_card => "Smart Card Controller",
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const serial = struct {
|
||||||
|
pub const ProgIf = enum(u8) {
|
||||||
|
compatible_8250 = 0x00,
|
||||||
|
compatible_16450 = 0x01,
|
||||||
|
compatible_16550 = 0x02,
|
||||||
|
compatible_16650 = 0x03,
|
||||||
|
compatible_16750 = 0x04,
|
||||||
|
compatible_16850 = 0x05,
|
||||||
|
compatible_16950 = 0x06,
|
||||||
|
|
||||||
|
pub fn name(self: ProgIf) []const u8 {
|
||||||
|
return switch (self) {
|
||||||
|
.compatible_8250 => "8250-Compatible (Generic XT)",
|
||||||
|
.compatible_16450 => "16450-Compatible",
|
||||||
|
.compatible_16550 => "16550-Compatible",
|
||||||
|
.compatible_16650 => "16650-Compatible",
|
||||||
|
.compatible_16750 => "16750-Compatible",
|
||||||
|
.compatible_16850 => "16850-Compatible",
|
||||||
|
.compatible_16950 => "16950-Compatible",
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const base_system_peripheral = struct {
|
||||||
|
pub const SubClass = enum(u8) {
|
||||||
|
pic = 0x00,
|
||||||
|
dma = 0x01,
|
||||||
|
timer = 0x02,
|
||||||
|
rtc = 0x03,
|
||||||
|
pci_hot_plug = 0x04,
|
||||||
|
sd_host = 0x05,
|
||||||
|
iommu = 0x06,
|
||||||
|
|
||||||
|
pub fn name(self: SubClass) []const u8 {
|
||||||
|
return switch (self) {
|
||||||
|
.pic => "PIC",
|
||||||
|
.dma => "DMA Controller",
|
||||||
|
.timer => "Timer",
|
||||||
|
.rtc => "RTC Controller",
|
||||||
|
.pci_hot_plug => "PCI Hot-Plug Controller",
|
||||||
|
.sd_host => "SD Host Controller",
|
||||||
|
.iommu => "IOMMU",
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const input_device = struct {
|
||||||
|
pub const SubClass = enum(u8) {
|
||||||
|
keyboard = 0x00,
|
||||||
|
digitizer_pen = 0x01,
|
||||||
|
mouse = 0x02,
|
||||||
|
scanner = 0x03,
|
||||||
|
gameport = 0x04,
|
||||||
|
|
||||||
|
pub fn name(self: SubClass) []const u8 {
|
||||||
|
return switch (self) {
|
||||||
|
.keyboard => "Keyboard Controller",
|
||||||
|
.digitizer_pen => "Digitizer Pen",
|
||||||
|
.mouse => "Mouse Controller",
|
||||||
|
.scanner => "Scanner Controller",
|
||||||
|
.gameport => "Gameport Controller",
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const serial_bus = struct {
|
||||||
|
pub const SubClass = enum(u8) {
|
||||||
|
firewire = 0x00,
|
||||||
|
access_bus = 0x01,
|
||||||
|
ssa = 0x02,
|
||||||
|
usb = 0x03,
|
||||||
|
fibre_channel = 0x04,
|
||||||
|
smbus = 0x05,
|
||||||
|
infiniband = 0x06,
|
||||||
|
ipmi = 0x07,
|
||||||
|
sercos = 0x08,
|
||||||
|
canbus = 0x09,
|
||||||
|
|
||||||
|
pub fn name(self: SubClass) []const u8 {
|
||||||
|
return switch (self) {
|
||||||
|
.firewire => "FireWire (IEEE 1394) Controller",
|
||||||
|
.access_bus => "ACCESS Bus Controller",
|
||||||
|
.ssa => "SSA",
|
||||||
|
.usb => "USB Controller",
|
||||||
|
.fibre_channel => "Fibre Channel",
|
||||||
|
.smbus => "SMBus Controller",
|
||||||
|
.infiniband => "InfiniBand Controller",
|
||||||
|
.ipmi => "IPMI Interface",
|
||||||
|
.sercos => "SERCOS Interface (IEC 61491)",
|
||||||
|
.canbus => "CANbus Controller",
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const usb = struct {
|
||||||
|
pub const ProgIf = enum(u8) {
|
||||||
|
uhci = 0x00,
|
||||||
|
ohci = 0x10,
|
||||||
|
ehci = 0x20,
|
||||||
|
xhci = 0x30,
|
||||||
|
unspecified = 0x80,
|
||||||
|
device = 0xFE,
|
||||||
|
|
||||||
|
pub fn name(self: ProgIf) []const u8 {
|
||||||
|
return switch (self) {
|
||||||
|
.uhci => "UHCI Controller",
|
||||||
|
.ohci => "OHCI Controller",
|
||||||
|
.ehci => "EHCI (USB2) Controller",
|
||||||
|
.xhci => "XHCI (USB3) Controller",
|
||||||
|
.unspecified => "Unspecified",
|
||||||
|
.device => "USB Device (not a host controller)",
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const wireless = struct {
|
||||||
|
pub const SubClass = enum(u8) {
|
||||||
|
irda = 0x00,
|
||||||
|
consumer_ir = 0x01,
|
||||||
|
rf = 0x10,
|
||||||
|
bluetooth = 0x11,
|
||||||
|
broadband = 0x12,
|
||||||
|
ethernet_802_1a = 0x20,
|
||||||
|
ethernet_802_1b = 0x21,
|
||||||
|
|
||||||
|
pub fn name(self: SubClass) []const u8 {
|
||||||
|
return switch (self) {
|
||||||
|
.irda => "iRDA Compatible Controller",
|
||||||
|
.consumer_ir => "Consumer IR Controller",
|
||||||
|
.rf => "RF Controller",
|
||||||
|
.bluetooth => "Bluetooth Controller",
|
||||||
|
.broadband => "Broadband Controller",
|
||||||
|
.ethernet_802_1a => "Ethernet Controller (802.1a)",
|
||||||
|
.ethernet_802_1b => "Ethernet Controller (802.1b)",
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
// --- Raw-byte decoding (what a function reports in its header) -------------------------
|
||||||
|
|
||||||
|
/// The name of an exhaustive class-code enum member, or null if `value` is not one — the
|
||||||
|
/// bridge from a raw config byte to a named taxonomy above.
|
||||||
|
fn enumName(comptime Enum: type, value: u8) ?[]const u8 {
|
||||||
|
return (std.enums.fromInt(Enum, value) orelse return null).name();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Name of the base class (byte 0x0B), e.g. `0x06` -> "Bridge".
|
||||||
|
pub fn className(base: u8) []const u8 {
|
||||||
|
return @as(BaseClass, @enumFromInt(base)).name();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Name of the subclass within its base class, e.g. `(0x06, 0x01)` -> "ISA Bridge".
|
||||||
|
/// Subclass `0x80` is "Other" by PCI convention; anything unlisted is "Unknown".
|
||||||
|
pub fn subclassName(base: u8, subclass: u8) []const u8 {
|
||||||
|
const named: ?[]const u8 = switch (@as(BaseClass, @enumFromInt(base))) {
|
||||||
|
.mass_storage => enumName(mass_storage.SubClass, subclass),
|
||||||
|
.network => enumName(network.SubClass, subclass),
|
||||||
|
.display => enumName(display.SubClass, subclass),
|
||||||
|
.multimedia => enumName(multimedia.SubClass, subclass),
|
||||||
|
.memory => enumName(memory.SubClass, subclass),
|
||||||
|
.bridge => enumName(bridge.SubClass, subclass),
|
||||||
|
.simple_communication => enumName(simple_communication.SubClass, subclass),
|
||||||
|
.base_system_peripheral => enumName(base_system_peripheral.SubClass, subclass),
|
||||||
|
.input_device => enumName(input_device.SubClass, subclass),
|
||||||
|
.serial_bus => enumName(serial_bus.SubClass, subclass),
|
||||||
|
.wireless => enumName(wireless.SubClass, subclass),
|
||||||
|
else => null,
|
||||||
|
};
|
||||||
|
return named orelse defaultSubclass(subclass);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn defaultSubclass(subclass: u8) []const u8 {
|
||||||
|
return if (subclass == 0x80) "Other" else "Unknown";
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Name of the programming interface, for the subclasses that define standard ones
|
||||||
|
/// (IDE modes, SATA/AHCI, NVMe, PCI-bridge decode, UART generation, USB host type).
|
||||||
|
/// Returns "" when the prog-IF carries no standard meaning for this class/subclass —
|
||||||
|
/// callers just print the hex byte in that case.
|
||||||
|
pub fn progIfName(base: u8, subclass: u8, prog_if: u8) []const u8 {
|
||||||
|
const named: ?[]const u8 = switch (@as(BaseClass, @enumFromInt(base))) {
|
||||||
|
.mass_storage => switch (std.enums.fromInt(mass_storage.SubClass, subclass) orelse return "") {
|
||||||
|
.serial_ata => enumName(mass_storage.serial_ata.ProgIf, prog_if),
|
||||||
|
.non_volatile_memory => enumName(mass_storage.non_volatile_memory.ProgIf, prog_if),
|
||||||
|
else => null,
|
||||||
|
},
|
||||||
|
.display => switch (std.enums.fromInt(display.SubClass, subclass) orelse return "") {
|
||||||
|
.vga_compatible => enumName(display.vga_compatible.ProgIf, prog_if),
|
||||||
|
else => null,
|
||||||
|
},
|
||||||
|
.bridge => switch (std.enums.fromInt(bridge.SubClass, subclass) orelse return "") {
|
||||||
|
.pci_to_pci => enumName(bridge.pci_to_pci.ProgIf, prog_if),
|
||||||
|
else => null,
|
||||||
|
},
|
||||||
|
.simple_communication => switch (std.enums.fromInt(simple_communication.SubClass, subclass) orelse return "") {
|
||||||
|
.serial => enumName(simple_communication.serial.ProgIf, prog_if),
|
||||||
|
else => null,
|
||||||
|
},
|
||||||
|
.serial_bus => switch (std.enums.fromInt(serial_bus.SubClass, subclass) orelse return "") {
|
||||||
|
.usb => enumName(serial_bus.usb.ProgIf, prog_if),
|
||||||
|
else => null,
|
||||||
|
},
|
||||||
|
else => null,
|
||||||
|
};
|
||||||
|
return named orelse "";
|
||||||
|
}
|
||||||
|
|
||||||
|
test "decodes the common class codes" {
|
||||||
|
const eq = std.testing.expectEqualStrings;
|
||||||
|
|
||||||
|
const isa = ClassCode.unpack(0x06_01_00);
|
||||||
|
try std.testing.expectEqual(@as(u8, 0x06), isa.base);
|
||||||
|
try std.testing.expectEqual(@as(u8, 0x01), isa.subclass);
|
||||||
|
try eq("Bridge", className(isa.base));
|
||||||
|
try eq("ISA Bridge", subclassName(isa.base, isa.subclass));
|
||||||
|
|
||||||
|
const ahci = ClassCode.unpack(0x01_06_01);
|
||||||
|
try eq("Mass Storage Controller", className(ahci.base));
|
||||||
|
try eq("Serial ATA Controller", subclassName(ahci.base, ahci.subclass));
|
||||||
|
try eq("AHCI 1.0", progIfName(ahci.base, ahci.subclass, ahci.prog_if));
|
||||||
|
|
||||||
|
const xhci = ClassCode.unpack(0x0C_03_30);
|
||||||
|
try eq("Serial Bus Controller", className(xhci.base));
|
||||||
|
try eq("USB Controller", subclassName(xhci.base, xhci.subclass));
|
||||||
|
try eq("XHCI (USB3) Controller", progIfName(xhci.base, xhci.subclass, xhci.prog_if));
|
||||||
|
}
|
||||||
|
|
||||||
|
test "unlisted codes fall back without a wrong name" {
|
||||||
|
const eq = std.testing.expectEqualStrings;
|
||||||
|
try eq("Unknown", className(0x77)); // no such base class
|
||||||
|
try eq("Other", subclassName(0x01, 0x80)); // 0x80 is the PCI "Other" convention
|
||||||
|
try eq("Unknown", subclassName(0x01, 0x7A)); // unlisted mass-storage subclass
|
||||||
|
try eq("", progIfName(0x01, 0x06, 0x7F)); // no standard SATA prog-IF for 0x7F
|
||||||
|
try eq("", progIfName(0x02, 0x00, 0x00)); // class with no prog-IF taxonomy at all
|
||||||
|
}
|
||||||
|
|
||||||
|
test "named parts pack to the raw triple" {
|
||||||
|
const xhci = ClassCode{
|
||||||
|
.base = @intFromEnum(BaseClass.serial_bus),
|
||||||
|
.subclass = @intFromEnum(serial_bus.SubClass.usb),
|
||||||
|
.prog_if = @intFromEnum(serial_bus.usb.ProgIf.xhci),
|
||||||
|
};
|
||||||
|
try std.testing.expectEqual(@as(u24, 0x0C_03_30), xhci.pack());
|
||||||
|
}
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user