Compare commits
104
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a53c2b0193 | ||
|
|
bf481c080c | ||
|
|
3a78dcab3f | ||
|
|
470f93a83d | ||
|
|
7798706b41 | ||
|
|
ad40de03c2 | ||
|
|
d8778b4b70 | ||
|
|
79d859a111 | ||
|
|
37fb09f75e | ||
|
|
34ebeb968d | ||
|
|
3cc1d38dd0 | ||
|
|
36e804b848 | ||
|
|
be83a42d42 | ||
|
|
650a1b1595 | ||
|
|
d8c55c6f2f | ||
|
|
2ebfb0c3b0 | ||
|
|
888eaa74e1 | ||
|
|
ed76cbbc79 | ||
|
|
140229b88d | ||
|
|
cb2379fd06 | ||
|
|
1665b239b0 | ||
|
|
70ed0337f8 | ||
|
|
116b8f6c41 | ||
|
|
77901bbba6 | ||
|
|
78582d24d2 | ||
|
|
1cdffe21b1 | ||
|
|
4df90bc212 | ||
|
|
713e77354b | ||
|
|
abb7b1b634 | ||
|
|
f5f0e15769 | ||
|
|
8652b4a724 | ||
|
|
e8233127c7 | ||
|
|
88e92254e9 | ||
|
|
5725d35e5b | ||
|
|
8c95525793 | ||
|
|
5bba5d3363 | ||
|
|
80b72db676 | ||
|
|
aa0c97353a | ||
|
|
dd93204b44 | ||
|
|
c7b17aaa0e | ||
|
|
d7a154a596 | ||
|
|
1bf91115dd | ||
|
|
65244e3103 | ||
|
|
75ccfff171 | ||
|
|
2a583d55a8 | ||
|
|
d218d93f79 | ||
|
|
a5fe63c1dd | ||
|
|
6b3ae0c997 | ||
|
|
59104dd988 | ||
|
|
e499f500c3 | ||
|
|
2ab0d129a2 | ||
|
|
d702d2e9ae | ||
|
|
3ec2d1828a | ||
|
|
21657943a8 | ||
|
|
e612d948d2 | ||
|
|
4ef21fa083 | ||
|
|
125a3b4993 | ||
|
|
e7c7e7b94c | ||
|
|
a581712b09 | ||
|
|
8eb4210251 | ||
|
|
56110b0019 | ||
|
|
afbf10f7fc | ||
|
|
b61b7775b9 | ||
|
|
be81394be3 | ||
|
|
47610e8ee2 | ||
|
|
193fd71a50 | ||
|
|
1c2b3ae64d | ||
|
|
d19a0ae38d | ||
|
|
3d1de37d0e | ||
|
|
ceacc6b514 | ||
|
|
8754d4e46a | ||
|
|
15b70856c9 | ||
|
|
83881641ca | ||
|
|
b0f894f50c | ||
|
|
750a73f050 | ||
|
|
eabb81684a | ||
|
|
0a8c81b17a | ||
|
|
9316f9f1c3 | ||
|
|
7384a730be | ||
|
|
b9d9e1e523 | ||
|
|
37fb3cb0cf | ||
|
|
5d57e7b01c | ||
|
|
8a7235c725 | ||
|
|
f4590fcc19 | ||
|
|
bb7597ea0b | ||
|
|
f57a73e8a1 | ||
|
|
724de7bbd0 | ||
|
|
75bc429aa1 | ||
|
|
91aff2bc4a | ||
|
|
546dd44a2a | ||
|
|
7501bd1703 | ||
|
|
1b47de5058 | ||
|
|
26ac97df31 | ||
|
|
37f72a4df8 | ||
|
|
f02259cae0 | ||
|
|
5940864958 | ||
|
|
91f2cfa17b | ||
|
|
debe815a5c | ||
|
|
dba3939a0f | ||
|
|
43afe6bf2e | ||
|
|
ed7f542006 | ||
|
|
36c29d2d6d | ||
|
|
941ab091db | ||
|
|
cf7c6df41c |
@@ -0,0 +1,16 @@
|
|||||||
|
# EditorConfig: https://editorconfig.org/
|
||||||
|
# Follows the Zig style guide: https://ziglang.org/documentation/0.16.0/#Style-Guide
|
||||||
|
|
||||||
|
root = true
|
||||||
|
|
||||||
|
[*]
|
||||||
|
charset = utf-8
|
||||||
|
end_of_line = lf
|
||||||
|
indent_style = space
|
||||||
|
indent_size = 4
|
||||||
|
trim_trailing_whitespace = true
|
||||||
|
insert_final_newline = true
|
||||||
|
|
||||||
|
[*.zig]
|
||||||
|
# "Line length: aim for 100; use common sense."
|
||||||
|
max_line_length = 100
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
*.zig text eol=lf
|
||||||
@@ -4,3 +4,6 @@ zig-out/
|
|||||||
|
|
||||||
# JetBrains IDE
|
# JetBrains IDE
|
||||||
.idea/
|
.idea/
|
||||||
|
|
||||||
|
.claude/
|
||||||
|
.github/
|
||||||
@@ -2,12 +2,31 @@
|
|||||||
Codename: Shodan
|
Codename: Shodan
|
||||||
Version: 1
|
Version: 1
|
||||||
|
|
||||||
A small operating system, written from scratch in Zig — a bootloader (`src/boot/`)
|
A small resilient operating system, written from scratch in Zig.
|
||||||
and a microkernel (`src/kernel/`), sharing a neutral handoff contract (`src/root.zig`).
|
|
||||||
It boots x86-64 via UEFI, and so far has a framebuffer console, a physical frame
|
## Zen of DanOS:
|
||||||
allocator, its own paging with W^X permissions, interrupt/exception handling, a
|
|
||||||
LAPIC timer, a kernel heap, a fixed-priority preemptive scheduler, and in-kernel IPC
|
- Resilient Micro-Kernel Architecture.
|
||||||
channels. See [`docs/`](docs/README.md) for how each piece works.
|
- Every process run in an isolated user space not kernel space.
|
||||||
|
- Processes cannot take down the entire OS with it when they die or is killed
|
||||||
|
- Stable public runtime library, private OS ABI.
|
||||||
|
- Keeps a stable runtime for user space processes between OS versions (great for backwards compatibility)
|
||||||
|
- Allows the underlying OS to be changed without effecting applications
|
||||||
|
- Provides a boundary to enable compatibility between OS's e.g. POSIX, MUSL etc
|
||||||
|
- Drivers are just isolated processes in user space.
|
||||||
|
- Thin binaries that can be restarted like applications.
|
||||||
|
- Useful during driver development.
|
||||||
|
- Drivers can claim MMIO / ports
|
||||||
|
- Driver resources (e.g. IRQ/Port/MMIO) claims are automatically cleaned up if the driver dies or is killed
|
||||||
|
- Drivers can also hook into the process lifecyle to clean up or reset hardware
|
||||||
|
- No legacy to deal with
|
||||||
|
- Zig code uses a clean coding style (Zen of Zig)
|
||||||
|
- Favor reading code over writing code.
|
||||||
|
- No magic numbers.
|
||||||
|
- No shortend names unless its for ABI compatibility or acronyms
|
||||||
|
- Inter-Process Communication (IPC)
|
||||||
|
- Publish and subscribe to Asynchronous Messages
|
||||||
|
- Talk to services and processes synchronously
|
||||||
|
|
||||||
## Prerequisites
|
## Prerequisites
|
||||||
|
|
||||||
@@ -30,8 +49,10 @@ channels. See [`docs/`](docs/README.md) for how each piece works.
|
|||||||
zig build
|
zig build
|
||||||
```
|
```
|
||||||
|
|
||||||
Produces the UEFI bootloader (`zig-out/bin/BOOTX64.efi`) and the kernel ELF
|
Produces a FHS-shaped `zig-out/` that *is* the danos filesystem and the boot volume:
|
||||||
(`zig-out/bin/kernel`).
|
the UEFI bootloader at `zig-out/EFI/BOOT/BOOTX64.efi`, the kernel at
|
||||||
|
`zig-out/system/kernel`, init at `zig-out/system/services/init`, drivers under
|
||||||
|
`zig-out/system/drivers/`, and the initial-ramdisk at `zig-out/boot/`.
|
||||||
|
|
||||||
## Run
|
## Run
|
||||||
|
|
||||||
@@ -58,7 +79,7 @@ straight into CI.
|
|||||||
|
|
||||||
## Documentation
|
## Documentation
|
||||||
|
|
||||||
Design notes explaining the *why* behind the code live in
|
Design notes explaining *why* behind the code live in
|
||||||
[`docs/`](docs/README.md) — start with [`docs/README.md`](docs/README.md).
|
[`docs/`](docs/README.md) — start with [`docs/README.md`](docs/README.md).
|
||||||
|
|
||||||
## Logo
|
## Logo
|
||||||
|
|||||||
+266
-53
@@ -1,15 +1,24 @@
|
|||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const uefi = std.os.uefi;
|
const uefi = std.os.uefi;
|
||||||
const elf = std.elf;
|
const elf = std.elf;
|
||||||
const danos = @import("danos");
|
const boot_handoff = @import("boot-handoff");
|
||||||
const BootInfo = danos.BootInfo;
|
const BootInformation = boot_handoff.BootInformation;
|
||||||
const GraphicsOutput = uefi.protocol.GraphicsOutput;
|
const GraphicsOutput = uefi.protocol.GraphicsOutput;
|
||||||
const EdidActive = uefi.protocol.edid.Active;
|
const EdidActive = uefi.protocol.edid.Active;
|
||||||
const MemoryMapSlice = uefi.tables.MemoryMapSlice;
|
const MemoryMapSlice = uefi.tables.MemoryMapSlice;
|
||||||
|
|
||||||
/// Name of the kernel ELF on the boot volume (installed to the ESP root by
|
// The boot volume is the FHS-shaped zig-out (see build.zig / docs/README.md), so the
|
||||||
/// build.zig). UEFI wants a UTF-16, null-terminated path.
|
// loader reads each artifact from its addressed FHS path. UEFI paths use backslashes;
|
||||||
const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("kernel");
|
// the FAT driver walks the components itself, so no per-directory dance is needed.
|
||||||
|
|
||||||
|
/// The kernel image: /system/kernel.
|
||||||
|
const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("system\\kernel");
|
||||||
|
|
||||||
|
/// The init program: /system/services/init.
|
||||||
|
const init_file_name = std.unicode.utf8ToUtf16LeStringLiteral("system\\services\\init");
|
||||||
|
|
||||||
|
/// The initial-ramdisk (the VFS server + drivers), in /boot.
|
||||||
|
const initial_ramdisk_file_name = std.unicode.utf8ToUtf16LeStringLiteral("boot\\initial-ramdisk.img");
|
||||||
|
|
||||||
/// Physical page size, and the sentinel UEFI uses to seek to end-of-file.
|
/// Physical page size, and the sentinel UEFI uses to seek to end-of-file.
|
||||||
const page_size = 4096;
|
const page_size = 4096;
|
||||||
@@ -33,10 +42,10 @@ fn boot() !noreturn {
|
|||||||
|
|
||||||
// Everything the kernel needs must be gathered *before* we exit boot
|
// Everything the kernel needs must be gathered *before* we exit boot
|
||||||
// services, since afterwards none of these calls are usable.
|
// services, since afterwards none of these calls are usable.
|
||||||
var boot_info: BootInfo = .{
|
var boot_information: BootInformation = .{
|
||||||
// A missing GOP (a headless machine) is not fatal — hand the kernel a
|
// A missing GOP (a headless machine) is not fatal — hand the kernel a
|
||||||
// "no framebuffer" descriptor (base 0) and let it log to serial instead.
|
// "no framebuffer" descriptor (base 0) and let it log to serial instead.
|
||||||
.framebuffer = queryFramebuffer(bs) catch danos.Framebuffer{
|
.framebuffer = queryFramebuffer(bs) catch boot_handoff.Framebuffer{
|
||||||
.base = 0,
|
.base = 0,
|
||||||
.width = 0,
|
.width = 0,
|
||||||
.height = 0,
|
.height = 0,
|
||||||
@@ -52,16 +61,38 @@ fn boot() !noreturn {
|
|||||||
.acpi_rsdp = if (acpiRootSystemDescriptorPointer()) |p| @intFromPtr(p) else 0,
|
.acpi_rsdp = if (acpiRootSystemDescriptorPointer()) |p| @intFromPtr(p) else 0,
|
||||||
};
|
};
|
||||||
|
|
||||||
const entry = try loadKernel(bs, &boot_info);
|
const entry = try loadKernel(bs, &boot_information);
|
||||||
|
|
||||||
|
// Best effort: a volume without /system/services/init still boots (kernel-only).
|
||||||
|
loadInit(bs, &boot_information) catch |err| {
|
||||||
|
log("danos: no /system/services/init (");
|
||||||
|
logBytes(@errorName(err));
|
||||||
|
log(") - booting without user space\r\n");
|
||||||
|
};
|
||||||
|
|
||||||
|
// Best effort: the initial_ramdisk (VFS server + drivers) is optional too.
|
||||||
|
loadInitialRamdisk(bs, &boot_information) catch |err| {
|
||||||
|
log("danos: no initial_ramdisk (");
|
||||||
|
logBytes(@errorName(err));
|
||||||
|
log(")\r\n");
|
||||||
|
};
|
||||||
|
|
||||||
|
// Build the page tables the kernel starts life on: identity + a physmap of
|
||||||
|
// low RAM, plus the higher-half kernel image once it links high. Allocated
|
||||||
|
// now, while boot services (and the memory map) are still stable — nothing
|
||||||
|
// is allocatable after ExitBootServices, and any allocation between fetching
|
||||||
|
// the map and exiting would invalidate the map key.
|
||||||
|
const cr3 = try buildBootstrapTables(bs, &boot_information);
|
||||||
|
|
||||||
log("danos: kernel loaded, exiting boot services\r\n");
|
log("danos: kernel loaded, exiting boot services\r\n");
|
||||||
boot_info.memory_map = try exitBootServices(bs);
|
boot_information.memory_map = try exitBootServices(bs);
|
||||||
|
|
||||||
// Hand control to the kernel. `danos.kernel_abi` is SysV, so the pointer is
|
// Switch onto our tables and jump to the kernel in one uninterruptible step.
|
||||||
// passed in RDI as the kernel expects — not RCX, which this UEFI binary's
|
// We load RDI explicitly (SystemV first arg) rather than trusting this UEFI
|
||||||
// default `.c` convention (Microsoft x64) would use.
|
// binary's Microsoft-x64 default, and jump straight to the (possibly
|
||||||
const kernel: *const fn (*const BootInfo) callconv(danos.kernel_abi) noreturn = @ptrFromInt(entry);
|
// higher-half) entry — the bootstrap tables map both the low loader code
|
||||||
kernel(&boot_info);
|
// executing this and the kernel's link address.
|
||||||
|
handoff(cr3, entry, &boot_information);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A display resolution in pixels.
|
/// A display resolution in pixels.
|
||||||
@@ -69,7 +100,7 @@ const Resolution = struct { width: u32, height: u32 };
|
|||||||
|
|
||||||
/// Switch the GPU to the monitor's native resolution (when we can determine it)
|
/// Switch the GPU to the monitor's native resolution (when we can determine it)
|
||||||
/// and read the resulting graphics mode into our own framebuffer description.
|
/// and read the resulting graphics mode into our own framebuffer description.
|
||||||
fn queryFramebuffer(bs: *uefi.tables.BootServices) !danos.Framebuffer {
|
fn queryFramebuffer(bs: *uefi.tables.BootServices) !boot_handoff.Framebuffer {
|
||||||
// Enumerate the handles carrying the Graphics Output Protocol. We go through
|
// Enumerate the handles carrying the Graphics Output Protocol. We go through
|
||||||
// handles (rather than locateProtocol) so we can also ask them for their EDID,
|
// handles (rather than locateProtocol) so we can also ask them for their EDID,
|
||||||
// which is what tells us the panel's native resolution.
|
// which is what tells us the panel's native resolution.
|
||||||
@@ -101,7 +132,7 @@ fn queryFramebuffer(bs: *uefi.tables.BootServices) !danos.Framebuffer {
|
|||||||
|
|
||||||
/// Map a GOP pixel format to ours. bit_mask / blt_only have no linear 32bpp
|
/// Map a GOP pixel format to ours. bit_mask / blt_only have no linear 32bpp
|
||||||
/// layout we can paint into, so they're rejected.
|
/// layout we can paint into, so they're rejected.
|
||||||
fn pixelFormat(fmt: GraphicsOutput.PixelFormat) !danos.PixelFormat {
|
fn pixelFormat(fmt: GraphicsOutput.PixelFormat) !boot_handoff.PixelFormat {
|
||||||
return switch (fmt) {
|
return switch (fmt) {
|
||||||
.red_green_blue_reserved_8_bit_per_color => .rgbx,
|
.red_green_blue_reserved_8_bit_per_color => .rgbx,
|
||||||
.blue_green_red_reserved_8_bit_per_color => .bgrx,
|
.blue_green_red_reserved_8_bit_per_color => .bgrx,
|
||||||
@@ -166,7 +197,7 @@ fn edidNative(edid: []const u8) ?Resolution {
|
|||||||
|
|
||||||
/// Open the kernel on the volume we booted from, read it into a pool buffer,
|
/// Open the kernel on the volume we booted from, read it into a pool buffer,
|
||||||
/// load its segments, and return the physical entry-point address.
|
/// load its segments, and return the physical entry-point address.
|
||||||
fn loadKernel(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !usize {
|
fn loadKernel(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !usize {
|
||||||
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
|
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
|
||||||
return error.NoLoadedImage;
|
return error.NoLoadedImage;
|
||||||
const device = loaded.device_handle orelse return error.NoBootDevice;
|
const device = loaded.device_handle orelse return error.NoBootDevice;
|
||||||
@@ -195,13 +226,190 @@ fn loadKernel(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !usize {
|
|||||||
read_total += n;
|
read_total += n;
|
||||||
}
|
}
|
||||||
|
|
||||||
return loadElf(bs, image, boot_info);
|
return loadElf(bs, image, boot_information);
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- bootstrap page tables -------------------------------------------------
|
||||||
|
// The kernel is (or will be) linked in the higher half but loaded low; the
|
||||||
|
// firmware's identity map doesn't cover the higher half, so the loader builds
|
||||||
|
// the first set of real page tables and switches CR3 before jumping in. They
|
||||||
|
// carry: an identity map of low RAM (so the loader's own code/stack executing
|
||||||
|
// the switch stays valid, and the low-linked kernel keeps working during the
|
||||||
|
// staged move), a physmap at boot_handoff.physmap_base (the kernel's permanent way to
|
||||||
|
// reach physical memory), and 4 KiB mappings of any higher-half kernel segment.
|
||||||
|
// The kernel later builds its own precise tables (paging.init) and abandons
|
||||||
|
// these; they leak as reserved LoaderData (~a handful of frames).
|
||||||
|
|
||||||
|
const pte_present: u64 = 1 << 0;
|
||||||
|
const pte_write: u64 = 1 << 1;
|
||||||
|
const pte_ps: u64 = 1 << 7; // page-size: a 2 MiB leaf at the PD level
|
||||||
|
const pte_address: u64 = 0x000F_FFFF_FFFF_F000;
|
||||||
|
const gib: u64 = 1 << 30;
|
||||||
|
|
||||||
|
/// A bump allocator over a pre-reserved block of zeroed frames, for page tables.
|
||||||
|
const TablePool = struct {
|
||||||
|
base: usize,
|
||||||
|
next: usize,
|
||||||
|
cap: usize,
|
||||||
|
|
||||||
|
fn alloc(self: *TablePool) !u64 {
|
||||||
|
if (self.next >= self.cap) return error.OutOfBootstrapFrames;
|
||||||
|
const frame = self.base + self.next * page_size;
|
||||||
|
self.next += 1;
|
||||||
|
@memset(@as(*[512]u64, @ptrFromInt(frame)), 0);
|
||||||
|
return frame;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn table(physical: u64) *[512]u64 {
|
||||||
|
return @ptrFromInt(physical);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Return the next-level table an entry points at, creating it if absent.
|
||||||
|
fn descend(self: *TablePool, entry: *u64) !u64 {
|
||||||
|
if (entry.* & pte_present != 0) return entry.* & pte_address;
|
||||||
|
const frame = try self.alloc();
|
||||||
|
entry.* = frame | pte_present | pte_write;
|
||||||
|
return frame;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn map2M(self: *TablePool, pml4: u64, virtual: u64, physical: u64) !void {
|
||||||
|
const pml4e = &table(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
|
const pdpt = try self.descend(pml4e);
|
||||||
|
const pdpte = &table(pdpt)[(virtual >> 30) & 0x1FF];
|
||||||
|
const pd = try self.descend(pdpte);
|
||||||
|
table(pd)[(virtual >> 21) & 0x1FF] = (physical & ~@as(u64, 0x1F_FFFF)) | pte_present | pte_write | pte_ps;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn map4K(self: *TablePool, pml4: u64, virtual: u64, physical: u64) !void {
|
||||||
|
const pml4e = &table(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
|
const pdpt = try self.descend(pml4e);
|
||||||
|
const pdpte = &table(pdpt)[(virtual >> 30) & 0x1FF];
|
||||||
|
const pd = try self.descend(pdpte);
|
||||||
|
const pde = &table(pd)[(virtual >> 21) & 0x1FF];
|
||||||
|
const pt = try self.descend(pde);
|
||||||
|
table(pt)[(virtual >> 12) & 0x1FF] = (physical & pte_address) | pte_present | pte_write;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Build the bootstrap tables and return the physical PML4 address (for CR3).
|
||||||
|
/// No NX bits are set anywhere, so EFER.NXE (still off here) is irrelevant.
|
||||||
|
fn buildBootstrapTables(bs: *uefi.tables.BootServices, boot_information: *const BootInformation) !u64 {
|
||||||
|
// 64 frames (256 KiB) — comfortably covers a PML4, two PDPTs, eight PDs for
|
||||||
|
// the 4 GiB identity+physmap ranges, plus the kernel image's PTs.
|
||||||
|
const pool_pages = 64;
|
||||||
|
const block = try bs.allocatePages(.any, .loader_data, pool_pages);
|
||||||
|
var pool = TablePool{ .base = @intFromPtr(block.ptr), .next = 0, .cap = pool_pages };
|
||||||
|
|
||||||
|
const pml4 = try pool.alloc();
|
||||||
|
|
||||||
|
// Identity + physmap for low RAM. 4 GiB covers all of QEMU's RAM and MMIO
|
||||||
|
// (LAPIC/IOAPIC/HPET/ECAM/framebuffer under q35); a machine with RAM or a
|
||||||
|
// framebuffer above 4 GiB would extend this — see the fb window below.
|
||||||
|
var address: u64 = 0;
|
||||||
|
while (address < 4 * gib) : (address += 2 << 20) {
|
||||||
|
try pool.map2M(pml4, address, address); // identity
|
||||||
|
try pool.map2M(pml4, boot_handoff.physicalToVirtual(address), address); // physmap
|
||||||
|
}
|
||||||
|
|
||||||
|
// A framebuffer above the 4 GiB window needs its own identity + physmap
|
||||||
|
// pages (the kernel touches fb.base before it builds its own tables).
|
||||||
|
const fb = boot_information.framebuffer;
|
||||||
|
if (fb.present() and fb.base + @as(u64, fb.pitch) * fb.height > 4 * gib) {
|
||||||
|
var p: u64 = fb.base & ~@as(u64, 0x1F_FFFF);
|
||||||
|
const fb_end = fb.base + @as(u64, fb.pitch) * fb.height;
|
||||||
|
while (p < fb_end) : (p += 2 << 20) {
|
||||||
|
try pool.map2M(pml4, p, p);
|
||||||
|
try pool.map2M(pml4, boot_handoff.physicalToVirtual(p), p);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Higher-half kernel segments (virtual != physical). While the kernel still links
|
||||||
|
// low its segments sit in the identity range and need no separate mapping
|
||||||
|
// (and 4 KiB-mapping them would collide with the 2 MiB identity leaves), so
|
||||||
|
// only map segments that actually live in the higher half.
|
||||||
|
for (boot_information.kernel_segments[0..boot_information.kernel_segment_count]) |seg| {
|
||||||
|
if (seg.virtual < boot_handoff.kernel_virt_base) continue;
|
||||||
|
var off: u64 = 0;
|
||||||
|
while (off < seg.pages * page_size) : (off += page_size) {
|
||||||
|
try pool.map4K(pml4, seg.virtual + off, seg.physical + off);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return pml4;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Switch onto `cr3` and jump to the kernel `entry` with `boot_information` in RDI,
|
||||||
|
/// interrupts off, in one block so nothing runs between the CR3 load and the
|
||||||
|
/// jump. The identity mapping keeps this low loader code valid across the CR3
|
||||||
|
/// load; the jump target is mapped (identity while low, higher-half once high).
|
||||||
|
fn handoff(cr3: u64, entry: usize, boot_information: *const BootInformation) noreturn {
|
||||||
|
asm volatile (
|
||||||
|
\\cli
|
||||||
|
\\movq %[cr3], %%cr3
|
||||||
|
\\movq %[bi], %%rdi
|
||||||
|
\\callq *%[entry]
|
||||||
|
:
|
||||||
|
: [cr3] "r" (cr3),
|
||||||
|
[bi] "r" (boot_information),
|
||||||
|
[entry] "r" (entry),
|
||||||
|
: .{ .memory = true });
|
||||||
|
unreachable;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read a whole file off the boot volume into a pool buffer that outlives the
|
||||||
|
/// loader. The buffer is deliberately NOT freed: it's LoaderData, which the
|
||||||
|
/// memory-map conversion classifies as reserved, so the kernel identity-maps it
|
||||||
|
/// and reads from there. Returns the buffer (pointer + length).
|
||||||
|
fn loadFile(bs: *uefi.tables.BootServices, name: [*:0]const u16) ![]u8 {
|
||||||
|
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
|
||||||
|
return error.NoLoadedImage;
|
||||||
|
const device = loaded.device_handle orelse return error.NoBootDevice;
|
||||||
|
const fs = (try bs.handleProtocol(uefi.protocol.SimpleFileSystem, device)) orelse
|
||||||
|
return error.NoFileSystem;
|
||||||
|
|
||||||
|
const root = try fs.openVolume();
|
||||||
|
defer _ = root.close() catch {};
|
||||||
|
|
||||||
|
const file = try root.open(name, .read, .{});
|
||||||
|
defer _ = file.close() catch {};
|
||||||
|
|
||||||
|
try file.setPosition(seek_end);
|
||||||
|
const size: usize = @intCast(try file.getPosition());
|
||||||
|
try file.setPosition(0);
|
||||||
|
if (size == 0) return error.EmptyFile;
|
||||||
|
|
||||||
|
const image = try bs.allocatePool(.loader_data, size); // survives the handoff
|
||||||
|
|
||||||
|
var read_total: usize = 0;
|
||||||
|
while (read_total < size) {
|
||||||
|
const n = try file.read(image[read_total..]);
|
||||||
|
if (n == 0) return error.UnexpectedEof;
|
||||||
|
read_total += n;
|
||||||
|
}
|
||||||
|
return image[0..size];
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Ferry the init program (/system/services/init) to the kernel. The kernel does the ELF
|
||||||
|
/// loading itself (into ring-3 mappings) — the loader just carries the bytes.
|
||||||
|
fn loadInit(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||||
|
const image = try loadFile(bs, init_file_name);
|
||||||
|
boot_information.init_base = @intFromPtr(image.ptr);
|
||||||
|
boot_information.init_len = image.len;
|
||||||
|
log("danos: /system/services/init loaded\r\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Ferry the initial_ramdisk (the VFS server + drivers) to the kernel, same as init.
|
||||||
|
fn loadInitialRamdisk(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||||
|
const image = try loadFile(bs, initial_ramdisk_file_name);
|
||||||
|
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
||||||
|
boot_information.initial_ramdisk_len = image.len;
|
||||||
|
log("danos: initial_ramdisk loaded\r\n");
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
||||||
/// record each segment's layout so the kernel can re-map itself with the right
|
/// record each segment's layout so the kernel can re-map itself with the right
|
||||||
/// permissions.
|
/// permissions.
|
||||||
fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !usize {
|
fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_information: *BootInformation) !usize {
|
||||||
if (image.len < @sizeOf(elf.Elf64_Ehdr)) return error.NotElf;
|
if (image.len < @sizeOf(elf.Elf64_Ehdr)) return error.NotElf;
|
||||||
const ehdr: *const elf.Elf64_Ehdr = @ptrCast(@alignCast(image.ptr));
|
const ehdr: *const elf.Elf64_Ehdr = @ptrCast(@alignCast(image.ptr));
|
||||||
|
|
||||||
@@ -218,7 +426,9 @@ fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !us
|
|||||||
|
|
||||||
// Reserve the exact physical pages this segment is linked at. This
|
// Reserve the exact physical pages this segment is linked at. This
|
||||||
// requires the segment's p_paddr to be free in the firmware memory map;
|
// requires the segment's p_paddr to be free in the firmware memory map;
|
||||||
// if it collides, adjust `image_base` in build.zig.
|
// if it collides, adjust `image_base` in build.zig. (Once the kernel
|
||||||
|
// links high — M2 step 4 — p_paddr becomes a separate low load address
|
||||||
|
// via the linker's AT(), and this stays a valid physical allocation.)
|
||||||
const mem_sz: usize = @intCast(phdr.p_memsz);
|
const mem_sz: usize = @intCast(phdr.p_memsz);
|
||||||
const pages = (mem_sz + page_size - 1) / page_size;
|
const pages = (mem_sz + page_size - 1) / page_size;
|
||||||
const dest: [*]align(page_size) uefi.Page = @ptrFromInt(phdr.p_paddr);
|
const dest: [*]align(page_size) uefi.Page = @ptrFromInt(phdr.p_paddr);
|
||||||
@@ -231,15 +441,18 @@ fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !us
|
|||||||
@memcpy(bytes[0..file_sz], image[off..][0..file_sz]);
|
@memcpy(bytes[0..file_sz], image[off..][0..file_sz]);
|
||||||
@memset(bytes[file_sz..mem_sz], 0);
|
@memset(bytes[file_sz..mem_sz], 0);
|
||||||
|
|
||||||
// Record it (identity-loaded: virtual == physical) for the kernel's VMM.
|
// Record the virtual link address and the physical load address so the
|
||||||
const n = boot_info.kernel_segment_count;
|
// kernel can map itself with the right permissions post-switch. They're
|
||||||
if (n < boot_info.kernel_segments.len) {
|
// equal while the kernel links low; they diverge once it links high.
|
||||||
boot_info.kernel_segments[n] = .{
|
const n = boot_information.kernel_segment_count;
|
||||||
.virt = phdr.p_vaddr,
|
if (n < boot_information.kernel_segments.len) {
|
||||||
|
boot_information.kernel_segments[n] = .{
|
||||||
|
.virtual = phdr.p_vaddr,
|
||||||
|
.physical = phdr.p_paddr,
|
||||||
.pages = pages,
|
.pages = pages,
|
||||||
.flags = phdr.p_flags,
|
.flags = phdr.p_flags,
|
||||||
};
|
};
|
||||||
boot_info.kernel_segment_count = n + 1;
|
boot_information.kernel_segment_count = n + 1;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -250,27 +463,27 @@ fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !us
|
|||||||
/// neutral form. Allocating the buffers can itself change the map (invalidating
|
/// neutral form. Allocating the buffers can itself change the map (invalidating
|
||||||
/// the key), so retry until it takes. Both buffers are LoaderData, which survives
|
/// the key), so retry until it takes. Both buffers are LoaderData, which survives
|
||||||
/// the exit, so the returned map stays valid for the kernel.
|
/// the exit, so the returned map stays valid for the kernel.
|
||||||
fn exitBootServices(bs: *uefi.tables.BootServices) !danos.MemoryMap {
|
fn exitBootServices(bs: *uefi.tables.BootServices) !boot_handoff.MemoryMap {
|
||||||
var attempts: usize = 0;
|
var attempts: usize = 0;
|
||||||
while (attempts < 8) : (attempts += 1) {
|
while (attempts < 8) : (attempts += 1) {
|
||||||
const info = try bs.getMemoryMapInfo();
|
const info = try bs.getMemoryMapInfo();
|
||||||
// Spare descriptors to absorb the growth from the allocations below.
|
// Spare descriptors to absorb the growth from the allocations below.
|
||||||
const cap = info.len + 8;
|
const cap = info.len + 8;
|
||||||
const map_buf = try bs.allocatePool(.loader_data, cap * info.descriptor_size);
|
const map_buffer = try bs.allocatePool(.loader_data, cap * info.descriptor_size);
|
||||||
const regions_buf = try bs.allocatePool(.loader_data, cap * @sizeOf(danos.MemoryRegion));
|
const regions_buffer = try bs.allocatePool(.loader_data, cap * @sizeOf(boot_handoff.MemoryRegion));
|
||||||
const map = bs.getMemoryMap(map_buf) catch {
|
const map = bs.getMemoryMap(map_buffer) catch {
|
||||||
_ = bs.freePool(map_buf.ptr) catch {};
|
_ = bs.freePool(map_buffer.ptr) catch {};
|
||||||
_ = bs.freePool(regions_buf.ptr) catch {};
|
_ = bs.freePool(regions_buffer.ptr) catch {};
|
||||||
continue;
|
continue;
|
||||||
};
|
};
|
||||||
bs.exitBootServices(uefi.handle, map.info.key) catch {
|
bs.exitBootServices(uefi.handle, map.info.key) catch {
|
||||||
_ = bs.freePool(map_buf.ptr) catch {};
|
_ = bs.freePool(map_buffer.ptr) catch {};
|
||||||
_ = bs.freePool(regions_buf.ptr) catch {};
|
_ = bs.freePool(regions_buffer.ptr) catch {};
|
||||||
continue;
|
continue;
|
||||||
};
|
};
|
||||||
// Boot services are gone; do not touch `bs` again. Converting the map is
|
// Boot services are gone; do not touch `bs` again. Converting the map is
|
||||||
// pure computation on memory we already hold, so it's safe here.
|
// pure computation on memory we already hold, so it's safe here.
|
||||||
return convertMemoryMap(map, regions_buf);
|
return convertMemoryMap(map, regions_buffer);
|
||||||
}
|
}
|
||||||
return error.ExitBootServicesFailed;
|
return error.ExitBootServicesFailed;
|
||||||
}
|
}
|
||||||
@@ -279,8 +492,8 @@ fn exitBootServices(bs: *uefi.tables.BootServices) !danos.MemoryMap {
|
|||||||
/// into `out` (sized for at least `map.info.len` regions). Adjacent regions of
|
/// into `out` (sized for at least `map.info.len` regions). Adjacent regions of
|
||||||
/// the same kind are coalesced. This is the loader's job precisely so the kernel
|
/// the same kind are coalesced. This is the loader's job precisely so the kernel
|
||||||
/// never sees UEFI's vocabulary — the same seam the framebuffer already uses.
|
/// never sees UEFI's vocabulary — the same seam the framebuffer already uses.
|
||||||
fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
fn convertMemoryMap(map: MemoryMapSlice, out: []u8) boot_handoff.MemoryMap {
|
||||||
const regions: [*]danos.MemoryRegion = @ptrCast(@alignCast(out.ptr));
|
const regions: [*]boot_handoff.MemoryRegion = @ptrCast(@alignCast(out.ptr));
|
||||||
// We're about to call boot-services memory `usable`, but our own stack lives
|
// We're about to call boot-services memory `usable`, but our own stack lives
|
||||||
// in it and the kernel starts out running on it. Keep the region holding the
|
// in it and the kernel starts out running on it. Keep the region holding the
|
||||||
// current stack pointer reserved so it's never handed out.
|
// current stack pointer reserved so it's never handed out.
|
||||||
@@ -297,16 +510,16 @@ fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
|||||||
if (d.number_of_pages == 0) continue;
|
if (d.number_of_pages == 0) continue;
|
||||||
var kind = classify(d);
|
var kind = classify(d);
|
||||||
// The descriptor we're executing on stays reserved (see rsp above).
|
// The descriptor we're executing on stays reserved (see rsp above).
|
||||||
const region_end = d.physical_start + d.number_of_pages * danos.page_size;
|
const region_end = d.physical_start + d.number_of_pages * page_size;
|
||||||
if (kind == .usable and rsp >= d.physical_start and rsp < region_end) kind = .reserved;
|
if (kind == .usable and rsp >= d.physical_start and rsp < region_end) kind = .reserved;
|
||||||
|
|
||||||
// Coalesce with the previous region if it's the same kind and contiguous.
|
// Coalesce with the previous region if it's the same kind and contiguous.
|
||||||
if (count > 0) {
|
if (count > 0) {
|
||||||
const prev = ®ions[count - 1];
|
const previous = ®ions[count - 1];
|
||||||
if (prev.kind == kind and
|
if (previous.kind == kind and
|
||||||
prev.base + prev.pages * danos.page_size == d.physical_start)
|
previous.base + previous.pages * page_size == d.physical_start)
|
||||||
{
|
{
|
||||||
prev.pages += d.number_of_pages;
|
previous.pages += d.number_of_pages;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -322,7 +535,7 @@ fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
|||||||
|
|
||||||
/// Map a UEFI descriptor to danos's neutral kind. A region that isn't
|
/// Map a UEFI descriptor to danos's neutral kind. A region that isn't
|
||||||
/// writeback-cacheable (`wb`) isn't backed by real RAM — it's device registers or
|
/// writeback-cacheable (`wb`) isn't backed by real RAM — it's device registers or
|
||||||
/// a reserved address-space window (e.g. PCIe config space) — so it's `mmio`
|
/// a reserved address-space window (e.g. PCIe configuration space) — so it's `mmio`
|
||||||
/// regardless of type. UEFI overloads `reserved_memory_type` for both reserved RAM
|
/// regardless of type. UEFI overloads `reserved_memory_type` for both reserved RAM
|
||||||
/// and such holes, and the cache attribute is what actually tells them apart.
|
/// and such holes, and the cache attribute is what actually tells them apart.
|
||||||
///
|
///
|
||||||
@@ -331,7 +544,7 @@ fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
|||||||
/// ever the firmware's (the one live piece, our stack, is reserved by the caller).
|
/// ever the firmware's (the one live piece, our stack, is reserved by the caller).
|
||||||
/// Anything unrecognised is `reserved` — the safe default; our own LoaderData (the
|
/// Anything unrecognised is `reserved` — the safe default; our own LoaderData (the
|
||||||
/// kernel image and these buffers) lands there and stays reserved.
|
/// kernel image and these buffers) lands there and stays reserved.
|
||||||
fn classify(d: *const uefi.tables.MemoryDescriptor) danos.MemoryKind {
|
fn classify(d: *const uefi.tables.MemoryDescriptor) boot_handoff.MemoryKind {
|
||||||
if (!d.attribute.wb) return .mmio;
|
if (!d.attribute.wb) return .mmio;
|
||||||
return switch (d.type) {
|
return switch (d.type) {
|
||||||
.conventional_memory, .boot_services_code, .boot_services_data => .usable,
|
.conventional_memory, .boot_services_code, .boot_services_data => .usable,
|
||||||
@@ -343,33 +556,33 @@ fn classify(d: *const uefi.tables.MemoryDescriptor) danos.MemoryKind {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Write a compile-time string to the console (best effort).
|
/// Write a compile-time string to the console (best effort).
|
||||||
fn log(comptime msg: []const u8) void {
|
fn log(comptime message: []const u8) void {
|
||||||
const out = uefi.system_table.con_out orelse return;
|
const out = uefi.system_table.con_out orelse return;
|
||||||
_ = out.outputString(std.unicode.utf8ToUtf16LeStringLiteral(msg)) catch {};
|
_ = out.outputString(std.unicode.utf8ToUtf16LeStringLiteral(message)) catch {};
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Write a runtime ASCII byte string (e.g. an @errorName) by widening to UTF-16.
|
/// Write a runtime ASCII byte string (e.g. an @errorName) by widening to UTF-16.
|
||||||
fn logBytes(bytes: []const u8) void {
|
fn logBytes(bytes: []const u8) void {
|
||||||
const out = uefi.system_table.con_out orelse return;
|
const out = uefi.system_table.con_out orelse return;
|
||||||
var buf: [128]u16 = undefined;
|
var buffer: [128]u16 = undefined;
|
||||||
var i: usize = 0;
|
var i: usize = 0;
|
||||||
for (bytes) |b| {
|
for (bytes) |b| {
|
||||||
if (i + 1 >= buf.len) break;
|
if (i + 1 >= buffer.len) break;
|
||||||
buf[i] = b;
|
buffer[i] = b;
|
||||||
i += 1;
|
i += 1;
|
||||||
}
|
}
|
||||||
buf[i] = 0;
|
buffer[i] = 0;
|
||||||
_ = out.outputString(buf[0..i :0].ptr) catch {};
|
_ = out.outputString(buffer[0..i :0].ptr) catch {};
|
||||||
}
|
}
|
||||||
|
|
||||||
fn acpiRootSystemDescriptorPointer() ?*const anyopaque {
|
fn acpiRootSystemDescriptorPointer() ?*const anyopaque {
|
||||||
const table_entries = uefi.system_table.number_of_table_entries;
|
const table_entries = uefi.system_table.number_of_table_entries;
|
||||||
const config_tables = uefi.system_table.configuration_table;
|
const configuration_tables = uefi.system_table.configuration_table;
|
||||||
const acpi2 = uefi.tables.ConfigurationTable.acpi_20_table_guid;
|
const acpi2 = uefi.tables.ConfigurationTable.acpi_20_table_guid;
|
||||||
const acpi1 = uefi.tables.ConfigurationTable.acpi_10_table_guid;
|
const acpi1 = uefi.tables.ConfigurationTable.acpi_10_table_guid;
|
||||||
|
|
||||||
for (0..table_entries) |i| {
|
for (0..table_entries) |i| {
|
||||||
const entry = config_tables[i];
|
const entry = configuration_tables[i];
|
||||||
if (entry.vendor_guid.eql(acpi2) or entry.vendor_guid.eql(acpi1)) {
|
if (entry.vendor_guid.eql(acpi2) or entry.vendor_guid.eql(acpi1)) {
|
||||||
return entry.vendor_table;
|
return entry.vendor_table;
|
||||||
}
|
}
|
||||||
@@ -48,50 +48,222 @@ fn timestamp(b: *std.Build) []const u8 {
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Build one user-space binary the same way for every program (init, and later
|
||||||
|
/// the VFS server + drivers): freestanding, ReleaseSmall, `.large` code model
|
||||||
|
/// (the image base is above 4 GiB — smaller models emit 32-bit relocations that
|
||||||
|
/// can't reach), linked against the `runtime` runtime library with the shared user
|
||||||
|
/// link script. Pinned to LLVM + LLD so the script's PHDRS (segment permissions)
|
||||||
|
/// are authoritative — the kernel's W^X user-ELF loader requires exact perms.
|
||||||
|
fn addUserBinary(
|
||||||
|
b: *std.Build,
|
||||||
|
target: std.Build.ResolvedTarget,
|
||||||
|
runtime_module: *std.Build.Module,
|
||||||
|
posix_module: *std.Build.Module,
|
||||||
|
mmio_module: *std.Build.Module,
|
||||||
|
xkeyboard_config_module: *std.Build.Module,
|
||||||
|
acpi_ids_module: *std.Build.Module,
|
||||||
|
name: []const u8,
|
||||||
|
root: []const u8,
|
||||||
|
) *std.Build.Step.Compile {
|
||||||
|
const exe = b.addExecutable(.{
|
||||||
|
.name = name,
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path(root),
|
||||||
|
.target = target,
|
||||||
|
.optimize = .ReleaseSmall,
|
||||||
|
.code_model = .large,
|
||||||
|
.single_threaded = true,
|
||||||
|
.sanitize_c = .off,
|
||||||
|
.stack_check = false,
|
||||||
|
.stack_protector = false,
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "runtime", .module = runtime_module },
|
||||||
|
// POSIX/C compatibility layer, available to any program that wants it
|
||||||
|
// (danos-native code uses `runtime` directly). See library/posix/.
|
||||||
|
.{ .name = "posix", .module = posix_module },
|
||||||
|
// Typed volatile MMIO + memory barriers, for drivers. See library/mmio/.
|
||||||
|
.{ .name = "mmio", .module = mmio_module },
|
||||||
|
// Keyboard layouts (keycode + modifiers -> keysym/character), available
|
||||||
|
// to any program that wants it. See library/xkeyboard-config/.
|
||||||
|
.{ .name = "xkeyboard-config", .module = xkeyboard_config_module },
|
||||||
|
// ACPI/PnP hardware-ID registry, so drivers name devices
|
||||||
|
// (HardwareId.ps2_keyboard) instead of magic "_HID" strings.
|
||||||
|
.{ .name = "acpi-ids", .module = acpi_ids_module },
|
||||||
|
},
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
exe.setLinkerScript(b.path("library/runtime/user.ld"));
|
||||||
|
exe.entry = .{ .symbol_name = "_start" };
|
||||||
|
exe.image_base = 0x7000_0000_0000;
|
||||||
|
exe.use_llvm = true;
|
||||||
|
exe.use_lld = true;
|
||||||
|
return exe;
|
||||||
|
}
|
||||||
|
|
||||||
pub fn build(b: *std.Build) void {
|
pub fn build(b: *std.Build) void {
|
||||||
ensureZigVersion();
|
ensureZigVersion();
|
||||||
|
|
||||||
const target = b.standardTargetOptions(.{});
|
const target = b.standardTargetOptions(.{});
|
||||||
const optimize = b.standardOptimizeOption(.{});
|
const optimize = b.standardOptimizeOption(.{});
|
||||||
|
|
||||||
// Shared handoff definitions (BootInfo, Framebuffer, ...). No target is set,
|
// The three shared contracts, each with its own audience so every import
|
||||||
// so the module inherits the target of whichever binary imports it — the
|
// declares which one it speaks (no target is set, so each inherits the target of
|
||||||
// freestanding kernel or the UEFI bootloader.
|
// whichever binary imports it). See docs/coding-standards.md.
|
||||||
const mod = b.addModule("danos", .{
|
// boot-handoff : loader <-> kernel (BootInformation, framebuffer, VM layout)
|
||||||
.root_source_file = b.path("src/root.zig"),
|
// abi : kernel <-> runtime, core (SystemCall, mmap prot flags, page_size)
|
||||||
|
// device-abi : kernel <-> user, devices (DeviceDescriptor, DeviceClass, ...)
|
||||||
|
const boot_handoff_module = b.addModule("boot-handoff", .{
|
||||||
|
.root_source_file = b.path("system/boot-handoff.zig"),
|
||||||
|
});
|
||||||
|
const abi_module = b.addModule("abi", .{
|
||||||
|
.root_source_file = b.path("system/abi.zig"),
|
||||||
|
});
|
||||||
|
// The devices sub-project's public interface (the flat wire types), exposed as
|
||||||
|
// its own module like vfs-protocol — importable by user space, unlike the
|
||||||
|
// kernel-internal device model it also feeds (system/devices/device-model.zig).
|
||||||
|
const device_abi_module = b.addModule("device-abi", .{
|
||||||
|
.root_source_file = b.path("system/devices/device-abi.zig"),
|
||||||
|
});
|
||||||
|
// PCI class-code decoding (class/subclass/prog-IF -> names). Pure reference data,
|
||||||
|
// shared by kernel discovery (the device-tree dump) and any user-space PCI tool.
|
||||||
|
const pci_class_module = b.addModule("pci-class", .{
|
||||||
|
.root_source_file = b.path("system/devices/pci-class.zig"),
|
||||||
|
});
|
||||||
|
// ACPI/PnP hardware-ID (_HID) names — the flat analog of pci-class for acpi_device
|
||||||
|
// nodes. Also shared reference data.
|
||||||
|
const acpi_ids_module = b.addModule("acpi-ids", .{
|
||||||
|
.root_source_file = b.path("system/devices/acpi-ids.zig"),
|
||||||
|
});
|
||||||
|
|
||||||
|
// Kernel tunables (maximum_cpus, stack sizes, tick rate). A dependency-free module of
|
||||||
|
// compile-time constants, imported wherever a knob is read; keeps the trade-offs
|
||||||
|
// in one place instead of scattered across the tree. See system/parameters.zig.
|
||||||
|
const parameters_module = b.addModule("parameters", .{
|
||||||
|
.root_source_file = b.path("system/parameters.zig"),
|
||||||
});
|
});
|
||||||
|
|
||||||
// Architecture-specific kernel code (CPU ops, entry, later GDT/IDT/paging).
|
// Architecture-specific kernel code (CPU ops, entry, later GDT/IDT/paging).
|
||||||
// The generic kernel imports this as "arch" and never names x86_64, so a new
|
// The generic kernel imports this as "architecture" and never names x86_64, so a new
|
||||||
// architecture is a matter of pointing this module at a different directory.
|
// architecture is a matter of pointing this module at a different directory.
|
||||||
const arch_mod = b.addModule("arch", .{
|
const architecture_module = b.addModule("architecture", .{
|
||||||
.root_source_file = b.path("src/kernel/arch/x86_64/cpu.zig"),
|
.root_source_file = b.path("system/kernel/architecture/x86_64/cpu.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
.{ .name = "danos", .module = mod }, // paging uses the shared BootInfo/memory-map types
|
.{ .name = "boot-handoff", .module = boot_handoff_module }, // paging uses BootInformation/memory-map + physicalToVirtual
|
||||||
|
.{ .name = "abi", .module = abi_module }, // paging works in page_size units
|
||||||
|
.{ .name = "parameters", .module = parameters_module }, // maximum_cpus, ist_stack_size, timer_hz
|
||||||
},
|
},
|
||||||
});
|
});
|
||||||
// CPU-exception stubs — real assembly, since they need cross-symbol
|
// CPU-exception stubs — real assembly, since they need cross-symbol
|
||||||
// jumps/calls that Zig inline asm can't express (see the file's header).
|
// jumps/calls that Zig inline asm can't express (see the file's header).
|
||||||
arch_mod.addAssemblyFile(b.path("src/kernel/arch/x86_64/isr.s"));
|
architecture_module.addAssemblyFile(b.path("system/kernel/architecture/x86_64/isr.s"));
|
||||||
|
// The AP bring-up trampoline: 16-/32-/64-bit mode-switch code that can't be
|
||||||
|
// inline asm (it runs relocated to a low page, not at its link address).
|
||||||
|
architecture_module.addAssemblyFile(b.path("system/kernel/architecture/x86_64/trampoline.s"));
|
||||||
|
|
||||||
// Firmware-agnostic device discovery. The generic kernel imports this as
|
// Firmware-agnostic device discovery. The generic kernel imports this as
|
||||||
// "platform" and asks it to enumerate hardware into a backend-neutral device
|
// "platform" and asks it to enumerate hardware into a backend-neutral device
|
||||||
// tree, never naming ACPI (or, later, device-tree) — the same discipline the
|
// tree, never naming ACPI (or, later, device-tree) — the same discipline the
|
||||||
// arch module applies to CPU code. The backend is selected at runtime from
|
// architecture module applies to CPU code. The backend is selected at runtime from
|
||||||
// the boot handoff (see src/device/platform.zig).
|
// the boot handoff (see system/devices/platform.zig).
|
||||||
const platform_mod = b.addModule("platform", .{
|
const platform_module = b.addModule("platform", .{
|
||||||
.root_source_file = b.path("src/device/platform.zig"),
|
.root_source_file = b.path("system/devices/platform.zig"),
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
.{ .name = "danos", .module = mod }, // BootInfo (carries the ACPI RSDP)
|
.{ .name = "boot-handoff", .module = boot_handoff_module }, // BootInformation (carries the ACPI RSDP), physicalToVirtual
|
||||||
|
.{ .name = "abi", .module = abi_module }, // acpi.zig works in page_size units
|
||||||
|
.{ .name = "device-abi", .module = device_abi_module }, // device-model's DeviceClass/ResourceKind live here
|
||||||
|
.{ .name = "pci-class", .module = pci_class_module }, // decode PCI class codes in the device dump
|
||||||
|
.{ .name = "acpi-ids", .module = acpi_ids_module }, // decode ACPI _HID names in the device dump
|
||||||
|
.{ .name = "parameters", .module = parameters_module }, // maximum_cpus (the discovery pool)
|
||||||
},
|
},
|
||||||
});
|
});
|
||||||
|
|
||||||
// Compile-time config the kernel reads as `@import("build_options")`. The
|
// The VFS wire protocol: the vfs sub-project's public interface, exposed as its
|
||||||
|
// own module. Both the vfs server and the runtime's file layer (unistd/stdio)
|
||||||
|
// depend on this contract by name — neither reaches into the other's files. This
|
||||||
|
// is the first "protocol module" (see docs/driver-model.md); usb/block will
|
||||||
|
// expose theirs the same way.
|
||||||
|
const vfs_protocol_module = b.addModule("vfs-protocol", .{
|
||||||
|
.root_source_file = b.path("system/services/vfs/protocol.zig"),
|
||||||
|
});
|
||||||
|
|
||||||
|
// The input wire protocol: the input service's public interface, exposed as its own
|
||||||
|
// module the same way vfs-protocol is. Shared by the input service, the runtime's
|
||||||
|
// `input` helper (subscribe/publish), and every source and subscriber.
|
||||||
|
const input_protocol_module = b.addModule("input-protocol", .{
|
||||||
|
.root_source_file = b.path("system/services/input/protocol.zig"),
|
||||||
|
});
|
||||||
|
|
||||||
|
// The danos-native user-space runtime: system_call wrappers, the C-convention
|
||||||
|
// heap, IPC helpers, the process start shim, device access. This is the stable
|
||||||
|
// application ABI; POSIX compatibility is a separate library on top (see below).
|
||||||
|
// Compiled into every user binary (see addUserBinary), so it inherits each exe's
|
||||||
|
// `.large` code model — do NOT set a target/code_model here. It imports `abi`
|
||||||
|
// for the shared SystemCall numbers / mmap flags, `device-abi` for the device
|
||||||
|
// types its `device` helper wraps, and re-exports `vfs-protocol` for the VFS
|
||||||
|
// server. It never touches `boot-handoff` — user space has no business with the
|
||||||
|
// loader↔kernel handoff.
|
||||||
|
const runtime_module = b.addModule("runtime", .{
|
||||||
|
.root_source_file = b.path("library/runtime/runtime.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "abi", .module = abi_module },
|
||||||
|
.{ .name = "device-abi", .module = device_abi_module },
|
||||||
|
.{ .name = "vfs-protocol", .module = vfs_protocol_module },
|
||||||
|
.{ .name = "input-protocol", .module = input_protocol_module },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
// The device-manager protocol: hello + (M18.2) tree reports, exposed as its
|
||||||
|
// own module like the other protocol modules. Imported through the runtime.
|
||||||
|
const device_manager_protocol_module = b.addModule("device-manager-protocol", .{
|
||||||
|
.root_source_file = b.path("system/services/device-manager/device-manager-protocol.zig"),
|
||||||
|
});
|
||||||
|
runtime_module.addImport("device-manager-protocol", device_manager_protocol_module);
|
||||||
|
|
||||||
|
// Typed volatile MMIO register access + memory-ordering barriers, for drivers on
|
||||||
|
// top of an mmio_map grant. Depends only on `builtin` (arch-conditional barriers);
|
||||||
|
// no target set, so it inherits each driver's. See library/mmio/mmio.zig.
|
||||||
|
const mmio_module = b.addModule("mmio", .{
|
||||||
|
.root_source_file = b.path("library/mmio/mmio.zig"),
|
||||||
|
});
|
||||||
|
|
||||||
|
// Keyboard layouts compiled from the X11 xkeyboard-config database into native Zig
|
||||||
|
// (keycode + modifiers -> keysym/character). The `layouts` tables are generated by
|
||||||
|
// tools/make-xkeyboard-config.py; `xkeyboard-config` is the hand-written API over them.
|
||||||
|
// No target set, so each inherits its importer's. See library/xkeyboard-config/.
|
||||||
|
const xkb_layouts_module = b.addModule("layouts", .{
|
||||||
|
.root_source_file = b.path("library/xkeyboard-config/generated/layouts.zig"),
|
||||||
|
});
|
||||||
|
const xkeyboard_config_module = b.addModule("xkeyboard-config", .{
|
||||||
|
.root_source_file = b.path("library/xkeyboard-config/xkeyboard-config.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "layouts", .module = xkb_layouts_module },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
// The POSIX / C compatibility layer, a separate library layered strictly over the
|
||||||
|
// runtime (it calls the runtime's IPC/heap, never system calls directly). This is
|
||||||
|
// the one place POSIX/C spellings are allowed verbatim — see docs/coding-standards.md
|
||||||
|
// and library/posix/posix.zig.
|
||||||
|
const posix_module = b.addModule("posix", .{
|
||||||
|
.root_source_file = b.path("library/posix/posix.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "runtime", .module = runtime_module },
|
||||||
|
.{ .name = "vfs-protocol", .module = vfs_protocol_module },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
// The initial_ramdisk container format, shared by the kernel (unpacks it) and the
|
||||||
|
// build-time packer tools/make-initial-ramdisk.py (produces it). No dependencies.
|
||||||
|
const initial_ramdisk_module = b.addModule("initial-ramdisk", .{
|
||||||
|
.root_source_file = b.path("system/initial-ramdisk.zig"),
|
||||||
|
});
|
||||||
|
|
||||||
|
// Compile-time configuration the kernel reads as `@import("build_options")`. The
|
||||||
// QEMU test harness sets -Dtest-case=<name> to run one self-test at boot.
|
// QEMU test harness sets -Dtest-case=<name> to run one self-test at boot.
|
||||||
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see src/kernel/tests.zig)");
|
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see system/kernel/tests.zig)");
|
||||||
const build_options = b.addOptions();
|
const build_options = b.addOptions();
|
||||||
build_options.addOption(?[]const u8, "test_case", test_case);
|
build_options.addOption(?[]const u8, "test_case", test_case);
|
||||||
const build_options_mod = build_options.createModule();
|
const build_options_module = build_options.createModule();
|
||||||
|
|
||||||
// --- Kernel: freestanding x86_64 ELF, jumped to by the bootloader ---
|
// --- Kernel: freestanding x86_64 ELF, jumped to by the bootloader ---
|
||||||
// SSE2 is part of the x86_64 baseline and UEFI leaves it enabled at handoff,
|
// SSE2 is part of the x86_64 baseline and UEFI leaves it enabled at handoff,
|
||||||
@@ -106,62 +278,190 @@ pub fn build(b: *std.Build) void {
|
|||||||
const exe = b.addExecutable(.{
|
const exe = b.addExecutable(.{
|
||||||
.name = "kernel",
|
.name = "kernel",
|
||||||
.root_module = b.createModule(.{
|
.root_module = b.createModule(.{
|
||||||
.root_source_file = b.path("src/kernel/main.zig"),
|
.root_source_file = b.path("system/kernel/kernel.zig"),
|
||||||
.target = kernel_target,
|
.target = kernel_target,
|
||||||
.optimize = optimize,
|
.optimize = optimize,
|
||||||
.code_model = .small, // kernel is linked in the low 2 GiB (see image_base)
|
.code_model = .kernel, // kernel runs in the top 2 GiB (higher half)
|
||||||
.red_zone = false, // interrupts would corrupt the SysV red zone
|
.red_zone = false, // interrupts would corrupt the SystemV red zone
|
||||||
.single_threaded = true, // no scheduler yet; avoids pulling in TLS/atomics
|
.single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores
|
||||||
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
||||||
.stack_check = false, // stack-probe calls have no runtime to land in
|
.stack_check = false, // stack-probe calls have no runtime to land in
|
||||||
.stack_protector = false,
|
.stack_protector = false,
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
.{ .name = "danos", .module = mod },
|
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
||||||
.{ .name = "arch", .module = arch_mod },
|
.{ .name = "abi", .module = abi_module },
|
||||||
.{ .name = "platform", .module = platform_mod },
|
.{ .name = "device-abi", .module = device_abi_module },
|
||||||
.{ .name = "build_options", .module = build_options_mod },
|
.{ .name = "architecture", .module = architecture_module },
|
||||||
|
.{ .name = "platform", .module = platform_module },
|
||||||
|
.{ .name = "parameters", .module = parameters_module },
|
||||||
|
.{ .name = "build_options", .module = build_options_module },
|
||||||
|
.{ .name = "initial-ramdisk", .module = initial_ramdisk_module },
|
||||||
},
|
},
|
||||||
}),
|
}),
|
||||||
});
|
});
|
||||||
exe.setLinkerScript(b.path("src/kernel/arch/x86_64/linker.ld"));
|
exe.setLinkerScript(b.path("system/kernel/architecture/x86_64/linker.ld"));
|
||||||
exe.entry = .{ .symbol_name = "_start" };
|
exe.entry = .{ .symbol_name = "_start" };
|
||||||
// Physical address the bootloader loads the kernel to (identity-mapped under
|
// The self-hosted linker ignores parts of the linker script (PHDRS,
|
||||||
// UEFI). Overrides Zig's default image base so the linker script's layout is
|
// /DISCARD/, AT(), section order); the higher-half layout depends on the
|
||||||
// honoured; adjust here if it collides with firmware-reserved memory.
|
// script being authoritative, so pin the kernel to LLVM + LLD.
|
||||||
exe.image_base = 0x100000; // 1 MiB
|
exe.use_llvm = true;
|
||||||
|
exe.use_lld = true;
|
||||||
|
// Higher-half virtual base (matches KERNEL_VIRT_BASE in linker.ld); the
|
||||||
|
// linker's AT() clauses give each segment a low physical load address
|
||||||
|
// (.text at 1 MiB), which the loader allocates and copies into.
|
||||||
|
exe.image_base = 0xFFFFFFFF80100000;
|
||||||
|
|
||||||
b.installArtifact(exe);
|
// Everything installs into a FHS-shaped zig-out: it IS the danos filesystem *and*
|
||||||
|
// the boot volume. Each binary lands at its addressed, leaf-collapsed path — the
|
||||||
|
// kernel at zig-out/system/kernel (from system/kernel/kernel.zig), init at
|
||||||
|
// zig-out/system/services/init, and so on (see docs/README.md). The bootloader
|
||||||
|
// then loads these FHS paths off the volume.
|
||||||
|
const kernel_install = b.addInstallArtifact(exe, .{ .dest_dir = .{ .override = .{ .custom = "system" } } });
|
||||||
|
b.getInstallStep().dependOn(&kernel_install.step);
|
||||||
|
|
||||||
// Boot methods live in src/boot/, one per way of getting the kernel running.
|
// --- init: the first user-space program (a system service) ---
|
||||||
|
// Built by the shared user-binary recipe (see addUserBinary): freestanding,
|
||||||
|
// linked into the kernel's user region against the `runtime` runtime library, and
|
||||||
|
// started in ring 3 by the kernel's user-ELF loader.
|
||||||
|
const init_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "init", "system/services/init/init.zig");
|
||||||
|
const init_install = b.addInstallArtifact(init_exe, .{ .dest_dir = .{ .override = .{ .custom = "system/services" } } });
|
||||||
|
b.getInstallStep().dependOn(&init_install.step);
|
||||||
|
|
||||||
|
// --- initial_ramdisk: a bundle of extra user binaries (VFS server + drivers) ---
|
||||||
|
// Each is built by the same user-binary recipe, then packed into one image by
|
||||||
|
// the host-side make-initial-ramdisk tool. The bootloader ferries the image to the kernel,
|
||||||
|
// which unpacks it and spawns each program (system/initial-ramdisk.zig).
|
||||||
|
const vfs_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs", "system/services/vfs/vfs.zig");
|
||||||
|
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs-test", "system/services/vfs/vfs-test.zig");
|
||||||
|
const hpet_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "hpet", "system/drivers/hpet/hpet.zig");
|
||||||
|
const bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "bus", "system/drivers/bus/bus.zig");
|
||||||
|
const ps2_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-bus", "system/drivers/ps2-bus/ps2-bus.zig");
|
||||||
|
const ps2_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig");
|
||||||
|
const ps2_mouse_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig");
|
||||||
|
const usb_xhci_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-xhci-bus", "system/drivers/usb-xhci-bus/usb-xhci-bus.zig");
|
||||||
|
// A test fixture, not a real driver: hellos to the device manager, then faults —
|
||||||
|
// what the driver-restart scenario drives the crash-loop cap with.
|
||||||
|
const crash_test_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "crash-test", "system/services/crash-test/crash-test.zig");
|
||||||
|
const device_list_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-list", "system/services/device-list/device-list.zig");
|
||||||
|
// The discovery service: one swappable process per firmware
|
||||||
|
// (docs/m19-m20-plan.md decision 7), bundled under the neutral ramdisk name
|
||||||
|
// "discovery" so the device manager never learns which firmware it is on.
|
||||||
|
// x86 boots describe hardware with ACPI; the Raspberry Pis hand over a
|
||||||
|
// flattened device tree — the aarch64 target flips the default when it
|
||||||
|
// lands (docs/arm.md). Both are placeholders until M20.1 (acpi) and the
|
||||||
|
// ARM bring-up (fdt).
|
||||||
|
const Discovery = enum { acpi, fdt };
|
||||||
|
const discovery = b.option(Discovery, "discovery", "Which discovery service fills the ramdisk's 'discovery' slot (default: acpi)") orelse Discovery.acpi;
|
||||||
|
const discovery_source: []const u8 = switch (discovery) {
|
||||||
|
.acpi => "system/services/acpi/acpi.zig",
|
||||||
|
.fdt => "system/services/fdt/fdt.zig",
|
||||||
|
};
|
||||||
|
const discovery_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "discovery", discovery_source);
|
||||||
|
const device_manager_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-manager", "system/services/device-manager/device-manager.zig");
|
||||||
|
// The input service and its exercisers: the fan-out server, a hardware-free synthetic
|
||||||
|
// source, and a subscriber that doubles as the `input` test's oracle. See docs/input.md.
|
||||||
|
const input_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input", "system/services/input/input.zig");
|
||||||
|
const input_source_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-source", "system/services/input-source/input-source.zig");
|
||||||
|
const input_test_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-test", "system/services/input-test/input-test.zig");
|
||||||
|
const args_echo_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "args-echo", "system/services/args-echo/args-echo.zig");
|
||||||
|
const process_test_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "process-test", "system/services/process-test/process-test.zig");
|
||||||
|
|
||||||
|
// Pack the user binaries into the initial_ramdisk image with the host-side Python tool
|
||||||
|
// (the container format is trivial, and Python sidesteps std API churn). Args:
|
||||||
|
// make-initial-ramdisk.py <out> [<name> <file>]... — one name/file pair per binary.
|
||||||
|
const mk_run = b.addSystemCommand(&.{"python3"});
|
||||||
|
mk_run.addFileArg(b.path("tools/make-initial-ramdisk.py"));
|
||||||
|
const initial_ramdisk_img = mk_run.addOutputFileArg("initial-ramdisk.img");
|
||||||
|
mk_run.addArg("vfs");
|
||||||
|
mk_run.addFileArg(vfs_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("vfs-test");
|
||||||
|
mk_run.addFileArg(vfstest_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("hpet");
|
||||||
|
mk_run.addFileArg(hpet_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("bus");
|
||||||
|
mk_run.addFileArg(bus_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("ps2-bus");
|
||||||
|
mk_run.addFileArg(ps2_bus_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("ps2-keyboard");
|
||||||
|
mk_run.addFileArg(ps2_keyboard_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("ps2-mouse");
|
||||||
|
mk_run.addFileArg(ps2_mouse_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("usb-xhci-bus");
|
||||||
|
mk_run.addFileArg(usb_xhci_bus_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("crash-test");
|
||||||
|
mk_run.addFileArg(crash_test_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("device-list");
|
||||||
|
mk_run.addFileArg(device_list_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("discovery");
|
||||||
|
mk_run.addFileArg(discovery_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("device-manager");
|
||||||
|
mk_run.addFileArg(device_manager_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("input");
|
||||||
|
mk_run.addFileArg(input_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("input-source");
|
||||||
|
mk_run.addFileArg(input_source_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("input-test");
|
||||||
|
mk_run.addFileArg(input_test_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("args-echo");
|
||||||
|
mk_run.addFileArg(args_echo_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("process-test");
|
||||||
|
mk_run.addFileArg(process_test_exe.getEmittedBin());
|
||||||
|
|
||||||
|
// Also install the packed binaries to their FHS homes, so zig-out is a true image
|
||||||
|
// of the filesystem — even though at boot they arrive inside the initial-ramdisk.
|
||||||
|
for ([_]struct { *std.Build.Step.Compile, []const u8 }{
|
||||||
|
.{ vfs_exe, "system/services" },
|
||||||
|
.{ device_manager_exe, "system/services" },
|
||||||
|
.{ input_exe, "system/services" },
|
||||||
|
.{ hpet_exe, "system/drivers" },
|
||||||
|
.{ bus_exe, "system/drivers" },
|
||||||
|
.{ ps2_bus_exe, "system/drivers" },
|
||||||
|
.{ ps2_keyboard_exe, "system/drivers" },
|
||||||
|
.{ ps2_mouse_exe, "system/drivers" },
|
||||||
|
.{ usb_xhci_bus_exe, "system/drivers" },
|
||||||
|
}) |entry| {
|
||||||
|
const step = b.addInstallArtifact(entry[0], .{ .dest_dir = .{ .override = .{ .custom = entry[1] } } });
|
||||||
|
b.getInstallStep().dependOn(&step.step);
|
||||||
|
}
|
||||||
|
|
||||||
|
// The initial-ramdisk itself installs to /boot (with the loaders).
|
||||||
|
const initial_ramdisk_install = b.addInstallFile(initial_ramdisk_img, "boot/initial-ramdisk.img");
|
||||||
|
b.getInstallStep().dependOn(&initial_ramdisk_install.step);
|
||||||
|
|
||||||
|
// Boot methods live in boot/, one per way of getting the kernel running.
|
||||||
// Each is its own binary/entry (a loader is built for its own target); today
|
// Each is its own binary/entry (a loader is built for its own target); today
|
||||||
// that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis.
|
// that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis.
|
||||||
const efiexe = b.addExecutable(.{
|
const efiexe = b.addExecutable(.{
|
||||||
.name = "BOOTX64",
|
.name = "BOOTX64",
|
||||||
.root_module = b.createModule(.{
|
.root_module = b.createModule(.{
|
||||||
.root_source_file = b.path("src/boot/efi.zig"),
|
.root_source_file = b.path("boot/efi.zig"),
|
||||||
.target = b.resolveTargetQuery(.{
|
.target = b.resolveTargetQuery(.{
|
||||||
.cpu_arch = .x86_64,
|
.cpu_arch = .x86_64,
|
||||||
.os_tag = .uefi,
|
.os_tag = .uefi,
|
||||||
}),
|
}),
|
||||||
.optimize = optimize,
|
.optimize = optimize,
|
||||||
.imports = &.{
|
.imports = &.{
|
||||||
.{ .name = "danos", .module = mod },
|
// The bootloader speaks only the handoff contract — never the user ABI.
|
||||||
|
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
||||||
},
|
},
|
||||||
}),
|
}),
|
||||||
});
|
});
|
||||||
|
|
||||||
b.installArtifact(efiexe);
|
// UEFI firmware requires the removable-media loader at exactly \EFI\BOOT\BOOTX64.efi,
|
||||||
|
// so that path is fixed by the firmware (it is /boot's EFI stub, conceptually).
|
||||||
|
const efi_install = b.addInstallArtifact(efiexe, .{ .dest_dir = .{ .override = .{ .custom = "EFI/BOOT" } } });
|
||||||
|
b.getInstallStep().dependOn(&efi_install.step);
|
||||||
|
|
||||||
// --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF ---
|
// --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF ---
|
||||||
// Firmware lives in different places per OS/distro, so probe the known
|
// Firmware lives in different places per OS/distro, so probe the known
|
||||||
// layouts (Arch, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
|
// layouts (Architecture, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
|
||||||
// that exists. Override with -Dovmf-code / -Dovmf-vars if yours is elsewhere.
|
// that exists. Override with -Dovmf-code / -Dovmf-vars if yours is elsewhere.
|
||||||
const ovmf_code = b.option(
|
const ovmf_code = b.option(
|
||||||
[]const u8,
|
[]const u8,
|
||||||
"ovmf-code",
|
"ovmf-code",
|
||||||
"Path to the OVMF_CODE firmware image",
|
"Path to the OVMF_CODE firmware image",
|
||||||
) orelse firstExisting(b.graph.io, &.{
|
) orelse firstExisting(b.graph.io, &.{
|
||||||
"/usr/share/edk2/x64/OVMF_CODE.4m.fd", // Arch
|
"/usr/share/edk2/x64/OVMF_CODE.4m.fd", // Architecture
|
||||||
"/usr/share/OVMF/OVMF_CODE_4M.fd", // Debian/Ubuntu
|
"/usr/share/OVMF/OVMF_CODE_4M.fd", // Debian/Ubuntu
|
||||||
"/usr/share/OVMF/OVMF_CODE.fd", // older Debian/Ubuntu
|
"/usr/share/OVMF/OVMF_CODE.fd", // older Debian/Ubuntu
|
||||||
"/usr/share/edk2-ovmf/x64/OVMF_CODE.fd", // Fedora
|
"/usr/share/edk2-ovmf/x64/OVMF_CODE.fd", // Fedora
|
||||||
@@ -173,7 +473,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
"ovmf-vars",
|
"ovmf-vars",
|
||||||
"Path to the OVMF_VARS firmware image (a writable copy is made)",
|
"Path to the OVMF_VARS firmware image (a writable copy is made)",
|
||||||
) orelse firstExisting(b.graph.io, &.{
|
) orelse firstExisting(b.graph.io, &.{
|
||||||
"/usr/share/edk2/x64/OVMF_VARS.4m.fd", // Arch
|
"/usr/share/edk2/x64/OVMF_VARS.4m.fd", // Architecture
|
||||||
"/usr/share/OVMF/OVMF_VARS_4M.fd", // Debian/Ubuntu
|
"/usr/share/OVMF/OVMF_VARS_4M.fd", // Debian/Ubuntu
|
||||||
"/usr/share/OVMF/OVMF_VARS.fd", // older Debian/Ubuntu
|
"/usr/share/OVMF/OVMF_VARS.fd", // older Debian/Ubuntu
|
||||||
"/usr/share/edk2-ovmf/x64/OVMF_VARS.fd", // Fedora
|
"/usr/share/edk2-ovmf/x64/OVMF_VARS.fd", // Fedora
|
||||||
@@ -181,15 +481,8 @@ pub fn build(b: *std.Build) void {
|
|||||||
"/usr/local/share/qemu/edk2-i386-vars.fd", // macOS Homebrew (Intel)
|
"/usr/local/share/qemu/edk2-i386-vars.fd", // macOS Homebrew (Intel)
|
||||||
});
|
});
|
||||||
|
|
||||||
// Assemble an EFI System Partition layout: esp/EFI/BOOT/BOOTX64.efi
|
// The FHS zig-out (installed above) *is* the boot volume — no separate ESP to
|
||||||
const efi_install = b.addInstallArtifact(efiexe, .{
|
// assemble. QEMU presents it to the guest as a FAT drive below.
|
||||||
.dest_dir = .{ .override = .{ .custom = "esp/EFI/BOOT" } },
|
|
||||||
});
|
|
||||||
// The bootloader loads the kernel by name from the volume root, so drop the
|
|
||||||
// kernel ELF at esp/kernel.
|
|
||||||
const kernel_install = b.addInstallArtifact(exe, .{
|
|
||||||
.dest_dir = .{ .override = .{ .custom = "esp" } },
|
|
||||||
});
|
|
||||||
|
|
||||||
// The firmware needs to write NVRAM, so give it a writable copy of the vars.
|
// The firmware needs to write NVRAM, so give it a writable copy of the vars.
|
||||||
const vars_copy = b.addSystemCommand(&.{ "cp", "-f", ovmf_vars });
|
const vars_copy = b.addSystemCommand(&.{ "cp", "-f", ovmf_vars });
|
||||||
@@ -197,6 +490,19 @@ pub fn build(b: *std.Build) void {
|
|||||||
|
|
||||||
const run_efi = b.addSystemCommand(&.{
|
const run_efi = b.addSystemCommand(&.{
|
||||||
"qemu-system-x86_64",
|
"qemu-system-x86_64",
|
||||||
|
"-device",
|
||||||
|
"qemu-xhci,id=xhci",
|
||||||
|
"-device",
|
||||||
|
"usb-mouse,bus=xhci.0",
|
||||||
|
"-device",
|
||||||
|
"usb-kbd,bus=xhci.0",
|
||||||
|
// "-usb",
|
||||||
|
// "-device",
|
||||||
|
// "usb-ehci,id=ehci",
|
||||||
|
// "-device",
|
||||||
|
// "usb-tablet,bus=usb-bus.0",
|
||||||
|
// "-device",
|
||||||
|
// "usb-mouse,bus=ehci.0",
|
||||||
"-machine",
|
"-machine",
|
||||||
"q35",
|
"q35",
|
||||||
"-m",
|
"-m",
|
||||||
@@ -206,10 +512,10 @@ pub fn build(b: *std.Build) void {
|
|||||||
});
|
});
|
||||||
run_efi.addArg("-drive");
|
run_efi.addArg("-drive");
|
||||||
run_efi.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
run_efi.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
||||||
// Present the ESP directory to the guest as a FAT drive.
|
// Present the FHS zig-out to the guest as a FAT drive — it is the boot volume.
|
||||||
run_efi.addArgs(&.{
|
run_efi.addArgs(&.{
|
||||||
"-drive",
|
"-drive",
|
||||||
b.fmt("format=raw,file=fat:rw:{s}/esp", .{b.install_path}),
|
b.fmt("format=raw,file=fat:rw:{s}", .{b.install_path}),
|
||||||
"-net",
|
"-net",
|
||||||
"none",
|
"none",
|
||||||
// Emulated display advertising 1280x720 as its native (EDID preferred)
|
// Emulated display advertising 1280x720 as its native (EDID preferred)
|
||||||
@@ -220,14 +526,19 @@ pub fn build(b: *std.Build) void {
|
|||||||
"-device",
|
"-device",
|
||||||
"VGA,edid=on,xres=1280,yres=720",
|
"VGA,edid=on,xres=1280,yres=720",
|
||||||
});
|
});
|
||||||
// Always capture the guest's serial0 (the kernel's machine-readable log) to a
|
// Capture the guest's serial0 (danos's machine-readable log) to the qemu-test
|
||||||
// timestamped file under zig-out, so each run leaves its own log behind.
|
// scratch area — a dev/host artifact, kept out of the FHS boot volume we mount.
|
||||||
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ b.install_path, timestamp(b) });
|
// (/var/log/system is reserved for the kernel's own logging system later.) One
|
||||||
|
// timestamped file per run.
|
||||||
|
const log_dir = b.fmt("{s}/qemu-test", .{b.install_path});
|
||||||
|
const make_log_dir = b.addSystemCommand(&.{ "mkdir", "-p", log_dir });
|
||||||
|
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
||||||
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
|
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
|
||||||
run_efi.step.dependOn(&efi_install.step);
|
// The whole FHS zig-out must be installed (and the scratch dir created) before we mount it.
|
||||||
run_efi.step.dependOn(&kernel_install.step);
|
run_efi.step.dependOn(b.getInstallStep());
|
||||||
|
run_efi.step.dependOn(&make_log_dir.step);
|
||||||
|
|
||||||
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/run-x86-64-serial0-<timestamp>.log");
|
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/qemu-test/run-x86-64-serial0-<timestamp>.log");
|
||||||
run_efi_step.dependOn(&run_efi.step);
|
run_efi_step.dependOn(&run_efi.step);
|
||||||
|
|
||||||
// const run_cmd = b.addRunArtifact(exe);
|
// const run_cmd = b.addRunArtifact(exe);
|
||||||
@@ -240,18 +551,50 @@ pub fn build(b: *std.Build) void {
|
|||||||
// }
|
// }
|
||||||
|
|
||||||
// Tests run on the host. The kernel and bootloader target freestanding/UEFI
|
// Tests run on the host. The kernel and bootloader target freestanding/UEFI
|
||||||
// and can't be executed natively, so only the shared module is unit-tested
|
// and can't be executed natively, so only the shared contracts are unit-tested
|
||||||
// here (compiled for the host rather than inheriting a freestanding target).
|
// here (compiled for the host rather than inheriting a freestanding target) —
|
||||||
|
// which also compile-checks that the three-way split stays self-consistent.
|
||||||
|
const test_step = b.step("test", "Run tests");
|
||||||
|
for ([_][]const u8{
|
||||||
|
"system/boot-handoff.zig",
|
||||||
|
"system/abi.zig",
|
||||||
|
"system/devices/device-abi.zig",
|
||||||
|
"system/devices/pci-class.zig", // class/subclass/prog-IF name decoding
|
||||||
|
"system/devices/acpi-ids.zig", // _HID name decoding
|
||||||
|
"system/devices/usb-abi.zig", // wire sizes + bit packings + set-up packet encodings
|
||||||
|
"system/devices/usb-ids.zig", // class/subclass/protocol code assignments
|
||||||
|
"library/mmio/mmio.zig", // barriers assemble + registers round-trip
|
||||||
|
"system/drivers/ps2-bus/scancode.zig", // set-2 decode + keyboard state machine
|
||||||
|
"system/drivers/ps2-bus/mouse-packet.zig", // 3-byte mouse packet assembly
|
||||||
|
}) |root| {
|
||||||
const mod_tests = b.addTest(.{
|
const mod_tests = b.addTest(.{
|
||||||
.root_module = b.createModule(.{
|
.root_module = b.createModule(.{
|
||||||
.root_source_file = b.path("src/root.zig"),
|
.root_source_file = b.path(root),
|
||||||
.target = target,
|
.target = target,
|
||||||
.optimize = optimize,
|
.optimize = optimize,
|
||||||
}),
|
}),
|
||||||
});
|
});
|
||||||
|
test_step.dependOn(&b.addRunArtifact(mod_tests).step);
|
||||||
|
}
|
||||||
|
|
||||||
const run_mod_tests = b.addRunArtifact(mod_tests);
|
// The xkeyboard-config keymap tests need its generated `layouts` import wired, so they
|
||||||
|
// don't fit the plain loop above. Its keycode->character assertions are the end-to-end
|
||||||
|
// proof that the xkb-data -> generator -> Zig-lookup pipeline is correct.
|
||||||
|
const xkb_tests = b.addTest(.{
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path("library/xkeyboard-config/xkeyboard-config.zig"),
|
||||||
|
.target = target,
|
||||||
|
.optimize = optimize,
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "layouts", .module = xkb_layouts_module },
|
||||||
|
},
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
test_step.dependOn(&b.addRunArtifact(xkb_tests).step);
|
||||||
|
|
||||||
const test_step = b.step("test", "Run tests");
|
// Convenience: `zig build gen-xkeyboard-config` regenerates the layout tables from the
|
||||||
test_step.dependOn(&run_mod_tests.step);
|
// vendored data (offline). `fetch` (the network step) stays a manual script run.
|
||||||
|
const gen_xkb = b.addSystemCommand(&.{ "python3", "tools/make-xkeyboard-config.py", "generate" });
|
||||||
|
const gen_xkb_step = b.step("gen-xkeyboard-config", "Regenerate library/xkeyboard-config/generated from the vendored data");
|
||||||
|
gen_xkb_step.dependOn(&gen_xkb.step);
|
||||||
}
|
}
|
||||||
|
|||||||
+123
-13
@@ -35,9 +35,39 @@ rather than restate it. Roughly in the order things happen at runtime:
|
|||||||
multitasking: kernel threads, the context switch, O(1) priority selection, and
|
multitasking: kernel threads, the context switch, O(1) priority selection, and
|
||||||
blocking (sleep, wait queues) — the leap to a running system.
|
blocking (sleep, wait queues) — the leap to a running system.
|
||||||
11. **[ipc.md](ipc.md) — inter-process communication.** Bounded blocking
|
11. **[ipc.md](ipc.md) — inter-process communication.** Bounded blocking
|
||||||
message-passing channels — the backbone the microkernel's isolated servers will
|
message-passing channels, then synchronous call/reply between *processes* over
|
||||||
talk over.
|
endpoints — the backbone the microkernel's isolated servers talk over.
|
||||||
12. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
12. **[syscall.md](syscall.md) — system calls.** How ring 3 asks the kernel for
|
||||||
|
something: the `syscall`/`sysret` fast path, the trap frame, and why the table is
|
||||||
|
deliberately tiny.
|
||||||
|
13. **[drivers.md](drivers.md) — writing a driver.** The payoff: a driver is an
|
||||||
|
ordinary ring-3 process that claims a device, maps its registers, and **sleeps
|
||||||
|
until its hardware interrupts it**. The claim is the capability; `irq_ack` is the
|
||||||
|
unmask.
|
||||||
|
14. **[driver-model.md](driver-model.md) — buses, classes and host controllers.** How
|
||||||
|
real driver stacks factor into three shapes, how families share code, and the
|
||||||
|
proposed ABI for the three primitives still missing (capability passing, DMA +
|
||||||
|
memory barriers, MSI).
|
||||||
|
15. **[process-management.md](process-management.md) — process management.** The
|
||||||
|
microkernel's `ps`/`kill`/SIGCHLD: enumerate as a table snapshot, the
|
||||||
|
supervision link as the kill authority, and child-exit notifications over the
|
||||||
|
same endpoints IRQs arrive on.
|
||||||
|
16. **[process-lifecycle.md](process-lifecycle.md) — the process lifecycle.** Design:
|
||||||
|
signals over IPC as the one lifecycle vocabulary every process speaks — the
|
||||||
|
POSIX.1-1990 words with message delivery instead of stack hijack, the stable
|
||||||
|
`runtime.process` interface, exit reasons, published exit events any stateful
|
||||||
|
service can subscribe to (the VFS releasing dead clients' handles), and the two
|
||||||
|
iron rules (cleanup is the kernel's job; kill is not a signal).
|
||||||
|
17. **[device-manager.md](device-manager.md) — the device manager.** Design: the
|
||||||
|
tree, the matcher, and the supervisor. Tree structure lives in the manager,
|
||||||
|
authority stays in the kernel; bus drivers report what they see; drivers are
|
||||||
|
restarted through the lifecycle vocabulary — the plan that turns
|
||||||
|
[resilience.md](resilience.md)'s restart goal into increments.
|
||||||
|
18. **[input.md](input.md) — the input module.** Broadcasting input events (keyboard,
|
||||||
|
mouse, joystick): why a synchronous rendezvous can't fan out to many listeners, the
|
||||||
|
asynchronous `ipc_send` primitive built to fix it, and the per-device subscribe/publish
|
||||||
|
service layered on top.
|
||||||
|
19. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
||||||
how `while (true) hlt` parks the CPU safely once there's nothing left to do.
|
how `while (true) hlt` parks the CPU safely once there's nothing left to do.
|
||||||
|
|
||||||
Start with the north star:
|
Start with the north star:
|
||||||
@@ -68,6 +98,10 @@ Cutting across all of these:
|
|||||||
- **[smp.md](smp.md) — multiple cores.** A design/research note on how microkernels
|
- **[smp.md](smp.md) — multiple cores.** A design/research note on how microkernels
|
||||||
(L4, seL4) handle SMP — big kernel lock vs per-CPU vs multikernel — and how the
|
(L4, seL4) handle SMP — big kernel lock vs per-CPU vs multikernel — and how the
|
||||||
right choice depends on whether danos is chasing real-time or resilience.
|
right choice depends on whether danos is chasing real-time or resilience.
|
||||||
|
- **[coding-standards.md](coding-standards.md) — coding standards.** The naming rule the
|
||||||
|
tree follows: non-acronyms are spelled out in full (`message`, not `msg`), files are
|
||||||
|
`kebab-case`, code follows Zig's case conventions, and the handful of exceptions
|
||||||
|
(POSIX/C ABI names, `init`/`len`/`ptr`, acronyms).
|
||||||
- **[sysv.md](sysv.md) — the calling convention.** What "the kernel is SysV" means,
|
- **[sysv.md](sysv.md) — the calling convention.** What "the kernel is SysV" means,
|
||||||
and why the loader→kernel boundary has to pin it (the RDI-vs-RCX handoff).
|
and why the loader→kernel boundary has to pin it (the RDI-vs-RCX handoff).
|
||||||
- **[testing.md](testing.md) — testing.** How the kernel is tested by booting it in
|
- **[testing.md](testing.md) — testing.** How the kernel is tested by booting it in
|
||||||
@@ -94,19 +128,95 @@ passing messages over **[IPC](ipc.md)** channels — runs, its CPU-specific bits
|
|||||||
behind the [arch](arch.md) boundary, and when idle, or on a panic, it **halts**
|
behind the [arch](arch.md) boundary, and when idle, or on a panic, it **halts**
|
||||||
([halting.md](halting.md)).
|
([halting.md](halting.md)).
|
||||||
|
|
||||||
|
Above that line the microkernel proper begins: **discovery** ([discovery.md](discovery.md),
|
||||||
|
[acpi.md](acpi.md)) learns what hardware exists, ring-3 processes ask the kernel for
|
||||||
|
things through the small **[syscall](syscall.md)** table, isolated servers reach each
|
||||||
|
other over IPC **endpoints** ([ipc.md](ipc.md)), and a **[driver](drivers.md)** claims
|
||||||
|
a device, maps its registers, and sleeps until the hardware interrupts it — which is
|
||||||
|
the whole reason for the arrangement ([vision.md](vision.md)).
|
||||||
|
|
||||||
|
## Repository layout
|
||||||
|
|
||||||
|
danos is a **monorepo of sub-projects**. Each service or driver is a directory that is
|
||||||
|
its own Zig module — it can hold as many files as it needs, and other sub-projects
|
||||||
|
reach it *by module name*, never by a path into its files. The source tree deliberately
|
||||||
|
**mirrors the runtime FHS** ([danos-file-system-hierarchy-FSH.md](danos-file-system-hierarchy-FSH.md)):
|
||||||
|
what you see under `system/` in the source is what a running danos represents under
|
||||||
|
`/system`.
|
||||||
|
|
||||||
|
**A sub-project is addressed by its directory; its entry point repeats the directory's
|
||||||
|
name.** `system/services/init/` contains `init.zig` (its root), and produces a binary
|
||||||
|
addressed as **`system/services/init`** — the repeated leaf resolves away:
|
||||||
|
|
||||||
|
| Source (root file) | Addressed as (module / binary / FHS path) |
|
||||||
|
|----------------------------------------|--------------------------------------------|
|
||||||
|
| `system/services/init/init.zig` | `system/services/init` → `/system/services/init` |
|
||||||
|
| `system/drivers/hpet/hpet.zig` | `system/drivers/hpet` → `/system/drivers/hpet` |
|
||||||
|
| `library/runtime/runtime.zig` | `library/runtime` (the `runtime` module) |
|
||||||
|
|
||||||
|
In **source**, a sub-project is a directory so it can hold many files — the entry is
|
||||||
|
`init/init.zig`, beside it `vfs/vfs-test.zig`, `vfs/protocol.zig`, and so on. When
|
||||||
|
**addressed or installed**, that collapses to the single canonical path: the `init`
|
||||||
|
binary installs to `/system/services/init` (a file at that path), not
|
||||||
|
`/system/services/init/init`. The repeated leaf exists only in source; the directory is
|
||||||
|
the identity, the entry file is its implementation. (Same idea as a Go package being its
|
||||||
|
directory, or a macOS `.app` bundle addressed by the bundle, not the executable within.)
|
||||||
|
A sub-project's extra files are reached through the module, never as separate paths.
|
||||||
|
|
||||||
|
```
|
||||||
|
system/ → /system danos's own internals (the self-representation)
|
||||||
|
boot-handoff.zig the loader↔kernel contract (the `boot-handoff` module)
|
||||||
|
abi.zig the private kernel↔runtime syscall ABI (the `abi` module)
|
||||||
|
parameters.zig initial-ramdisk.zig shared contracts
|
||||||
|
kernel/ IPC, memory, scheduling, the private syscall dispatch
|
||||||
|
architecture/x86_64/ the `architecture` module (never named by generic code)
|
||||||
|
devices/ the device model /system/devices reflects (+ aml/)
|
||||||
|
device-abi.zig the device wire types (the `device-abi` module)
|
||||||
|
drivers/ hpet/ bus/ one sub-project per driver → /system/drivers
|
||||||
|
services/ init/ vfs/ device-manager/ system servers → /system/services (vfs/ holds
|
||||||
|
vfs.zig, vfs-test.zig, protocol.zig)
|
||||||
|
library/ → /lib libraries, one sub-directory each
|
||||||
|
runtime/ the danos-native runtime — the stable application ABI
|
||||||
|
posix/ POSIX/C compatibility, layered over runtime
|
||||||
|
boot/ → /boot the loaders
|
||||||
|
tools/ test/ host-side build + QEMU test harness
|
||||||
|
```
|
||||||
|
|
||||||
|
A sub-project exposes its **public interface as a module**: `system/services/vfs/` owns
|
||||||
|
the VFS wire protocol (`protocol.zig`, the `vfs-protocol` module), which the POSIX
|
||||||
|
layer imports by name. `usb`/`block` drivers will expose their protocols the same way.
|
||||||
|
|
||||||
|
`library/posix/` is special: it is the **one place** POSIX/C spellings are allowed
|
||||||
|
verbatim (`stat`, `O_CREAT`, `fopen`, `errno`). Everywhere else follows the danos
|
||||||
|
naming rule with no exception — see [coding-standards.md](coding-standards.md). The
|
||||||
|
POSIX layer calls the runtime, never the kernel's system calls directly, so it never
|
||||||
|
appears in the private-ABI path.
|
||||||
|
|
||||||
## Source map
|
## Source map
|
||||||
|
|
||||||
| Area | Code |
|
| Area | Code |
|
||||||
|------|------|
|
|------|------|
|
||||||
| Boot methods (one per way of booting the kernel) | `src/boot/` — `efi.zig` (UEFI) → `BOOTX64.efi` |
|
| Boot methods (one per way of booting the kernel) | `boot/` — `efi.zig` (UEFI) → `BOOTX64.efi` |
|
||||||
| Kernel entry, panic, bring-up | `src/kernel/main.zig` |
|
| Kernel entry, panic, bring-up | `system/kernel/kernel.zig` |
|
||||||
| Shared loader↔kernel contract (`BootInfo`, `Framebuffer`, `MemoryMap`, ABI) | `src/root.zig` |
|
| Loader↔kernel handoff (`BootInfo`, `Framebuffer`, `MemoryMap`, VM layout) | `system/boot-handoff.zig` |
|
||||||
| Physical frame allocator | `src/kernel/pmm.zig` |
|
| Private kernel↔runtime syscall ABI (`SystemCall`, mmap prot flags, `page_size`) — the runtime speaks it, not apps | `system/abi.zig` |
|
||||||
| Kernel heap (`std.mem.Allocator`) | `src/kernel/heap.zig` |
|
| Device wire types (`DeviceDescriptor`, `DeviceClass`, …) | `system/devices/device-abi.zig` |
|
||||||
| Scheduler (fixed-priority preemptive; blocking, wait queues) | `src/kernel/sched.zig` |
|
| Physical frame allocator | `system/kernel/pmm.zig` |
|
||||||
| IPC channels (message passing) | `src/kernel/ipc.zig` |
|
| Kernel heap (`std.mem.Allocator`) | `system/kernel/heap.zig` |
|
||||||
| Framebuffer text console (mirrors to serial) | `src/kernel/console.zig` |
|
| Scheduler (fixed-priority preemptive; blocking, wait queues) | `system/kernel/scheduler.zig` |
|
||||||
| In-kernel test cases | `src/kernel/tests.zig` |
|
| Big kernel lock + interrupt-safe critical sections | `system/kernel/sync.zig` |
|
||||||
| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/timer, serial, linker script) | `src/kernel/arch/x86_64/` |
|
| IPC channels between kernel threads (message passing) | `system/kernel/ipc.zig` |
|
||||||
|
| IPC endpoints: cross-address-space call/reply, handles, notifications | `system/kernel/ipc-synchronous.zig` |
|
||||||
|
| User processes: ELF loading, address spaces, the syscall table | `system/kernel/process.zig` |
|
||||||
|
| Device tree + claim capability + `device_register` containment | `system/kernel/devices-broker.zig` |
|
||||||
|
| IRQ-as-IPC: routing a device interrupt to a driver's endpoint | `system/kernel/irq.zig` |
|
||||||
|
| Hardware discovery (ACPI/device tree) behind one neutral device model | `system/devices/` |
|
||||||
|
| Framebuffer text console (mirrors to serial) | `system/kernel/console.zig` |
|
||||||
|
| In-kernel test cases | `system/kernel/tests.zig` |
|
||||||
|
| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/IO-APIC/timer, serial, linker script) | `system/kernel/architecture/x86_64/` |
|
||||||
|
| danos-native runtime (`runtime`): syscall wrappers, heap, IPC, device access — the stable application ABI | `library/runtime/` |
|
||||||
|
| POSIX/C compatibility (`posix`): unistd, stdio — the one place POSIX names are allowed | `library/posix/` |
|
||||||
|
| System services (init, the VFS server + `protocol`, the device-manager) | `system/services/` |
|
||||||
|
| Device drivers, one sub-project each (`hpet` leaf driver, `bus` bus driver) | `system/drivers/` |
|
||||||
| Build + `run-x86-64` (QEMU/OVMF) | `build.zig` |
|
| Build + `run-x86-64` (QEMU/OVMF) | `build.zig` |
|
||||||
| QEMU integration test harness | `test/qemu_test.py` |
|
| QEMU integration test harness | `test/qemu_test.py` |
|
||||||
|
|||||||
+6
-6
@@ -16,13 +16,13 @@ the RSDT's address is a field *inside* the RSDP. The platform follows that point
|
|||||||
UEFI configuration table
|
UEFI configuration table
|
||||||
│ the loader reads the RSDP's physical address
|
│ the loader reads the RSDP's physical address
|
||||||
▼
|
▼
|
||||||
BootInfo.acpi_rsdp (u64, in the shared `danos` module) src/root.zig
|
BootInfo.acpi_rsdp (u64, in the loader↔kernel handoff) system/boot-handoff.zig
|
||||||
│ the kernel forwards the whole BootInfo
|
│ the kernel forwards the whole BootInfo
|
||||||
▼
|
▼
|
||||||
platform.discover(boot_info, …) src/device/platform.zig
|
platform.discover(boot_info, …) system/devices/platform.zig
|
||||||
│ reads boot_info.acpi_rsdp, hands it to the ACPI backend
|
│ reads boot_info.acpi_rsdp, hands it to the ACPI backend
|
||||||
▼
|
▼
|
||||||
acpi.discover(rsdp_phys, …) src/device/acpi.zig
|
acpi.discover(rsdp_phys, …) system/devices/acpi.zig
|
||||||
│ dereferences the RSDP, reads the pointer it contains
|
│ dereferences the RSDP, reads the pointer it contains
|
||||||
▼
|
▼
|
||||||
RSDP ──(a field in the struct)──► RSDT / XSDT ──► SDTs (MADT, MCFG, FADT, HPET, DSDT…)
|
RSDP ──(a field in the struct)──► RSDT / XSDT ──► SDTs (MADT, MCFG, FADT, HPET, DSDT…)
|
||||||
@@ -35,7 +35,7 @@ successor the **XSDT** (ACPI 2.0+) — which in turn lists every other SDT.
|
|||||||
## Step 1 — the loader finds the RSDP
|
## Step 1 — the loader finds the RSDP
|
||||||
|
|
||||||
Only the firmware knows where ACPI lives, so the RSDP must be grabbed while UEFI is
|
Only the firmware knows where ACPI lives, so the RSDP must be grabbed while UEFI is
|
||||||
still up. `acpiRootSystemDescriptorPointer()` in `src/boot/efi.zig` walks the UEFI
|
still up. `acpiRootSystemDescriptorPointer()` in `boot/efi.zig` walks the UEFI
|
||||||
**configuration table** for the ACPI GUID and returns the vendor pointer — the same
|
**configuration table** for the ACPI GUID and returns the vendor pointer — the same
|
||||||
"grab it before `ExitBootServices`" pattern as the [framebuffer](framebuffer.md) and
|
"grab it before `ExitBootServices`" pattern as the [framebuffer](framebuffer.md) and
|
||||||
the [memory map](memory-map.md).
|
the [memory map](memory-map.md).
|
||||||
@@ -44,11 +44,11 @@ the [memory map](memory-map.md).
|
|||||||
|
|
||||||
The loader can't just call the device module: the bootloader binary and the kernel
|
The loader can't just call the device module: the bootloader binary and the kernel
|
||||||
binary are compiled separately, and **the loader isn't linked against the `platform`
|
binary are compiled separately, and **the loader isn't linked against the `platform`
|
||||||
module at all** (it imports only the shared `danos` module). So instead of a call, it
|
module at all** (it imports only the `boot-handoff` contract). So instead of a call, it
|
||||||
deposits a value in the handoff struct:
|
deposits a value in the handoff struct:
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
// src/boot/efi.zig — while boot services are still up
|
// boot/efi.zig — while boot services are still up
|
||||||
.acpi_rsdp = if (acpiRootSystemDescriptorPointer()) |p| @intFromPtr(p) else 0,
|
.acpi_rsdp = if (acpiRootSystemDescriptorPointer()) |p| @intFromPtr(p) else 0,
|
||||||
```
|
```
|
||||||
|
|
||||||
|
|||||||
+13
-13
@@ -13,7 +13,7 @@ runtime dispatch. `build.zig` exposes one architecture's code as a module called
|
|||||||
|
|
||||||
```zig
|
```zig
|
||||||
const arch_mod = b.addModule("arch", .{
|
const arch_mod = b.addModule("arch", .{
|
||||||
.root_source_file = b.path("src/kernel/arch/x86_64/cpu.zig"),
|
.root_source_file = b.path("system/kernel/architecture/x86_64/cpu.zig"),
|
||||||
});
|
});
|
||||||
```
|
```
|
||||||
|
|
||||||
@@ -26,7 +26,7 @@ arch.halt(); // never says "x86_64"
|
|||||||
```
|
```
|
||||||
|
|
||||||
Adding a second architecture is then a build-time choice: create
|
Adding a second architecture is then a build-time choice: create
|
||||||
`src/kernel/arch/aarch64/`, and point the `arch` module at it when the target CPU is
|
`system/kernel/arch/aarch64/`, and point the `arch` module at it when the target CPU is
|
||||||
AArch64. `main.zig` and `console.zig` don't change. **That compiler-checked module
|
AArch64. `main.zig` and `console.zig` don't change. **That compiler-checked module
|
||||||
boundary _is_ the architecture interface** — when a new arch is missing a function
|
boundary _is_ the architecture interface** — when a new arch is missing a function
|
||||||
the generic kernel calls, the build fails and names exactly what's missing.
|
the generic kernel calls, the build fails and names exactly what's missing.
|
||||||
@@ -36,7 +36,7 @@ the generic kernel calls, the build fails and names exactly what's missing.
|
|||||||
The split follows a simple test: does it name a CPU instruction, a hardware
|
The split follows a simple test: does it name a CPU instruction, a hardware
|
||||||
register, or a memory-management structure? If so, it's arch-specific.
|
register, or a memory-management structure? If so, it's arch-specific.
|
||||||
|
|
||||||
| Arch-specific — `src/kernel/arch/x86_64/` | Generic — kernel core |
|
| Arch-specific — `system/kernel/architecture/x86_64/` | Generic — kernel core |
|
||||||
|---|---|
|
|---|---|
|
||||||
| `cpu.zig`: `halt()` (`hlt`), later GDT/IDT/paging | `console.zig` — pure pixel math, works anywhere |
|
| `cpu.zig`: `halt()` (`hlt`), later GDT/IDT/paging | `console.zig` — pure pixel math, works anywhere |
|
||||||
| `linker.ld` — link layout, load address | `main.zig` — `kmain` orchestration, panic handler |
|
| `linker.ld` — link layout, load address | `main.zig` — `kmain` orchestration, panic handler |
|
||||||
@@ -51,9 +51,9 @@ should end up on the generic side; the arch module stays small.
|
|||||||
There are really two independent questions, and it's worth not conflating them:
|
There are really two independent questions, and it's worth not conflating them:
|
||||||
|
|
||||||
- **CPU architecture** (x86_64 vs AArch64): instructions, MMU, interrupts →
|
- **CPU architecture** (x86_64 vs AArch64): instructions, MMU, interrupts →
|
||||||
`src/kernel/arch/<cpu>/`.
|
`system/kernel/arch/<cpu>/`.
|
||||||
- **Boot protocol** (UEFI vs Raspberry Pi firmware + device tree): handled
|
- **Boot protocol** (UEFI vs Raspberry Pi firmware + device tree): handled
|
||||||
*separately*, because loaders are their own binaries. `src/boot/efi.zig` builds
|
*separately*, because loaders are their own binaries. `boot/efi.zig` builds
|
||||||
`BOOTX64.efi`, a distinct executable from the kernel ELF. On a Pi there is no
|
`BOOTX64.efi`, a distinct executable from the kernel ELF. On a Pi there is no
|
||||||
separate loader at all — the firmware jumps straight into the kernel with a
|
separate loader at all — the firmware jumps straight into the kernel with a
|
||||||
device-tree pointer, so that entry work would live in the AArch64 arch code.
|
device-tree pointer, so that entry work would live in the AArch64 arch code.
|
||||||
@@ -61,29 +61,29 @@ There are really two independent questions, and it's worth not conflating them:
|
|||||||
|
|
||||||
## Current x86_64 contents
|
## Current x86_64 contents
|
||||||
|
|
||||||
- **`src/kernel/arch/x86_64/cpu.zig`** — the `arch` module root. Exposes `halt()` (see
|
- **`system/kernel/architecture/x86_64/cpu.zig`** — the `arch` module root. Exposes `halt()` (see
|
||||||
[halting.md](halting.md)), `init()` (bring up the descriptor tables),
|
[halting.md](halting.md)), `init()` (bring up the descriptor tables),
|
||||||
`enablePaging()`, `setFaultHandler`, `readCr2`/`readCr3`, and the `CpuState`
|
`enablePaging()`, `setFaultHandler`, `readCr2`/`readCr3`, and the `CpuState`
|
||||||
trap frame.
|
trap frame.
|
||||||
- **`src/kernel/arch/x86_64/gdt.zig`** / **`idt.zig`** / **`tss.zig`** — the GDT, IDT and
|
- **`system/kernel/architecture/x86_64/gdt.zig`** / **`idt.zig`** / **`tss.zig`** — the GDT, IDT and
|
||||||
TSS plus CPU-exception handling (see [interrupts.md](interrupts.md)).
|
TSS plus CPU-exception handling (see [interrupts.md](interrupts.md)).
|
||||||
- **`src/kernel/arch/x86_64/paging.zig`** — the kernel's page tables (see
|
- **`system/kernel/architecture/x86_64/paging.zig`** — the kernel's page tables (see
|
||||||
[paging.md](paging.md)).
|
[paging.md](paging.md)).
|
||||||
- **`src/kernel/arch/x86_64/apic.zig`** — the Local APIC and its timer, the source of
|
- **`system/kernel/architecture/x86_64/apic.zig`** — the Local APIC and its timer, the source of
|
||||||
device interrupts (see [device-interrupts.md](device-interrupts.md)).
|
device interrupts (see [device-interrupts.md](device-interrupts.md)).
|
||||||
- **`src/kernel/arch/x86_64/serial.zig`** / **`io.zig`** — the COM1 UART (the kernel's
|
- **`system/kernel/architecture/x86_64/serial.zig`** / **`io.zig`** — the COM1 UART (the kernel's
|
||||||
machine-readable log channel, see [testing.md](testing.md)) and the shared
|
machine-readable log channel, see [testing.md](testing.md)) and the shared
|
||||||
port-I/O + MSR primitives.
|
port-I/O + MSR primitives.
|
||||||
- **`src/kernel/arch/x86_64/isr.s`** — the exception stubs, the `lgdt`/`lidt`/`ltr` load
|
- **`system/kernel/architecture/x86_64/isr.s`** — the exception stubs, the `lgdt`/`lidt`/`ltr` load
|
||||||
helpers, and the context switch (`switch_context` / `task_trampoline`, see
|
helpers, and the context switch (`switch_context` / `task_trampoline`, see
|
||||||
[scheduling.md](scheduling.md)) — real assembly, since Zig inline asm can't
|
[scheduling.md](scheduling.md)) — real assembly, since Zig inline asm can't
|
||||||
express them.
|
express them.
|
||||||
- **`src/kernel/arch/x86_64/linker.ld`** — the kernel link layout (fixed low load
|
- **`system/kernel/architecture/x86_64/linker.ld`** — the kernel link layout (fixed low load
|
||||||
address, one PT_LOAD per permission set).
|
address, one PT_LOAD per permission set).
|
||||||
|
|
||||||
The kernel entry point `_start` currently still lives in the generic `main.zig` as
|
The kernel entry point `_start` currently still lives in the generic `main.zig` as
|
||||||
a thin trampoline into `kmain`. It's arch-adjacent (its calling convention is
|
a thin trampoline into `kmain`. It's arch-adjacent (its calling convention is
|
||||||
x86_64 [SysV](sysv.md), via the shared `danos.kernel_abi`), but it's three lines
|
x86_64 [SysV](sysv.md), via the shared `system.kernel_abi`), but it's three lines
|
||||||
and mostly generic, so it stays put for now. When AArch64 arrives — where entry means setting
|
and mostly generic, so it stays put for now. When AArch64 arrives — where entry means setting
|
||||||
up a stack and reading a device-tree pointer from a register — the entry work will
|
up a stack and reading a device-tree pointer from a register — the entry work will
|
||||||
be substantial and per-arch, and *that* is when we extract an entry interface into
|
be substantial and per-arch, and *that* is when we extract an entry interface into
|
||||||
|
|||||||
+4
-4
@@ -18,7 +18,7 @@ matters for understanding why. This page maps the landscape so the
|
|||||||
new ISA.
|
new ISA.
|
||||||
|
|
||||||
They are as different from each other as either is from x86-64: separate registers,
|
They are as different from each other as either is from x86-64: separate registers,
|
||||||
page-table formats, and calling conventions. Each needs its own `src/kernel/arch/<name>/`.
|
page-table formats, and calling conventions. Each needs its own `system/kernel/arch/<name>/`.
|
||||||
|
|
||||||
## The Raspberry Pi models
|
## The Raspberry Pi models
|
||||||
|
|
||||||
@@ -57,16 +57,16 @@ the DTB/ACPI tells you what devices exist.
|
|||||||
|
|
||||||
## What danos needs, layer by layer
|
## What danos needs, layer by layer
|
||||||
|
|
||||||
- **One CPU arch module: `src/kernel/arch/aarch64/`** — covering the Zero 2 W and Pi 3-5,
|
- **One CPU arch module: `system/kernel/arch/aarch64/`** — covering the Zero 2 W and Pi 3-5,
|
||||||
providing the same `arch` interface as x86_64: `halt`, context switch,
|
providing the same `arch` interface as x86_64: `halt`, context switch,
|
||||||
interrupt/exception vectors, page tables, a UART, a timer. No `src/kernel/arch/arm/` is
|
interrupt/exception vectors, page tables, a UART, a timer. No `system/kernel/arch/arm/` is
|
||||||
planned (see the decision above), so there's a single ARM backend to write.
|
planned (see the decision above), so there's a single ARM backend to write.
|
||||||
- **A device-tree boot path.** Since stock Pis boot via DTB, danos needs an entry
|
- **A device-tree boot path.** Since stock Pis boot via DTB, danos needs an entry
|
||||||
that parses the DTB's `/memory` and `/reserved-memory` into the neutral
|
that parses the DTB's `/memory` and `/reserved-memory` into the neutral
|
||||||
[`MemoryMap`](memory-map.md) — the same neutral handoff `efi.zig` produces, just
|
[`MemoryMap`](memory-map.md) — the same neutral handoff `efi.zig` produces, just
|
||||||
from a different source. This is where keeping boot-protocol knowledge on the
|
from a different source. This is where keeping boot-protocol knowledge on the
|
||||||
loader side (as we did for the UEFI memory-map classification) pays off.
|
loader side (as we did for the UEFI memory-map classification) pays off.
|
||||||
- **The UEFI loader mostly carries over.** `src/boot/efi.zig` is largely
|
- **The UEFI loader mostly carries over.** `boot/efi.zig` is largely
|
||||||
boot-*protocol* code (`std.os.uefi` protocol calls), not x86 code. Its only truly
|
boot-*protocol* code (`std.os.uefi` protocol calls), not x86 code. Its only truly
|
||||||
x86-specific bits are the ELF machine check (`.X86_64`) and the SysV calling
|
x86-specific bits are the ELF machine check (`.X86_64`) and the SysV calling
|
||||||
convention for the kernel jump. So an `aarch64`-UEFI target (QEMU `virt` + AAVMF)
|
convention for the kernel jump. So an `aarch64`-UEFI target (QEMU `virt` + AAVMF)
|
||||||
|
|||||||
@@ -0,0 +1,168 @@
|
|||||||
|
# Coding standards
|
||||||
|
|
||||||
|
Conventions for danos source. The overriding one, from which most of the rest follows:
|
||||||
|
|
||||||
|
> **Names are spelled out in full. An identifier is not abbreviated unless the
|
||||||
|
> abbreviation is an acronym.**
|
||||||
|
|
||||||
|
`interruptDispatch`, not `intDisp`. `message_len`, not `message_len` (`msg` expands, `len`
|
||||||
|
is a Zig idiom — see the exceptions). `devices_broker`, not `devices_broker`. `scheduler`, not
|
||||||
|
`sched`. The cost of a longer name is paid once, at the keyboard; the cost of a
|
||||||
|
cryptic one is paid every time the code is read, by everyone who reads it. In a
|
||||||
|
microkernel whose whole argument is that a human can hold each piece in their head,
|
||||||
|
that trade is not close.
|
||||||
|
|
||||||
|
## The rule, precisely
|
||||||
|
|
||||||
|
**Acronyms and initialisms stay.** They *are* the full name — expanding them would make
|
||||||
|
the code worse, not better. `IPC`, `MMIO`, `DMA`, `IRQ`, `TSS`, `GDT`, `IDT`, `APIC`,
|
||||||
|
`GSI`, `HPET`, `ACPI`, `PCI`, `EOI`, `BAR`, `ECAM`, `MSI`, `CPU`, `ELF`, `ABI`, `UEFI`,
|
||||||
|
`MMU`, `TLB`, `ISR`, `ISA`, `GAS`, `HAL`, `PMM`, `VMM`, `VFS`, `HID`, `HCD`, `SMP`,
|
||||||
|
`AML`, `MADT`, `MCFG`, `FADT`, `RSDP`, `XSDT`, `RSDT`, `GOP`, `EDID`, `TSC`, `PIT`,
|
||||||
|
`RTC`, `LAPIC`, `SIPI`. In code they carry whatever case the surrounding convention
|
||||||
|
demands: `Hal` the type, `hal` the variable, `mapMmio` the function.
|
||||||
|
|
||||||
|
**Everything else is spelled out.** If it's a word with letters removed, restore them:
|
||||||
|
|
||||||
|
| Abbreviation | Full |
|
||||||
|
|---|---|
|
||||||
|
| `proto` | `protocol` |
|
||||||
|
| `msg` | `message` |
|
||||||
|
| `desc` | `descriptor` |
|
||||||
|
| `res` | `resource` |
|
||||||
|
| `recv` | `receive` |
|
||||||
|
| `buf` | `buffer` |
|
||||||
|
| `cur` | `current` |
|
||||||
|
| `src` / `dst` | `source` / `destination` |
|
||||||
|
| `idx` | `index` |
|
||||||
|
| `addr` | `address` |
|
||||||
|
| `reg` | `register` |
|
||||||
|
| `prev` | `previous` |
|
||||||
|
| `cfg` / `config` | `configuration` |
|
||||||
|
| `arch` | `architecture` |
|
||||||
|
| `sched` | `scheduler` |
|
||||||
|
| `dev` | `device` |
|
||||||
|
| `sys` / `syscall` | `system` / `system_call` |
|
||||||
|
| `info` | `information` |
|
||||||
|
| `dt` | `device_tree` |
|
||||||
|
| `ep` | `endpoint` |
|
||||||
|
| `rt` | `runtime` |
|
||||||
|
| `func` | `function` |
|
||||||
|
| `phys` / `virt` | `physical` / `virtual` |
|
||||||
|
| `wq` | `wait_queue` |
|
||||||
|
|
||||||
|
This list is illustrative, not exhaustive. The rule is the rule; when you meet a new
|
||||||
|
abbreviation, expand it.
|
||||||
|
|
||||||
|
## Exceptions
|
||||||
|
|
||||||
|
Three, and only three.
|
||||||
|
|
||||||
|
1. **Foreign ABI names are spelled exactly as the ABI spells them — but only inside
|
||||||
|
the layer that *is* that ABI.** A function that *is* the C or POSIX interface keeps
|
||||||
|
its name: `fopen`, `fwrite`, `fread`, `malloc`, `calloc`, `realloc`, `free`,
|
||||||
|
`memcpy`, `mmap`, `munmap`, `open`, `read`, `write`, `close`, `lseek`, `stat`,
|
||||||
|
`errno`, `O_CREAT`. We don't get to rename `fwrite` to `fileWrite` — it wouldn't be
|
||||||
|
`fwrite` any more.
|
||||||
|
|
||||||
|
**This exception is scoped to one place: `library/posix/`.** A file under
|
||||||
|
`library/posix/` *is* the foreign ABI, so it keeps the ABI's spellings — that is the
|
||||||
|
whole rule for that directory. **Everywhere else, Zig/danos naming applies with no
|
||||||
|
POSIX exception**, so there is nothing to get wrong: if you're not in
|
||||||
|
`library/posix/`, expand it. A concept POSIX also has gets a danos name outside that
|
||||||
|
layer — the VFS wire protocol carries a `FileStatus`, not a `Stat`, and a `create`
|
||||||
|
flag, not `O_CREAT`; `library/posix/` is what maps `stat`→`status` and
|
||||||
|
`O_CREAT`→`create` at the boundary. (The `syscall` *wrappers* elsewhere are not an
|
||||||
|
exception to this — they wrap the private danos ABI, so they use danos names.)
|
||||||
|
|
||||||
|
2. **Zig idioms are spelled the way Zig spells them.** Three names are the language's,
|
||||||
|
not ours, and are left alone:
|
||||||
|
- **`init` / `deinit`** — the constructor convention (`std.ArrayList.init`), not a
|
||||||
|
shortening of "initialize".
|
||||||
|
- **`len` / `ptr`** — the slice field names (`slice.len`, `slice.ptr`). Our own
|
||||||
|
structs use bare `len`/`ptr` fields to mirror them, so a reader carries one
|
||||||
|
mental model. (Compounds still expand: a field is `message_len`, not
|
||||||
|
`message_length` — `len` is kept, `msg` is not.)
|
||||||
|
- The builtins (`@min`, `@max`, `@memcpy`) and `allocator.alloc` / `.create` are
|
||||||
|
Zig's spelling.
|
||||||
|
|
||||||
|
The rule governs the names *we* coin.
|
||||||
|
|
||||||
|
3. **Single-letter variables in a trivial local scope.** `for (items) |item, i|` may
|
||||||
|
keep `i`; a coordinate may be `x`, `y`. The moment the scope is big enough that the
|
||||||
|
letter's meaning isn't obvious on sight, give it a real name. When in doubt, name it.
|
||||||
|
|
||||||
|
That's all — no Unix-abbreviation exception. The source directories are full words
|
||||||
|
(`system`, `library`, not `src`/`lib`), and there is no daemon `d` suffix: a driver
|
||||||
|
lives in `system/drivers/` and a service in `system/services/`, so the *location*
|
||||||
|
already says what it is. Encoding the role in the name too (`busd`, `vfsd`) is
|
||||||
|
redundant — the program is just `bus`, `vfs`. Don't put in a name what its directory
|
||||||
|
already tells you.
|
||||||
|
|
||||||
|
## A note on collisions
|
||||||
|
|
||||||
|
Two identifiers can legitimately expand to the same word. When they do, keep both
|
||||||
|
meaningful by renaming one to its *specific* identity rather than the generic
|
||||||
|
expansion. Two cases resolved this way:
|
||||||
|
|
||||||
|
- The `config` module (compile-time tunables — `maximum_cpus`, `timer_hz`) would
|
||||||
|
collide with `cfg` (a `PlatformConfiguration` value) at `configuration`. The module
|
||||||
|
became **`parameters`**, which is what it holds.
|
||||||
|
- The kernel `device.zig` module would collide with `dev` (a device value) at
|
||||||
|
`device`. The module alias became **`device_model`**, which is what it is — the
|
||||||
|
device data model (`Device`, `DeviceTree`, `ResourceKind`).
|
||||||
|
- The `Namespace` module alias (`ns`/`nsp` across the AML files) collides with a
|
||||||
|
`Namespace` **instance**. Resolved by dropping the module alias entirely — the two
|
||||||
|
types it provided are imported directly (`const Node = @import("namespace.zig").Node;`)
|
||||||
|
— which frees `namespace` for the instance.
|
||||||
|
|
||||||
|
A related case is one abbreviation with two meanings. In the AML code, `op` means
|
||||||
|
**opcode** (`opcodes.zig`, the `*_opcode` constants) but `Op` in `BinaryOperation` /
|
||||||
|
`LogicOperation` means **operation** — distinguished by case. The per-opcode parser
|
||||||
|
handlers, formerly `opName`/`opField`, are `parseName`/`parseField`: they *parse* the
|
||||||
|
opcode's structure, which says what they do without overloading "op".
|
||||||
|
|
||||||
|
## Case and file names
|
||||||
|
|
||||||
|
Within those spelling rules, follow Zig's own conventions:
|
||||||
|
|
||||||
|
- **Types** — `PascalCase`: `DeviceDescriptor`, `Endpoint`, `WaitQueue`.
|
||||||
|
- **Functions** — `camelCase`: `mapUserDeviceInto`, `notifyFromIsr`.
|
||||||
|
- **Variables, fields, constants** — `snake_case`: `message_length`, `devices_broker`,
|
||||||
|
`notify_badge_bit`.
|
||||||
|
|
||||||
|
**File names are `kebab-case`.** A file named for a multi-word thing hyphenates it:
|
||||||
|
`device-tree.zig`, `ipc-synchronous.zig`, `vfs-protocol.zig`, `devices-broker.zig`. A
|
||||||
|
single word or acronym needs no hyphen: `scheduler.zig`, `paging.zig`, `apic.zig`,
|
||||||
|
`idt.zig`. (The module *alias* a file is imported under still follows the code
|
||||||
|
conventions above — `snake_case` — because it's an identifier, not a filename.)
|
||||||
|
|
||||||
|
**A sub-project's entry point repeats its directory's name** — `init/init.zig`,
|
||||||
|
`runtime/runtime.zig`, `hpet/hpet.zig` — and the sub-project is addressed by the
|
||||||
|
*directory* (`system/services/init`, `library/runtime`), with the repeated leaf
|
||||||
|
resolving away. See the repository-layout section of [README.md](README.md).
|
||||||
|
|
||||||
|
## Why acronyms are the line
|
||||||
|
|
||||||
|
Because an acronym has no letters to restore. `MMIO` doesn't become "memory mapped
|
||||||
|
input output" in code — that expansion is what the acronym *is for*. But `msg` is just
|
||||||
|
`message` with three letters stolen, and stealing them buys nothing a reader wants. The
|
||||||
|
test for "is this an abbreviation I must expand" is simply: *is there a longer word this
|
||||||
|
is a clipped form of?* If yes, write the word. If it's an initialism standing in for a
|
||||||
|
phrase, leave it.
|
||||||
|
|
||||||
|
## Zen of Zig
|
||||||
|
|
||||||
|
* Communicate intent precisely.
|
||||||
|
* Edge cases matter.
|
||||||
|
* Favor reading code over writing code.
|
||||||
|
* Only one obvious way to do things.
|
||||||
|
* Runtime crashes are better than bugs.
|
||||||
|
* Compile errors are better than runtime crashes.
|
||||||
|
* Incremental improvements.
|
||||||
|
* Avoid local maximums.
|
||||||
|
* Reduce the amount one must remember.
|
||||||
|
* Focus on code rather than style.
|
||||||
|
* Resource allocation may fail; resource deallocation must succeed.
|
||||||
|
* Memory is a resource.
|
||||||
|
* Together we serve the users.
|
||||||
@@ -0,0 +1,121 @@
|
|||||||
|
# DanOS Filesystem Hierarchy Standard (DFHS)
|
||||||
|
|
||||||
|
Most modern Unix and Unix-like operating systems follow the FHS. DanOS has its own FHS structure which extends the unix FHS. This is provided by virtual file system driver (VFS).
|
||||||
|
|
||||||
|
## Directory structure
|
||||||
|
|
||||||
|
| Path | Description |
|
||||||
|
|------------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||||
|
| / | Primary hierarchy root and root directory of the entire file system hierarchy. |
|
||||||
|
| /bin | Essential command binaries that need to be available in single-user mode, including to bring up the system or repair it, for all users (e.g., cat, ls, cp). |
|
||||||
|
| /boot | Boot loader files (e.g., EFI, initial-ramdisk.img ). |
|
||||||
|
| /dev | POSIX Device files (e.g., /dev/null, /dev/disk0, /dev/tty, /dev/random). |
|
||||||
|
| /etc | Host-specific system-wide configuration files. |
|
||||||
|
| /home | Users' home directories, containing saved files, personal settings, etc. |
|
||||||
|
| /lib | Libraries essential for the binaries in /bin and /sbin. eg realtime, system, ipc etc. |
|
||||||
|
| /sbin | Essential system binaries (e.g init) |
|
||||||
|
| /srv | Site-specific data served by this system, such as data and scripts for web servers, data offered by FTP servers, and repositories for version control systems |
|
||||||
|
| /system | DanOS operating system files (similar idea to C:\Windows). A true representation of danos — its layout mirrors the source tree, so `/system` is what danos *is*. |
|
||||||
|
| /system/devices | danos virtual device tree e.g. similar to /sys on linux but with danos device tree conventions (the structures in the devices module) |
|
||||||
|
| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/hpet) |
|
||||||
|
| /system/services | system-service binaries — the VFS server, init, and other user-mode servers (e.g. /system/services/vfs, /system/services/init) |
|
||||||
|
| /system/kernel | the kernel image |
|
||||||
|
| /tmp | Directory for temporary files (see also /var/tmp). Often not preserved between system reboots and may be severely size-restricted. |
|
||||||
|
| /usr | Secondary hierarchy for read-only user data; contains the majority of (multi-)user utilities and applications. Should be shareable and read-only. |
|
||||||
|
| /var | Variable files: files whose content is expected to continually change during normal operation of the system, such as logs, spool files, and temporary e-mail files. |
|
||||||
|
|
||||||
|
## File types
|
||||||
|
|
||||||
|
POSIX specifies the long format of the ls command to represent the Unix file type as the first letter for an entry.
|
||||||
|
|
||||||
|
| type | symbol | Description |
|
||||||
|
|-------------------|--------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||||
|
| regular | - | An ordinary file holding an uninterpreted byte stream. Reads and writes are positional, and the file grows on demand (e.g., a binary in /bin, a config file in /etc). |
|
||||||
|
| directory | d | A container mapping names to other files. It may only be modified through directory operations, never written to directly. |
|
||||||
|
| symbolic link | l | A file whose contents are a path that is resolved in its place. The target need not exist, and may cross mount points. |
|
||||||
|
| FIFO special | p | A named pipe: an in-order byte stream between processes, where writers block until a reader opens the other end. |
|
||||||
|
| block special | b | A device node addressed in fixed-size blocks with the kernel free to buffer and reorder access (e.g., /dev/disk0). |
|
||||||
|
| character special | c | A device node addressed as an unbuffered byte stream, delivered to the driver in order (e.g., /dev/tty, /dev/null). |
|
||||||
|
| socket | s | A named endpoint for bidirectional message-passing between processes, bound to a path rather than an address. |
|
||||||
|
|
||||||
|
## /dev
|
||||||
|
|
||||||
|
`/dev` holds the names through which processes reach devices. It is deliberately not
|
||||||
|
the device tree: the tree — every node discovered by ACPI or PCI enumeration, with its
|
||||||
|
resources and its parent — lives under [/system/devices](#directory-structure) and is
|
||||||
|
addressed by device id. `/dev` is the much smaller set of devices that have a driver
|
||||||
|
willing to serve them, addressed by name.
|
||||||
|
|
||||||
|
A device node is not a file the VFS can read. The bytes live in a driver process
|
||||||
|
([drivers.md](drivers.md)), so opening a `/dev` name has to resolve to that driver's
|
||||||
|
IPC endpoint, and subsequent reads and writes are calls against it. This is what
|
||||||
|
`system/services/vfs/vfs.zig` reserves for M10 and what the `Stat.kind` field is for; **none of it is
|
||||||
|
implemented today.** The current VFS is a flat, in-memory ramfs of eight nodes, with no
|
||||||
|
directories at all and `kind` hardcoded to zero. The three sections below describe the
|
||||||
|
intended shape, and are honest about which parts the kernel can already support.
|
||||||
|
|
||||||
|
### Character devices
|
||||||
|
|
||||||
|
A character device is a byte stream with no addressable position: bytes are delivered
|
||||||
|
to the driver in the order written, and a read consumes what is there. Terminals,
|
||||||
|
serial lines, keyboards and mice are all of this shape. These are the natural first
|
||||||
|
device nodes in danos, because a character driver needs nothing the kernel doesn't
|
||||||
|
already provide — it claims its device, maps its registers with `mmio_map`, and blocks
|
||||||
|
on `replyWait` for either an interrupt or a client request. `system/drivers/hpet/hpet.zig` is already
|
||||||
|
that program, minus the client half.
|
||||||
|
|
||||||
|
The obstacle was never the file type; it is which hardware a ring-3 driver can reach.
|
||||||
|
Direct `in`/`out` from user space is still a #GP (no TSS I/O bitmap, IOPL never raised),
|
||||||
|
but a driver no longer needs it: **`io_read`/`io_write`** grant port access the same way
|
||||||
|
`mmio_map` grants memory — gated by `device_claim` and the device's discovered `io_port`
|
||||||
|
resource. So the 16550 UART at `0x3F8` and the PS/2 controller at `0x60`/`0x64` (and thus
|
||||||
|
`/dev/ttyS0` and a keyboard node) are now writable as ordinary ring-3 drivers; the
|
||||||
|
low-rate legacy hardware that needs port I/O is fine with a syscall per access. A
|
||||||
|
memory-mapped device such as the framebuffer, needing no port I/O at all, remains the
|
||||||
|
easiest first entry.
|
||||||
|
|
||||||
|
### Block devices
|
||||||
|
|
||||||
|
A block device is addressed in fixed-size blocks and, unlike a character device, the
|
||||||
|
layer above is free to buffer, reorder, coalesce and retry requests against it. Disks
|
||||||
|
and other persistent storage are the whole population of this class.
|
||||||
|
|
||||||
|
A block driver is now **writable, but not yet memory-safe.** Every storage controller
|
||||||
|
worth naming is a bus master: it is programmed by handing it the physical address of a
|
||||||
|
descriptor ring and left to read and write memory on its own. That ring is exactly what
|
||||||
|
**`dma_alloc`** now provides — physically contiguous, pinned, uncacheable, with its
|
||||||
|
physical address disclosed — and **`/lib/mmio`**'s barriers order the descriptor writes
|
||||||
|
against the doorbell, and **`msi_bind`** delivers completions. So an AHCI or NVMe driver
|
||||||
|
can be written today (the M14/M15 work in [driver-model.md](driver-model.md); the earlier
|
||||||
|
"cannot host a block driver at all" is no longer true).
|
||||||
|
|
||||||
|
What is *not* yet true is that it is safe. A device programmed with an arbitrary physical
|
||||||
|
address writes to arbitrary physical memory, and page tables do not sit between a device
|
||||||
|
and RAM — an IOMMU does. The IOMMU is now *detected* (M16), but no translation domains
|
||||||
|
are programmed, so granting a DMA-capable device to a driver process is still equivalent
|
||||||
|
to granting ring 0. Until per-device domains confine a driver's DMA to the buffers it
|
||||||
|
`dma_alloc`'d, a block driver works but forfeits the isolation that motivates user-space
|
||||||
|
drivers — enforcement is the next step, and lands with that first driver. A ramdisk over
|
||||||
|
the initial ramdisk remains the one block-shaped thing that needs no driver process at all.
|
||||||
|
|
||||||
|
### Pseudo-devices
|
||||||
|
|
||||||
|
A pseudo-device has the interface of a device and no hardware behind it: `/dev/null`
|
||||||
|
discarding writes and reading as end-of-file, `/dev/zero` reading as an endless run of
|
||||||
|
zero bytes, `/dev/full` failing writes with `ENOSPC`, `/dev/random` and `/dev/urandom`
|
||||||
|
yielding unpredictable bytes.
|
||||||
|
|
||||||
|
These are the only `/dev` entries danos can implement immediately, and they are the
|
||||||
|
sensible place to start, because they are exactly the entries that need no driver
|
||||||
|
process, no `device_claim`, no MMIO grant and no interrupt. The VFS server answers them
|
||||||
|
out of its own address space — `null` and `zero` are a few lines each in
|
||||||
|
`system/services/vfs/vfs.zig`'s `read` and `write` handlers. Doing so forces the two pieces of
|
||||||
|
structure that every later device node depends on and that the flat ramfs currently
|
||||||
|
lacks: a directory, so that `/dev/null` is a path rather than a name; and a populated
|
||||||
|
`Stat.kind`, so that a caller can tell a character device from a regular file.
|
||||||
|
|
||||||
|
`/dev/random` is the one that is not free. It needs an entropy source, and the honest
|
||||||
|
options on this kernel are `RDRAND`/`RDSEED` where CPUID advertises them, and the HPET
|
||||||
|
counter's low bits as a poor fallback. Neither is a seeded CSPRNG, and a `/dev/random`
|
||||||
|
that is merely unpredictable-looking is worse than none — nothing should be keyed from
|
||||||
|
it until it is a real one.
|
||||||
+34
-12
@@ -18,7 +18,7 @@ Interrupt delivery on modern x86 goes through the **APIC**, not the legacy 8259
|
|||||||
PIC. There are two halves; we only need one so far:
|
PIC. There are two halves; we only need one so far:
|
||||||
|
|
||||||
- The **Local APIC** (per-CPU, memory-mapped at physical `0xFEE00000`) handles the
|
- The **Local APIC** (per-CPU, memory-mapped at physical `0xFEE00000`) handles the
|
||||||
CPU's own timer and receives interrupts routed to it. `src/kernel/arch/x86_64/apic.zig`.
|
CPU's own timer and receives interrupts routed to it. `system/kernel/architecture/x86_64/apic.zig`.
|
||||||
- The **IO-APIC** routes *external* device lines (keyboard, etc.) to LAPIC vectors.
|
- The **IO-APIC** routes *external* device lines (keyboard, etc.) to LAPIC vectors.
|
||||||
Not needed for the timer — it'll arrive with the keyboard.
|
Not needed for the timer — it'll arrive with the keyboard.
|
||||||
|
|
||||||
@@ -89,7 +89,6 @@ if (state.vector < 32) {
|
|||||||
on_fault(state); // exception: report and halt (never returns)
|
on_fault(state); // exception: report and halt (never returns)
|
||||||
} else if (handlers[state.vector]) |handler| {
|
} else if (handlers[state.vector]) |handler| {
|
||||||
handler(); // device: run the registered handler
|
handler(); // device: run the registered handler
|
||||||
apic.eoi(); // ...acknowledge the LAPIC
|
|
||||||
}
|
}
|
||||||
// else: spurious/unhandled — deliberately no EOI
|
// else: spurious/unhandled — deliberately no EOI
|
||||||
```
|
```
|
||||||
@@ -100,10 +99,23 @@ Two things make device interrupts *return* where exceptions don't:
|
|||||||
flows back to `isr_common`, which restores every register it saved and executes
|
flows back to `isr_common`, which restores every register it saved and executes
|
||||||
`iretq` — resuming the interrupted instruction exactly. (This is why the stub
|
`iretq` — resuming the interrupted instruction exactly. (This is why the stub
|
||||||
saves *all* the general registers.)
|
saves *all* the general registers.)
|
||||||
2. **End-of-interrupt.** After handling, we write the LAPIC's EOI register. Miss
|
2. **End-of-interrupt.** Somewhere in there we write the LAPIC's EOI register. Miss
|
||||||
this and the LAPIC thinks we're still busy and never delivers the next
|
this and the LAPIC thinks we're still busy and never delivers the next
|
||||||
interrupt. It's the single most common "my timer fired once and stopped" bug.
|
interrupt. It's the single most common "my timer fired once and stopped" bug.
|
||||||
|
|
||||||
|
**Each handler issues its own EOI**, rather than the dispatcher doing it around the
|
||||||
|
call. That looks like a needless devolution while the timer is the only device, and
|
||||||
|
`apic.timerTick` indeed does nothing but `eoi()` before bumping its counter (early,
|
||||||
|
because the tick hook is the scheduler, which may switch tasks and not return
|
||||||
|
promptly — the LAPIC mustn't wait on it).
|
||||||
|
|
||||||
|
It stops looking needless with the second device. A *routed* interrupt — one arriving
|
||||||
|
through the I/O APIC from a real device line — must be **masked before it is
|
||||||
|
acknowledged**, because a level-triggered line is still asserted at EOI time and would
|
||||||
|
redeliver instantly, forever. Only the handler knows which discipline its source
|
||||||
|
needs, so only the handler can sequence it. See [drivers.md](drivers.md), where the
|
||||||
|
device is quieted by a driver in ring 3, long after the ISR has returned.
|
||||||
|
|
||||||
A device handler is a plain `fn () void` — a timer or keyboard handler doesn't need
|
A device handler is a plain `fn () void` — a timer or keyboard handler doesn't need
|
||||||
the interrupted registers. (Note: the stubs don't save the SSE/vector registers, so
|
the interrupted registers. (Note: the stubs don't save the SSE/vector registers, so
|
||||||
a handler must not use them; ours don't.)
|
a handler must not use them; ours don't.)
|
||||||
@@ -131,14 +143,24 @@ If the APIC weren't enabled, or `sti` were missing, or EOI were forgotten, the
|
|||||||
count would stay put and the test would fail. That it advances — while the CPU was
|
count would stay put and the test would fail. That it advances — while the CPU was
|
||||||
spinning in unrelated code — is the whole mechanism working end to end.
|
spinning in unrelated code — is the whole mechanism working end to end.
|
||||||
|
|
||||||
|
## Since (done elsewhere)
|
||||||
|
|
||||||
|
- **Preemption**: the timer handler is where the scheduler decides to switch — the
|
||||||
|
reason a *returning* interrupt matters. See [scheduling.md](scheduling.md).
|
||||||
|
- **`sleep()` / timeouts** built on the calibrated clock.
|
||||||
|
- **The I/O APIC, routed**: external device lines now reach a vector, and the
|
||||||
|
interrupt is delivered onward to a *user-space* driver as an IPC message. See
|
||||||
|
[drivers.md](drivers.md).
|
||||||
|
- **Uncacheable MMIO**: device grants are mapped `PCD|PWT` (strong-uncacheable) for
|
||||||
|
user drivers — see [paging.md](paging.md).
|
||||||
|
|
||||||
## What's next (not done here)
|
## What's next (not done here)
|
||||||
|
|
||||||
- **The keyboard**: bring up the IO-APIC, route its IRQ to a vector, and read
|
- **The keyboard**: the PS/2 controller is port-mapped (`0x60`/`0x64`), and port I/O is
|
||||||
scancodes from the PS/2 controller — the first *input* device.
|
now available to ring 3 via the claim-gated `io_read`/`io_write` syscalls
|
||||||
- **`sleep()` / timeouts** built on the calibrated clock (the monotonic
|
([drivers.md](drivers.md)) — so the first *input* device is unblocked; it just needs
|
||||||
`uptimeMs()` is in place).
|
writing (claim the controller, `irq_bind` GSI 1, read scancodes from `0x60`).
|
||||||
- **Uncacheable MMIO**: the LAPIC page is currently mapped writeback-cacheable like
|
- **MSI-X**: `msi_bind` gives one per-device edge-triggered vector (M15); MSI-X's
|
||||||
the rest of the identity map. QEMU tolerates it, but real hardware wants MMIO
|
multi-vector table (many queues per device, e.g. NVMe) is the remaining extension.
|
||||||
marked uncacheable (via the page's cache bits or an MTRR).
|
- **The LAPIC's own page** is still mapped writeback-cacheable like the rest of the
|
||||||
- **Preemption**: once there are tasks, the timer handler is where the scheduler
|
identity map. QEMU tolerates it; real hardware wants it uncacheable.
|
||||||
decides to switch — the reason a *returning* interrupt matters.
|
|
||||||
|
|||||||
@@ -0,0 +1,154 @@
|
|||||||
|
# The device manager
|
||||||
|
|
||||||
|
**Status: the protocol and supervision are built** (M18.1, 2026-07-13): `hello`
|
||||||
|
with its deadline, supervised spawn, restart with backoff, and the crash-loop
|
||||||
|
cap are in — usb-xhci-bus is the first conforming driver, and the
|
||||||
|
`driver-restart` scenario proves fault → backoff → re-claim → cap end to end.
|
||||||
|
Tree reports are built too (M18.2, 2026-07-13): the xHCI driver scans its
|
||||||
|
root-hub ports and reports each connected device (`child_added`); the manager
|
||||||
|
mirrors them and prunes a dead reporter's children, and the `usb-report`
|
||||||
|
scenario proves report → prune → respawn → re-report. The application surface is built (M18.3, 2026-07-13):
|
||||||
|
`enumerate` and `subscribe` over IPC, with `device-list` as the first client —
|
||||||
|
the manager is now the one answer to "what devices exist" for applications.
|
||||||
|
The primitives underneath are real ([process-management.md](process-management.md):
|
||||||
|
spawn/supervise/kill/exit-notification; [driver-model.md](driver-model.md): the device
|
||||||
|
table as a capability system; [drivers.md](drivers.md): claim/map/IRQ), and the first
|
||||||
|
per-device driver spawn works (the device manager matches the xHCI controller by PCI
|
||||||
|
class and spawns `usb-xhci-bus` with the device id as argv[1]). This document designs
|
||||||
|
the rest: the device manager as **the tree, the matcher, and the supervisor** — the
|
||||||
|
policy process that turns [resilience.md](resilience.md)'s restart goal into practice
|
||||||
|
for drivers.
|
||||||
|
|
||||||
|
How processes stop, reload, and report their deaths is deliberately **not** in this
|
||||||
|
document: that is the universal lifecycle every danos process speaks —
|
||||||
|
[process-lifecycle.md](process-lifecycle.md), signals over IPC and the stable
|
||||||
|
`runtime.process` interface. The device manager is that design's first serious
|
||||||
|
customer, not its owner. Its own protocol contains nothing lifecycle-shaped; a
|
||||||
|
driver is stopped, health-checked, and buried exactly like any other process.
|
||||||
|
|
||||||
|
## The tree: structure in the manager, authority in the kernel
|
||||||
|
|
||||||
|
The device tree is two things fused: *information* (what exists, how it nests) and
|
||||||
|
*authority* (a descriptor is a licence to map physical memory). They separate:
|
||||||
|
|
||||||
|
- The **kernel keeps the capability system** — device, I/O-port, and interrupt
|
||||||
|
claims, resource containment on `device_register`, the
|
||||||
|
`mmio_map`/`irq_bind`/`msi_bind` gates — and **cleans all of it up when a process
|
||||||
|
dies** (settled; it is increment 1 of
|
||||||
|
[process-lifecycle.md](process-lifecycle.md)). The three invariants in
|
||||||
|
[driver-model.md](driver-model.md) stay exactly where they are. A device manager
|
||||||
|
that could mint MMIO mappings by its own say-so would be a second kernel, and a
|
||||||
|
buggy one would un-earn everything the microkernel bought.
|
||||||
|
- The **device manager owns the tree as data** — identity, topology, naming, driver
|
||||||
|
matching, hotplug events, and being the one process everything else asks about
|
||||||
|
devices. Firmware discovery seeds it (today via the kernel's snapshot); **bus
|
||||||
|
drivers grow it** by reporting what they see; applications query and watch it.
|
||||||
|
`device_enumerate` fades to a manager-internal (then deleted) seam.
|
||||||
|
|
||||||
|
Long-term, discovery itself leaves the kernel — but not *into* the manager. PCI
|
||||||
|
enumeration is a **pci-bus driver**: the manager spawns it against the host bridge
|
||||||
|
(already a device with the ECAM window as a resource), it scans, it reports functions
|
||||||
|
like any bus reports children. ACPI becomes an **acpi service** that interprets the
|
||||||
|
tables and reports the namespace. The manager only orchestrates and merges. Moving
|
||||||
|
AML interpretation out of ring 0 is its own project on its own track; nothing here
|
||||||
|
depends on when it lands.
|
||||||
|
|
||||||
|
## The protocol
|
||||||
|
|
||||||
|
A `device-manager-protocol` module (the vfs-protocol pattern): extern-struct
|
||||||
|
messages, a version in the handshake, reserved fields everywhere. The manager is a
|
||||||
|
well-known endpoint (`ipc.register(.device_manager)`); the badge tells it who is
|
||||||
|
talking; the same endpoint receives its children's exit notifications — one loop,
|
||||||
|
one world.
|
||||||
|
|
||||||
|
| Direction | Message | Purpose |
|
||||||
|
|---|---|---|
|
||||||
|
| driver → manager | `hello { version, role, device_id }` | confirms the argv assignment, starts the deadline clock |
|
||||||
|
| bus → manager | `child_added { parent, identity, resources }` | one node the bus discovered |
|
||||||
|
| bus → manager | `child_removed { id }` | unplug, or the bus lost it |
|
||||||
|
| app → manager | `enumerate` | snapshot of the tree (read-only) |
|
||||||
|
| app → manager | `subscribe` | receive published add/remove events |
|
||||||
|
|
||||||
|
`hello` is the one deadline the manager enforces itself: spawned and silent past the
|
||||||
|
deadline means wrong binary, wrong protocol version, or wedged before main — apply
|
||||||
|
the stop sequence and the restart policy. Everything else lifecycle-shaped
|
||||||
|
(terminate, the common `ping` liveness call, exit reasons) arrives through
|
||||||
|
[process-lifecycle.md](process-lifecycle.md)'s vocabulary, not this protocol.
|
||||||
|
|
||||||
|
Assignment stays argv (`usb-xhci-bus <device id>`) for now — simple, and it works.
|
||||||
|
The step after `hello` exists is delegation: the manager claims (or is granted) the
|
||||||
|
devices and passes the claim to the driver over IPC (the M13 capability-transfer
|
||||||
|
mechanism), replacing first-come-first-served `device_claim` with policy. Identity in
|
||||||
|
`child_added` is per-bus: PCI children carry the class triple (`pci_class`, as the
|
||||||
|
xHCI match already uses); USB children carry the (class, subclass, protocol) triple
|
||||||
|
from usb-ids.zig — each bus's native language, decoded by the shared ids modules.
|
||||||
|
|
||||||
|
## Supervision and restart
|
||||||
|
|
||||||
|
Every driver is spawned with the manager's exit endpoint (`spawnSupervised` — built).
|
||||||
|
On a death notification:
|
||||||
|
|
||||||
|
1. **Read the reason** ([process-lifecycle.md](process-lifecycle.md) increment 2).
|
||||||
|
Clean exit → it meant to; don't restart. Fault or missed `hello` deadline →
|
||||||
|
restart with **backoff**, and a crash-loop cap (three fast deaths → mark failed,
|
||||||
|
stop respawning, log loudly; a later `reload` to the manager can retry).
|
||||||
|
2. **Prune the subtree** the dead bus driver reported. Its children describe
|
||||||
|
protocol state (xHCI slot ids, transfer rings) that died with the process;
|
||||||
|
keeping the nodes would be keeping a lie. Watchers receive `child_removed` — the
|
||||||
|
input service losing, then regaining, a keyboard is the *honest* description of
|
||||||
|
what happened. The restarted instance rediscovers and re-reports.
|
||||||
|
3. **The claim is already free** because the kernel released it at death — the
|
||||||
|
restarted instance claims the same controller and comes up.
|
||||||
|
|
||||||
|
Who supervises the supervisor: **init** (PID 1), which already supervises the
|
||||||
|
services it starts. If the manager dies, drivers keep running (they hold their
|
||||||
|
claims; the kernel doesn't care who their supervisor was — though their exit
|
||||||
|
notifications now dangle harmlessly). The restarted manager re-learns the world:
|
||||||
|
kernel snapshot, then a re-`hello` round — drivers answer a broadcast or are stopped
|
||||||
|
and respawned. Full state handoff is deliberately not attempted.
|
||||||
|
|
||||||
|
## Thin drivers, class protocols
|
||||||
|
|
||||||
|
The [driver-model.md](driver-model.md) three-shape split, restated as processes:
|
||||||
|
|
||||||
|
- A **bus driver** (usb-xhci-bus) owns its controller — claim, MMIO, IRQ/MSI, DMA
|
||||||
|
rings — and offers a *transfer* protocol ("submit a control transfer to device N",
|
||||||
|
built from the usb-abi request constructors) plus tree reports to the manager.
|
||||||
|
- A **class driver** (usb-hid, usb-storage) owns nothing: it is matched to a reported
|
||||||
|
child by its identity triple, speaks the bus's transfer protocol downward and its
|
||||||
|
service's protocol upward — HID reports to the input service, blocks to the block
|
||||||
|
service. It works unchanged over any controller.
|
||||||
|
- **Services** (input, display, block) aggregate class drivers and face applications.
|
||||||
|
|
||||||
|
Each arrow is a protocol module. The manager routes none of the data plane — it
|
||||||
|
introduces the parties (matching), supervises them (lifecycle), and gets out of the
|
||||||
|
way.
|
||||||
|
|
||||||
|
## Increments
|
||||||
|
|
||||||
|
Increments 1–4 are the lifecycle prerequisites and live in
|
||||||
|
[process-lifecycle.md](process-lifecycle.md) (claim cleanup on death, exit reasons,
|
||||||
|
published exit events, signals + `runtime.process`). On top of those:
|
||||||
|
|
||||||
|
5. **device-manager-protocol**: `hello`, supervised spawn with restart policy;
|
||||||
|
usb-xhci-bus becomes the first conforming driver.
|
||||||
|
6. **Tree reports**: `child_added`/`child_removed`; the manager mirrors; xHCI reports
|
||||||
|
the mouse and keyboard QEMU already hangs off it.
|
||||||
|
7. **App surface**: `enumerate`/`subscribe` over IPC; `device_enumerate` retreats
|
||||||
|
to a manager-internal seam.
|
||||||
|
8. **Discovery migration**: pci-bus driver first, acpi service second, kernel scan
|
||||||
|
retired last. (AML-in-user-space is its own track.)
|
||||||
|
|
||||||
|
## Settled questions (2026-07-12)
|
||||||
|
|
||||||
|
- **Stateful buses**: pruning the subtree on bus-driver death is right for USB. A
|
||||||
|
future storage bus with in-flight writes wants drain-before-terminate — which is
|
||||||
|
exactly the `deadline_ms` parameter `stop()` already has; a per-driver deadline
|
||||||
|
is one value in the manager's policy table when such a bus arrives. No design
|
||||||
|
change.
|
||||||
|
- **Manager death**: drivers survive the manager; the restarted manager re-learns
|
||||||
|
the world (above). Checkpointing driver state with the manager is deferred until
|
||||||
|
something demonstrates the need.
|
||||||
|
- **Matching stays code until the third bus.** `driverFor`/`pciDriverFor` are
|
||||||
|
honest at two bus types; the third triggers the manifest (a driver declares what
|
||||||
|
it binds: a PCI class triple, a USB class triple, an ACPI `_HID`).
|
||||||
@@ -0,0 +1,367 @@
|
|||||||
|
# The driver model: buses, classes, and host controllers
|
||||||
|
|
||||||
|
[drivers.md](drivers.md) shows how to write *a* driver — claim a device, map its
|
||||||
|
registers, sleep on its interrupt. That's enough for a leaf device like the HPET. It is
|
||||||
|
not enough for a disk, a keyboard, or a network card, because those hang off a
|
||||||
|
*controller*, on a *bus*, speaking a *protocol*, and no single process should have to
|
||||||
|
know all three.
|
||||||
|
|
||||||
|
Real driver stacks factor into three shapes. This document is about what each one is,
|
||||||
|
what the kernel must give it, how they share code — and precisely which primitive each
|
||||||
|
is still blocked on.
|
||||||
|
|
||||||
|
## Three shapes
|
||||||
|
|
||||||
|
| Shape | Owns | Reaches hardware by | Talks to |
|
||||||
|
|---|---|---|---|
|
||||||
|
| **Host controller driver** (HCD) | a controller — an xHCI PCI function, an AHCI port block | `mmio_map` + `irq_bind` + DMA | the devices behind it, in its bus's language |
|
||||||
|
| **Bus driver** | a bus — a PCI bridge, a USB hub | `device_register`, to publish what it finds | class drivers, over IPC |
|
||||||
|
| **Class / protocol driver** | *nothing* | *nothing* | its bus driver, over IPC |
|
||||||
|
|
||||||
|
The last row is the surprising one and the whole point. A USB keyboard driver touches
|
||||||
|
no registers, takes no interrupts, and maps no memory. It sends HID protocol messages
|
||||||
|
to whatever published the device, and it works identically whether the controller
|
||||||
|
below is xHCI, EHCI, or a Raspberry Pi's DWC2. That is what buys you drivers that
|
||||||
|
outlive the hardware they were written for.
|
||||||
|
|
||||||
|
In practice **HCD and bus driver are usually the same process**. An xHCI driver is a
|
||||||
|
host controller driver (it owns the PCI function, its BARs, its interrupt, its DMA
|
||||||
|
rings) *and* a bus driver (it enumerates USB devices and publishes them). Splitting
|
||||||
|
them is a fiction; what matters is that both *roles* have kernel support, because a
|
||||||
|
plain bus driver with no controller — a USB hub — is also a real thing.
|
||||||
|
|
||||||
|
## The device table is the spine
|
||||||
|
|
||||||
|
danos already has the right central structure. `system/kernel/devices-broker.zig` holds a table of
|
||||||
|
`DeviceDesc`, each with a parent, a class, and a set of resources. Firmware discovery
|
||||||
|
seeds it ([discovery.md](discovery.md)); `device_register` grows it.
|
||||||
|
|
||||||
|
Three invariants make it a capability system rather than a directory:
|
||||||
|
|
||||||
|
1. **A claim is exclusive.** `device_claim(id)` succeeds once. Everything downstream —
|
||||||
|
`mmio_map`, `irq_bind`, `device_register` — checks `devices_broker.ownerOf(id) == me`.
|
||||||
|
2. **A descriptor is a licence to map physical memory.** Whoever claims a device may
|
||||||
|
map its `.memory` resources and bind its `.irq` resources. This is why
|
||||||
|
`device_register` cannot be a free-for-all.
|
||||||
|
3. **Therefore: containment.** Every resource of a registered child must lie inside a
|
||||||
|
resource of the same kind on its parent (`devices_broker.contains`). A bus driver can only
|
||||||
|
ever *subdivide* what it already holds. Without this, `device_register` would be a
|
||||||
|
syscall named "map any physical page you like."
|
||||||
|
|
||||||
|
Containment is transitive by construction: a grandchild is contained in its child,
|
||||||
|
which is contained in the bus. Nothing can be laundered through a chain.
|
||||||
|
|
||||||
|
Note that firmware topology does **not** obey containment, and isn't asked to — a PCI
|
||||||
|
function's BAR is not inside its host bridge's `bus_range`, because a bus-number range
|
||||||
|
is not an address window. Discovery is trusted; user space is not.
|
||||||
|
|
||||||
|
### What a bus driver looks like
|
||||||
|
|
||||||
|
`system/drivers/bus/bus.zig` is the smallest honest one. Its "bus" is the HPET's register block and
|
||||||
|
its "devices" are the block's comparators:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
_ = dev.claim(bus.id); // 1. own the bus
|
||||||
|
const base = dev.mmioMap(bus.id, 0).?; // 2. enumerate it — from the hardware
|
||||||
|
const n = ((cap.* >> 8) & 0x1F) + 1; // GENERAL_CAP says how many children
|
||||||
|
|
||||||
|
for (0..n) |i| { // 3. publish each child
|
||||||
|
var child = std.mem.zeroes(dev.DeviceDesc);
|
||||||
|
child.class = @intFromEnum(dev.DeviceClass.timer);
|
||||||
|
child.resource_count = 1;
|
||||||
|
child.resources[0] = .{ .kind = memory,
|
||||||
|
.start = bus_mmio.start + 0x100 + 0x20 * i,
|
||||||
|
.len = 0x20 };
|
||||||
|
_ = dev.register(bus.id, &child).?; // kernel checks containment
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Each child is left **unclaimed**, which is the handoff: a comparator driver can now
|
||||||
|
`device_claim` one and `mmio_map` it, and will see only its own 0x20-byte window. A child
|
||||||
|
whose window escapes the bus is refused — `bus` asserts that, and the `bus` test
|
||||||
|
asserts the kernel's table upholds it.
|
||||||
|
|
||||||
|
A USB device has *no* resources at all: `resource_count = 0`, because it's addressed
|
||||||
|
through its controller, not by MMIO. That case is allowed and is the common one.
|
||||||
|
|
||||||
|
## Families: sharing code between drivers
|
||||||
|
|
||||||
|
A "family" is two modules, not one:
|
||||||
|
|
||||||
|
- **A logic module** — the parts of the bus that every driver on it re-derives. Config
|
||||||
|
space walking and BAR decode for PCI. Descriptor parsing, control transfers, and hub
|
||||||
|
protocol for USB.
|
||||||
|
- **A protocol module** — the IPC message types that let a class driver talk to
|
||||||
|
*whatever* published its device. This is the part that makes class drivers portable.
|
||||||
|
|
||||||
|
danos already has one of each: `library/runtime/device.zig` is a logic module,
|
||||||
|
[`system/services/vfs/protocol.zig`](system/services/vfs/protocol.zig) is a protocol module shared by `system/services/vfs/vfs.zig`
|
||||||
|
and its clients. The pattern generalises directly:
|
||||||
|
|
||||||
|
```
|
||||||
|
library/
|
||||||
|
runtime/ module "runtime" — syscalls, heap, ipc, device, stdio
|
||||||
|
mmio/ module "mmio" — volatile register access + barriers [M14]
|
||||||
|
bus/
|
||||||
|
pci/ module "pci" — ECAM, BAR decode, capability walk
|
||||||
|
usb/ module "usb" — descriptors, control transfers, hubs
|
||||||
|
proto/
|
||||||
|
vfs/ module "vfs-protocol" (today: system/services/vfs/protocol.zig)
|
||||||
|
block/ module "block-protocol"
|
||||||
|
hid/ module "hid-protocol"
|
||||||
|
|
||||||
|
system/drivers/ one sub-project each → /system/drivers (no `d` suffix)
|
||||||
|
xhci/ HCD + bus driver imports runtime, pci, usb, mmio
|
||||||
|
usb-hid/ class driver imports runtime, usb, hid-protocol
|
||||||
|
block/ class driver imports runtime, block-protocol
|
||||||
|
```
|
||||||
|
|
||||||
|
The only build change needed: [`addUserBinary`](build.zig) currently takes exactly one
|
||||||
|
module (`rt_mod`) and injects it. It should take a slice of modules. That's a
|
||||||
|
five-line change, and it's the *entire* mechanism — Zig modules already give you
|
||||||
|
everything else.
|
||||||
|
|
||||||
|
The discipline that makes this work: **a class driver must not import a bus's logic
|
||||||
|
module.** `usbhid` imports `proto.hid` and `usb` (for descriptor types), never `pci`.
|
||||||
|
If a class driver needs `mmio`, it has become an HCD and should be one.
|
||||||
|
|
||||||
|
## What exists today
|
||||||
|
|
||||||
|
- **M10** — `device_enumerate`, `device_claim`, `mmio_map`. Strong-uncacheable device
|
||||||
|
grants, `device_grant` teardown.
|
||||||
|
- **M11** — `irq_bind` / `irq_ack`. IRQ delivered as an IPC notification; mask before
|
||||||
|
EOI; `irq_ack` is the unmask.
|
||||||
|
- **M12** — `parent` in `DeviceDesc`, `device_register` with resource containment.
|
||||||
|
- **M13** — capability passing. `ipc_call` / `ipc_reply_wait` grew a `send_cap` argument
|
||||||
|
and a `received_cap` return (r8): an endpoint travels with a message, installed into
|
||||||
|
the receiver's handle table (shared, refcount-bumped — a copy, not a move). A full
|
||||||
|
table fails `-ENOSPC` and does not half-deliver. This is the "open" primitive — a bus
|
||||||
|
driver mints a per-device endpoint and hands it to a class driver. The runtime exposes
|
||||||
|
`callCap` and `replyWait(..., send_cap)`; no class driver consumes it yet.
|
||||||
|
- **M14** — DMA memory + the memory-ordering layer. `/lib/mmio` gives drivers typed
|
||||||
|
volatile access and `mb`/`rmb`/`wmb` (per-arch); `dma_alloc`/`dma_free` grant
|
||||||
|
physically-contiguous, pinned, uncacheable, reclaim-on-teardown buffers with the
|
||||||
|
physical address exposed (`pmm.allocContiguous`, a DMA arena, `mapUserDmaInto`).
|
||||||
|
`dma_below_4g` caps the address for legacy engines; `dma_write_combining` is accepted
|
||||||
|
but falls back to coherent until PAT is programmed. hpet is refactored onto `/lib/mmio`;
|
||||||
|
no DMA driver consumes `dma_alloc` yet.
|
||||||
|
- **M15** — interrupts for PCI devices, the MSI half. Discovery now gives every PCI
|
||||||
|
function its 4 KiB ECAM config space as resource 0 (unblocking the capability walk
|
||||||
|
with no new syscall), and `msi_bind(device_id, endpoint) -> address, data` allocates a
|
||||||
|
per-device edge-triggered vector, delivered as an IPC notification with no mask and no
|
||||||
|
ack cycle. Legacy INTx (`_PRT` parsing + shared lines) is deliberately skipped — MSI
|
||||||
|
is the real answer. QEMU's HPET has no MSI, so delivery is proven with a self-IPI; the
|
||||||
|
first PCI driver is the first real consumer.
|
||||||
|
- **Port I/O** — `io_read`/`io_write(device_id, resource_index, offset, width[, value])`:
|
||||||
|
a claimed device's `io_port` resource lets a driver read/write its ports, gated exactly
|
||||||
|
like `mmio_map` gates memory (direct ring-3 `in`/`out` stays a #GP). This is what makes
|
||||||
|
a PS/2 or 16550 driver possible; the low-rate legacy hardware that needs it is fine with
|
||||||
|
a syscall per access. `io_port` resources were recorded by discovery and ignored — now
|
||||||
|
they're used.
|
||||||
|
- **M16 (detection)** — the IOMMU is now *found*: discovery parses the ACPI DMAR table,
|
||||||
|
maps the first VT-d unit, and reads its version + capabilities (`iommu_present` in the
|
||||||
|
platform info). This is detection only — **no translation domains are programmed, so
|
||||||
|
DMA is still unprotected** (the caveat below). Enforcement lands with the first DMA
|
||||||
|
driver, which is what there is to protect and test against. Proven in the `iommu` test,
|
||||||
|
booted with an emulated `intel-iommu`.
|
||||||
|
- **`system_spawn`** — a user-space supervisor starts a driver:
|
||||||
|
`system_spawn(name, arguments)` loads a binary bundled in the initial-ramdisk as a
|
||||||
|
fresh ring-3 process; `name` becomes the child's argv[0] and the optional
|
||||||
|
NUL-separated `arguments` blob its argv[1..], delivered on a SysV entry stack
|
||||||
|
([sysv.md](sysv.md)). This is what
|
||||||
|
turned the device manager from "log the match" into "run the driver": the kernel now
|
||||||
|
spawns only `init`, `init` spawns the services, and the **device-manager** discovers
|
||||||
|
the hardware and spawns each driver ([drivers.md](drivers.md)). Ungated for now — a
|
||||||
|
spawn capability is future work.
|
||||||
|
|
||||||
|
So: **bus drivers work now, and they're started by the device manager, not the kernel.**
|
||||||
|
HCDs and class drivers do not work yet. Here is exactly why, and exactly what would fix it.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
# Proposed ABI
|
||||||
|
|
||||||
|
## M13 — capability passing, for class drivers ✅ done
|
||||||
|
|
||||||
|
*Implemented as described below (see "What exists today"). The signatures landed
|
||||||
|
verbatim: `send_cap` in r9, `received_cap` returned in r8, `-ENOSPC` on a full receiver
|
||||||
|
table with no delivery. The rest of this section is the original design note.*
|
||||||
|
|
||||||
|
**The blocker.** A class driver has to reach *its* device. Today the only way to find
|
||||||
|
an endpoint is the name registry: `ipc_register(service_id, h)` / `ipc_lookup(id)`,
|
||||||
|
where `ServiceId` is a global integer namespace with `max_services = 8`. You cannot
|
||||||
|
mint one endpoint per USB device that way, and there is no way for a bus driver to
|
||||||
|
*hand* a class driver an endpoint. M7 deferred this deliberately.
|
||||||
|
|
||||||
|
**The fix.** Let a message carry one handle. Sender names a handle in its own table;
|
||||||
|
the kernel installs the endpoint into the receiver's table (bumping `refcount`) and
|
||||||
|
tells the receiver the index it landed at.
|
||||||
|
|
||||||
|
```
|
||||||
|
ipc_call(h, msg, message_len, reply, reply_cap, send_cap) -> reply_len
|
||||||
|
ipc_reply_wait(h, reply, reply_len, recv, recv_cap, send_cap)
|
||||||
|
-> recv_len (rax), badge (rdx), received_cap (r8)
|
||||||
|
```
|
||||||
|
|
||||||
|
`send_cap` is a handle or `no_cap` (`~0`). `received_cap` is the index the transferred
|
||||||
|
endpoint was installed at in the receiver's table, or `no_cap`.
|
||||||
|
|
||||||
|
- Both calls grow from 5 args to 6, which fits: `syscall5` uses `rdi/rsi/rdx/r10/r8`,
|
||||||
|
leaving `r9`. `ipc_reply_wait` already returns two values via `setSyscallResult2`;
|
||||||
|
this needs a third (`setSyscallResult3`).
|
||||||
|
- If the receiver's handle table is full, the call fails `-ENOSPC` and **the message is
|
||||||
|
not delivered** — a half-delivered capability is worse than a failed send.
|
||||||
|
- `closeHandles` already drops references on exit, so the lifetime story is unchanged.
|
||||||
|
|
||||||
|
That single primitive gives you the standard `open` pattern:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
// class driver // bus driver
|
||||||
|
const h = ipc.lookup(.usb).?; const r = ipc.replyWait(ep, ...);
|
||||||
|
const dev_ep = ipc.callCap(h, // ... mint a per-device endpoint,
|
||||||
|
.{ .op = .open, .id = dev_id }); // reply with it as send_cap
|
||||||
|
// now dev_ep is a private channel to that one device
|
||||||
|
```
|
||||||
|
|
||||||
|
## M14 — DMA memory and the memory-ordering contract, for HCDs ✅ done
|
||||||
|
|
||||||
|
*Implemented: `/lib/mmio` (typed volatile access + `mb`/`rmb`/`wmb`, per-arch) and
|
||||||
|
`dma_alloc`/`dma_free` (contiguous, pinned, uncacheable, reclaim-on-teardown, physical
|
||||||
|
address exposed). `dma_write_combining` still falls back to coherent — real WC needs
|
||||||
|
PAT, a small follow-up. The rest of this section is the original design note.*
|
||||||
|
|
||||||
|
**The blocker.** An HCD is a DMA-engine programmer. It needs a descriptor ring the
|
||||||
|
device can read, which means memory that is (a) physically contiguous, (b) at a
|
||||||
|
physical address the driver knows, (c) of the right cacheability, and (d) pinned.
|
||||||
|
[`sysMmap`](system/kernel/process.zig) gives you *none* of the four: it calls `pmm.alloc()`
|
||||||
|
once per page, maps writeback-cached, and never reveals a physical address.
|
||||||
|
|
||||||
|
**The fix.**
|
||||||
|
|
||||||
|
```
|
||||||
|
dma_alloc(len, flags) -> vaddr (rax), paddr (rdx)
|
||||||
|
dma_free(vaddr, len) -> 0
|
||||||
|
|
||||||
|
flags: dma_coherent (1) uncacheable; the default and the only one that's portable
|
||||||
|
dma_wc (2) write-combining — needs PAT programmed; for framebuffers
|
||||||
|
dma_below_4g (4) for devices with 32-bit DMA addressing
|
||||||
|
```
|
||||||
|
|
||||||
|
Guarantees: page-aligned, physically contiguous, zeroed, pinned for the life of the
|
||||||
|
mapping, and the physical address is stable. It needs one thing the kernel lacks —
|
||||||
|
`pmm.allocContiguous(n, max_phys)`; today `pmm.alloc()` hands out one frame at a time
|
||||||
|
with no adjacency guarantee.
|
||||||
|
|
||||||
|
**The memory-ordering contract.** danos has, at the time of writing, **zero memory
|
||||||
|
barriers anywhere in the tree.** That is currently correct-by-accident and won't
|
||||||
|
survive the first DMA driver, or the first ARM boot.
|
||||||
|
|
||||||
|
`volatile` is not a barrier. In Zig it means: don't elide this access, and don't
|
||||||
|
reorder it against *other volatile* accesses. It says nothing about your *ordinary*
|
||||||
|
stores — the descriptor you just filled in normal WB memory — which LLVM may freely
|
||||||
|
sink past a volatile MMIO write. The canonical bug:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
ring[i] = descriptor; // ordinary store to WB RAM
|
||||||
|
doorbell.* = i; // volatile store to UC MMIO
|
||||||
|
// nothing stops the compiler reordering these; the device reads a stale descriptor
|
||||||
|
```
|
||||||
|
|
||||||
|
So the rules, which belong in `library/mmio.zig` and behind `arch`:
|
||||||
|
|
||||||
|
| Situation | Required |
|
||||||
|
|---|---|
|
||||||
|
| MMIO register read/write | `mmio.read` / `mmio.write` (volatile) |
|
||||||
|
| Fill DMA descriptor, then ring doorbell | `wmb()` between them |
|
||||||
|
| Woken by IRQ, then read what the device wrote | `rmb()` before the read |
|
||||||
|
| MMIO write that must complete before the next read | `mb()` |
|
||||||
|
|
||||||
|
And the per-arch lowering — the reason this must be an `arch` primitive and not a
|
||||||
|
sprinkling of `asm volatile`:
|
||||||
|
|
||||||
|
| | x86_64 | aarch64 |
|
||||||
|
|---|---|---|
|
||||||
|
| `mb()` | `mfence` | `dsb sy` |
|
||||||
|
| `rmb()` | `lfence` | `dsb ld` |
|
||||||
|
| `wmb()` | `sfence` | `dsb st` |
|
||||||
|
| DMA cache coherency | coherent; nothing to do | **not guaranteed**; needs non-cacheable buffers or cache maintenance |
|
||||||
|
|
||||||
|
x86 is forgiving here — TSO plus strong-uncacheable MMIO means you usually get away
|
||||||
|
with a compiler barrier alone. ARM is not, and [vision.md](vision.md) makes ARM the win
|
||||||
|
condition. Build the abstraction while there is one caller to fix.
|
||||||
|
|
||||||
|
(Zig note: `@fence` was **removed in 0.16**. Use `@atomicRmw(..., .seq_cst)` for a full
|
||||||
|
barrier, or per-arch inline asm — which is what `library/mmio.zig` should hide.)
|
||||||
|
|
||||||
|
## M15 — interrupts for PCI devices ✅ done (MSI)
|
||||||
|
|
||||||
|
*Implemented the MSI half: ECAM config space per PCI function (resource 0) and
|
||||||
|
`msi_bind` (per-device edge-triggered vector, delivered as a notification). Legacy INTx
|
||||||
|
`_PRT` parsing is skipped on purpose. `msi_bind` returns (address, data) as two values
|
||||||
|
rather than an out-struct. The rest of this section is the original design note.*
|
||||||
|
|
||||||
|
**The blocker, and it's a hard one.** No PCI device can take an interrupt today.
|
||||||
|
[`addBars`](system/devices/acpi.zig) records `.memory` and `.io_port` BARs and never an
|
||||||
|
`.irq`; there is no `_PRT` parsing anywhere in the tree. `hpet` only works because the
|
||||||
|
HPET advertises its own routing options in its own registers — a privilege no ordinary
|
||||||
|
device has.
|
||||||
|
|
||||||
|
**The fix, in two halves.**
|
||||||
|
|
||||||
|
*Legacy INTx*: parse `_PRT` from the DSDT to map (device, INTA–D) → GSI, and record it
|
||||||
|
as an `.irq` resource. Then `irq_bind` works unchanged. But INTx lines are **shared**,
|
||||||
|
and `irq.bound[gsi]` holds one endpoint. Sharing needs a list, and every driver on the
|
||||||
|
line must be polled on each interrupt — the reason everyone left INTx behind.
|
||||||
|
|
||||||
|
*MSI/MSI-X*, which is the real answer: per-device vectors, edge-triggered, unshared, no
|
||||||
|
mask/ack cycle, no 24-GSI ceiling. The kernel allocates a vector and hands the driver
|
||||||
|
the (address, data) pair to program into its own MSI capability:
|
||||||
|
|
||||||
|
```
|
||||||
|
msi_bind(dev_id, endpoint, out) -> 0 // out: extern struct { addr: u64, data: u32 }
|
||||||
|
```
|
||||||
|
|
||||||
|
The driver writes those into config space itself — which means it needs config space,
|
||||||
|
which means **discovery should give each `pci_device` a `.memory` resource for its
|
||||||
|
4 KiB ECAM slot**. That's a small change to `parseMcfg` and it unblocks the whole
|
||||||
|
capability walk (MSI, MSI-X, PCIe extended caps) without any new syscall.
|
||||||
|
|
||||||
|
Note QEMU's HPET reports `Tn_FSB_INT_DEL_CAP = 0` — no MSI — so `hpet` can never
|
||||||
|
exercise this path. The first MSI driver will be the first PCI driver.
|
||||||
|
|
||||||
|
## M16 — the IOMMU, and the honest caveat ◑ detection done, enforcement pending
|
||||||
|
|
||||||
|
*The IOMMU is now detected (DMAR parsed, VT-d unit mapped and read — see the `iommu`
|
||||||
|
test), but **enforcement is not built**: no translation domains are programmed, so the
|
||||||
|
caveat below still holds in full. Detection can't be taken further usefully until there
|
||||||
|
is a DMA driver to protect and QEMU's `intel-iommu` to test the protection against —
|
||||||
|
building the per-device domains alongside that first driver is both the natural order
|
||||||
|
and the only way to verify them. The rest of this section is the original caveat.*
|
||||||
|
|
||||||
|
Everything above is capability-gated at the *CPU*. None of it is gated at the *device*.
|
||||||
|
A driver that can program a bus-mastering engine can make that device write to any
|
||||||
|
physical address, because page tables sit between the CPU and RAM, not between a device
|
||||||
|
and RAM. Until VT-d/DMAR (or SMMU on ARM) is programmed from the DMAR table, **`device_claim`
|
||||||
|
on any DMA-capable device is equivalent to granting ring 0.**
|
||||||
|
|
||||||
|
This does not make the model useless — it's the same position Linux is in with the
|
||||||
|
IOMMU off, and every other guarantee (crash isolation, restart, no shared address
|
||||||
|
space) still holds. But "user-space drivers are memory-safe" is not true yet, and the
|
||||||
|
gap should be named rather than implied.
|
||||||
|
|
||||||
|
## Ordering
|
||||||
|
|
||||||
|
`M13` (capability passing) is independent of `M14`/`M15` and is the cheapest. It
|
||||||
|
unlocks class drivers, which are the shape with no hardware requirements at all — you
|
||||||
|
could write a real one against `bus`'s comparators tomorrow.
|
||||||
|
|
||||||
|
`M14` and `M15` together unlock the first HCD. `M14`'s barrier layer is worth landing
|
||||||
|
on its own regardless: it's small, obviously correct, and stops every future driver
|
||||||
|
from hand-rolling `*volatile` and getting ARM wrong.
|
||||||
|
|
||||||
|
## See also
|
||||||
|
|
||||||
|
- [drivers.md](drivers.md) — how to write one, concretely.
|
||||||
|
- [discovery.md](discovery.md) / [acpi.md](acpi.md) — where the device table comes from.
|
||||||
|
- [ipc.md](ipc.md) — endpoints, badges, and the notification path an IRQ arrives on.
|
||||||
|
- [resilience.md](resilience.md) — restart, the reason any of this is worth the trouble.
|
||||||
+364
@@ -0,0 +1,364 @@
|
|||||||
|
# Writing a driver
|
||||||
|
|
||||||
|
In a monolithic kernel a driver is a function call away from everything: it runs in
|
||||||
|
ring 0, dereferences any physical address, and its interrupt handler *is* the ISR. In
|
||||||
|
danos a driver is **an ordinary ring-3 process**. It has its own address space, it
|
||||||
|
can crash without taking the kernel with it, and — the point of this document — it
|
||||||
|
can be restarted ([resilience](resilience.md)).
|
||||||
|
|
||||||
|
That leaves three questions the kernel has to answer, because a process can't answer
|
||||||
|
them for itself:
|
||||||
|
|
||||||
|
1. **What hardware exists?** → `device_enumerate`, over the device table discovery built
|
||||||
|
([discovery](discovery.md), [acpi](acpi.md)).
|
||||||
|
2. **How do I touch its registers?** → `device_claim` + `mmio_map`: the kernel maps the
|
||||||
|
device's physical MMIO window into your address space, and from then on it's plain
|
||||||
|
memory. No syscall per register access.
|
||||||
|
3. **How do I find out it wants something?** → `irq_bind`: the interrupt is delivered
|
||||||
|
to you as an IPC notification. You block; the hardware wakes you.
|
||||||
|
|
||||||
|
A driver is, in one sentence, *a process that sleeps until its device has something to
|
||||||
|
say.*
|
||||||
|
|
||||||
|
## How a driver gets started: discover, match, spawn
|
||||||
|
|
||||||
|
Nothing in the kernel decides that the HPET needs the `hpet` driver — that is policy,
|
||||||
|
and policy lives in user space. Boot brings user space up as a three-level supervision
|
||||||
|
hierarchy, each level owning one job:
|
||||||
|
|
||||||
|
```
|
||||||
|
kernel ──spawns──► init (PID 1) ──spawns──► device-manager ──spawns──► hpet
|
||||||
|
| | |
|
||||||
|
spawns only init, the service supervisor: the driver supervisor: enumerates
|
||||||
|
publishes the starts the system /system/devices, matches each device
|
||||||
|
initial-ramdisk services (vfs, the to a driver, and system_spawn's it
|
||||||
|
so user space can device-manager). Its
|
||||||
|
system_spawn from it list is init policy.
|
||||||
|
```
|
||||||
|
|
||||||
|
The kernel launches exactly one process — `init` — and hands it nothing but the raw
|
||||||
|
ability to start more (`system_spawn(name, arguments)`, which loads a binary bundled
|
||||||
|
in the initial-ramdisk as a fresh ring-3 process — `name` becoming its argv[0],
|
||||||
|
the optional arguments its argv[1..], on a SysV entry stack, see sysv.md). Everything else is a user-space decision:
|
||||||
|
|
||||||
|
- **init** ([system/services/init](system/services/init/init.zig)) is the **service
|
||||||
|
supervisor**. It spawns the system services danos brings up at boot — today `vfs` and
|
||||||
|
the `device-manager` — from a small list. Drivers are deliberately *not* its job.
|
||||||
|
- **device-manager** ([system/services/device-manager](system/services/device-manager/device-manager.zig))
|
||||||
|
is the **driver supervisor**. It does the three steps a monolithic kernel would do in
|
||||||
|
its probe path, entirely from ring 3:
|
||||||
|
1. **Discover** — `device_enumerate` snapshots the device table the kernel built from
|
||||||
|
ACPI/PCI ([discovery](discovery.md)).
|
||||||
|
2. **Match** — for each device it looks up a driver by `DeviceClass`. The match policy
|
||||||
|
is a table (`driverFor`): today a static `timer → hpet` map; a fuller system reads
|
||||||
|
what each driver *binds* (a manifest under `/system/drivers`, or the driver
|
||||||
|
describing its own match).
|
||||||
|
3. **Spawn** — `system_spawn(driver_name, arguments)` starts the matched driver (the
|
||||||
|
arguments can carry *which* device it matched), which then claims
|
||||||
|
its device and runs the event loop below.
|
||||||
|
|
||||||
|
So "how is a driver discovered and configured" has two halves: **discovery** is the
|
||||||
|
kernel's device table, read by anyone; **configuration** is two user-space policies —
|
||||||
|
init's service list and the device-manager's match table. Both are hardcoded in their
|
||||||
|
respective programs today; the natural next step is to move them into `/etc` (see the
|
||||||
|
milestone notes in [driver-model.md](driver-model.md)). `system_spawn` is currently
|
||||||
|
ungated — any process may spawn any bundled binary — because there is no spawn
|
||||||
|
capability yet.
|
||||||
|
|
||||||
|
## The capability: claim before touch
|
||||||
|
|
||||||
|
The driver syscall numbers (`system/abi.zig`) with the device types they carry
|
||||||
|
(`system/devices/device-abi.zig`), dispatched in `system/kernel/process.zig`:
|
||||||
|
|
||||||
|
| # | Call | Meaning |
|
||||||
|
|---|------|---------|
|
||||||
|
| 11 | `device_enumerate(buf, max) -> total` | Snapshot the device table |
|
||||||
|
| 12 | `device_claim(id) -> ok` | Take **exclusive** ownership |
|
||||||
|
| 13 | `mmio_map(id, res_idx) -> vaddr` | Map a claimed device's register window |
|
||||||
|
| 14 | `irq_bind(id, res_idx, endpoint)` | Deliver that device's IRQ as a notification |
|
||||||
|
| 15 | `irq_ack(id, res_idx)` | Re-arm the IRQ after servicing the device |
|
||||||
|
| 16 | `device_register(parent_id, desc) -> id` | Publish a child of a device you claimed |
|
||||||
|
|
||||||
|
Notice that **nothing takes a physical address or an interrupt number.** Every call
|
||||||
|
names a device by id and a resource by index. That indirection is the entire security
|
||||||
|
model. If `mmio_map` took a physical address, any process could map the kernel's
|
||||||
|
memory; if `irq_bind` took a GSI, any process could bind the keyboard's line and
|
||||||
|
silently intercept it. Instead the kernel checks two things (`process.ownedGsi`, and
|
||||||
|
the same check at the top of `sysMmioMap`):
|
||||||
|
|
||||||
|
- `devices_broker.ownerOf(dev_id) == me` — you claimed it, and claims are exclusive
|
||||||
|
- the resource at `res_idx` is of the right *kind* — `memory` for `mmio_map`, `irq`
|
||||||
|
for `irq_bind`
|
||||||
|
|
||||||
|
The claim is the capability. Everything else follows from it.
|
||||||
|
|
||||||
|
## Registers: `mmio_map`
|
||||||
|
|
||||||
|
`mmio_map` walks the caller's page tables and installs the device's physical frames
|
||||||
|
with `present | user | writable | nx | pcd | pwt`
|
||||||
|
(`arch/x86_64/paging.zig:mapUserDeviceInto`). Two of those bits are load-bearing:
|
||||||
|
|
||||||
|
- **`pcd | pwt`** — strong-uncacheable. A device register is not memory; a cached read
|
||||||
|
would return a stale value and a write might never leave the CPU.
|
||||||
|
- **`device_grant`** (bit 9, one of the PTE's available bits) — marks the leaf as MMIO
|
||||||
|
rather than RAM, so `freeSubtree` skips `pmm.free` on it when the address space is
|
||||||
|
destroyed. Without this, killing a driver would hand the HPET's registers back to
|
||||||
|
the frame allocator as if they were free RAM. The `iopass` test guards it.
|
||||||
|
|
||||||
|
Grants land in their own arena, `0x0000_7100_0000_0000` (PML4[226]), so device pages
|
||||||
|
never widen an existing mapping.
|
||||||
|
|
||||||
|
Then you just… use it:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
const base = dev.mmioMap(dev_id, mmio_res) orelse return;
|
||||||
|
const counter: *volatile u64 = @ptrFromInt(base + 0xF0);
|
||||||
|
const now = counter.*; // a load, straight to the hardware. no kernel involved.
|
||||||
|
```
|
||||||
|
|
||||||
|
## Interrupts: the cycle, and why it has that shape
|
||||||
|
|
||||||
|
An interrupt handler in a microkernel has a problem. The code that knows how to quiet
|
||||||
|
the device is in ring 3, in another address space, and it will not run for
|
||||||
|
microseconds or milliseconds — after a context switch, when the scheduler gets to it.
|
||||||
|
But the CPU wants an EOI *now*, and a **level-triggered** line stays asserted until
|
||||||
|
the device is quieted. EOI a still-asserted line and the I/O APIC redelivers
|
||||||
|
immediately. Forever. The driver never gets to run at all.
|
||||||
|
|
||||||
|
The way out is to mask the line before acknowledging it:
|
||||||
|
|
||||||
|
```
|
||||||
|
kernel ISR irqMask(gsi) // line still asserted; stop it reaching a CPU
|
||||||
|
irqEoi() // now safe to tell the LAPIC we're done
|
||||||
|
notifyFromIsr() // wake the driver — it runs much later
|
||||||
|
|
||||||
|
driver replyWait() -> badge with the notify bit set
|
||||||
|
<clear the device's status register> // NOW the line deasserts
|
||||||
|
irq_ack(dev, res) // kernel unmasks: quiet, so it can't refire
|
||||||
|
|
||||||
|
```
|
||||||
|
|
||||||
|
`irq_ack` is not bookkeeping you could skip. **It is the unmask.** Forget it and the
|
||||||
|
interrupt fires exactly once, ever; call it before the device is quiet and you get an
|
||||||
|
interrupt storm. That single fact explains why `irq_bind` and `irq_ack` are two
|
||||||
|
syscalls and not one.
|
||||||
|
|
||||||
|
This is also why `interruptDispatch` (`arch/x86_64/idt.zig`) no longer issues the EOI
|
||||||
|
itself. It used to, before running the handler — correct for the LAPIC timer, and
|
||||||
|
impossible for a routed device line. Each handler now owns its EOI, because only the
|
||||||
|
handler knows which discipline its source needs.
|
||||||
|
|
||||||
|
### The driver side is an event loop, not a callback
|
||||||
|
|
||||||
|
`IPC_ReplyWait` returns *either* a client request *or* a notification, told apart by
|
||||||
|
the top bit of the badge (`ipc_sync.notify_badge_bit`). So a driver is one
|
||||||
|
single-threaded loop over both of its event sources:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
while (true) {
|
||||||
|
const r = ipc.replyWait(endpoint, reply, &recv);
|
||||||
|
if (r.isNotification()) { // r.source() is the GSI
|
||||||
|
service_device(); // clear the status register
|
||||||
|
_ = dev.irqAck(id, irq_res); // re-arm
|
||||||
|
} else {
|
||||||
|
handle_client_request(recv[0..r.len]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
No reentrancy, no "what am I allowed to call from an interrupt handler", no shared
|
||||||
|
state between ISR and task context. The interrupt is just a message.
|
||||||
|
|
||||||
|
Two properties worth knowing:
|
||||||
|
|
||||||
|
- **An interrupt taken while you're elsewhere is not lost.** If the driver is off in
|
||||||
|
an `ipc_call` to another server when the IRQ fires, `wakeLocked` finds nobody
|
||||||
|
waiting, but the badge is already on the endpoint's notify ring. The next
|
||||||
|
`replyWait` pops it (`ipc_sync.replyWait` checks `popNotify` before the sender FIFO).
|
||||||
|
- **Notifications coalesce, they don't count.** The ring is 8 deep and drops on
|
||||||
|
overflow. That's correct: an IRQ notification is a *level* ("the device wants
|
||||||
|
attention"), not a tally. Re-read the device's status register; never assume one
|
||||||
|
notification means exactly one event. Because the ISR masks the line until you
|
||||||
|
`irq_ack`, at most one badge per GSI can be outstanding — so the ring can only
|
||||||
|
overflow if you bind more than eight GSIs to a single endpoint. Don't.
|
||||||
|
|
||||||
|
## A whole driver
|
||||||
|
|
||||||
|
`system/drivers/hpet/hpet.zig` is ~150 lines and does all of it. The shape:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
const hpet = findHpet(buf) orelse return; // device_enumerate, look for
|
||||||
|
// class=timer with memory + irq
|
||||||
|
_ = dev.claim(hpet.dev_id); // the capability
|
||||||
|
const base = dev.mmioMap(hpet.dev_id, hpet.mmio).?;
|
||||||
|
const endpoint = ipc.createIpcEndpoint().?;
|
||||||
|
|
||||||
|
// program the hardware over the mapping we were just handed
|
||||||
|
reg(base, 0x100).* = level | int_enb | (hpet.gsi << 9); // timer 0 config
|
||||||
|
reg(base, 0x108).* = reg(base, 0xF0).* + period; // comparator
|
||||||
|
reg(base, 0x010).* |= 1; // ENABLE
|
||||||
|
|
||||||
|
_ = dev.irqBind(hpet.dev_id, hpet.irq, endpoint);
|
||||||
|
|
||||||
|
while (...) {
|
||||||
|
const r = ipc.replyWait(endpoint, &.{}, &recv); // blocked. not polling.
|
||||||
|
if (r.badge & notify_bit == 0) continue;
|
||||||
|
reg(base, 0x020).* = 1; // clear status -> deassert
|
||||||
|
reg(base, 0x108).* = reg(base, 0xF0).* + period; // re-arm
|
||||||
|
_ = dev.irqAck(hpet.dev_id, hpet.irq); // unmask
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
The HPET is a good first driver for a reason that isn't obvious. Its *counter* is a
|
||||||
|
clocksource — the only way to use it is to read it, so it proved `mmio_map` without
|
||||||
|
needing interrupts at all. Its *comparators* are a clockevent, and can be configured
|
||||||
|
**level-triggered** (`Tn_INT_TYPE_CNF`), which asserts a bit in `GENERAL_INT_STATUS`
|
||||||
|
that the driver must write-1-to-clear. That's a genuine deassert step, so the full
|
||||||
|
mask/ack cycle above is exercised for real rather than being decoration on an
|
||||||
|
edge-triggered line that would have been fine without it.
|
||||||
|
|
||||||
|
One wrinkle it also demonstrates: the ACPI HPET table carries **no interrupt number**.
|
||||||
|
Which I/O APIC inputs a comparator may drive is a bitmask in `Tn_INT_ROUTE_CAP`, in
|
||||||
|
the device's own registers. So discovery (`acpi.parseHpet`) maps the block, reads the
|
||||||
|
mask, and records one concrete GSI as an `irq` resource. The driver then programs
|
||||||
|
`Tn_INT_ROUTE_CNF` to raise exactly that line — and the kernel will only bind the one
|
||||||
|
it recorded. Hardware that describes itself at runtime still has to fit through a
|
||||||
|
static capability.
|
||||||
|
|
||||||
|
## Publishing children: `device_register`
|
||||||
|
|
||||||
|
A device that *contains other devices* — a PCI bridge, a USB hub, or the HPET's block
|
||||||
|
of comparators — needs a driver that enumerates it and tells the kernel what it found.
|
||||||
|
That's `device_register`, and it makes the device table a tree rather than a list
|
||||||
|
(`DeviceDesc.parent`).
|
||||||
|
|
||||||
|
```zig
|
||||||
|
var child = std.mem.zeroes(dev.DeviceDesc);
|
||||||
|
child.class = @intFromEnum(dev.DeviceClass.timer);
|
||||||
|
child.resource_count = 1;
|
||||||
|
child.resources[0] = .{ .kind = memory, .start = bus_base + 0x100, .len = 0x20 };
|
||||||
|
const child_id = dev.register(bus_id, &child).?;
|
||||||
|
```
|
||||||
|
|
||||||
|
The child is left **unclaimed**, which is the whole point: another process claims it and
|
||||||
|
`mmio_map`s it, and sees only that 0x20-byte window.
|
||||||
|
|
||||||
|
The rule the kernel enforces is **containment**: every resource of a child must lie
|
||||||
|
inside a resource of the same kind on its parent. Ranges must nest; an IRQ must match
|
||||||
|
exactly. This isn't bureaucracy — a `DeviceDesc` is a licence to map physical memory, so
|
||||||
|
without containment `device_register` would be a syscall for mapping any page you like. A
|
||||||
|
bus driver may only ever subdivide what it already owns.
|
||||||
|
|
||||||
|
A device with **no resources** is legal and common. A USB device is reached through its
|
||||||
|
controller, not by MMIO, so it gets `resource_count = 0`.
|
||||||
|
|
||||||
|
See [`system/drivers/bus/bus.zig`](../system/drivers/bus/bus.zig) for a complete one, and
|
||||||
|
[driver-model.md](driver-model.md) for how bus drivers, class drivers and host
|
||||||
|
controller drivers fit together.
|
||||||
|
|
||||||
|
## What the kernel does not do for you
|
||||||
|
|
||||||
|
- **It does not quiet your device.** That's the whole reason `irq_ack` exists.
|
||||||
|
- **It does not know your registers.** `mmio_map` hands you a base address; every
|
||||||
|
offset in this document came from the HPET spec, not from danos.
|
||||||
|
- **It does not serialise your driver.** Two clients calling one driver endpoint are
|
||||||
|
serialised by `replyWait`, but nothing stops your driver from being preempted.
|
||||||
|
|
||||||
|
## Limits, today
|
||||||
|
|
||||||
|
Worth knowing before you write the second driver:
|
||||||
|
|
||||||
|
Several things this list used to warn about are now available (see
|
||||||
|
[driver-model.md](driver-model.md)): **port I/O** (`io_read`/`io_write`, claim-gated by
|
||||||
|
the device's `io_port` resource — direct ring-3 `in`/`out` is still a #GP, so a PS/2 or
|
||||||
|
16550 driver goes through these), **DMA memory** (`dma_alloc`: contiguous, pinned,
|
||||||
|
uncacheable, physical address exposed), and **memory barriers** (`/lib/mmio`'s
|
||||||
|
`mb`/`rmb`/`wmb`). What remains:
|
||||||
|
|
||||||
|
- **Page granularity.** `mmio_map` rounds to 4 KiB. Two devices sharing a page means
|
||||||
|
granting one grants the other. A `device_register`ed child's *resource* can be narrower
|
||||||
|
than a page, but its *mapping* can't.
|
||||||
|
- **DMA is not contained.** A driver that can program a bus-mastering device can make
|
||||||
|
that device write to *any* physical address — page tables don't sit between a device
|
||||||
|
and RAM; an IOMMU does. The IOMMU is now *detected* (M16), but no translation domains
|
||||||
|
are programmed, so `device_claim` on a DMA-capable device is still effectively
|
||||||
|
equivalent to granting ring 0. This is the largest gap between the design's promise and
|
||||||
|
what it delivers; enforcement lands with the first DMA driver.
|
||||||
|
- **No `dev_release`.** A claim is never dropped (only IRQ/MSI bindings are, on exit), so
|
||||||
|
a device stays owned for the life of its driver — which blocks restart.
|
||||||
|
- **One endpoint per GSI**, so shared legacy PCI INTx lines can't be split between two
|
||||||
|
drivers. MSI/MSI-X — one vector per device, edge-triggered, unshared — is the real
|
||||||
|
answer, and QEMU's HPET doesn't offer it (`Tn_FSB_INT_DEL_CAP = 0`).
|
||||||
|
- **Polarity is hardcoded** active-high in `irq.bind`. A device whose MADT override
|
||||||
|
says active-low needs that threaded through from discovery.
|
||||||
|
- **14 device vectors** (33–46) and **24 GSIs**, bounded by the stubs `isr.s` emits and
|
||||||
|
by a single I/O APIC.
|
||||||
|
- **Don't bind more than 8 GSIs to one endpoint.** The notify ring is 8 deep and drops
|
||||||
|
on overflow. With one GSI per endpoint that's unreachable — the line is masked from
|
||||||
|
the ISR until `irq_ack`, so at most one badge is ever outstanding. Bind nine devices
|
||||||
|
to one endpoint, though, and a dropped badge leaves that line masked with nobody
|
||||||
|
left to ack it.
|
||||||
|
- **A faulting driver still kills the machine.** There is no per-process kill path: a
|
||||||
|
ring-3 page fault halts the kernel, so `releaseIrqs` runs only on a voluntary
|
||||||
|
`exit`. Fault isolation is the whole premise ([vision](vision.md)) and it is
|
||||||
|
[not built yet](resilience.md).
|
||||||
|
- **A dead driver's device is not reclaimed.** `releaseIrqs` unbinds and masks the
|
||||||
|
line on exit, but the claim is never released — restart is
|
||||||
|
[not built](resilience.md).
|
||||||
|
- **On real hardware, the mask/EOI cycle may need a remote-IRR flush.** Masking a
|
||||||
|
level-triggered redirection entry with remote-IRR set doesn't clear it on some
|
||||||
|
chipsets, and the line never fires again. QEMU clears it on EOI regardless, so the
|
||||||
|
tests can't see this. Linux flushes remote-IRR by toggling the entry to edge and
|
||||||
|
back. See the note at the top of `system/kernel/irq.zig`.
|
||||||
|
|
||||||
|
## Verifying it
|
||||||
|
|
||||||
|
The `hpet` test spawns `hpet` from the initial ramdisk and watches the serial log. The driver
|
||||||
|
prints `hpet: ok` only after being woken five times, and its loop's only exit is
|
||||||
|
through `replyWait` returning a notification — it cannot reach that line by polling.
|
||||||
|
|
||||||
|
The last check doesn't trust the driver's self-report at all: the kernel reads the I/O
|
||||||
|
APIC redirection entry back and asserts the line really is routed to a device vector,
|
||||||
|
really is level-triggered, and really was left unmasked by the driver's final
|
||||||
|
`irq_ack`.
|
||||||
|
|
||||||
|
```
|
||||||
|
$ python3 test/qemu_test.py hpet irqfree iopass
|
||||||
|
hpet ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
|
irqfree ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
|
iopass ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||||
|
```
|
||||||
|
|
||||||
|
Two companions cover what `hpet` can't, because it never exits:
|
||||||
|
|
||||||
|
- **`irqfree`** — the teardown path. Binds two owners to one shared endpoint, releases
|
||||||
|
one, and reads the I/O APIC back: the departing owner's line is masked, the sibling's
|
||||||
|
is not. That second half is why bindings are keyed on the owning *task* and not on
|
||||||
|
the endpoint pointer — endpoints are shared, so releasing "everything pointing at
|
||||||
|
this endpoint" would silently mask a live driver's device.
|
||||||
|
- **`iopass`** — the `device_grant` teardown rule, so destroying a driver's address
|
||||||
|
space never returns MMIO frames to the RAM pool.
|
||||||
|
|
||||||
|
## What's next (not done here)
|
||||||
|
|
||||||
|
The big driver-model pieces — capability passing (class drivers), DMA + barriers, MSI,
|
||||||
|
and IOMMU detection — are **now done** ([driver-model.md](driver-model.md), M13–M16), as
|
||||||
|
is **port I/O** (`io_read`/`io_write`, the claim-gated syscalls that make a PS/2 or 16550
|
||||||
|
driver possible). What's left is IOMMU *enforcement* (per-device domains — it waits on
|
||||||
|
the first DMA driver to protect and test against) and these smaller items:
|
||||||
|
|
||||||
|
- **Releasing a claim.** There is no `dev_release`, and `devices_broker` never drops a claim on
|
||||||
|
exit — only IRQ bindings are released. A dead driver's device stays owned forever,
|
||||||
|
which blocks restart.
|
||||||
|
- **Unregistering children.** `device_register` only appends. A USB device that is
|
||||||
|
unplugged cannot be removed, and a bus driver in a loop can exhaust the 64-entry
|
||||||
|
table.
|
||||||
|
- **Restart.** A supervisor that *spawns* drivers now exists — the device-manager starts
|
||||||
|
them with `system_spawn` — but a supervisor that *restarts* them does not. A driver that
|
||||||
|
dies should release its claim, have its device quiesced, and be respawned; today nothing
|
||||||
|
notices the death. Some pieces (`releaseIrqs`, `device_grant` teardown, the claim table)
|
||||||
|
exist, and `dev_release` (below) is the missing mechanism; the restart policy is the
|
||||||
|
resilience track ([resilience.md](resilience.md)).
|
||||||
|
- **Interrupt priority / threaded IRQ latency.** `notifyFromIsr` enqueues the woken
|
||||||
|
driver but doesn't preempt (`wakeLocked` deliberately leaves that to the caller), so
|
||||||
|
a woken driver waits for the next scheduling point.
|
||||||
+25
-20
@@ -10,7 +10,7 @@ that hands us a working CPU, a memory map, and a screen, and then gets out of th
|
|||||||
way.
|
way.
|
||||||
|
|
||||||
The key thing to understand: **UEFI is not our OS, it's a stepping stone.** It
|
The key thing to understand: **UEFI is not our OS, it's a stepping stone.** It
|
||||||
exists to load *us*. Our `src/boot/efi.zig` is a UEFI *application* — a normal program
|
exists to load *us*. Our `boot/efi.zig` is a UEFI *application* — a normal program
|
||||||
that the firmware runs — and its entire purpose is to gather what the kernel needs
|
that the firmware runs — and its entire purpose is to gather what the kernel needs
|
||||||
and then jump into the kernel.
|
and then jump into the kernel.
|
||||||
|
|
||||||
@@ -20,14 +20,17 @@ UEFI boots by looking for a FAT-formatted partition called the **EFI System
|
|||||||
Partition (ESP)** and running a file at a well-known fallback path:
|
Partition (ESP)** and running a file at a well-known fallback path:
|
||||||
|
|
||||||
```
|
```
|
||||||
esp/EFI/BOOT/BOOTX64.efi <- the "removable media" default for x86-64
|
EFI/BOOT/BOOTX64.efi <- the "removable media" default for x86-64
|
||||||
```
|
```
|
||||||
|
|
||||||
That's exactly the layout `build.zig` assembles. It builds `src/boot/efi.zig` for the
|
The boot volume is the **FHS-shaped `zig-out`** itself (see the repository-layout note
|
||||||
`uefi` target, installs it to `esp/EFI/BOOT/BOOTX64.efi`, and drops the kernel ELF
|
in [README.md](README.md)): `build.zig` installs `boot/efi.zig` (built for the `uefi`
|
||||||
at `esp/kernel`. The `run-x86-64` step then points QEMU at OVMF (UEFI firmware for
|
target) to `zig-out/EFI/BOOT/BOOTX64.efi` — the one path UEFI firmware fixes — and lays
|
||||||
virtual machines) and presents that `esp/` directory to the guest as a FAT drive.
|
the rest out by FHS path: the kernel at `zig-out/system/kernel`, init at
|
||||||
The firmware finds `BOOTX64.efi` and runs it — that's our `main()`.
|
`zig-out/system/services/init`, the initial-ramdisk at `zig-out/boot/`. The
|
||||||
|
`run-x86-64` step points QEMU at OVMF (UEFI firmware for virtual machines) and presents
|
||||||
|
`zig-out` to the guest as a FAT drive. The firmware finds `BOOTX64.efi` and runs it —
|
||||||
|
that's our `main()`, which then loads the kernel and init from their FHS paths.
|
||||||
|
|
||||||
## Boot services: the firmware's API
|
## Boot services: the firmware's API
|
||||||
|
|
||||||
@@ -83,9 +86,9 @@ All of this *must* happen now, because after exit there's no GOP to ask. (See
|
|||||||
|
|
||||||
- Use the **LoadedImage** protocol to discover which device we booted from, then
|
- Use the **LoadedImage** protocol to discover which device we booted from, then
|
||||||
**SimpleFileSystem** to open that volume.
|
**SimpleFileSystem** to open that volume.
|
||||||
- Open the file named `danos`, seek to the end to learn its size, rewind, and read
|
- Open the kernel ELF at its FHS path (`system\kernel`), seek to the end to learn its
|
||||||
the whole ELF into a firmware-allocated pool buffer. (`read` may return short, so
|
size, rewind, and read the whole ELF into a firmware-allocated pool buffer. (`read`
|
||||||
we loop.)
|
may return short, so we loop.)
|
||||||
- Parse the ELF: validate the `\x7fELF` magic and the `x86_64` machine type, then
|
- Parse the ELF: validate the `\x7fELF` magic and the `x86_64` machine type, then
|
||||||
walk the program headers. For every `PT_LOAD` segment we:
|
walk the program headers. For every `PT_LOAD` segment we:
|
||||||
- reserve the exact physical pages it's linked at (`p_paddr`) via
|
- reserve the exact physical pages it's linked at (`p_paddr`) via
|
||||||
@@ -121,7 +124,7 @@ entirely ours.
|
|||||||
### 4. Jump to the kernel
|
### 4. Jump to the kernel
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
const kernel: *const fn (*const BootInfo) callconv(danos.kernel_abi) noreturn =
|
const kernel: *const fn (*const BootInfo) callconv(boot_handoff.kernel_abi) noreturn =
|
||||||
@ptrFromInt(entry);
|
@ptrFromInt(entry);
|
||||||
kernel(&boot_info);
|
kernel(&boot_info);
|
||||||
```
|
```
|
||||||
@@ -139,17 +142,19 @@ kernel is freestanding and uses the **SysV AMD64** convention (first argument in
|
|||||||
read garbage.
|
read garbage.
|
||||||
|
|
||||||
So both sides pin the convention explicitly to SysV via the shared
|
So both sides pin the convention explicitly to SysV via the shared
|
||||||
`danos.kernel_abi` (defined in `src/root.zig`). The loader's function-pointer type
|
`boot_handoff.kernel_abi` (defined in `system/boot-handoff.zig`). The loader's
|
||||||
and the kernel's `_start` both reference it, so the pointer lands in the register
|
function-pointer type and the kernel's `_start` both reference it, so the pointer lands
|
||||||
the kernel expects. This is the whole reason `kernel_abi` lives in the shared
|
in the register the kernel expects. This is the whole reason `kernel_abi` lives in the
|
||||||
`danos` module: it's a contract both binaries must agree on. See
|
shared `boot-handoff` module: it's a contract both binaries must agree on. See
|
||||||
[sysv.md](sysv.md) for what "SysV" means and where else it shows up.
|
[sysv.md](sysv.md) for what "SysV" means and where else it shows up.
|
||||||
|
|
||||||
## The handoff contract
|
## The handoff contract
|
||||||
|
|
||||||
The loader and kernel are two *separate* binaries built for two different targets,
|
The loader and kernel are two *separate* binaries built for two different targets,
|
||||||
so everything they exchange must have an identically-defined memory layout. That's
|
so everything they exchange must have an identically-defined memory layout. That's
|
||||||
what `src/root.zig` provides — imported by both as the `danos` module:
|
what `system/boot-handoff.zig` provides — imported by both as the `boot-handoff` module.
|
||||||
|
It is *only* the handoff: the kernel↔user ABI (`system/abi.zig`) and the device types
|
||||||
|
(`system/devices/device-abi.zig`) are separate contracts the bootloader never sees.
|
||||||
|
|
||||||
- `BootInfo` — the top-level struct passed to the kernel (currently just the
|
- `BootInfo` — the top-level struct passed to the kernel (currently just the
|
||||||
framebuffer; this is where future handoff data like the memory map will go).
|
framebuffer; this is where future handoff data like the memory map will go).
|
||||||
@@ -164,16 +169,16 @@ the loader writes are the bytes the kernel reads.
|
|||||||
```
|
```
|
||||||
power on
|
power on
|
||||||
-> UEFI firmware initialises hardware
|
-> UEFI firmware initialises hardware
|
||||||
-> finds esp/EFI/BOOT/BOOTX64.efi, runs it (our efi.zig main)
|
-> finds EFI/BOOT/BOOTX64.efi on the FHS volume, runs it (our efi.zig main)
|
||||||
-> grab boot services
|
-> grab boot services
|
||||||
-> queryFramebuffer (via GOP: EDID native res, setMode, describe fb)
|
-> queryFramebuffer (via GOP: EDID native res, setMode, describe fb)
|
||||||
-> loadKernel (read danos ELF, load PT_LOAD segments to 0x100000)
|
-> loadKernel (read system/kernel ELF, load PT_LOAD segments to 0x100000)
|
||||||
-> exitBootServices (retry until the memory-map key holds)
|
-> exitBootServices (retry until the memory-map key holds)
|
||||||
-> jump to e_entry, boot_info pointer in RDI
|
-> jump to e_entry, boot_info pointer in RDI
|
||||||
-> kernel _start (src/kernel/main.zig: framebuffer console, then halt)
|
-> kernel _start (system/kernel/kernel.zig: framebuffer console, then halt)
|
||||||
```
|
```
|
||||||
|
|
||||||
Bottom line: **UEFI's job is to give us a CPU, memory, and a framebuffer, then
|
Bottom line: **UEFI's job is to give us a CPU, memory, and a framebuffer, then
|
||||||
disappear.** `src/boot/efi.zig` is the thin bridge that collects those gifts into a
|
disappear.** `boot/efi.zig` is the thin bridge that collects those gifts into a
|
||||||
`BootInfo`, tears down the firmware, and jumps into the kernel — after which we're
|
`BootInfo`, tears down the firmware, and jumps into the kernel — after which we're
|
||||||
on our own.
|
on our own.
|
||||||
|
|||||||
@@ -3,12 +3,12 @@
|
|||||||
Once the kernel knows what RAM exists ([memory-map.md](memory-map.md)), it needs a
|
Once the kernel knows what RAM exists ([memory-map.md](memory-map.md)), it needs a
|
||||||
way to *hand out* that RAM: give me a free page of physical memory, and later,
|
way to *hand out* that RAM: give me a free page of physical memory, and later,
|
||||||
here's one back. That's the **physical frame allocator** (a "physical memory
|
here's one back. That's the **physical frame allocator** (a "physical memory
|
||||||
manager", hence `src/kernel/pmm.zig`). It deals only in fixed 4 KiB **frames** — the
|
manager", hence `system/kernel/pmm.zig`). It deals only in fixed 4 KiB **frames** — the
|
||||||
natural unit because that's the granularity the CPU's paging hardware maps — and
|
natural unit because that's the granularity the CPU's paging hardware maps — and
|
||||||
it is the primitive everything above it stands on: page tables, the kernel heap,
|
it is the primitive everything above it stands on: page tables, the kernel heap,
|
||||||
per-process memory all ultimately ask the frame allocator for pages.
|
per-process memory all ultimately ask the frame allocator for pages.
|
||||||
|
|
||||||
It's **generic kernel code**: it operates on the neutral `danos.MemoryRegion`
|
It's **generic kernel code**: it operates on the neutral `system.MemoryRegion`
|
||||||
array, so there's no UEFI in it and nothing architecture-specific beyond the 4 KiB
|
array, so there's no UEFI in it and nothing architecture-specific beyond the 4 KiB
|
||||||
page. (Contrast [arch.md](arch.md), which is where CPU-specific code lives.)
|
page. (Contrast [arch.md](arch.md), which is where CPU-specific code lives.)
|
||||||
|
|
||||||
@@ -33,7 +33,7 @@ RAM is 32768 frames — a **4 KiB bitmap, a single frame**. Even 64 GiB needs on
|
|||||||
|
|
||||||
## How it works
|
## How it works
|
||||||
|
|
||||||
State lives in `src/kernel/pmm.zig`: the `bitmap` slice, `total_frames`, `used_frames`,
|
State lives in `system/kernel/pmm.zig`: the `bitmap` slice, `total_frames`, `used_frames`,
|
||||||
and a `next_hint` marking where the next allocation scan should start.
|
and a `next_hint` marking where the next allocation scan should start.
|
||||||
|
|
||||||
### init(map) — building it from the memory map
|
### init(map) — building it from the memory map
|
||||||
|
|||||||
+2
-2
@@ -9,10 +9,10 @@ write a 32-bit value to the right address, and a pixel changes color. That's
|
|||||||
exactly what `Console.pixel` does:
|
exactly what `Console.pixel` does:
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
self.rowPtr(y)[x] = color; // src/kernel/console.zig
|
self.rowPtr(y)[x] = color; // system/kernel/console.zig
|
||||||
```
|
```
|
||||||
|
|
||||||
Our `Framebuffer` struct (`src/root.zig`) is the four facts you need to
|
Our `Framebuffer` struct (`system/boot-handoff.zig`) is the four facts you need to
|
||||||
address it:
|
address it:
|
||||||
|
|
||||||
| Field | Meaning |
|
| Field | Meaning |
|
||||||
|
|||||||
+3
-3
@@ -16,7 +16,7 @@ safely, until the machine is reset or powered off.
|
|||||||
## The core of it: `hlt`
|
## The core of it: `hlt`
|
||||||
|
|
||||||
Everything comes down to one x86 instruction. It's CPU-specific, so it lives in
|
Everything comes down to one x86 instruction. It's CPU-specific, so it lives in
|
||||||
the arch module, `src/kernel/arch/x86_64/cpu.zig` (see [arch.md](arch.md)), and the
|
the arch module, `system/kernel/architecture/x86_64/cpu.zig` (see [arch.md](arch.md)), and the
|
||||||
generic kernel calls it as `arch.halt()`:
|
generic kernel calls it as `arch.halt()`:
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
@@ -83,7 +83,7 @@ treats the call:
|
|||||||
signature for a kernel entry point — the bootloader jumps in and nothing ever
|
signature for a kernel entry point — the bootloader jumps in and nothing ever
|
||||||
jumps back out.
|
jumps back out.
|
||||||
|
|
||||||
You can see the chain in `src/kernel/main.zig`: `_start` is `noreturn`, it calls
|
You can see the chain in `system/kernel/kernel.zig`: `_start` is `noreturn`, it calls
|
||||||
`kmain` which is `noreturn`, which ends by calling `arch.halt()` which is
|
`kmain` which is `noreturn`, which ends by calling `arch.halt()` which is
|
||||||
`noreturn`. The "never returns" property is threaded all the way down.
|
`noreturn`. The "never returns" property is threaded all the way down.
|
||||||
|
|
||||||
@@ -106,7 +106,7 @@ There are three halt sites, and they're all the same idea:
|
|||||||
`arch.halt()`. A panic is unrecoverable here, so stopping the machine — rather
|
`arch.halt()`. A panic is unrecoverable here, so stopping the machine — rather
|
||||||
than limping on with corrupted state — is the safe response.
|
than limping on with corrupted state — is the safe response.
|
||||||
|
|
||||||
3. **Bootloader failure** — in `src/boot/efi.zig`, if `boot()` fails *before* handing
|
3. **Bootloader failure** — in `boot/efi.zig`, if `boot()` fails *before* handing
|
||||||
off to the kernel, `main` logs the error and parks the machine with the same
|
off to the kernel, `main` logs the error and parks the machine with the same
|
||||||
loop so the message stays on screen:
|
loop so the message stays on screen:
|
||||||
|
|
||||||
|
|||||||
+1
-1
@@ -7,7 +7,7 @@ top of both to provide what the rest of the kernel actually wants: `alloc(n)` /
|
|||||||
the thing that unlocks dynamic data structures — lists, hash maps, driver state,
|
the thing that unlocks dynamic data structures — lists, hash maps, driver state,
|
||||||
eventually a process table.
|
eventually a process table.
|
||||||
|
|
||||||
It's generic kernel code (`src/kernel/heap.zig`): the allocator logic is
|
It's generic kernel code (`system/kernel/heap.zig`): the allocator logic is
|
||||||
architecture-neutral, using `arch.mapPage` and the frame allocator underneath.
|
architecture-neutral, using `arch.mapPage` and the frame allocator underneath.
|
||||||
|
|
||||||
## A growable free-list allocator
|
## A growable free-list allocator
|
||||||
|
|||||||
+161
@@ -0,0 +1,161 @@
|
|||||||
|
# The input module: broadcasting input events
|
||||||
|
|
||||||
|
A keyboard driver has one keystroke and *many* programs that might want it — a shell, a
|
||||||
|
window server, a logger. None of them owns the hardware, and the driver should not know
|
||||||
|
who is listening. So between the drivers and the listeners sits the **input service**
|
||||||
|
(`system/services/input/`): drivers **publish** events to it, programs **subscribe**, and
|
||||||
|
it fans each event out to every interested subscriber. It is an ordinary ring-3 process
|
||||||
|
reached over IPC, like the [VFS server](../system/services/vfs/vfs.zig) — no kernel knows
|
||||||
|
what a key is.
|
||||||
|
|
||||||
|
## One service, several device classes
|
||||||
|
|
||||||
|
The service carries three device classes today — **keyboard**, **mouse**, and
|
||||||
|
**joystick/gamepad** — and is built to take more
|
||||||
|
([protocol.zig](../system/services/input/protocol.zig)). Each class has its own typed
|
||||||
|
event:
|
||||||
|
|
||||||
|
- `KeyEvent` — `key_down`/`key_up` (physical make/break) and `key_press` (a character was
|
||||||
|
produced, carrying the Unicode scalar); plus a layout-independent `keycode` and a
|
||||||
|
`modifiers` bitmask.
|
||||||
|
- `MouseEvent` — relative `motion` (`dx`/`dy`), `button_down`/`button_up`, and `scroll`.
|
||||||
|
- `JoystickEvent` — `axis` moves (a signed value on a `control` index) and
|
||||||
|
`button_down`/`button_up`.
|
||||||
|
|
||||||
|
All three travel in one **`InputEvent` envelope** tagged with a `DeviceKind`, so the
|
||||||
|
fan-out is a single code path and a subscriber can take a mix of classes on one stream.
|
||||||
|
Decode an envelope with `asKeyboard()` / `asMouse()` / `asJoystick()` (each returns null
|
||||||
|
unless the tag matches). A subscriber names the classes it wants with a **`device_mask`**,
|
||||||
|
and the service routes each event only to subscribers whose mask includes its class — so a
|
||||||
|
mouse-only listener never wakes for keystrokes.
|
||||||
|
|
||||||
|
## Why this needed a new kernel primitive
|
||||||
|
|
||||||
|
The interesting part is delivery, and it runs straight into the shape of danos IPC.
|
||||||
|
[ipc.md](ipc.md) describes a **synchronous rendezvous**: a server holds exactly one
|
||||||
|
pending reply (`Task.ipc_client`) and *must* answer it on its next `replyWait`. Two
|
||||||
|
consequences decide the whole design:
|
||||||
|
|
||||||
|
1. **You cannot block N subscribers waiting for "the next event".** A server can hold only
|
||||||
|
one caller at a time, so the natural "subscriber calls `next_event()` and blocks" API
|
||||||
|
is impossible for more than one subscriber. Delivery therefore has to be **push** — the
|
||||||
|
service reaching out to subscribers — not pull.
|
||||||
|
|
||||||
|
2. **A synchronous push can hang the whole service.** If the service delivered with
|
||||||
|
`ipc_call`, it would block until each subscriber replied. `ipc_call` has no timeout, and
|
||||||
|
the kernel does **not** wake a caller parked on a *dead* peer's endpoint (it only fails a
|
||||||
|
peer that was mid-reply — see [process.zig](../system/kernel/process.zig)
|
||||||
|
`releaseTaskResourcesLocked`). One subscriber that exits mid-delivery would wedge input
|
||||||
|
for everyone. That is the opposite of the resilience the microkernel is for.
|
||||||
|
|
||||||
|
The fix is the asynchronous send that [ipc.md](ipc.md) had already earmarked as future
|
||||||
|
work ("asynchronous / buffered send … for notifications between servers"):
|
||||||
|
|
||||||
|
```
|
||||||
|
ipc_send(handle, message_ptr, message_len) -> 0 / -errno
|
||||||
|
```
|
||||||
|
|
||||||
|
`ipc_send` copies a small payload into the endpoint's **bounded queue** and wakes a
|
||||||
|
receiver, then returns immediately — it never blocks and so can never hang on a dead or
|
||||||
|
slow subscriber. The receiver picks it up through the same `replyWait` it already runs:
|
||||||
|
the wake arrives as a **buffered message** — `notify_badge_bit | notify_message_bit` set in
|
||||||
|
the badge (distinguishing it from a bare IRQ/child-exit notification), the sender's task id
|
||||||
|
in the low bits, and the payload in the receive buffer, with no reply owed. The queue holds
|
||||||
|
16 messages per endpoint; a full queue **drops the oldest**, because a buffered message is
|
||||||
|
discrete data, not a coalescing "level" like an interrupt. See
|
||||||
|
[ipc-synchronous.zig](../system/kernel/ipc-synchronous.zig) (`sendLocked`, `popPost`, and
|
||||||
|
the `replyWait` receive loop).
|
||||||
|
|
||||||
|
This is the async counterpart of `ipc_call`, and the input service is its first consumer.
|
||||||
|
|
||||||
|
## How the pieces fit
|
||||||
|
|
||||||
|
```
|
||||||
|
keyboard/mouse driver, input-source input service subscriber(s)
|
||||||
|
----------------------------------- ------------- -------------
|
||||||
|
connectSource(); loop: replyWait: subscribeKeyboard()/…All:
|
||||||
|
publishKeyboardEvent(k) ─ ipc_call ─▶ publish → broadcast: createIpcEndpoint()
|
||||||
|
publishMouseEvent(m) for each sub whose callCap(subscribe,
|
||||||
|
publishJoystickEvent(j) mask matches event.device: send_cap = ep,
|
||||||
|
ipc_send(sub_ep) ──────▶ device_mask)
|
||||||
|
reply ok loop: next()
|
||||||
|
subscribe → store {ep cap, └─ replyWait(ep)
|
||||||
|
task id, device_mask} → InputEvent
|
||||||
|
```
|
||||||
|
|
||||||
|
- A **subscriber** calls `input.subscribe(mask)` — or a typed helper: `subscribeKeyboard()`,
|
||||||
|
`subscribeMouse()`, `subscribeJoystick()` (one class, `next()` returns the decoded event),
|
||||||
|
or `subscribeAll()` (every class, `next()` returns a tagged `InputEvent`)
|
||||||
|
([library/runtime/input.zig](../library/runtime/input.zig)). It creates its own endpoint
|
||||||
|
and hands it to the service as a **capability** (M13 capability passing — the input
|
||||||
|
service is that feature's first real user), along with its `device_mask`. Then it loops on
|
||||||
|
`next()`, a `replyWait` on that endpoint returning each pushed event.
|
||||||
|
- A **source** (a keyboard, mouse, or joystick driver) calls `input.connectSource()` and the
|
||||||
|
method for its class: `publishKeyboardEvent`, `publishMouseEvent`, or
|
||||||
|
`publishJoystickEvent`. Publishing is a short synchronous `ipc_call` the service answers at
|
||||||
|
once; the service's own fan-out is asynchronous, so publishing never blocks on a slow
|
||||||
|
subscriber.
|
||||||
|
- The **service** ([input.zig](../system/services/input/input.zig)) keeps a small subscriber
|
||||||
|
table (endpoint handle + owning task id + `device_mask`). On `publish` it `ipc_send`s the
|
||||||
|
event to every subscriber whose mask includes the event's device class. On `subscribe` it
|
||||||
|
stores the passed capability and mask and, as housekeeping, prunes any slot whose owning
|
||||||
|
process has exited (checked against `process_enumerate`) — not for correctness (an async
|
||||||
|
send to an orphaned endpoint is harmless) but to reclaim the slot.
|
||||||
|
|
||||||
|
Publisher and subscriber must be **separate processes**: a single thread that both
|
||||||
|
published and serviced its own subscription would deadlock (its `publish` call blocks until
|
||||||
|
the service delivers to its endpoint, which only the same thread could receive).
|
||||||
|
|
||||||
|
## Status and follow-ups
|
||||||
|
|
||||||
|
- **The keyboard is real.** The `ps2-bus` driver owns PNP0303, which carries *both* the
|
||||||
|
0x60/0x64 ports and IRQ1, so reading the hardware lives in the bus, not in
|
||||||
|
[keyboard.zig](../system/drivers/ps2-bus/keyboard.zig): the bus binds IRQ1 and, on each
|
||||||
|
interrupt, drains port 0x60, routing every byte by the status register's
|
||||||
|
auxiliary-output bit to whichever child driver **attached** for that device (an
|
||||||
|
`AttachRequest` to the well-known `ps2_bus` service, carrying the child's endpoint as a
|
||||||
|
capability; the bytes then arrive as asynchronous `ForwardedByte` messages, so the IRQ
|
||||||
|
path never blocks on a child). The keyboard driver decodes the stream — scancode **set 2**,
|
||||||
|
what the keyboard sends with the 8042's legacy translation off, decoded by
|
||||||
|
[scancode.zig](../system/drivers/ps2-bus/scancode.zig) into USB HID usage keycodes with
|
||||||
|
make/break, typematic-repeat, and modifier tracking (host-tested under `zig build test`) —
|
||||||
|
and publishes real `key_down`/`key_press`/`key_up` events.
|
||||||
|
- **Keycode → character** is wired in: the keyboard driver fills a `key_press` event's
|
||||||
|
`character` through [`library/xkeyboard-config`](../library/xkeyboard-config/README.md)
|
||||||
|
(`xkb.map(layout, keycode, mods)` → keysym + Unicode character), synthesizing the ASCII
|
||||||
|
control characters for Enter/Tab/Backspace/Escape, whose keysyms map to no Unicode. The
|
||||||
|
layout defaults to `us`; the bus can pass another as the driver's argv[2] — the seam for
|
||||||
|
a future settings source.
|
||||||
|
- **The mouse is real too.** IRQ12 is enumerated on the auxiliary device's own ACPI node
|
||||||
|
(PNP0F13), so the bus claims that node alongside the controller and routes both IRQs to
|
||||||
|
its one endpoint, acking whichever line the notification's badge names.
|
||||||
|
[mouse.zig](../system/drivers/ps2-bus/mouse.zig) attaches the way the keyboard does and
|
||||||
|
assembles the forwarded bytes with
|
||||||
|
[mouse-packet.zig](../system/drivers/ps2-bus/mouse-packet.zig) (three-byte stream-mode
|
||||||
|
packets: sync/overflow handling, nine-bit movement, screen-convention `dy` — host-tested
|
||||||
|
under `zig build test`) into `button_down`/`button_up` transitions and `motion` events.
|
||||||
|
**Follow-up:** the IntelliMouse magic-knock for a scroll wheel (four-byte packets) and
|
||||||
|
`scroll` events. The hardware-free `input-source` still rotates through all three classes
|
||||||
|
synthetically (including a joystick, which has no driver yet) via the
|
||||||
|
`input.synthetic*Event` helpers.
|
||||||
|
- **Drop-oldest under overflow** is a defined loss; the 16-slot ring absorbs normal bursts.
|
||||||
|
Real backpressure/flow-control is future work.
|
||||||
|
- **`publish` is unauthenticated** — any process may publish, consistent with the current
|
||||||
|
bring-up trust model (see [driver-model.md](driver-model.md)). A source capability is
|
||||||
|
future work.
|
||||||
|
|
||||||
|
## Verifying it
|
||||||
|
|
||||||
|
The `input` case (`python3 test/qemu_test.py input`, in
|
||||||
|
[tests.zig](../system/kernel/tests.zig) `inputTest`) boots the real kernel and spawns the
|
||||||
|
service, the synthetic source (which cycles keyboard, mouse, and joystick events), and a
|
||||||
|
subscriber that took all three classes. It passes only when the subscriber heartbeats
|
||||||
|
`input-test: ok` — proof that an event travelled source → service → subscriber over IPC,
|
||||||
|
exercising `ipc_send`, capability-passing subscription, and per-device routing. Each
|
||||||
|
serial line names the class received, so the log shows all three arriving on one stream.
|
||||||
|
|
||||||
|
## See also
|
||||||
|
|
||||||
|
- [ipc.md](ipc.md) — the synchronous rendezvous and the notification path `ipc_send` extends.
|
||||||
|
- [syscall.md](syscall.md) — the system-call surface, including `ipc_send`.
|
||||||
|
- [driver-model.md](driver-model.md) — class drivers, capability passing (M13), the trust model.
|
||||||
+20
-9
@@ -9,7 +9,7 @@ reboot is miserable.
|
|||||||
|
|
||||||
This is the machinery that catches those faults and prints what happened instead.
|
This is the machinery that catches those faults and prints what happened instead.
|
||||||
It's all x86_64-specific, so it lives behind the [arch](arch.md) boundary in
|
It's all x86_64-specific, so it lives behind the [arch](arch.md) boundary in
|
||||||
`src/kernel/arch/x86_64/`. Only the 32 CPU-defined exception vectors are wired up so far;
|
`system/kernel/architecture/x86_64/`. Only the 32 CPU-defined exception vectors are wired up so far;
|
||||||
device interrupts (timer, keyboard, via the APIC) come later, on the same IDT.
|
device interrupts (timer, keyboard, via the APIC) come later, on the same IDT.
|
||||||
|
|
||||||
## First the GDT
|
## First the GDT
|
||||||
@@ -20,7 +20,7 @@ IDT gate names a code-segment *selector* that must resolve in the current GDT. T
|
|||||||
firmware left a GDT in place, but we don't control it, so we install our own with
|
firmware left a GDT in place, but we don't control it, so we install our own with
|
||||||
known selectors: `0x08` kernel code, `0x10` kernel data.
|
known selectors: `0x08` kernel code, `0x10` kernel data.
|
||||||
|
|
||||||
`src/kernel/arch/x86_64/gdt.zig` holds three flat descriptors — a required null entry,
|
`system/kernel/architecture/x86_64/gdt.zig` holds three flat descriptors — a required null entry,
|
||||||
plus code and data — where the only bits that matter in long mode are the access
|
plus code and data — where the only bits that matter in long mode are the access
|
||||||
byte and the code segment's long-mode (`L`) flag. Loading it (`gdt_flush` in
|
byte and the code segment's long-mode (`L`) flag. Loading it (`gdt_flush` in
|
||||||
`isr.s`) does two things: `lgdt`, then reload the segment registers. The data
|
`isr.s`) does two things: `lgdt`, then reload the segment registers. The data
|
||||||
@@ -33,7 +33,7 @@ into CS:RIP.
|
|||||||
The **Interrupt Descriptor Table** maps each of 256 vectors to a handler. Each
|
The **Interrupt Descriptor Table** maps each of 256 vectors to a handler. Each
|
||||||
entry is a 16-byte *gate* holding the handler's address (split across three
|
entry is a 16-byte *gate* holding the handler's address (split across three
|
||||||
fields, a quirk of the format), the code selector (`0x08`), and flags: `0x8E`
|
fields, a quirk of the format), the code selector (`0x08`), and flags: `0x8E`
|
||||||
means present, ring 0, 64-bit interrupt gate. `src/kernel/arch/x86_64/idt.zig` builds the
|
means present, ring 0, 64-bit interrupt gate. `system/kernel/architecture/x86_64/idt.zig` builds the
|
||||||
table, points the first 32 vectors at their stubs, and loads it with `lidt`
|
table, points the first 32 vectors at their stubs, and loads it with `lidt`
|
||||||
(`idt_flush`).
|
(`idt_flush`).
|
||||||
|
|
||||||
@@ -49,7 +49,7 @@ hit a fault *while trying to deliver another fault* — very often because the
|
|||||||
current stack pointer is bad, so pushing the exception frame itself faulted. If
|
current stack pointer is bad, so pushing the exception frame itself faulted. If
|
||||||
the #DF handler then tried to push onto that same bad stack, it would fault a
|
the #DF handler then tried to push onto that same bad stack, it would fault a
|
||||||
third time and **triple-fault** — an instant reset. So the #DF gate is pointed at
|
third time and **triple-fault** — an instant reset. So the #DF gate is pointed at
|
||||||
**IST1**, a small dedicated stack (`src/kernel/arch/x86_64/tss.zig`) that's always valid.
|
**IST1**, a small dedicated stack (`system/kernel/architecture/x86_64/tss.zig`) that's always valid.
|
||||||
|
|
||||||
Bringing it up: fill in the TSS's IST1 pointer, publish the TSS through a
|
Bringing it up: fill in the TSS's IST1 pointer, publish the TSS through a
|
||||||
descriptor in the GDT (`gdt.setTss`), and load it into the task register with
|
descriptor in the GDT (`gdt.setTss`), and load it into the task register with
|
||||||
@@ -60,7 +60,7 @@ which is why the GDT grew from three entries to five.
|
|||||||
|
|
||||||
On an exception the CPU pushes a small frame (SS, RSP, RFLAGS, CS, RIP) and, for
|
On an exception the CPU pushes a small frame (SS, RSP, RFLAGS, CS, RIP) and, for
|
||||||
*some* vectors, an **error code**. That inconsistency is a nuisance, so each stub
|
*some* vectors, an **error code**. That inconsistency is a nuisance, so each stub
|
||||||
in `src/kernel/arch/x86_64/isr.s` normalises it: vectors that don't get a hardware error
|
in `system/kernel/architecture/x86_64/isr.s` normalises it: vectors that don't get a hardware error
|
||||||
code push a dummy `0`, then every stub pushes its **vector number** and jumps to a
|
code push a dummy `0`, then every stub pushes its **vector number** and jumps to a
|
||||||
shared tail, `isr_common`. The tail pushes all the general registers and calls the
|
shared tail, `isr_common`. The tail pushes all the general registers and calls the
|
||||||
Zig handler with a pointer to the whole thing.
|
Zig handler with a pointer to the whole thing.
|
||||||
@@ -80,11 +80,22 @@ inline). `build.zig` adds `isr.s` to the arch module.
|
|||||||
## Reporting a fault
|
## Reporting a fault
|
||||||
|
|
||||||
`isr_common` calls `exceptionHandler`, which forwards to a swappable `on_fault`
|
`isr_common` calls `exceptionHandler`, which forwards to a swappable `on_fault`
|
||||||
hook. The generic kernel installs a reporter (`onException` in `main.zig`) that
|
hook. The generic kernel installs a reporter (`onException` in `kernel.zig`) that
|
||||||
prints, in red, the exception name and vector, the error code, the faulting RIP
|
prints the exception name and vector, the error code, the faulting RIP
|
||||||
and RSP, and — for a page fault (#PF, vector 14) — the faulting address from
|
and RSP, and — for a page fault (#PF, vector 14) — the faulting address from
|
||||||
**CR2**. Then it halts. There's no fault *recovery* yet, so every exception is
|
**CR2**. What happens next depends on where the fault came from:
|
||||||
terminal; the point is that it's now **visible** instead of a silent reset.
|
|
||||||
|
- **User mode (CPL 3): kill the process, keep the machine.** The kernel is intact
|
||||||
|
(the CPU trapped onto the task's kernel stack), so the faulting process is
|
||||||
|
killed — address space, IRQ bindings, and IPC handles reclaimed; a client it
|
||||||
|
owed a reply to is failed with `-EPEER` — and the core reschedules. A crashing
|
||||||
|
driver takes itself down, never the OS. This is fault recovery step 2 of
|
||||||
|
[resilience.md](resilience.md). NMI, double fault, and machine check are
|
||||||
|
excluded: they report machine trouble regardless of what was running.
|
||||||
|
- **Kernel mode: halt this core.** The trusted base itself is broken, so there is
|
||||||
|
nothing safe to kill; the fault is still *contained* to the core (an
|
||||||
|
application-processor fault leaves the rest of the system running), and the
|
||||||
|
report makes it **visible** instead of a silent reset.
|
||||||
|
|
||||||
The hook is set before `arch.init()` in `kmain`, so a fault during setup is still
|
The hook is set before `arch.init()` in `kmain`, so a fault during setup is still
|
||||||
caught.
|
caught.
|
||||||
|
|||||||
+79
-12
@@ -6,12 +6,20 @@ just call each other — a request becomes a **message**. In a microkernel, what
|
|||||||
was a function call across a monolithic kernel is IPC, so it's a first-class
|
was a function call across a monolithic kernel is IPC, so it's a first-class
|
||||||
concern, not an afterthought.
|
concern, not an afterthought.
|
||||||
|
|
||||||
This first form is a **bounded blocking channel** (`src/kernel/ipc.zig`): a fixed-size
|
There are two layers, built a milestone apart:
|
||||||
ring buffer of messages with a producer/consumer rendezvous, built on the
|
|
||||||
scheduler's [wait queues](scheduling.md).
|
- **`system/kernel/ipc.zig`** — a bounded blocking channel between *kernel threads*,
|
||||||
|
described below. The primitive, and where the blocking discipline was worked out.
|
||||||
|
- **`system/kernel/ipc-synchronous.zig`** — synchronous call/reply between *processes*, across
|
||||||
|
address spaces. What user-space servers and drivers actually talk over. It's the
|
||||||
|
second half of this document.
|
||||||
|
|
||||||
## The channel
|
## The channel
|
||||||
|
|
||||||
|
The first form is a **bounded blocking channel** (`system/kernel/ipc.zig`): a fixed-size
|
||||||
|
ring buffer of messages with a producer/consumer rendezvous, built on the
|
||||||
|
scheduler's [wait queues](scheduling.md).
|
||||||
|
|
||||||
`Channel(T, capacity)` is generic over the message type and buffer size. It holds a
|
`Channel(T, capacity)` is generic over the message type and buffer size. It holds a
|
||||||
ring buffer, a count, and two wait queues:
|
ring buffer, a count, and two wait queues:
|
||||||
|
|
||||||
@@ -42,16 +50,75 @@ full and empty over and over, so both the blocking-send and blocking-recv paths
|
|||||||
exercised heavily. The messages arrive intact and in order (their sum is the
|
exercised heavily. The messages arrive intact and in order (their sum is the
|
||||||
expected `5050`), and neither task busy-waits — they block and wake each other.
|
expected `5050`), and neither task busy-waits — they block and wake each other.
|
||||||
|
|
||||||
|
## Endpoints: call/reply across address spaces
|
||||||
|
|
||||||
|
A channel connects two kernel threads sharing one address space. Real servers are
|
||||||
|
*processes*, so the payload has to cross an address-space boundary. That's
|
||||||
|
`system/kernel/ipc-synchronous.zig`, and its shape is L4's: a synchronous **rendezvous** at an
|
||||||
|
`Endpoint`, with the message copied directly from the sender's pages to the receiver's
|
||||||
|
(`copyAcross` walks both sets of page tables through the physmap — no CR3 switch, no
|
||||||
|
bounce buffer).
|
||||||
|
|
||||||
|
Two syscalls carry it:
|
||||||
|
|
||||||
|
- **`ipc_call(h, msg, reply)`** — copy `msg` to the server, block until it replies.
|
||||||
|
- **`ipc_reply_wait(h, reply, recv)`** — reply to the client you're still holding (if
|
||||||
|
any), then block for the next request. One syscall, because a server's steady state
|
||||||
|
is *always* "finish the last one, wait for the next".
|
||||||
|
|
||||||
|
An endpoint is reached by **handle** — a small integer index into the process's handle
|
||||||
|
table (`Task.handles`), exactly like a file descriptor, and just as unforgeable. The
|
||||||
|
bootstrap problem (how do you get the first handle?) is solved by a tiny name registry:
|
||||||
|
a server calls `ipc_register(service_id, h)` under a well-known small integer, and a
|
||||||
|
client calls `ipc_lookup(service_id)`.
|
||||||
|
|
||||||
|
The server never learns the client's identity beyond a **badge**, delivered alongside
|
||||||
|
the message: the caller's task id.
|
||||||
|
|
||||||
|
### Interrupts are messages too
|
||||||
|
|
||||||
|
`notifyFromIsr` posts an *asynchronous* notification to an endpoint — no payload, no
|
||||||
|
reply owed — and wakes whoever is blocked in `reply_wait`. Its badge has the top bit
|
||||||
|
set (`notify_badge_bit`), which is how a driver's single event loop distinguishes "a
|
||||||
|
client wants something" from "the hardware wants something". Notifications sit in a
|
||||||
|
small coalescing ring on the endpoint, so an interrupt taken while the driver was busy
|
||||||
|
elsewhere is not lost.
|
||||||
|
|
||||||
|
This is what makes a user-space driver possible at all, and it's the subject of
|
||||||
|
[drivers.md](drivers.md).
|
||||||
|
|
||||||
## What's next (not done here)
|
## What's next (not done here)
|
||||||
|
|
||||||
- **Across address spaces.** Today both endpoints are kernel threads sharing the
|
|
||||||
kernel's memory, so the message is copied within one address space. When user
|
|
||||||
mode arrives, the same channel carries messages between *isolated* processes,
|
|
||||||
copying the payload across the boundary — which is where IPC earns its place as
|
|
||||||
the microkernel's backbone.
|
|
||||||
- **Synchronous call/reply.** A request/response pattern (send-and-wait-for-reply)
|
|
||||||
on top of channels, the shape most driver/service calls take.
|
|
||||||
- **Interrupts as messages.** A hardware interrupt delivered to the driver task
|
|
||||||
that owns the device, as an IPC message.
|
|
||||||
- **Priority inheritance** through IPC, so a high-priority client blocked on a
|
- **Priority inheritance** through IPC, so a high-priority client blocked on a
|
||||||
low-priority server doesn't suffer unbounded priority inversion.
|
low-priority server doesn't suffer unbounded priority inversion.
|
||||||
|
- **Handle transfer.** A server can't hand a client a handle to a third endpoint, so
|
||||||
|
every capability is either well-known (the registry) or inherited — there's no way
|
||||||
|
to delegate one.
|
||||||
|
- **Asynchronous / buffered send** for the cases where a rendezvous is the wrong
|
||||||
|
shape (logging, notifications between servers). *Landed as `ipc_send`* — a
|
||||||
|
non-blocking post to an endpoint's bounded payload queue, delivered through
|
||||||
|
`reply_wait` as a buffered message (badge bit `notify_message_bit`). Built for, and
|
||||||
|
first used by, the [input service](input.md)'s keyboard-event broadcast, where a
|
||||||
|
synchronous push would let one dead subscriber hang the fan-out. A full queue drops
|
||||||
|
the oldest (discrete messages, not a coalescing level like the notification ring).
|
||||||
|
- **A bounded reply.** `MSG_MAX` is 256 bytes and the copy runs under the big kernel
|
||||||
|
lock; a bulk transfer wants shared pages, not a copy.
|
||||||
|
|
||||||
|
## Lifecycle conventions over IPC (M17)
|
||||||
|
|
||||||
|
Three conventions from [process-lifecycle.md](process-lifecycle.md) ride the
|
||||||
|
notification mechanism:
|
||||||
|
|
||||||
|
- **Signals** arrive as notifications on the endpoint a process nominated with
|
||||||
|
`signal_bind` (`runtime.process.bindSignals`): badge = the signal bit plus the
|
||||||
|
coalesced pending mask (`runtime.process.signalsFrom` decodes). Statements,
|
||||||
|
never questions; no payload, no reply.
|
||||||
|
- **One-shot timers** (`timer_bind`, `runtime.system.timerOnce`) land as a
|
||||||
|
timer-bit notification — the timed wait: a service arms a deadline and keeps
|
||||||
|
serving, instead of blocking in sleep.
|
||||||
|
- **The universal ping**: a **zero-length request is the liveness probe**,
|
||||||
|
answered with a zero-length reply by the service harness itself
|
||||||
|
(`runtime.service.run`). No protocol's requests start at length zero, so the
|
||||||
|
encoding cannot collide, and a wedged service simply fails to answer — which
|
||||||
|
is the diagnosis. Deep health ("can I reach my hardware?") stays a per-service
|
||||||
|
protocol message.
|
||||||
|
|||||||
+3
-3
@@ -13,7 +13,7 @@ kernel follows.
|
|||||||
|
|
||||||
## The log is multi-sink
|
## The log is multi-sink
|
||||||
|
|
||||||
`src/kernel/log.zig` is the diagnostic log. It fans a message out to a set of
|
`system/kernel/log.zig` is the diagnostic log. It fans a message out to a set of
|
||||||
registered **sinks**, each best-effort and self-guarding:
|
registered **sinks**, each best-effort and self-guarding:
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
@@ -36,7 +36,7 @@ Properties that matter:
|
|||||||
## The framebuffer is *not* a log sink
|
## The framebuffer is *not* a log sink
|
||||||
|
|
||||||
The framebuffer is a general graphics surface, **not inherently a text terminal**.
|
The framebuffer is a general graphics surface, **not inherently a text terminal**.
|
||||||
Today `src/kernel/console.zig` paints a text grid on it as a *bootstrap* console, but
|
Today `system/kernel/console.zig` paints a text grid on it as a *bootstrap* console, but
|
||||||
that's a stop-gap: once the driver machinery exists the framebuffer becomes a proper
|
that's a stop-gap: once the driver machinery exists the framebuffer becomes a proper
|
||||||
**graphics device driver**, and the text crutch goes away. So the log must not assume
|
**graphics device driver**, and the text crutch goes away. So the log must not assume
|
||||||
it — routing the verbose log through a pixel console would bake in "the OS is text".
|
it — routing the verbose log through a pixel console would bake in "the OS is text".
|
||||||
@@ -58,7 +58,7 @@ screen. `console.write` is a no-op when the firmware gave us no framebuffer.
|
|||||||
A framebuffer is not guaranteed — a headless server exposes no UEFI Graphics Output
|
A framebuffer is not guaranteed — a headless server exposes no UEFI Graphics Output
|
||||||
Protocol. That used to be *fatal* (the loader failed the boot). Now the loader hands
|
Protocol. That used to be *fatal* (the loader failed the boot). Now the loader hands
|
||||||
over a "no framebuffer" descriptor (`base == 0`) rather than failing, and
|
over a "no framebuffer" descriptor (`base == 0`) rather than failing, and
|
||||||
`Framebuffer.present()` (in `src/root.zig`) gates every on-screen path. A headless,
|
`Framebuffer.present()` (in `system/boot-handoff.zig`) gates every on-screen path. A headless,
|
||||||
serial-less machine boots and runs correctly — it just goes quiet.
|
serial-less machine boots and runs correctly — it just goes quiet.
|
||||||
|
|
||||||
## Last-resort channels (no text output at all)
|
## Last-resort channels (no text output at all)
|
||||||
|
|||||||
@@ -0,0 +1,206 @@
|
|||||||
|
# M17–M18 execution plan: process lifecycle + device manager
|
||||||
|
|
||||||
|
The operational plan for building [process-lifecycle.md](process-lifecycle.md)
|
||||||
|
(M17) and [device-manager.md](device-manager.md) increments 5–7 (M18). Design is
|
||||||
|
settled in those documents; this file is the build order — one phase at a time,
|
||||||
|
each phase green before the next starts. Delete or archive this file when M18
|
||||||
|
lands.
|
||||||
|
|
||||||
|
**Definition of green, every phase:** `zig build` clean, `zig build test` clean,
|
||||||
|
`python3 test/qemu_test.py` passes (existing scenarios plus the phase's new one),
|
||||||
|
and the relevant design doc's "known gaps" / status lines updated. Commit per
|
||||||
|
green phase (no co-author trailers).
|
||||||
|
|
||||||
|
**Workflow (settled 2026-07-12):** work happens in a dedicated git worktree, on
|
||||||
|
feature branches cut from `main` — `feat/process-lifecycle` (M17.1–17.4),
|
||||||
|
`feat/device-manager` (M18.1), `feat/usb-xhci-bus` (M18.2–18.3). When a branch's
|
||||||
|
phases are all green it is **auto-merged into `main`**; branches are kept after
|
||||||
|
merge, not deleted. Merges and branches are pushed to origin. Phase 0 (once):
|
||||||
|
commit the design docs, merge the outstanding `feat/usb` work into `main`, and
|
||||||
|
run the existing QEMU suite green before any new work starts.
|
||||||
|
|
||||||
|
**Numbering note:** continues the milestone sequence (driver track ended at M16).
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
The loop marks a phase `[x]` in the same commit that lands it. A phase is marked
|
||||||
|
only when its definition of green holds.
|
||||||
|
|
||||||
|
- [x] **Phase 0** — baseline: docs committed, feat/usb merged to main, pushed;
|
||||||
|
`usb-xhci-libary.zig` renamed to `usb-xhci-library.zig`; existing QEMU
|
||||||
|
suite green from the worktree (48/48, 2026-07-12).
|
||||||
|
- [x] **M17.1** — kernel releases claims/MSI on death (claims: `releaseAllOwnedBy`
|
||||||
|
in the reap; MSI was already swept by `irq.releaseOwner`; `claim-release`
|
||||||
|
test; suite 49/49)
|
||||||
|
- [x] **M17.2** — exit reasons (`ExitReason` recorded at exit/fault/kill before
|
||||||
|
the notification; `process_exit_reason` supervisor-gated;
|
||||||
|
`runtime.process.exitReason`; kernel + ring-3 assertions; suite 49/49)
|
||||||
|
- [x] **M17.3** — published exit events + VFS subscriber (`process_subscribe`,
|
||||||
|
bounded ref-counted table, publish on every death;
|
||||||
|
`runtime.process.subscribeExits`; VFS handles carry owners and are swept on
|
||||||
|
the owner's death; `vfs-client-death` test; suite 50/50)
|
||||||
|
- [x] **M17.4** — signals, timer notifications, `runtime.process`, the service
|
||||||
|
harness (signal_bind/process_signal + coalescing pending mask; timer_bind
|
||||||
|
on the tick; bindSignals/signalsFrom/sendSignal/stop + timerOnce;
|
||||||
|
runtime.service.run with the zero-length ping; VFS converted; `signals`
|
||||||
|
scenario; suite 51/51)
|
||||||
|
- [x] **merge** `feat/process-lifecycle` → main, push (merged 2026-07-13)
|
||||||
|
- [x] **M18.1** — device-manager protocol: hello + restart policy
|
||||||
|
(device-manager-protocol module; the manager as a harness service:
|
||||||
|
supervised spawns, hello deadline via timer sweep, restart with
|
||||||
|
300/600/1200ms backoff, exit reasons deciding restart-vs-stopped,
|
||||||
|
crash-loop cap; usb-xhci-bus first conforming driver; crash-test fixture
|
||||||
|
re-proving claim release each respawn; `driver-restart` scenario;
|
||||||
|
maximum_tasks 16→32 — the sweep was overflowing the pool; suite 52/52)
|
||||||
|
- [x] **merge** `feat/device-manager` → main, push (merged 2026-07-13)
|
||||||
|
- [x] **M18.2** — xHCI port scan + tree reports (child_added/child_removed in
|
||||||
|
the protocol; the manager's child mirror with death-pruning; xHCI maps the
|
||||||
|
register BAR — resource 0 is ECAM — reads CAPLENGTH/HCSPARAMS1, scans
|
||||||
|
PORTSC, reports connected ports with speed-class identity; `usb-report`
|
||||||
|
scenario proves report → prune → respawn → re-report; suite 53/53)
|
||||||
|
- [x] **M18.3** — app surface: enumerate/subscribe over IPC (subscriber
|
||||||
|
endpoint rides as the call's capability; events are the same structs the
|
||||||
|
buses send); device-list first client; protocol capped at the kernel's
|
||||||
|
IPC MESSAGE_MAXIMUM (256); the startUserTask debug print removed — it
|
||||||
|
sheared concurrent serial lines and was the scenario-flake root cause;
|
||||||
|
`device-list` scenario; suite 54/54)
|
||||||
|
- [x] **merge** `feat/usb-xhci-bus` → main, push (merged 2026-07-13) — **plan complete**
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## M17.1 — the kernel releases a dead process's claims
|
||||||
|
|
||||||
|
The cleanup half of iron rule 1; the prerequisite for every restart story.
|
||||||
|
|
||||||
|
- `system/kernel/devices-broker.zig`: `releaseAllOwnedBy(owner: u32)` — clear
|
||||||
|
every `claimed[]` slot holding this task id.
|
||||||
|
- `system/kernel/process.zig`: call it from the reap path, alongside the existing
|
||||||
|
IRQ-binding release (the ordering comment there says why IRQs go first — claims
|
||||||
|
slot in after them, before the exit notification).
|
||||||
|
- MSI vectors: find where `msi_bind` records per-device vectors (interrupts
|
||||||
|
module) and release those by owner in the same pass.
|
||||||
|
- Docs: remove the claims bullet from process-management.md "Known gaps".
|
||||||
|
|
||||||
|
**Test:** new QEMU scenario `claim-release` — a test child claims an unclaimed
|
||||||
|
device, is killed, is respawned, and claims the same device again successfully;
|
||||||
|
assert both claims in the serial log. Kernel-side unit coverage in
|
||||||
|
`system/kernel/tests.zig` for `releaseAllOwnedBy` (claim two devices as two owners,
|
||||||
|
release one owner, verify exactly its claims freed).
|
||||||
|
|
||||||
|
## M17.2 — exit reasons
|
||||||
|
|
||||||
|
- `system/abi.zig`: `ExitReason` (exited, aborted, segmentation_fault,
|
||||||
|
illegal_instruction, arithmetic_fault, killed).
|
||||||
|
- Kernel: record the reason at every death site — clean exit path, each fault
|
||||||
|
class in `onException`, the kill path. Bounded recent-exits table (ids are never
|
||||||
|
reused, so a small ring keyed by id is enough).
|
||||||
|
- New system call `process_exit_reason(id)` — supervisor-gated, like kill; returns
|
||||||
|
the recorded reason or `-ESRCH` once evicted.
|
||||||
|
- `library/runtime/process.zig`: `ExitReason` + `exitReason(id: u32)`.
|
||||||
|
- Docs: remove the no-exit-status bullet from process-management.md.
|
||||||
|
|
||||||
|
**Test:** extend the `supervision` scenario — three children: one exits cleanly,
|
||||||
|
one faults (the fault-recovery pattern), one is killed; the supervisor asserts all
|
||||||
|
three reasons.
|
||||||
|
|
||||||
|
## M17.3 — published exit events
|
||||||
|
|
||||||
|
- Kernel: bounded subscriber table (endpoints); new system call
|
||||||
|
`process_subscribe(endpoint)` (ungated, like `process_enumerate`); every death
|
||||||
|
posts `notify_exit_bit | id` to each subscriber — the same post the supervisor
|
||||||
|
path already uses.
|
||||||
|
- `library/runtime/process.zig`: `subscribeExits(endpoint)`.
|
||||||
|
- VFS becomes the first subscriber: on an exit event, release every handle keyed
|
||||||
|
by that task id (badges already are task ids). Log the release.
|
||||||
|
- Docs: note the convention in ipc.md (exit events reuse the exit-notification
|
||||||
|
badge encoding).
|
||||||
|
|
||||||
|
**Test:** new QEMU scenario `vfs-client-death` — a client opens a file and is
|
||||||
|
killed without closing; assert the VFS logs the handle release and its open-handle
|
||||||
|
count returns to baseline.
|
||||||
|
|
||||||
|
## M17.4 — signals and the service harness
|
||||||
|
|
||||||
|
- Kernel: per-task pending mask + bound endpoint; system calls
|
||||||
|
`signal_bind(endpoint)` and `process_signal(id, signal)` (supervisor-or-self
|
||||||
|
gated); delivery posts `notify_signal_bit | pending mask`, coalescing; pending
|
||||||
|
signals with no bound endpoint pend silently.
|
||||||
|
- `library/runtime/process.zig`: `Signal`, `SignalSet`, `bindSignals`,
|
||||||
|
`signalsFrom`, `sendSignal`, `stop(id, deadline_ms)` (terminate → wait for exit
|
||||||
|
notification → kill). Implement `terminate`, `reload`, `user_1`, `user_2`;
|
||||||
|
`interrupt`/`quit` are enum members with no sender yet; `alarm` stays unbuilt.
|
||||||
|
- Kernel: **one-shot timer notifications** — `timer_bind(endpoint, ms)` posts a
|
||||||
|
notification badge when the deadline lands (IRQ-as-IPC again, on the timer
|
||||||
|
wheel `sleep` already uses). This is the missing timed-wait primitive:
|
||||||
|
`replyWait` blocks forever and `sleep` blocks the whole process, but `stop()`'s
|
||||||
|
escalation, the device manager's `hello` deadline (M18.1), and restart backoff
|
||||||
|
all need a deadline while staying responsive. It is also the mechanism `alarm`
|
||||||
|
gets for free later.
|
||||||
|
- New `library/runtime/service.zig`: the harness — `run(callbacks)` owning the
|
||||||
|
replyWait loop, folding protocol messages, signals, and child-exit notifications
|
||||||
|
into `init` / `on_message` / `on_reload` / `on_terminate`; answers the common
|
||||||
|
`ping` automatically. Define the reserved `ping` request encoding here and
|
||||||
|
document it in ipc.md (one obvious encoding; smallest that cannot collide with
|
||||||
|
existing protocols).
|
||||||
|
- Convert one existing service (input-source or hpet) to the harness as proof it
|
||||||
|
subtracts code rather than adding it.
|
||||||
|
|
||||||
|
**Test:** extend `supervision` — a harness-built child: `sendSignal(reload)`
|
||||||
|
observed in its log, `ping` answered, `stop()` produces a clean exit with reason
|
||||||
|
`exited`; a second child that ignores signals (no bind) is killed by `stop()`'s
|
||||||
|
deadline with reason `killed`.
|
||||||
|
|
||||||
|
## M18.1 — device-manager protocol: hello + restart policy
|
||||||
|
|
||||||
|
- New `system/services/device-manager/device-manager-protocol.zig` module
|
||||||
|
(vfs-protocol pattern): `hello { version, role, device_id }`; version constant;
|
||||||
|
reserved fields.
|
||||||
|
- Device manager: register the `.device_manager` endpoint; spawn drivers with its
|
||||||
|
exit endpoint; enforce the hello deadline; restart policy — backoff, crash-loop
|
||||||
|
cap (three fast deaths → mark failed, log, stop), reasons from M17.2 deciding
|
||||||
|
restart vs not.
|
||||||
|
- usb-xhci-bus: adopt the harness + send hello. hpet/ps2-bus follow only if the
|
||||||
|
conversion is mechanical; otherwise they keep working unconverted (the manager
|
||||||
|
only enforces hello on drivers spawned with an assignment).
|
||||||
|
- build.zig: test-loop entry for the protocol module if it grows pure logic.
|
||||||
|
|
||||||
|
**Test:** new QEMU scenario `driver-restart` — the xHCI driver takes a test-only
|
||||||
|
argv flag to fault after hello on its first run; assert: fault, exit reason
|
||||||
|
recorded, manager respawns with backoff, second run claims the controller
|
||||||
|
(M17.1) and hellos clean. Assert the crash-loop cap by a driver that always
|
||||||
|
faults (a tiny test driver, not xhci).
|
||||||
|
|
||||||
|
## M18.2 — bus tree reports
|
||||||
|
|
||||||
|
- Protocol: `child_added { parent, identity, resources }` / `child_removed { id }`.
|
||||||
|
- usb-xhci-bus: bring-up to **port scan only** — map the MMIO window (claimed in
|
||||||
|
M16-era work), controller reset/start per xHCI spec, walk the port registers,
|
||||||
|
report one `child_added` per connected port with speed + port number as
|
||||||
|
identity. **No transfer rings, no descriptors** — reading device/interface
|
||||||
|
descriptors (and therefore USB class triples for matching) is the follow-on USB
|
||||||
|
track, not this plan.
|
||||||
|
- Device manager: mirror reports into its tree; prune the subtree (emitting
|
||||||
|
`child_removed`) when a bus driver dies; assert re-report on restart.
|
||||||
|
|
||||||
|
**Test:** QEMU already attaches usb-kbd + usb-mouse on xhci.0 — assert two
|
||||||
|
`child_added` events reach the manager and appear in its tree dump; kill the
|
||||||
|
driver, assert two `child_removed` then two fresh `child_added` after respawn.
|
||||||
|
|
||||||
|
## M18.3 — the application surface
|
||||||
|
|
||||||
|
- Protocol: `enumerate` (tree snapshot) + `subscribe` (published add/remove
|
||||||
|
events, input-service pattern).
|
||||||
|
- A small client (`device-list`, the `ps` analog) exercising both; the manager
|
||||||
|
becomes the one answer to "what devices exist" for user space.
|
||||||
|
`device_enumerate` stays for drivers/kernel seeding — its retreat is tied to the
|
||||||
|
discovery migration, out of this plan.
|
||||||
|
|
||||||
|
**Test:** QEMU scenario — `device-list` shows the tree including USB children;
|
||||||
|
during a driver restart the subscribing client logs remove + add events.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
**Explicitly out of scope** (own tracks, after M18): discovery migration (pci-bus
|
||||||
|
driver, acpi service, retiring the kernel scan), USB control transfers +
|
||||||
|
descriptors + class-driver matching, the musl layer, `interrupt`/`quit` senders
|
||||||
|
(needs a console), job control.
|
||||||
@@ -0,0 +1,187 @@
|
|||||||
|
# M19–M20 execution plan: discovery migration
|
||||||
|
|
||||||
|
The operational plan for [device-manager.md](device-manager.md)'s increment 8:
|
||||||
|
discovery leaves the kernel — a **pci-bus driver** (M19) and an **acpi service**
|
||||||
|
(M20), with the kernel's device enumeration retired behind them. Same rules as
|
||||||
|
[m17-m18-plan.md](m17-m18-plan.md): one phase at a time, each green before the
|
||||||
|
next; this file is the build order and the checklist.
|
||||||
|
|
||||||
|
**Definition of green, every phase:** `zig build` clean, `zig build test` clean,
|
||||||
|
`python3 test/qemu_test.py` passes (existing scenarios plus the phase's new
|
||||||
|
one), and the relevant design doc updated. Commit per green phase (no co-author
|
||||||
|
trailers). The full suite is the regression net — the existing
|
||||||
|
`driver-restart` / `usb-report` / `device-list` / `input` scenarios must stay
|
||||||
|
green *through* the migration, which is the whole point: the system must not be
|
||||||
|
able to tell who enumerated it.
|
||||||
|
|
||||||
|
**Workflow:** dedicated worktree; branches off `main` — `feat/pci-bus`
|
||||||
|
(M19.0–19.3), `feat/acpi-service` (M20.1–20.3); auto-merge to main when a
|
||||||
|
branch is green; keep branches; push everything.
|
||||||
|
|
||||||
|
## Settled decisions (2026-07-13 — veto before the loop starts)
|
||||||
|
|
||||||
|
1. **What "retiring the kernel scan" means.** The kernel keeps, forever, the
|
||||||
|
parses it needs before user space exists: RSDP/XSDT location, MADT (SMP),
|
||||||
|
the HPET table (the tick), FADT + the AML `\_S5` evaluation (poweroff — the
|
||||||
|
power tests prove it), and MCFG (the host bridge node). What retires is
|
||||||
|
**device enumeration**: the ECAM function walk (M19.3) and the DSDT/SSDT
|
||||||
|
namespace walk that builds device nodes (M20.3). The AML module stays a
|
||||||
|
shared build module compiled into both the kernel (for `\_S5`) and the acpi
|
||||||
|
service (for everything else) — same source, two builds, no fork.
|
||||||
|
2. **Bridge apertures come from the firmware memory map, not AML.** Registered
|
||||||
|
PCI functions carry BAR resources, and containment demands the bridge own
|
||||||
|
windows that cover them. The apertures are derived kernel-side from the
|
||||||
|
boot memory map's MMIO holes (regions that are neither RAM nor tables) —
|
||||||
|
mechanical, AML-free, and available at boot regardless of what later moved
|
||||||
|
to user space. (The bridge today carries only ECAM + bus range; this is the
|
||||||
|
prerequisite M19.0 exists for.)
|
||||||
|
3. **`device_register` becomes idempotent on exact match.** A re-registration
|
||||||
|
with identical (parent, class, resources) returns the existing id instead
|
||||||
|
of appending. The kernel table has no unregister, so without this a
|
||||||
|
restarted registering bus would duplicate its children on every respawn —
|
||||||
|
idempotence makes restart-and-re-report safe for every future bus, not just
|
||||||
|
PCI.
|
||||||
|
4. **The manager matches from reports.** `ChildAdded` gains a `device_id`
|
||||||
|
field (the kernel-registered id, `no_device` for unregistered leaves like
|
||||||
|
USB ports). After the M19.3 flip, PCI driver matching keys off reported
|
||||||
|
identity (the class triple) instead of the manager's boot-time snapshot —
|
||||||
|
the snapshot match remains only for what the kernel still seeds. One flip
|
||||||
|
phase changes both sides at once so no device is ever matched twice.
|
||||||
|
5. **The acpi service's authority is one node.** The kernel publishes an
|
||||||
|
`acpi-tables` device: memory resources covering the table blobs plus a
|
||||||
|
broad `io_port` resource — the documented trust grant to exactly one
|
||||||
|
process (AML OperationRegions reach EC/PM ports; the claim-gated
|
||||||
|
io_read/io_write calls already exist). The service claims it, maps the
|
||||||
|
tables, and runs the shared AML module in ring 3 behind a `Hal` backed by
|
||||||
|
`mmio_map` + `io_read`/`io_write`.
|
||||||
|
6. **Both new processes are protocol drivers** under the manager: hello,
|
||||||
|
supervision, restart with backoff — all inherited from M18.1 for free.
|
||||||
|
Registration idempotence (decision 3) is what makes their restarts sound.
|
||||||
|
7. **Firmware neutrality is the contract** (2026-07-13). The generic layer is
|
||||||
|
everything at and above the device-manager protocol — descriptors,
|
||||||
|
containment, reports, matching, supervision — and none of it may become
|
||||||
|
x86-specific. Discovery is one swappable process per firmware: the acpi
|
||||||
|
service on x86; an **fdt service** on the Raspberry Pis (claims a
|
||||||
|
`devicetree-blob` node, reports children from the flattened device tree —
|
||||||
|
pure data, no bytecode, no port grant, strictly simpler than ACPI). The
|
||||||
|
manager owns the tree as *data* and touches no hardware, ever — AML runs in
|
||||||
|
a crashable, supervised discoverer precisely so a firmware-bytecode fault
|
||||||
|
can never take down the supervisor. Two consequences recorded now:
|
||||||
|
`DeviceDescriptor`'s 8-byte `hid` cannot hold an FDT `compatible` string
|
||||||
|
("brcm,bcm2835-aux-uart") — identity widens before the fdt service exists;
|
||||||
|
and cross-firmware surfaces are named by **domain, not firmware** (M21
|
||||||
|
defines a *power* protocol, not an "ACPI events" protocol — PSCI/mailbox
|
||||||
|
sources feed the same subscribers on ARM). **Landed early (2026-07-13):**
|
||||||
|
both services exist as placeholders (system/services/acpi, system/services/
|
||||||
|
fdt) and the build's `-Ddiscovery=acpi|fdt` option fills the ramdisk's
|
||||||
|
neutral `discovery` slot — the manager will spawn "discovery" by that name
|
||||||
|
in M20.3 and never learn which firmware it is on.
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
- [ ] **M19.0** — prerequisites on `feat/pci-bus`: bridge MMIO apertures from
|
||||||
|
the memory-map holes; `device_register` idempotence (+ kernel unit
|
||||||
|
checks); `ChildAdded.device_id`; archive note on m17-m18-plan.md.
|
||||||
|
- [ ] **M19.1** — pci-bus driver, scan only: claim the host bridge, map the
|
||||||
|
ECAM window, walk bus/device/function headers, log what it finds.
|
||||||
|
Scenario `pci-scan`: the kernel test compares the driver's reported count
|
||||||
|
against the broker table's `pci_device` count — equivalence, per class.
|
||||||
|
- [ ] **M19.2** — register + report: each function registered under the bridge
|
||||||
|
(config-space slice + BARs, `pci_class` in the descriptor), reported with
|
||||||
|
`child_added { device_id, identity = class triple }`. Manager mirrors;
|
||||||
|
spawn-from-reports stays **off**. Scenario extends `pci-scan`:
|
||||||
|
registered ids resolve, no duplicates after a forced driver restart
|
||||||
|
(idempotence proven end to end).
|
||||||
|
- [ ] **M19.3** — the flip: kernel `enumeratePci` call removed (bridge node
|
||||||
|
stays); manager matches PCI drivers from reports. One commit. The
|
||||||
|
existing xHCI scenarios (`driver-restart`, `usb-report`, `device-list`)
|
||||||
|
are the assertion — xhci must come up spawned off a pci-bus report, and
|
||||||
|
the suite must not be able to tell the difference. discovery.md updated.
|
||||||
|
- [ ] **merge** `feat/pci-bus` → main, push.
|
||||||
|
- [ ] **M20.1** — acpi service, parse only (fills the existing placeholder at
|
||||||
|
system/services/acpi/acpi.zig): kernel publishes `acpi-tables`
|
||||||
|
(decision 5) — memory over the table blobs, the broad io_port grant, and
|
||||||
|
**the SCI as an irq resource** (from the FADT; unused until M21 but free
|
||||||
|
to record now). The service claims it, maps the blobs, runs the shared
|
||||||
|
AML module in ring 3, logs the namespace device count and `_HID`s.
|
||||||
|
Scenario `acpi-parse`: user-space count equals the kernel walk's count.
|
||||||
|
- [ ] **M20.2** — register + report: namespace devices with `_HID` + `_CRS`
|
||||||
|
resources registered under `acpi-tables` (its io_port + the memory-map
|
||||||
|
holes give containment), reported to the manager. Spawn-from-reports for
|
||||||
|
ACPI matches stays off. Scenario: the reported set includes the PS/2
|
||||||
|
keyboard and mouse nodes with their IRQ resources.
|
||||||
|
- [ ] **M20.3** — the flip: kernel DSDT device-node building removed (static
|
||||||
|
tables + `\_S5` stay, decision 1); manager matches ACPI-hid drivers
|
||||||
|
(ps2-bus) from reports. The `input` and `device-manager` scenarios are
|
||||||
|
the assertion. discovery.md + acpi.md + device-manager.md updated;
|
||||||
|
device-manager.md increment 8 closed.
|
||||||
|
- [ ] **merge** `feat/acpi-service` → main, push — **loop ends here**.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Phase notes
|
||||||
|
|
||||||
|
**M19.0 apertures:** the boot memory map already crosses the handoff
|
||||||
|
([boot-handoff]); the holes computation belongs where the bridge node is built
|
||||||
|
(`parseMcfg`). Sanity-check on QEMU q35: the xHCI BAR (`0xc0000000`-region
|
||||||
|
values seen in the M18 logs) must land inside a derived aperture, asserted in
|
||||||
|
the kernel unit test.
|
||||||
|
|
||||||
|
**M19.1 scanning without owning config access twice:** the driver reads config
|
||||||
|
space through its ECAM mmio_map grant of the *bridge* window — the same bytes
|
||||||
|
the kernel walk read. Vendor-id `0xFFFF` skip, header-type multifunction rule,
|
||||||
|
no bridge recursion (matches the kernel's current single-segment walk).
|
||||||
|
|
||||||
|
**M19.2 BAR sizing:** the classic size probe (write all-ones, read mask,
|
||||||
|
restore) is deferred — the BARs' current programmed values and types are
|
||||||
|
enough for containment-checked registration at bring-up; sizing lands with the
|
||||||
|
first driver that needs to *move* a BAR. Log what is registered so the
|
||||||
|
scenario can assert it.
|
||||||
|
|
||||||
|
**M19.3 what the manager still seeds from the snapshot:** everything the
|
||||||
|
kernel still enumerates (timers, ACPI nodes until M20.3). The PCI arm of
|
||||||
|
`pciDriverFor` switches source; `driverFor` doesn't move until M20.3.
|
||||||
|
|
||||||
|
**M20.1 Hal in ring 3:** `mapMmio` → `device.mmioMap` over the claimed
|
||||||
|
acpi-tables node (plus a table-offset map for blobs); `pioRead`/`pioWrite` →
|
||||||
|
`device.ioRead`/`ioWrite` against its io_port resource. The interpreter cannot
|
||||||
|
tell it moved — that is the assertion of `acpi-parse`.
|
||||||
|
|
||||||
|
**M20.2 containment for `_CRS`:** io ports fall inside the node's broad
|
||||||
|
io_port resource; MMIO windows (HPET, LAPIC ranges some firmwares list) fall
|
||||||
|
inside the memory-map holes added to the node in M20.1. Anything that doesn't
|
||||||
|
fit is logged and skipped, loudly — bring-up honesty over silent drops.
|
||||||
|
|
||||||
|
**M20.3 ps2 ordering:** ps2-bus binds nodes the acpi service now reports, so
|
||||||
|
its spawn moves behind the report (the manager's matching handles this once
|
||||||
|
the source flips); the `input` scenario proves the keyboard still types.
|
||||||
|
|
||||||
|
**Explicitly out of scope:** PCI bridge recursion (single segment, flat bus
|
||||||
|
walk stays); BAR reprogramming/sizing; disk/PCIe hotplug; interrupt routing
|
||||||
|
changes (`_PRT` stays wherever it is today); the USB descriptor track;
|
||||||
|
multi-segment ECAM; per-device power states (D-states, `_PSx`/`_PRx`,
|
||||||
|
suspend/resume — a future *lifecycle-vocabulary* extension, since "suspend"
|
||||||
|
has the shape of a signal every driver must answer, and it has no consumer
|
||||||
|
until laptop sleep); CPU P/C-states.
|
||||||
|
|
||||||
|
## M21 preview — ACPI events + system power (planned next, not in this loop)
|
||||||
|
|
||||||
|
The acpi service grows the event side (settled direction 2026-07-13; detailed
|
||||||
|
phases when M20 lands):
|
||||||
|
|
||||||
|
- **21.1 SCI + fixed events**: irq_bind the SCI (the resource M20.1 already
|
||||||
|
records), read/clear PM1 status, publish the power-button event to
|
||||||
|
subscribers (the same pub/sub shape the manager uses).
|
||||||
|
- **21.2 GPE + Notify**: Notify dispatch in the shared AML interpreter, GPE
|
||||||
|
block handling, `Notify(device, code)` published per reported node. The
|
||||||
|
acpi service is a **bus** here: battery (PNP0C0A), AC (ACPI0003), and lid
|
||||||
|
(PNP0C0D) nodes are reported children; small class drivers bind them and
|
||||||
|
speak an evaluate/subscribe protocol to the service — the xHCI split,
|
||||||
|
repeated. The embedded controller (`_Qxx` queries) rides this phase;
|
||||||
|
QEMU emulates no battery/EC, so those paths are interface-complete and
|
||||||
|
validated on real hardware (the laptop is the win condition), while the
|
||||||
|
plumbing is proven by the power button.
|
||||||
|
- **21.3 the capstone**: QEMU `system_powerdown` → acpi service event → init
|
||||||
|
runs the M17 stop sequence over its children → kernel `\_S5` — orderly
|
||||||
|
shutdown as the scenario that proves lifecycle + events compose. (The
|
||||||
|
harness grows a QMP poke to inject the event.)
|
||||||
+3
-3
@@ -27,7 +27,7 @@ danos's own neutral format, and the kernel only ever sees that.**
|
|||||||
|
|
||||||
## The neutral format
|
## The neutral format
|
||||||
|
|
||||||
Defined in `src/root.zig`, the shared loader↔kernel contract:
|
Defined in `system/boot-handoff.zig`, the shared loader↔kernel contract:
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
pub const MemoryKind = enum(u32) {
|
pub const MemoryKind = enum(u32) {
|
||||||
@@ -68,7 +68,7 @@ pub const BootInfo = extern struct {
|
|||||||
|
|
||||||
## The loader side (UEFI)
|
## The loader side (UEFI)
|
||||||
|
|
||||||
Two functions in `src/boot/efi.zig`, called from `exitBootServices`:
|
Two functions in `boot/efi.zig`, called from `exitBootServices`:
|
||||||
|
|
||||||
- **`classify`** maps each UEFI descriptor to a `MemoryKind`:
|
- **`classify`** maps each UEFI descriptor to a `MemoryKind`:
|
||||||
`conventional_memory` **and** `boot_services_code`/`boot_services_data → usable`;
|
`conventional_memory` **and** `boot_services_code`/`boot_services_data → usable`;
|
||||||
@@ -121,7 +121,7 @@ The kernel receives a plain array and reads it with zero UEFI knowledge:
|
|||||||
|
|
||||||
```zig
|
```zig
|
||||||
const mm = boot_info.memory_map;
|
const mm = boot_info.memory_map;
|
||||||
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
const regions = @as([*]const system.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
||||||
for (regions) |r| {
|
for (regions) |r| {
|
||||||
if (r.kind == .usable) usable_pages += r.pages;
|
if (r.kind == .usable) usable_pages += r.pages;
|
||||||
}
|
}
|
||||||
|
|||||||
+49
-14
@@ -7,7 +7,7 @@ which live in memory we'd like to reclaim and don't control), switches CR3 onto
|
|||||||
them, and — crucially — maps with **real permissions**.
|
them, and — crucially — maps with **real permissions**.
|
||||||
|
|
||||||
It's x86_64-specific (the 4-level table format is an Intel/AMD thing), so it lives
|
It's x86_64-specific (the 4-level table format is an Intel/AMD thing), so it lives
|
||||||
behind the [arch](arch.md) boundary in `src/kernel/arch/x86_64/paging.zig`.
|
behind the [arch](arch.md) boundary in `system/kernel/architecture/x86_64/paging.zig`.
|
||||||
|
|
||||||
## The format
|
## The format
|
||||||
|
|
||||||
@@ -17,20 +17,54 @@ the final 4 KiB page. Each entry holds a physical address plus flag bits —
|
|||||||
present, writable, and (bit 63) **no-execute**. danos maps everything with 4 KiB
|
present, writable, and (bit 63) **no-execute**. danos maps everything with 4 KiB
|
||||||
pages: precise, and the extra table memory is negligible against available RAM.
|
pages: precise, and the extra table memory is negligible against available RAM.
|
||||||
|
|
||||||
|
## Higher half: the address-space layout
|
||||||
|
|
||||||
|
danos is a **higher-half kernel**. The kernel is linked to run at
|
||||||
|
`0xFFFF_FFFF_8000_0000` but loaded low (the linker script's `AT()` gives each
|
||||||
|
segment a physical load address at 1 MiB up; the bootloader maps the high link
|
||||||
|
address to the low load address in its bootstrap tables and jumps in). The entire
|
||||||
|
**low canonical half is reserved for user space**; the kernel lives in the top half
|
||||||
|
alongside a **physmap** — a straight window onto all of physical memory at
|
||||||
|
`physmap_base + phys`. Wherever the kernel needs to touch a physical address (a
|
||||||
|
page-table frame, an ACPI table, a device register), it adds that constant:
|
||||||
|
`system.physToVirt(phys)`. The layout constants live in `system/boot-handoff.zig`:
|
||||||
|
|
||||||
|
| region | virtual base | PML4 slot |
|
||||||
|
|--------|--------------|-----------|
|
||||||
|
| user image + stack | `0x0000_7000_0000_0000` | 224 (low half) |
|
||||||
|
| kernel heap | `0xFFFF_8000_0000_0000` | 256 |
|
||||||
|
| physmap (all RAM + MMIO windows) | `0xFFFF_8800_0000_0000` + phys | 272 |
|
||||||
|
| kernel image | `0xFFFF_FFFF_8000_0000` | 511 |
|
||||||
|
|
||||||
|
The bootloader builds temporary **bootstrap tables** (identity + a 4 GiB physmap +
|
||||||
|
the high kernel) so it can switch CR3 and jump to the high entry; the kernel then
|
||||||
|
builds its own precise tables below and abandons them. Because both use the same
|
||||||
|
`physmap_base`, any physmap pointer minted before the switch stays valid after it.
|
||||||
|
|
||||||
## What gets mapped, and with what permissions
|
## What gets mapped, and with what permissions
|
||||||
|
|
||||||
The address space is built in three passes (`init`):
|
The address space is built in four passes (`init`):
|
||||||
|
|
||||||
1. **All RAM, identity-mapped RW + NX.** Every non-MMIO region from the
|
1. **All RAM in the physmap, RW + NX.** Every non-MMIO region from the
|
||||||
[memory map](memory-map.md) is mapped virtual == physical, read-write and
|
[memory map](memory-map.md) is mapped at `physToVirt(phys)`, read-write and
|
||||||
*non-executable*. Identity mapping keeps everything already running valid across
|
*non-executable*. There is **no low/identity mapping** — the low half is user
|
||||||
the CR3 switch (the frame allocator addresses frames by physical address, page
|
space. (Frames the kernel touches while still building these tables are reached
|
||||||
tables are reached the same way, the stack stays put).
|
through the loader's bootstrap physmap, which covers the low 4 GiB; both the
|
||||||
2. **The framebuffer and the Local APIC**, the device memory we actually touch,
|
frame allocator and the table builder scan low-address-up, so those frames stay
|
||||||
also RW + NX. Everything else — unbacked address space, other MMIO — is simply
|
under that limit.)
|
||||||
left unmapped, so a stray access faults instead of silently succeeding.
|
2. **The framebuffer and the Local APIC**, the device memory the kernel touches
|
||||||
3. **The kernel's own segments, overlaid with their true ELF permissions.** This is
|
directly, as physmap windows (RW + NX). Other MMIO is mapped on demand by
|
||||||
|
`mapMmio`, also into the physmap; everything else is left unmapped, so a stray
|
||||||
|
access faults instead of silently succeeding.
|
||||||
|
3. **The kernel's own segments, overlaid with their true ELF permissions**, at
|
||||||
|
their high link addresses mapped to their low physical load addresses. This is
|
||||||
the interesting part.
|
the interesting part.
|
||||||
|
4. **Every higher-half PML4 entry pre-created** (an empty PDPT where none exists
|
||||||
|
yet). The kernel half is then a fixed set of top-level slots, so a per-process
|
||||||
|
address space can share it by copying `PML4[256..512)` once — growth beneath
|
||||||
|
those slots (heap, on-demand MMIO) propagates to every address space because
|
||||||
|
they share the PDPTs. `init` asserts no new higher-half PML4 entry appears
|
||||||
|
afterward.
|
||||||
|
|
||||||
### W^X from the ELF program headers
|
### W^X from the ELF program headers
|
||||||
|
|
||||||
@@ -55,9 +89,10 @@ reserved bit and fault.
|
|||||||
|
|
||||||
### The null guard
|
### The null guard
|
||||||
|
|
||||||
Page 0 is deliberately left unmapped. A null (or near-null) pointer dereference now
|
The whole low half is unmapped except for explicit user mappings, so page 0 (and
|
||||||
takes a page fault instead of quietly reading or writing real memory — turning a
|
every near-null address) is unmapped by construction. A null (or near-null) pointer
|
||||||
whole class of silent bugs into an immediate, located crash.
|
dereference in the kernel takes a page fault instead of quietly reading or writing
|
||||||
|
real memory — turning a whole class of silent bugs into an immediate, located crash.
|
||||||
|
|
||||||
## Switching on, and the on-demand API
|
## Switching on, and the on-demand API
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,327 @@
|
|||||||
|
# Process lifecycle: signals over IPC
|
||||||
|
|
||||||
|
**Status: increments 1–4 built** (2026-07-12): claim release on death, exit
|
||||||
|
reasons, published exit events, and signals + one-shot timers + the service
|
||||||
|
harness are all in — the interface below is as-built. The primitives underneath
|
||||||
|
predate this design ([process-management.md](process-management.md):
|
||||||
|
spawn, the supervision link, kill, child-exit notifications); this document designs
|
||||||
|
the layer above them — the standard vocabulary a danos process speaks about its own
|
||||||
|
life, and the stable `runtime.process` interface that carries it. Nothing here is
|
||||||
|
device- or driver-specific: a driver, the VFS, and a user application all stop,
|
||||||
|
reload, and die the same way. The device manager is simply this design's first
|
||||||
|
serious customer ([device-manager.md](device-manager.md)).
|
||||||
|
|
||||||
|
**"POSIX" in this document means the concepts, never the letter of the standard.**
|
||||||
|
danos borrows the ideas and the hard-won lessons (what SIGTERM *means*, why SIGPIPE
|
||||||
|
was a mistake) without inheriting the mechanism, the API, or the names. The naming
|
||||||
|
rule is danos's own and it is strict: plain words that communicate intent
|
||||||
|
(`terminate`, `reload`, `exited`) and the IPC vocabulary the system already speaks
|
||||||
|
(`bind`, `subscribe`, `publish`, `endpoint`) — never `SIG*`, never a second word for
|
||||||
|
a concept that already has one. Literal POSIX arrives later and lives elsewhere: a
|
||||||
|
**musl-based C layer** (growing out of library/posix) that wires C programs to the
|
||||||
|
danos runtime — musl's syscall surface retargeted at danos system calls and IPC
|
||||||
|
protocols (files onto the VFS protocol, `sigaction`/`wait` onto this lifecycle,
|
||||||
|
sockets onto whatever networking becomes). Ported programs see POSIX; the system
|
||||||
|
underneath never does.
|
||||||
|
|
||||||
|
## Why a standard vocabulary
|
||||||
|
|
||||||
|
A supervisor can only manage processes it has never heard of if "please exit" means
|
||||||
|
the same thing to all of them. That is the one thing POSIX signals got deeply right:
|
||||||
|
`SIGTERM` means the same thing to nginx and to a five-line script, which is why
|
||||||
|
process supervision on Unix (init systems, container runtimes) is possible at all.
|
||||||
|
danos wants that property from day one, because supervision-and-restart is the
|
||||||
|
system's core motivation ([resilience.md](resilience.md)).
|
||||||
|
|
||||||
|
What POSIX got wrong — for a system like this — is the **delivery mechanism**:
|
||||||
|
asynchronous control-flow hijack. A Unix handler runs on a stolen stack at an
|
||||||
|
arbitrary instruction boundary, which is why the async-signal-safe function list
|
||||||
|
exists, why `errno` must be saved, and why the canonical signal bug is a SIGTERM
|
||||||
|
handler innocently calling `printf` mid-`malloc`. That entire bug class comes from
|
||||||
|
the mechanism, not the vocabulary, and none of it is worth importing.
|
||||||
|
|
||||||
|
A microkernel already has the right channel: **a signal is a message.** QNX delivers
|
||||||
|
POSIX signals over its message passing; seL4 has notification objects; Erlang turned
|
||||||
|
"death is a message to whoever linked" into a reliability philosophy. danos has
|
||||||
|
already done it once without naming it: a child's death arrives as a notification
|
||||||
|
badge on the supervisor's endpoint — the microkernel's SIGCHLD, the IRQ-as-IPC
|
||||||
|
pattern reused. Signals are the same pattern reused a third time.
|
||||||
|
|
||||||
|
## The mechanism
|
||||||
|
|
||||||
|
- **`signal_bind(endpoint)`** — a process nominates the endpoint its signals arrive
|
||||||
|
on, exactly as `irq_bind` nominates where a device's interrupts land. The runtime
|
||||||
|
does this at startup for any program that opts in.
|
||||||
|
- **`process_signal(id, signal)`** — posts the signal as an asynchronous
|
||||||
|
notification to the target's bound endpoint: badge = `notify_badge_bit |
|
||||||
|
notify_signal_bit | pending signals`. Non-blocking for the sender, always.
|
||||||
|
- **Pending signals coalesce** in a per-process bitmask until the target next waits
|
||||||
|
— exactly like interrupt notifications, and exactly POSIX's own semantics for
|
||||||
|
non-realtime signals (two pending SIGTERMs are one SIGTERM). The bitmask *is* the
|
||||||
|
design: signals carry no payload. Anything with a payload is a protocol message.
|
||||||
|
- **Authority**: the supervisor may signal its children — the same link that is
|
||||||
|
already the kill authority. A process may signal itself. Anything broader waits
|
||||||
|
for transferable process handles.
|
||||||
|
- **No binding, no problem**: a process that never calls `signal_bind` is not
|
||||||
|
broken — its signals pend unread and only `process_kill` works on it. Simple
|
||||||
|
programs stay simple; the vocabulary is opt-in, the kill authority is not.
|
||||||
|
|
||||||
|
Because delivery is a message into the process's own event loop, there is no
|
||||||
|
async-signal-safe list in danos: a handler is ordinary code running at a point the
|
||||||
|
process chose. The bug class is gone by construction, not by discipline.
|
||||||
|
|
||||||
|
## The vocabulary: POSIX.1-1990, sorted honestly
|
||||||
|
|
||||||
|
The full 1990 set, and what each becomes. Two intrinsically problematic cases get a
|
||||||
|
defense below the table.
|
||||||
|
|
||||||
|
| POSIX.1-1990 | danos disposition | Notes |
|
||||||
|
|---|---|---|
|
||||||
|
| SIGTERM | signal `terminate` | finish up and exit; the supervisor's polite half |
|
||||||
|
| SIGHUP | signal `reload` | re-read configuration / re-scan |
|
||||||
|
| SIGINT | signal `interrupt` | interactive interrupt; meaningful once a console can send it, in the vocabulary now so numbering is stable |
|
||||||
|
| SIGQUIT | signal `quit` | as SIGINT, without the core-dump baggage |
|
||||||
|
| SIGALRM | signal `alarm` | timer expiry as a message; the Unix SIGALRM+`longjmp` timeout hacks are impossible here. In the vocabulary, unbuilt: no consumer yet, and when one appears it is runtime sugar over the existing timer — zero kernel work |
|
||||||
|
| SIGUSR1, SIGUSR2 | signals `user_1`, `user_2` | service-defined |
|
||||||
|
| SIGCHLD | **already exists** — the exit notification | the badge carries the child id, dodging the classic coalescing bug (Unix code must loop `waitpid`) |
|
||||||
|
| SIGKILL | `process_kill` — kernel mechanism | its definition is "cannot be handled"; it was never really a signal |
|
||||||
|
| SIGABRT | exit reason `abort` | `abort()` is synchronous self-termination, not an event |
|
||||||
|
| SIGSEGV, SIGILL, SIGFPE | exit reasons, **never delivered** | see below |
|
||||||
|
| SIGPIPE | **an error return**, not a signal | see below |
|
||||||
|
| SIGSTOP, SIGTSTP, SIGTTIN, SIGTTOU, SIGCONT | deferred | job control needs terminals, sessions, and process groups; stop/continue is scheduler territory |
|
||||||
|
|
||||||
|
**The fault signals (SIGSEGV, SIGILL, SIGFPE) are intrinsically wrong for messages.**
|
||||||
|
They are *synchronous* — raised at a specific faulting instruction, not "sometime
|
||||||
|
soon". A message cannot be delivered to a process whose next instruction re-faults;
|
||||||
|
it never reaches its event loop to read it. POSIX only makes fault handlers "work"
|
||||||
|
via the async hijack (run the handler *instead of* the instruction), and even there,
|
||||||
|
returning from a SIGSEGV handler without curing the cause is undefined behavior.
|
||||||
|
danos's architecture already has the better answer: fault → the kernel kills the
|
||||||
|
process ([resilience.md](resilience.md) step 2, built) → the supervisor reads the
|
||||||
|
reason → restart. Recovery is restart, not a handler. This is also truer to the 1990
|
||||||
|
standard than handling is: the standard's default action for all three was
|
||||||
|
"terminate the process".
|
||||||
|
|
||||||
|
**SIGPIPE deserves special contempt.** Its default kills a process that writes to a
|
||||||
|
closed pipe — which is why "the whole server died because one client disconnected"
|
||||||
|
is roughly every network daemon's first production bug, and why every mature codebase
|
||||||
|
contains the same fix: ignore SIGPIPE, handle the `EPIPE` error return. danos made
|
||||||
|
the right choice natively already — a reply owed to a dead peer fails with `-EPEER`.
|
||||||
|
Errors from operations are error returns from those operations. The posix layer can
|
||||||
|
synthesize SIGPIPE for ported code that expects it.
|
||||||
|
|
||||||
|
### Statements, not questions
|
||||||
|
|
||||||
|
A signal and a protocol message both travel over IPC — the difference is the
|
||||||
|
**contract**, not the transport. danos IPC has two primitives, both already in
|
||||||
|
daily use: the **asynchronous notification** (a badge — bits that coalesce into a
|
||||||
|
pending mask; the sender never blocks; no payload, *no reply path*; how IRQs and
|
||||||
|
exit events arrive) and the **synchronous call** (a rendezvous — payload both
|
||||||
|
ways, the caller waits for the reply; how VFS requests work). A signal is the
|
||||||
|
first kind: a *statement*. `terminate` wants no reply — the exit notification is
|
||||||
|
its acknowledgement.
|
||||||
|
|
||||||
|
A health probe is the second kind: a *question*, worthless without its answer —
|
||||||
|
and the answer's absence within a deadline is the very thing being measured.
|
||||||
|
Asked as a signal it has no reply channel (a coalescing bit can't carry an answer,
|
||||||
|
and the authority rule forbids a child signalling its supervisor back); asked as a
|
||||||
|
call, the timeout-is-the-diagnosis semantics come free. So there is no `health`
|
||||||
|
signal. Liveness is the common **`ping`**: a reserved request every harness-run
|
||||||
|
service answers automatically on its main endpoint — still free for the service
|
||||||
|
author, still one obvious way — and a supervisor's probe is a `ping` call with a
|
||||||
|
deadline.
|
||||||
|
|
||||||
|
## The two iron rules
|
||||||
|
|
||||||
|
1. **Cleanup is the kernel's job.** A process can die with no warning — fault,
|
||||||
|
kill, power. Correctness must never depend on a `terminate` handler running. On
|
||||||
|
any death the kernel releases the address space, IPC handles, IRQ bindings, and
|
||||||
|
owed replies (built), and must also release **device, I/O-port, and interrupt
|
||||||
|
claims and MSI vectors** (the known gap in
|
||||||
|
[process-management.md](process-management.md); increment 1). A signal handler is
|
||||||
|
for *graceful* work — flushing, deregistering, saving — never for *necessary*
|
||||||
|
work.
|
||||||
|
2. **Kill is not a signal, and exit reasons are load-bearing.** The standard stop
|
||||||
|
sequence is *terminate → deadline → `process_kill`*; the unhandleable kill stays
|
||||||
|
a kernel mechanism. And a supervisor deciding whether to restart must know *how*
|
||||||
|
the child died: clean exit (meant to — don't restart), fault (restart with
|
||||||
|
backoff), killed (the supervisor did it). The exit notification today carries
|
||||||
|
only the id; it grows a reason. Restart policy cannot be written without it.
|
||||||
|
|
||||||
|
## Who learns of a death
|
||||||
|
|
||||||
|
A death has three audiences, and conflating them is how systems end up with either
|
||||||
|
zombie state or privileged snooping:
|
||||||
|
|
||||||
|
1. **The supervisor** — gets the exit notification on the endpoint it gave at spawn
|
||||||
|
(built), which grows the `ExitReason` (increment 2). The supervisor is the only
|
||||||
|
audience that needs the *reason*, because it is the only one deciding whether to
|
||||||
|
restart.
|
||||||
|
2. **The peer owed a reply** — already built: a client that dies mid-request fails
|
||||||
|
the server's reply with `-EPEER`; a server that dies fails its waiting clients
|
||||||
|
the same way. This covers the *synchronous* case only.
|
||||||
|
3. **The subscribers** — the new piece, and it is the input service's
|
||||||
|
publish/subscribe shape ([input.md](input.md)) applied to exits. A stateful
|
||||||
|
service accumulates per-client state across many requests: the VFS holds a dead
|
||||||
|
client's open file handles, the input service holds its subscriptions, a future
|
||||||
|
network stack holds its sockets. None of these are the client's supervisor, and
|
||||||
|
none learn anything from a failed reply if the client simply never calls again.
|
||||||
|
So the kernel **publishes every exit** to whoever subscribed:
|
||||||
|
`process_subscribe(endpoint)` adds a subscriber, and each death posts a
|
||||||
|
notification to every subscriber (badge = `notify_exit_bit | process id` — the
|
||||||
|
same encoding supervisors already decode, the IRQ-as-IPC pattern once more). The
|
||||||
|
subscriber filters for ids it holds state for and releases what the dead client
|
||||||
|
held. Correlating is free of bookkeeping: an IPC sender's badge already *is* its
|
||||||
|
task id (`runtime.ipc.Received`), so the id a service has been keying client
|
||||||
|
state by all along is the id the exit event carries.
|
||||||
|
|
||||||
|
Subscription, not broadcast-to-everyone: only processes that asked receive
|
||||||
|
events, the kernel keeps a bounded subscriber table, and delivery is the same
|
||||||
|
non-blocking coalescing notification as everything else — a dying process never
|
||||||
|
waits on its mourners. Subscribing is ungated, like `process_enumerate`: what is
|
||||||
|
running (and dying) is not a secret between cooperating processes. Subscribers
|
||||||
|
do not receive the exit reason — the VFS does not care *why* the client died.
|
||||||
|
|
||||||
|
This is the service-side mirror of iron rule 1: **a service must never depend on
|
||||||
|
its clients cleaning up after themselves.** Handle release on client death is the
|
||||||
|
service's job, triggered by the published exit event — never by a courtesy
|
||||||
|
"closing now" message that a crashed client will never send.
|
||||||
|
|
||||||
|
## The stable interface: `runtime.process`
|
||||||
|
|
||||||
|
`runtime.process` already owns what a process receives at birth (`Init`, the
|
||||||
|
argv contract). It grows to own the other end of life.
|
||||||
|
|
||||||
|
**The runtime is the stable interface; the numbers are not.** danos applications do
|
||||||
|
not make system calls — they call the runtime library, and the system-call numbers,
|
||||||
|
notification bits, and signal bit positions beneath it are a **private kernel ↔
|
||||||
|
runtime contract** that may change at any time (settled 2026-07-12). This is why
|
||||||
|
the runtime exists. Today kernel and runtime ship from one tree in one image, so
|
||||||
|
"stability" is simply building them together. When driver binaries start shipping
|
||||||
|
as separately-versioned applications — the whole point of the restart design — the
|
||||||
|
binary's embedded runtime version becomes compatibility metadata (the same idea as
|
||||||
|
the protocol version in the device manager's `hello`), and the kernel refuses what
|
||||||
|
it cannot serve. Signals therefore need no reserved numbering scheme: the enum
|
||||||
|
below is vocabulary, not ABI.
|
||||||
|
|
||||||
|
```zig
|
||||||
|
/// The signal vocabulary. The value is the bit position in the pending mask — a
|
||||||
|
/// private kernel/runtime detail, free to change while they ship together.
|
||||||
|
pub const Signal = enum(u5) {
|
||||||
|
terminate = 0, // SIGTERM: finish up and exit
|
||||||
|
reload = 1, // SIGHUP: re-read configuration
|
||||||
|
interrupt = 2, // SIGINT
|
||||||
|
quit = 3, // SIGQUIT
|
||||||
|
alarm = 4, // SIGALRM
|
||||||
|
user_1 = 5, // SIGUSR1
|
||||||
|
user_2 = 6, // SIGUSR2
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A decoded pending mask: the coalesced set of signals a notification delivered.
|
||||||
|
pub const SignalSet = struct {
|
||||||
|
pending: u32,
|
||||||
|
pub fn has(set: SignalSet, signal: Signal) bool { ... }
|
||||||
|
pub fn iterate(set: SignalSet) Iterator { ... }
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Nominate `endpoint` as this process's signal endpoint (signal_bind). The
|
||||||
|
/// runtime's service harness calls this; a bare program may call it directly and
|
||||||
|
/// fold signals into its own replyWait loop.
|
||||||
|
pub fn bindSignals(endpoint: usize) bool { ... }
|
||||||
|
|
||||||
|
/// Decode a received badge into signals, or null if the badge is not a signal
|
||||||
|
/// notification (mirrors ipc.Received.isChildExit).
|
||||||
|
pub fn signalsFrom(badge: usize) ?SignalSet { ... }
|
||||||
|
|
||||||
|
/// Send `signal` to process `id`. Supervisor-gated, like kill; non-blocking.
|
||||||
|
pub fn sendSignal(id: u32, signal: Signal) bool { ... }
|
||||||
|
|
||||||
|
/// The standard stop sequence: terminate, wait up to `deadline_ms` for the exit
|
||||||
|
/// notification, then process_kill. The one call a supervisor needs.
|
||||||
|
pub fn stop(id: u32, deadline_ms: u64) void { ... }
|
||||||
|
|
||||||
|
/// Subscribe `endpoint` to published exit events (process_subscribe). Every
|
||||||
|
/// process death posts an asynchronous notification: badge = notify_exit_bit |
|
||||||
|
/// process id — the same encoding a supervisor's exit notification uses, decoded
|
||||||
|
/// by the same ipc.Received helpers. For stateful services: release what the dead
|
||||||
|
/// client held (file handles, subscriptions, sockets). Ungated, like
|
||||||
|
/// process_enumerate.
|
||||||
|
pub fn subscribeExits(endpoint: usize) bool { ... }
|
||||||
|
|
||||||
|
/// How a process ended — queried after the exit notification (the kernel records
|
||||||
|
/// it first, so the two never race). What restart policy reads. (Built in M17.2.)
|
||||||
|
pub const ExitReason = enum(u8) {
|
||||||
|
exited, // returned from main / clean exit
|
||||||
|
aborted, // abort() — deliberate self-termination (SIGABRT's ghost; reserved)
|
||||||
|
segmentation_fault, // SIGSEGV's ghost
|
||||||
|
illegal_instruction, // SIGILL's ghost
|
||||||
|
arithmetic_fault, // SIGFPE's ghost
|
||||||
|
protection_fault, // general protection fault
|
||||||
|
fault, // any other CPU exception
|
||||||
|
killed, // process_kill
|
||||||
|
};
|
||||||
|
```
|
||||||
|
|
||||||
|
Two deliberate absences. There is no `mask`/`block` API — a process that is not
|
||||||
|
ready for a signal simply has not waited on its endpoint yet; the pending mask *is*
|
||||||
|
the blocked set. And there is no per-signal handler registration at this layer —
|
||||||
|
dispatch is the process's own `switch` over `SignalSet`, or the service harness's
|
||||||
|
callbacks (`on_terminate`, `on_reload`) for programs that want defaults.
|
||||||
|
|
||||||
|
### The service harness
|
||||||
|
|
||||||
|
`runtime.service` owns the `replyWait` loop and folds every event source — signals,
|
||||||
|
child exits, protocol messages — into callbacks, with the vocabulary's defaults:
|
||||||
|
`terminate` returns from the loop (clean exit), the common `ping` is answered automatically,
|
||||||
|
`reload` is ignored unless overridden. One loop, no locking, nothing reentrant. A
|
||||||
|
service author writes domain logic; the lifecycle contract is satisfied by the
|
||||||
|
harness. A process that bypasses the harness and ignores its signals meets the
|
||||||
|
deadline-then-kill escalation — you cannot force a process to implement an
|
||||||
|
interface, but you can make compliance free and non-compliance fatal.
|
||||||
|
|
||||||
|
### The musl layer later
|
||||||
|
|
||||||
|
The POSIX C layer is a **musl port**: musl's arch/syscall layer retargeted so that
|
||||||
|
what musl believes are kernel syscalls become danos runtime calls and IPC — `open`
|
||||||
|
and `read` onto the VFS protocol, `kill`/`sigaction`/`waitpid` onto this document's
|
||||||
|
vocabulary, `exit` onto the runtime's exit path. `sigaction` handlers registered
|
||||||
|
through it are invoked by the runtime's loop when the signal message arrives —
|
||||||
|
synchronous underneath, async-looking to ported code, delivered at wait boundaries
|
||||||
|
the way most Unix programs already experience signals (at syscalls). No stack hijack
|
||||||
|
ever happens, `SA_RESTART` semantics come free because nothing was interrupted, and
|
||||||
|
SIGPIPE can be synthesized from `-EPEER` for the programs that expect it. C programs
|
||||||
|
get POSIX; danos-native programs never pay for it.
|
||||||
|
|
||||||
|
## Increments
|
||||||
|
|
||||||
|
1. **Kernel: release device/port/IRQ claims and MSI vectors on death** — the
|
||||||
|
cleanup half of iron rule 1, and the prerequisite for any restart story. Test:
|
||||||
|
kill a claiming driver, spawn it again, the claim succeeds.
|
||||||
|
2. **Exit reason in the death notification** (`ExitReason` above).
|
||||||
|
3. **Exit events**: `process_subscribe` in the kernel (bounded subscriber table,
|
||||||
|
publishes on every death), `runtime.process.subscribeExits`; the VFS becomes the
|
||||||
|
first subscriber — releasing a dead client's handles is its proof test.
|
||||||
|
4. **Signals**: `signal_bind` + `process_signal` + the pending mask in the kernel;
|
||||||
|
`runtime.process` grows the interface above; the service harness handles
|
||||||
|
`terminate` and answers the common `ping`; `stop()` for supervisors.
|
||||||
|
|
||||||
|
[device-manager.md](device-manager.md) builds directly on all four.
|
||||||
|
|
||||||
|
## Settled questions (2026-07-12)
|
||||||
|
|
||||||
|
- **Signal numbering is not ABI**: the runtime is the stable interface; the numbers
|
||||||
|
beneath it are a private kernel ↔ runtime contract (see "The stable interface").
|
||||||
|
- **Liveness is a `ping` call, not a signal**: signals are statements, questions
|
||||||
|
are synchronous calls (see "Statements, not questions"). A service wanting *deep*
|
||||||
|
health ("can I reach my hardware?") defines its own protocol message on top.
|
||||||
|
- **Process handles: deferred.** Pids + the supervisor gate cover everything
|
||||||
|
planned; transferable handles (Fuchsia-style, delegating signalling without
|
||||||
|
delegating kill) wait for the capability table to grow types beyond endpoints.
|
||||||
|
- **`alarm`: in the vocabulary, unbuilt.** No consumer yet; when one appears it is
|
||||||
|
runtime sugar over the existing timer (arm a timer that posts your own signal) —
|
||||||
|
zero kernel work, so deferring costs nothing.
|
||||||
|
- **Subscription granularity: all exits**, subscriber-side filtering — one
|
||||||
|
subscription per service, a bounded kernel table. Per-id subscriptions only if
|
||||||
|
event volume ever matters (hundreds of processes, not before).
|
||||||
|
- **Client identity across the exit boundary: no convention needed** — an IPC
|
||||||
|
sender's badge already is its task id (see "Who learns of a death").
|
||||||
@@ -0,0 +1,120 @@
|
|||||||
|
# Process Management
|
||||||
|
|
||||||
|
How danos lists, supervises, and kills processes — the microkernel answer to
|
||||||
|
`ps`, `kill`, and `SIGCHLD`/`wait`.
|
||||||
|
|
||||||
|
## Why system calls, not `/proc`
|
||||||
|
|
||||||
|
Unix systems sit on a spectrum. Classic BSD/macOS list processes through
|
||||||
|
syscalls (`sysctl(KERN_PROC)`) and kill through `kill(2)`; Linux renders the
|
||||||
|
process table as `/proc` for *reading* but still kills through a syscall; Plan 9
|
||||||
|
made the file tree the whole interface (`echo kill > /proc/n/ctl`). Microkernels
|
||||||
|
mostly abandon ambient PIDs: Minix and QNX route everything through a user-space
|
||||||
|
process-manager server, and Fuchsia/seL4 control processes only through handles.
|
||||||
|
|
||||||
|
danos rules out `/proc` **as the primitive**: here a `/proc` would be served by
|
||||||
|
the VFS server — a user process — which would put the VFS in the path of process
|
||||||
|
control. If the VFS (or anything under it) hangs, nothing could be listed or
|
||||||
|
killed, *including the hung VFS*. The control plane for processes must not
|
||||||
|
depend on a process. So the primitives are kernel system calls; a read-only
|
||||||
|
`/proc` rendering can be layered on later, and a POSIX-style process-manager
|
||||||
|
server can be built *from* these primitives when one is needed.
|
||||||
|
|
||||||
|
## The three primitives
|
||||||
|
|
||||||
|
### `process_enumerate(buffer, maximum) -> total`
|
||||||
|
|
||||||
|
A snapshot of the task table into a caller buffer of `abi.ProcessDescriptor`
|
||||||
|
(id, supervisor, state, priority, name) — the exact shape of
|
||||||
|
`device_enumerate`, so `ps` is a user program over a snapshot, not a kernel
|
||||||
|
service. The total may exceed what fit; call again with a larger buffer. Kernel
|
||||||
|
tasks are included with an empty name — an honest listing shows the idle tasks
|
||||||
|
too. Ungated and read-only: what is running is not a secret between cooperating
|
||||||
|
bring-up processes.
|
||||||
|
|
||||||
|
### `system_spawn(..., exit_endpoint) -> child id`, and the supervision link
|
||||||
|
|
||||||
|
`system_spawn` records the caller as the child's **supervisor** and returns the
|
||||||
|
child's process id (ids are monotonic, never reused — a stale id can only miss).
|
||||||
|
That link is the kill authority: it answers "who may kill process 7?" without
|
||||||
|
inventing users or permissions, the same way a device *claim* is the capability
|
||||||
|
for `mmio_map`. It composes with the supervision hierarchy the device manager
|
||||||
|
already forms: init supervises the services it starts, the device manager
|
||||||
|
supervises the drivers it matches. (A transferable process *handle* — Fuchsia
|
||||||
|
style — can replace the id once the handle table grows types beyond endpoints.)
|
||||||
|
|
||||||
|
`exit_endpoint` (a handle, or `abi.no_cap`) is the supervisor's death-watch: when
|
||||||
|
the child ends — clean exit, CPU fault, or `process_kill` — the kernel posts an
|
||||||
|
asynchronous notification to that endpoint, exactly like a bound IRQ. The badge
|
||||||
|
carries `abi.notify_badge_bit | abi.notify_exit_bit | child_id`, so one endpoint
|
||||||
|
supervises many children and can even share with IRQ notifications. This is the
|
||||||
|
microkernel's SIGCHLD: no new mechanism, just the IRQ-as-IPC pattern reused, and
|
||||||
|
a supervisor's event loop (`ipc.replyWait`) already knows how to receive it. The
|
||||||
|
child holds a reference to the endpoint from birth, so the notification cannot
|
||||||
|
dangle even if the supervisor dies first.
|
||||||
|
|
||||||
|
### `process_kill(id) -> 0 / -ESRCH / -EPERM`
|
||||||
|
|
||||||
|
Only the supervisor may kill; kernel tasks are not killable processes. Like a
|
||||||
|
signal, delivery is prompt but asynchronous — 0 means the kill is accepted and
|
||||||
|
irrevocable; the exit notification confirms completion.
|
||||||
|
|
||||||
|
## How a kill lands (the kernel mechanics)
|
||||||
|
|
||||||
|
Everything below runs under the big kernel lock, where task states cannot move.
|
||||||
|
|
||||||
|
- **Target ready or blocked** (not on any core): reaped on the killer's own
|
||||||
|
call. The reap releases what death always releases (IRQ bindings first, then
|
||||||
|
a client the target still owed a reply to is failed with `-EPEER`, IPC handles
|
||||||
|
closed, the exit notification posted last) — plus the unlinking only a
|
||||||
|
*remote* death needs: out of the ready queue, out of an endpoint's sender FIFO
|
||||||
|
(`Task.ipc_wait_endpoint`), out of a receive wait queue (`Task.wait_queue`),
|
||||||
|
and out of any server's owed-reply slot, so nothing ever dequeues a dangling
|
||||||
|
pointer. Destroying the address space is safe because no core can have it
|
||||||
|
loaded: every switch away from a task loads the next task's tables.
|
||||||
|
- **Target running on another core**: it cannot be torn down mid-instruction,
|
||||||
|
so it is condemned (`Task.kill_pending`) and dies at whichever comes first:
|
||||||
|
- its next **system_call entry** — checked before dispatch, so a condemned
|
||||||
|
process cannot spawn, claim, or message anything on its way out;
|
||||||
|
- its core's next **timer tick** — but only when the task is not inside one
|
||||||
|
of its own system calls (`Task.in_system_call`): the tick may have
|
||||||
|
interrupted kernel code mid-operation, where teardown would leak whatever
|
||||||
|
the operation held. User-mode execution is always a safe kill point. The
|
||||||
|
tick-time terminate abandons the interrupt frame exactly like the fault
|
||||||
|
path (the LAPIC is acknowledged before the tick hook runs);
|
||||||
|
- any core's tick finding it **blocked or ready** (it entered a syscall and
|
||||||
|
parked after being condemned) — reaped by the same remote-reap path.
|
||||||
|
|
||||||
|
A pure user-mode spin loop that never makes a system call therefore dies
|
||||||
|
within one tick; nothing a process does can outrun the kill.
|
||||||
|
|
||||||
|
The scheduler stays below the process layer: finishing a kill (IRQ bindings,
|
||||||
|
handles, the notification) is called *up* through two hooks process.zig
|
||||||
|
registers at boot (`terminate_current_hook`, `reap_task_hook`), mirroring how
|
||||||
|
the architecture layer calls up into `tick`.
|
||||||
|
|
||||||
|
## Known gaps (bring-up honesty)
|
||||||
|
|
||||||
|
- ~~Device claims are not released on death~~ Closed (M17.1): every path out of a
|
||||||
|
process releases its device claims alongside its IRQ and MSI bindings
|
||||||
|
(`releaseTaskResourcesLocked`), so a restarted driver can claim its hardware
|
||||||
|
again — the cleanup half of [process-lifecycle.md](process-lifecycle.md)'s iron
|
||||||
|
rule 1. The `claim-release` test proves the kill → release → re-claim cycle.
|
||||||
|
- Kernel stacks of dead tasks are leaked, as on every exit path (no reaper yet).
|
||||||
|
- ~~There is no exit status in the notification~~ Closed (M17.2): the kernel
|
||||||
|
records how every process ends — exited, a fault class, or killed — before it
|
||||||
|
posts the exit notification, and the supervisor reads it with
|
||||||
|
`process_exit_reason` (`runtime.process.exitReason`). This is the input to
|
||||||
|
restart policy ([process-lifecycle.md](process-lifecycle.md)); an exit *code*
|
||||||
|
for the clean case can still ride alongside later.
|
||||||
|
- Enumerate writes through the caller's raw pointer under the bring-up trust
|
||||||
|
model, like `device_enumerate` (an unmapped page is a self-DoS, not an
|
||||||
|
isolation break).
|
||||||
|
|
||||||
|
## Tests
|
||||||
|
|
||||||
|
`process-list` (enumerate), `process-kill` (kernel-level kill paths, refusals,
|
||||||
|
notifications), `supervision` (the whole user-side surface via the process-test
|
||||||
|
service: spawn supervised → enumerate → kill blocked and spinning children →
|
||||||
|
notifications → gone), `claim-release` (a killed claim-holder's device is
|
||||||
|
claimable again). See test/qemu_test.py.
|
||||||
+9
-3
@@ -1,6 +1,10 @@
|
|||||||
# Resilience: fault isolation and live restart
|
# Resilience: fault isolation and live restart
|
||||||
|
|
||||||
A design/research note, not built yet. This is the property danos is really chasing:
|
Steps 1–2 of the ordering below are **built**: user-mode isolation, and fault →
|
||||||
|
kill the process → keep the core (`onException` in `system/kernel/kernel.zig`; the
|
||||||
|
`fault-recovery` test proves a crashing ring-3 process dies alone while the system
|
||||||
|
keeps running). The supervisor notification and restart policy (steps 3+) are
|
||||||
|
still design. This is the property danos is really chasing:
|
||||||
**if a part of the OS breaks, isolate it, and re-initialise it — without rebooting.**
|
**if a part of the OS breaks, isolate it, and re-initialise it — without rebooting.**
|
||||||
A crashed driver gets restarted; a wedged service gets killed and brought back. It's
|
A crashed driver gets restarted; a wedged service gets killed and brought back. It's
|
||||||
the reason the [microkernel](vision.md) shape was chosen, and it's a *separate* goal
|
the reason the [microkernel](vision.md) shape was chosen, and it's a *separate* goal
|
||||||
@@ -111,9 +115,11 @@ Honest boundaries:
|
|||||||
## Suggested ordering
|
## Suggested ordering
|
||||||
|
|
||||||
1. **User mode + address-space isolation** — the shared prerequisite (also on the
|
1. **User mode + address-space isolation** — the shared prerequisite (also on the
|
||||||
path for everything else).
|
path for everything else). **Done.**
|
||||||
2. **Kernel: fault → kill process → notify.** Turn today's "halt on fault" into
|
2. **Kernel: fault → kill process → notify.** Turn today's "halt on fault" into
|
||||||
"confine to the process and report it."
|
"confine to the process and report it." **Done** (the kill and reclaim; the
|
||||||
|
supervisor notification waits for step 3's supervisor). A killed server's
|
||||||
|
pending client is unblocked with `-EPEER` rather than hung.
|
||||||
3. **A minimal supervisor server** that can (re)start a process.
|
3. **A minimal supervisor server** that can (re)start a process.
|
||||||
4. **Resource cleanup on death** — reclaim memory/MMIO/IPC/IRQ, via caps or a grant
|
4. **Resource cleanup on death** — reclaim memory/MMIO/IPC/IRQ, via caps or a grant
|
||||||
table.
|
table.
|
||||||
|
|||||||
+23
-2
@@ -6,8 +6,8 @@ ready task always runs, and tasks at the same priority take turns. That model is
|
|||||||
chosen for [real-time](vision.md) — it's predictable (you can reason about which
|
chosen for [real-time](vision.md) — it's predictable (you can reason about which
|
||||||
task runs when) and its decisions are O(1), unlike a fair-share scheduler.
|
task runs when) and its decisions are O(1), unlike a fair-share scheduler.
|
||||||
|
|
||||||
The scheduler proper (`src/kernel/sched.zig`) is generic; the context switch and new-task
|
The scheduler proper (`system/kernel/sched.zig`) is generic; the context switch and new-task
|
||||||
stack setup are architecture-specific (`src/kernel/arch/x86_64/`, see [arch](arch.md)).
|
stack setup are architecture-specific (`system/kernel/architecture/x86_64/`, see [arch](arch.md)).
|
||||||
|
|
||||||
## Tasks
|
## Tasks
|
||||||
|
|
||||||
@@ -67,6 +67,27 @@ exist, which is what a real-time scheduler needs.
|
|||||||
- **Round-robin within a level.** When a task is descheduled it goes to the *back*
|
- **Round-robin within a level.** When a task is descheduled it goes to the *back*
|
||||||
of its level's queue, so equal-priority tasks share the CPU fairly.
|
of its level's queue, so equal-priority tasks share the CPU fairly.
|
||||||
|
|
||||||
|
## Affinity: pinning a task to a core
|
||||||
|
|
||||||
|
By default a task runs on **any** core — the ready queue above is global, and any
|
||||||
|
idle core pulls the highest-priority task from it (work-conserving; see
|
||||||
|
[smp.md](smp.md)). A task can instead be **pinned** to one core with
|
||||||
|
`spawnOn(entry, priority, cpu)`, giving it an *affinity*: it will only ever run
|
||||||
|
there, never migrating.
|
||||||
|
|
||||||
|
Mechanically, each core has its **own** pinned queue (same 8-level FIFO + bitmap)
|
||||||
|
alongside the global one. A pinned task is enqueued only into its core's pinned
|
||||||
|
queue; selection compares the top of the global queue and the running core's pinned
|
||||||
|
queue and takes the higher priority (still O(1) — two bit-scans and a compare), with
|
||||||
|
a pinned task winning an equal-priority tie so it can't be starved by global work.
|
||||||
|
Because every queue is mutated under the [big kernel lock](smp.md), one core enqueuing
|
||||||
|
into another core's pinned queue is safe.
|
||||||
|
|
||||||
|
This is the *explicit-affinity* model (no surprise migration mid-deadline), which is
|
||||||
|
the more real-time-predictable direction. `spawnOn` refuses to pin to an offline or
|
||||||
|
out-of-range core — it creates the task unpinned instead, so it still runs somewhere
|
||||||
|
rather than stranding in a queue no core services, and returns whether the pin took.
|
||||||
|
|
||||||
## Sleeping and the idle task
|
## Sleeping and the idle task
|
||||||
|
|
||||||
A task can **block** — give up the CPU until an event, rather than busy-wait
|
A task can **block** — give up the CPU until an event, rather than busy-wait
|
||||||
|
|||||||
+114
-7
@@ -16,10 +16,14 @@ danos specifics):
|
|||||||
- **Real-time** — whether timing is *predictable*. Comes from bounded operations
|
- **Real-time** — whether timing is *predictable*. Comes from bounded operations
|
||||||
(our O(1) scheduler), not from core count.
|
(our O(1) scheduler), not from core count.
|
||||||
|
|
||||||
danos is uniprocessor today: one global `current` task, one set of ready queues, one
|
danos now runs on multiple cores. The firmware starts only the **bootstrap processor
|
||||||
timer. Even on an 8-core CPU, the firmware starts only the **bootstrap processor
|
(BSP)**; the kernel wakes the other cores (**application processors**, APs) with
|
||||||
(BSP)**; the other cores (**application processors**, APs) sit parked until the
|
INIT–SIPI–SIPI, brings each up into 64-bit long mode with its own descriptor tables,
|
||||||
kernel wakes them, which it doesn't yet.
|
LAPIC and timer, and drops it into the scheduler. Tasks run **genuinely in parallel** —
|
||||||
|
the `smp` self-test confirms worker tasks executing on all four cores at once under
|
||||||
|
QEMU `-smp 4`. Shared kernel state (scheduler queues, IPC) is serialised behind a big
|
||||||
|
kernel lock. What's left is refinement, not first-light: per-core run queues, IPIs,
|
||||||
|
and thread-to-core affinity (see [Implementation status](#implementation-status)).
|
||||||
|
|
||||||
## The common microkernel instinct: don't share kernel state
|
## The common microkernel instinct: don't share kernel state
|
||||||
|
|
||||||
@@ -114,14 +118,20 @@ active reconsideration in favour of resilience — see [vision.md](vision.md).)
|
|||||||
Whatever the top goal, the *sequence* is the same and seL4 validates starting simple:
|
Whatever the top goal, the *sequence* is the same and seL4 validates starting simple:
|
||||||
|
|
||||||
1. **Enumerate cores** — needs [device discovery](discovery.md) (ACPI MADT on x86,
|
1. **Enumerate cores** — needs [device discovery](discovery.md) (ACPI MADT on x86,
|
||||||
device tree on ARM). SMP is a concrete consumer of that work.
|
device tree on ARM). SMP is a concrete consumer of that work. **Done on x86:** the
|
||||||
|
MADT parse records every usable Local APIC — with the `apic_id` an AP wake targets —
|
||||||
|
and `platform.cpus()` returns the list (see [discovery.md](discovery.md)). The boot
|
||||||
|
log reports the count; the ARM (device-tree) path still needs it.
|
||||||
2. **Wake the APs** — INIT–SIPI–SIPI on x86; PSCI/spin-tables on ARM. Each core brings
|
2. **Wake the APs** — INIT–SIPI–SIPI on x86; PSCI/spin-tables on ARM. Each core brings
|
||||||
up its own tables, timer, and idle task.
|
up its own tables, timer, and idle task. **Done on x86** — cores climb to long mode,
|
||||||
|
set up their own GDT/TSS, and enter the scheduler; tasks run in parallel across all
|
||||||
|
cores ([status](#implementation-status)).
|
||||||
3. **Start with a big kernel lock.** It's a legitimate first design, not a shortcut —
|
3. **Start with a big kernel lock.** It's a legitimate first design, not a shortcut —
|
||||||
philosophically aligned with a tiny kernel, and it lets the single-core correctness
|
philosophically aligned with a tiny kernel, and it lets the single-core correctness
|
||||||
model you already have (the interrupt-flag discipline in
|
model you already have (the interrupt-flag discipline in
|
||||||
[scheduling.md](scheduling.md)) stay largely intact: one lock around kernel entry
|
[scheduling.md](scheduling.md)) stay largely intact: one lock around kernel entry
|
||||||
instead of rethinking every critical section.
|
instead of rethinking every critical section. **Done** — see
|
||||||
|
`system/kernel/sync.zig`.
|
||||||
4. **Later, if contention bites,** evolve toward **per-core run queues + explicit
|
4. **Later, if contention bites,** evolve toward **per-core run queues + explicit
|
||||||
affinity** (the Fiasco.OC direction) — also the more real-time-predictable model.
|
affinity** (the Fiasco.OC direction) — also the more real-time-predictable model.
|
||||||
5. **Placement stays a user-space policy** — the kernel runs a thread on the core it's
|
5. **Placement stays a user-space policy** — the kernel runs a thread on the core it's
|
||||||
@@ -130,6 +140,103 @@ Whatever the top goal, the *sequence* is the same and seL4 validates starting si
|
|||||||
Big-lock-first → per-core-later. The affinity/MCS depth is only worth it if real-time
|
Big-lock-first → per-core-later. The affinity/MCS depth is only worth it if real-time
|
||||||
turns out to be the actual goal.
|
turns out to be the actual goal.
|
||||||
|
|
||||||
|
## Implementation status
|
||||||
|
|
||||||
|
The "wake + schedule" build (real parallel task execution) is going in as a sequence
|
||||||
|
of green checkpoints — each step keeps the single-core test suite passing before the
|
||||||
|
next lands.
|
||||||
|
|
||||||
|
**Done:**
|
||||||
|
|
||||||
|
- **Core enumeration** — the MADT parse records every usable Local APIC (with its
|
||||||
|
`apic_id`, which an AP wake targets); `platform.cpus()` returns the list. See
|
||||||
|
[discovery.md](discovery.md).
|
||||||
|
- **The big kernel lock** (`system/kernel/sync.zig`) — one coarse spinlock guarding the
|
||||||
|
scheduler queues and IPC, always held with local interrupts disabled. It is held
|
||||||
|
*across* a context switch and released by whichever task resumes (the hand-off
|
||||||
|
rule); `task_trampoline` releases it for a freshly-spawned task. `scheduler.zig` and
|
||||||
|
`ipc.zig` run every critical section under it. Uncontended on one core, so behaviour
|
||||||
|
is identical to the old interrupt-flag model.
|
||||||
|
- **Per-CPU state** — a `PerCpu` struct (running task, idle task, APIC id) per core,
|
||||||
|
its pointer kept in the x86 **GS base** (`IA32_GS_BASE`; no `swapgs`, since there's
|
||||||
|
no user mode yet). The old global `current` is now `thisCpu().current`. The ready
|
||||||
|
queues stay **global** under the lock — work-conserving, so any idle core will pull
|
||||||
|
the highest-priority ready task; per-core queues are a later optimisation.
|
||||||
|
- **AP wake to long mode** — `arch.startSecondary` drives INIT–SIPI–SIPI (via the
|
||||||
|
LAPIC ICR) to wake each parked core one at a time. A woken core starts in 16-bit
|
||||||
|
real mode at a low page and runs the [trampoline](../system/kernel/architecture/x86_64/trampoline.s)
|
||||||
|
up through protected mode into 64-bit long mode, then lands in `smp.zig:apEntry`,
|
||||||
|
publishes its per-CPU pointer, and reports in. Verified in QEMU with `-smp 4`:
|
||||||
|
all four cores report `online`.
|
||||||
|
|
||||||
|
The trampoline earns its complexity from four hardware facts:
|
||||||
|
- a STARTUP IPI vectors a core to physical `vector << 12` (a *byte* vector), so the
|
||||||
|
trampoline must live **below 1 MiB** — the kernel reserves that page from the frame
|
||||||
|
allocator at boot, before paging/heap draw down the scarce low frames;
|
||||||
|
- the blanket RAM identity map is **NX** (W^X), but the AP fetches the trampoline
|
||||||
|
from it under paging, so that one page is made executable for bring-up;
|
||||||
|
- the blob is copied to a page whose address isn't known at link time, so it is
|
||||||
|
**position-independent**: it derives its own base from `CS` and, crucially,
|
||||||
|
addresses data *segment-relative in real mode* (where the segment base already
|
||||||
|
supplies the page base) but *base-register-relative in protected/long mode* (flat
|
||||||
|
segments, base 0). Getting that distinction wrong was the first bug found;
|
||||||
|
- an AP starts with a bare `CR0`/`CR4`, but the kernel is built **with SSE** (the
|
||||||
|
x86_64 baseline) and the compiler emits SSE for things as ordinary as a struct
|
||||||
|
copy — so the trampoline must set `CR4.OSFXSR`/`OSXMMEXCPT` and fix `CR0.EM`/`MP`,
|
||||||
|
or the first SSE instruction on the AP `#UD`s. The BSP inherited those bits from
|
||||||
|
UEFI; the AP has to set them itself. This was the second bug — it masqueraded as a
|
||||||
|
fault in `lgdt` (the first kernel code after entry that the compiler vectorised).
|
||||||
|
|
||||||
|
- **Per-core tables + scheduler entry** — each AP loads **its own GDT** (with its own
|
||||||
|
TSS descriptor) and **its own TSS** (its own IST/`rsp0` stack), loads the shared
|
||||||
|
IDT, enables its LAPIC and timer, then calls the generic `secondaryMain`: it turns
|
||||||
|
its bring-up context into the core's idle task (as task 0 is for the BSP), marks the
|
||||||
|
core online, and enters the run loop. With interrupts on, each core's own timer tick
|
||||||
|
preempts its idle context into whatever the global ready queue offers — so all cores
|
||||||
|
pull real work in parallel. The `smp` test spawns CPU-bound workers and confirms they
|
||||||
|
execute on all four cores at once, and `fault-ap-df` pins a #DF to an AP and checks
|
||||||
|
that core catches it on **its own** IST (a broken per-core TSS would triple-fault) —
|
||||||
|
reported as "core N: …", so a fault is always attributed to the core it happened on,
|
||||||
|
and is contained to that core (the rest of the system keeps running).
|
||||||
|
- **Thread affinity** — `spawnOn(entry, priority, cpu)` pins a task to a core (its own
|
||||||
|
per-core pinned queue, merged with the global queue at selection; see
|
||||||
|
[scheduling.md](scheduling.md#affinity-pinning-a-task-to-a-core)). The `affinity`
|
||||||
|
test confirms a pinned task never migrates. This is the mechanism the fault-on-AP
|
||||||
|
test rides on, and the *explicit-affinity* real-time-predictable model.
|
||||||
|
- **Right-sized footprint** — the per-CPU ceiling (`system.max_cpus`, one constant
|
||||||
|
shared by discovery, the scheduler, and the per-core GDT/TSS) is generous (128), but
|
||||||
|
the *large* per-core resources — the kernel and IST (double-fault) stacks — are
|
||||||
|
**heap-allocated at bring-up**, only for cores that actually come online. Only the
|
||||||
|
BSP's IST stack is static, because it must exist before the frame allocator does.
|
||||||
|
This kept the kernel image small (a static `[128][16 KiB]` IST array would have been
|
||||||
|
2 MiB of `.bss`); it's a few tens of KiB instead.
|
||||||
|
- **`single_threaded` off** — the kernel was built `single_threaded = true`, which
|
||||||
|
compiles `std.atomic` down to plain non-atomic ops. Harmless on one core, but it
|
||||||
|
quietly breaks the big kernel lock across cores; it's now `false`.
|
||||||
|
- **Re-armable wake + retry** — the trampoline frame is reserved for the system's
|
||||||
|
life, but kept **inert between wakes**: zeroed and non-executable, armed (blob
|
||||||
|
copied in, page made executable) only for the moment a core is actually climbing,
|
||||||
|
then disarmed again. So there's never a dormant executable page, and a core can be
|
||||||
|
(re)woken at any time — `arch.startSecondary` is one self-contained attempt (arm →
|
||||||
|
INIT–SIPI–SIPI → disarm), and its `INIT` resets a wedged core, so retrying just
|
||||||
|
works. Boot retries a non-responding core up to three times; the same primitive is
|
||||||
|
the groundwork a future **power manager** would drive to bring cores up (and,
|
||||||
|
eventually, its counterpart to take them offline — which additionally needs the
|
||||||
|
core's tasks migrated off first).
|
||||||
|
|
||||||
|
**Next (refinement, not first-light):**
|
||||||
|
|
||||||
|
- **IPIs** — cross-core wake/preempt. Not needed for correctness: an idle core wakes
|
||||||
|
on its own timer tick and pulls ready work then; IPIs only cut that latency from
|
||||||
|
≤1 ms to near-instant.
|
||||||
|
- **Per-core run queues** — the Fiasco.OC direction, if the single global queue's lock
|
||||||
|
contention ever bites. (Thread *affinity* already exists — see above; this is the
|
||||||
|
further step of giving each core its own primary run queue for load distribution.)
|
||||||
|
- **Fault recovery** — today a fault halts (only) the faulting core. Turning that into
|
||||||
|
"kill the task, keep the core running" is the [resilience](resilience.md) track (it
|
||||||
|
needs the task's lock/resource state handled), and for taking a core fully offline,
|
||||||
|
its tasks migrated first.
|
||||||
|
|
||||||
## Further reading
|
## Further reading
|
||||||
|
|
||||||
**Microkernel SMP & scheduling**
|
**Microkernel SMP & scheduling**
|
||||||
|
|||||||
@@ -1,6 +1,16 @@
|
|||||||
# System Calls
|
# System Calls
|
||||||
System calls (syscalls) are the bridge between your programs and the operating system's restricted core (kernel).
|
System calls (syscalls) are the bridge between your programs and the operating system's restricted core (kernel).
|
||||||
|
|
||||||
|
> **Status:** danos has real user processes (M3). User programs enter the kernel
|
||||||
|
> via the `syscall` instruction (STAR/LSTAR/SFMASK set per core; the entry stub in
|
||||||
|
> `isr.s` does the `swapgs` + kernel-stack switch and reuses the interrupt
|
||||||
|
> dispatcher). The `int 0x80` gate is kept alongside as a minimal test path. The
|
||||||
|
> current call set is still a placeholder — `0 = exit(code)`, `1 = ping`,
|
||||||
|
> `2 = write(ptr, len)`, `3 = sleep(ms)` (see `system/kernel/process.zig`); the
|
||||||
|
> handler dispatches on whether the caller is a scheduled process (its own address
|
||||||
|
> space) or a borrowed test thread. The microkernel set below (IPC_Call /
|
||||||
|
> IPC_ReplyWait / Yield) replaces it once a second user server exists.
|
||||||
|
|
||||||
## The Mechanism of a Syscall
|
## The Mechanism of a Syscall
|
||||||
|
|
||||||
A system call follows a highly orchestrated, 6-step sequence to ensure hardware safety and process security:
|
A system call follows a highly orchestrated, 6-step sequence to ensure hardware safety and process security:
|
||||||
@@ -37,6 +47,8 @@ Everything else---including`read()`,`write()`,`malloc()`, and`fork()`---will run
|
|||||||
- **What it does:**Used strictly by your background user-space servers (like your disk driver or filesystem). It sends a reply to the last client that called it, and immediately puts the server to sleep until the next request arrives.[[1](https://news.ycombinator.com/item?id=33078441)]
|
- **What it does:**Used strictly by your background user-space servers (like your disk driver or filesystem). It sends a reply to the last client that called it, and immediately puts the server to sleep until the next request arrives.[[1](https://news.ycombinator.com/item?id=33078441)]
|
||||||
3. **`Yield()`/`Thread_Ctrl()`**
|
3. **`Yield()`/`Thread_Ctrl()`**
|
||||||
- **What it does:**Allows a thread to voluntarily give up its CPU time slice, or allows a root task to spawn/kill threads.
|
- **What it does:**Allows a thread to voluntarily give up its CPU time slice, or allows a root task to spawn/kill threads.
|
||||||
|
4. **`ipc_send(endpoint, message_buffer)`(Asynchronous Send)**
|
||||||
|
- **What it does:**Posts a small payload to an endpoint's bounded queue and returns *without* blocking — no rendezvous, no reply. The receiver picks it up through the same `IPC_ReplyWait`, as a buffered message. It is the async counterpart of `IPC_Call`, for one-to-many broadcasts where a synchronous rendezvous would let one dead or slow receiver hang the sender. The [input service](input.md) — keyboard-event fan-out — is its first user. A full queue drops the oldest message (a buffered message is discrete data, unlike a coalescing interrupt notification).
|
||||||
|
|
||||||
* * * * *
|
* * * * *
|
||||||
|
|
||||||
|
|||||||
+36
-4
@@ -1,6 +1,6 @@
|
|||||||
# SysV: the kernel's calling convention
|
# SysV: the kernel's calling convention
|
||||||
|
|
||||||
Several places in danos say "the kernel is SysV" — most visibly `src/root.zig`:
|
Several places in danos say "the kernel is SysV" — most visibly `system/boot-handoff.zig`:
|
||||||
|
|
||||||
```zig
|
```zig
|
||||||
pub const kernel_abi: std.builtin.CallingConvention = .{ .x86_64_sysv = .{} };
|
pub const kernel_abi: std.builtin.CallingConvention = .{ .x86_64_sysv = .{} };
|
||||||
@@ -52,19 +52,51 @@ argument arrives in **RCX**, not RDI.
|
|||||||
|
|
||||||
danos's two binaries default to different conventions:
|
danos's two binaries default to different conventions:
|
||||||
|
|
||||||
- `src/boot/efi.zig` is built for the UEFI target, so its default C convention is
|
- `boot/efi.zig` is built for the UEFI target, so its default C convention is
|
||||||
Microsoft x64 (first argument → RCX).
|
Microsoft x64 (first argument → RCX).
|
||||||
- The kernel is freestanding, so its convention is SysV (first argument → RDI).
|
- The kernel is freestanding, so its convention is SysV (first argument → RDI).
|
||||||
|
|
||||||
When the loader jumps to the kernel passing the `BootInfo` pointer, both sides have
|
When the loader jumps to the kernel passing the `BootInfo` pointer, both sides have
|
||||||
to agree *which register that pointer lands in*. Left to their defaults, the loader
|
to agree *which register that pointer lands in*. Left to their defaults, the loader
|
||||||
would place it in RCX while the kernel looked in RDI — and the kernel would read
|
would place it in RCX while the kernel looked in RDI — and the kernel would read
|
||||||
garbage. So both sides reference the same `danos.kernel_abi` (SysV): the loader's
|
garbage. So both sides reference the same `system.kernel_abi` (SysV): the loader's
|
||||||
function-pointer type and the kernel's `_start` both carry
|
function-pointer type and the kernel's `_start` both carry
|
||||||
`callconv(danos.kernel_abi)`, and the pointer reliably arrives in RDI. That is the
|
`callconv(system.kernel_abi)`, and the pointer reliably arrives in RDI. That is the
|
||||||
whole reason `kernel_abi` lives in the shared contract — see [efi.md](efi.md) for
|
whole reason `kernel_abi` lives in the shared contract — see [efi.md](efi.md) for
|
||||||
the handoff it governs.
|
the handoff it governs.
|
||||||
|
|
||||||
|
## The process-entry stack (argc/argv)
|
||||||
|
|
||||||
|
The SysV ABI also fixes what a *fresh process* finds on its stack — and danos
|
||||||
|
follows it, so its own runtime and any future C libc read arguments the same way.
|
||||||
|
At the first user instruction, `rsp` is 16-byte aligned and points at (addresses
|
||||||
|
growing upward):
|
||||||
|
|
||||||
|
```
|
||||||
|
rsp → argc u64
|
||||||
|
argv[0] … argv[argc-1] pointers into the strings area below
|
||||||
|
NULL argv terminator
|
||||||
|
NULL envp terminator (no environment yet)
|
||||||
|
{AT_PAGESZ, page size} auxiliary vector
|
||||||
|
{AT_NULL, 0} auxiliary-vector terminator
|
||||||
|
argv string bytes NUL-terminated
|
||||||
|
───────────────────────── stack top (stack_top_virtual)
|
||||||
|
```
|
||||||
|
|
||||||
|
The kernel builds this block at the top of the process's stack — 8 pages (32 KiB,
|
||||||
|
`parameters.user_stack_pages`) mapped RW+NX below a fixed top, with the page below
|
||||||
|
them left unmapped as a **guard**, so a stack overflow faults (killing only that
|
||||||
|
process) instead of silently corrupting the image
|
||||||
|
(`buildEntryStack` in `system/kernel/process.zig`); `argv[0]` is always the path
|
||||||
|
or initial-ramdisk name the process was spawned as, and `system_spawn`'s optional
|
||||||
|
argument blob becomes `argv[1..]`. The runtime's `_start`
|
||||||
|
(`library/runtime/start.zig`) hands the block to `rt_start`, which builds a
|
||||||
|
`runtime.process.Init` from it and passes that to the program's `main`
|
||||||
|
(`pub fn main(init: runtime.process.Init)`; a parameterless `main()` is also
|
||||||
|
accepted). A C runtime's `crt0` would walk
|
||||||
|
the identical layout unmodified — that's the compatibility being bought. The
|
||||||
|
`args` test proves the round trip.
|
||||||
|
|
||||||
## Where else it surfaces
|
## Where else it surfaces
|
||||||
|
|
||||||
- **The red zone → `red_zone = false`.** `build.zig` disables the red zone for the
|
- **The red zone → `red_zone = false`.** `build.zig` disables the red zone for the
|
||||||
|
|||||||
+7
-6
@@ -8,8 +8,9 @@ without a human staring at the screen.
|
|||||||
There are two layers:
|
There are two layers:
|
||||||
|
|
||||||
- **Host unit tests** (`zig build test`) — for pure, platform-independent logic in
|
- **Host unit tests** (`zig build test`) — for pure, platform-independent logic in
|
||||||
the shared `danos` module (the handoff layout in `src/root.zig`). These compile
|
the shared contracts (`system/boot-handoff.zig`, `system/abi.zig`,
|
||||||
for the host and run natively.
|
`system/devices/device-abi.zig`), which also compile-checks the three-way split
|
||||||
|
stays self-consistent. These compile for the host and run natively.
|
||||||
- **QEMU integration tests** (`python3 test/qemu_test.py`) — boot the real kernel
|
- **QEMU integration tests** (`python3 test/qemu_test.py`) — boot the real kernel
|
||||||
and check its behaviour. This is the interesting part.
|
and check its behaviour. This is the interesting part.
|
||||||
|
|
||||||
@@ -17,7 +18,7 @@ There are two layers:
|
|||||||
|
|
||||||
The framebuffer console draws pixels, which a test can't read without
|
The framebuffer console draws pixels, which a test can't read without
|
||||||
screen-scraping. So the kernel also writes everything to a **serial port**
|
screen-scraping. So the kernel also writes everything to a **serial port**
|
||||||
(`src/kernel/arch/x86_64/serial.zig`, a 16550 UART on COM1). `Console.write` mirrors every
|
(`system/kernel/architecture/x86_64/serial.zig`, a 16550 UART on COM1). `Console.write` mirrors every
|
||||||
byte to it, so all kernel output — boot log, memory summary, exception reports —
|
byte to it, so all kernel output — boot log, memory summary, exception reports —
|
||||||
appears on serial as plain text.
|
appears on serial as plain text.
|
||||||
|
|
||||||
@@ -29,7 +30,7 @@ a new architecture's UART is what makes the same tests run there.
|
|||||||
## In-kernel test cases
|
## In-kernel test cases
|
||||||
|
|
||||||
Building with `-Dtest-case=<name>` makes the kernel, after normal bring-up, run one
|
Building with `-Dtest-case=<name>` makes the kernel, after normal bring-up, run one
|
||||||
self-test from `src/kernel/tests.zig` instead of idling. Each case writes structured
|
self-test from `system/kernel/tests.zig` instead of idling. Each case writes structured
|
||||||
markers to serial:
|
markers to serial:
|
||||||
|
|
||||||
```
|
```
|
||||||
@@ -108,7 +109,7 @@ firmware, boot method, serial device). The cases are architecture-neutral —
|
|||||||
So bringing up a second architecture — an AArch64 Raspberry Pi is the motivating
|
So bringing up a second architecture — an AArch64 Raspberry Pi is the motivating
|
||||||
one — means:
|
one — means:
|
||||||
|
|
||||||
1. implement `src/kernel/arch/aarch64/` (CPU ops, its UART, exception vectors, page
|
1. implement `system/kernel/arch/aarch64/` (CPU ops, its UART, exception vectors, page
|
||||||
tables) behind the same `arch` interface,
|
tables) behind the same `arch` interface,
|
||||||
2. add an `aarch64` entry to `ARCHES` with its `qemu-system-aarch64` invocation,
|
2. add an `aarch64` entry to `ARCHES` with its `qemu-system-aarch64` invocation,
|
||||||
|
|
||||||
@@ -118,7 +119,7 @@ architectures".
|
|||||||
|
|
||||||
## Writing a new case
|
## Writing a new case
|
||||||
|
|
||||||
1. Add a function to `src/kernel/tests.zig` and dispatch it in `run` on its name.
|
1. Add a function to `system/kernel/tests.zig` and dispatch it in `run` on its name.
|
||||||
2. Emit `[PASS]/[FAIL]` lines and a `DANOS-TEST-RESULT:` line (non-faulting cases),
|
2. Emit `[PASS]/[FAIL]` lines and a `DANOS-TEST-RESULT:` line (non-faulting cases),
|
||||||
or trigger the condition and rely on the handler's output (faulting cases).
|
or trigger the condition and rely on the handler's output (faulting cases).
|
||||||
3. Add an entry to `CASES` in `test/qemu_test.py` with the regex that proves it.
|
3. Add an entry to `CASES` in `test/qemu_test.py` with the regex that proves it.
|
||||||
|
|||||||
+13
-4
@@ -81,11 +81,20 @@ prerequisites.
|
|||||||
(with boot-services memory reclaimed), [paging](paging.md) with W^X, [exceptions and
|
(with boot-services memory reclaimed), [paging](paging.md) with W^X, [exceptions and
|
||||||
interrupts](interrupts.md), a [calibrated timer + ns clock](device-interrupts.md), a
|
interrupts](interrupts.md), a [calibrated timer + ns clock](device-interrupts.md), a
|
||||||
[heap](heap.md), a [fixed-priority preemptive scheduler](scheduling.md) with blocking,
|
[heap](heap.md), a [fixed-priority preemptive scheduler](scheduling.md) with blocking,
|
||||||
and in-kernel [IPC channels](ipc.md) — plus a [test harness](testing.md).
|
in-kernel [IPC channels](ipc.md), SMP (all cores scheduling, with affinity), a
|
||||||
|
**higher-half kernel** with a physmap, and **user space**: per-process address
|
||||||
|
spaces, `syscall`/`sysret` with the `swapgs` discipline, a user-ELF loader, and
|
||||||
|
`/system/services/init` — a real user ELF built from `system/services/init/`, running at CPL 3 as PID 1 on its
|
||||||
|
own page tables — plus a [test harness](testing.md).
|
||||||
|
|
||||||
- **Isolation track** — **user mode + address-space isolation** (higher-half kernel,
|
- **Isolation track** — **user mode + address-space isolation**. *Done: a
|
||||||
ring 3, per-process page tables). The substrate everything else needs. *Next, and a
|
higher-half kernel with a physmap (the low half is user space), per-process
|
||||||
prerequisite for the resilience and driver tracks.*
|
address spaces with CR3 switched on context switch, the `swapgs` discipline,
|
||||||
|
`syscall`/`sysret`, a user-ELF loader, and `/system/services/init` running as a real
|
||||||
|
preemptive ring-3 process (PID 1). Remaining polish: an address-space/stack
|
||||||
|
reaper for exited tasks, SMAP + fault-recovering copy-in/out, the real IPC
|
||||||
|
syscalls (IPC_Call/IPC_ReplyWait — they arrive with the second user server),
|
||||||
|
and TLB shootdown once a process has more than one thread.*
|
||||||
- **Resilience track** — fault → kill → notify, a supervisor/reincarnation server,
|
- **Resilience track** — fault → kill → notify, a supervisor/reincarnation server,
|
||||||
resource cleanup on death, then a restartable driver as proof. Needs isolation.
|
resource cleanup on death, then a restartable driver as proof. Needs isolation.
|
||||||
See [resilience.md](resilience.md).
|
See [resilience.md](resilience.md).
|
||||||
|
|||||||
@@ -0,0 +1,80 @@
|
|||||||
|
//! /lib/mmio — typed volatile MMIO register access, plus the memory-ordering
|
||||||
|
//! barriers a device driver needs. Used by drivers on top of an `mmio_map` grant.
|
||||||
|
//!
|
||||||
|
//! **`volatile` is not a barrier.** In Zig it means only: don't elide this access, and
|
||||||
|
//! don't reorder it against *other volatile* accesses. It says nothing about ordinary
|
||||||
|
//! stores — the DMA descriptor you just filled in write-back RAM — which the compiler
|
||||||
|
//! (and, on weakly-ordered hardware, the CPU) may freely move past a volatile MMIO
|
||||||
|
//! write. The canonical bug:
|
||||||
|
//!
|
||||||
|
//! ring[i] = descriptor; // ordinary store to WB RAM
|
||||||
|
//! doorbell.* = i; // volatile store to UC MMIO
|
||||||
|
//! // nothing orders these; the device can read a stale descriptor
|
||||||
|
//!
|
||||||
|
//! Put a `wmb()` between them. The barriers lower per-architecture — which is the whole
|
||||||
|
//! reason they are a named primitive and not scattered `asm volatile`:
|
||||||
|
//!
|
||||||
|
//! x86_64 aarch64
|
||||||
|
//! mb() mfence dsb sy
|
||||||
|
//! rmb() lfence dsb ld
|
||||||
|
//! wmb() sfence dsb st
|
||||||
|
//!
|
||||||
|
//! x86 is forgiving (TSO + strong-uncacheable MMIO), so a compiler barrier usually
|
||||||
|
//! suffices; ARM is not, and ARM is the win condition (docs/vision.md) — so the
|
||||||
|
//! abstraction exists now, while there is one caller (hpet) to get right. See
|
||||||
|
//! docs/driver-model.md (M14) for the full ordering contract.
|
||||||
|
|
||||||
|
const builtin = @import("builtin");
|
||||||
|
|
||||||
|
/// Read a register of type `T` at absolute virtual address `addr` — a location inside
|
||||||
|
/// a device's `mmio_map` grant. `volatile`: never elided, never reordered against
|
||||||
|
/// another volatile access.
|
||||||
|
pub inline fn read(comptime T: type, addr: usize) T {
|
||||||
|
return @as(*const volatile T, @ptrFromInt(addr)).*;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write `value` of type `T` to the register at absolute virtual address `addr`.
|
||||||
|
pub inline fn write(comptime T: type, addr: usize, value: T) void {
|
||||||
|
@as(*volatile T, @ptrFromInt(addr)).* = value;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Full barrier: all loads and stores before it are globally visible before any after
|
||||||
|
/// it. Use when an MMIO write must complete before a following read.
|
||||||
|
pub inline fn mb() void {
|
||||||
|
switch (builtin.target.cpu.arch) {
|
||||||
|
.x86_64 => asm volatile ("mfence" ::: .{ .memory = true }),
|
||||||
|
.aarch64 => asm volatile ("dsb sy" ::: .{ .memory = true }),
|
||||||
|
else => @compileError("mmio.mb: unsupported architecture"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read barrier: loads before it complete before loads after it. Use after an IRQ
|
||||||
|
/// wake, before reading what the device wrote to shared memory.
|
||||||
|
pub inline fn rmb() void {
|
||||||
|
switch (builtin.target.cpu.arch) {
|
||||||
|
.x86_64 => asm volatile ("lfence" ::: .{ .memory = true }),
|
||||||
|
.aarch64 => asm volatile ("dsb ld" ::: .{ .memory = true }),
|
||||||
|
else => @compileError("mmio.rmb: unsupported architecture"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write barrier: stores before it become visible before stores after it. Use between
|
||||||
|
/// filling a DMA descriptor in RAM and ringing the device's doorbell.
|
||||||
|
pub inline fn wmb() void {
|
||||||
|
switch (builtin.target.cpu.arch) {
|
||||||
|
.x86_64 => asm volatile ("sfence" ::: .{ .memory = true }),
|
||||||
|
.aarch64 => asm volatile ("dsb st" ::: .{ .memory = true }),
|
||||||
|
else => @compileError("mmio.wmb: unsupported architecture"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
test "barriers emit and registers round-trip through a RAM cell" {
|
||||||
|
// The barriers must at least assemble for the host arch; ordering can't be unit
|
||||||
|
// tested, but a missing/mistyped mnemonic is caught here.
|
||||||
|
wmb();
|
||||||
|
rmb();
|
||||||
|
mb();
|
||||||
|
var cell: u64 = 0;
|
||||||
|
write(u64, @intFromPtr(&cell), 0xDEAD_BEEF);
|
||||||
|
try @import("std").testing.expectEqual(@as(u64, 0xDEAD_BEEF), read(u64, @intFromPtr(&cell)));
|
||||||
|
}
|
||||||
@@ -0,0 +1,13 @@
|
|||||||
|
//! DanOS's POSIX / C compatibility layer — `unistd`, `stdio`, and (later) the C
|
||||||
|
//! `errno` / `struct stat` / `extern "C"` surface. This is the *one* place POSIX and
|
||||||
|
//! C spellings are allowed to appear verbatim (see docs/coding-standards.md): a file
|
||||||
|
//! under library/posix/ *is* the foreign ABI, so it keeps the ABI's names. Everything
|
||||||
|
//! it touches on the danos side (the VFS protocol, the runtime) uses danos names,
|
||||||
|
//! which this layer translates to at the boundary.
|
||||||
|
//!
|
||||||
|
//! It is layered strictly *over* the runtime: it calls the runtime's IPC and heap,
|
||||||
|
//! never the kernel's system calls directly. danos-native applications use the
|
||||||
|
//! runtime; this exists so *POSIX* software can too.
|
||||||
|
|
||||||
|
pub const unistd = @import("unistd.zig");
|
||||||
|
pub const stdio = @import("stdio.zig");
|
||||||
@@ -0,0 +1,115 @@
|
|||||||
|
//! A small C stdio layer over the POSIX-style file API (unistd.zig). Unbuffered
|
||||||
|
//! for now — each fread/fwrite is one VFS round trip; an internal buffer (fewer
|
||||||
|
//! IPC calls) is a later optimisation. Both a Zig-callable API and `extern "C"`
|
||||||
|
//! symbols are provided, so Zig and future C programs share it.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const unistd = @import("unistd.zig");
|
||||||
|
const heap = @import("runtime").heap;
|
||||||
|
|
||||||
|
pub const SEEK_SET = unistd.SEEK_SET;
|
||||||
|
pub const SEEK_CURRENT = unistd.SEEK_CURRENT;
|
||||||
|
pub const SEEK_END = unistd.SEEK_END;
|
||||||
|
|
||||||
|
/// A C `FILE`: an fd plus sticky end-of-file / error flags. Allocated on the
|
||||||
|
/// heap; `fclose` frees it.
|
||||||
|
pub const FILE = extern struct {
|
||||||
|
fd: i32,
|
||||||
|
eof: c_int = 0,
|
||||||
|
err: c_int = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
fn flagsFor(mode: []const u8) u32 {
|
||||||
|
if (mode.len == 0) return 0;
|
||||||
|
return switch (mode[0]) {
|
||||||
|
'w', 'a' => unistd.O_CREAT,
|
||||||
|
else => 0,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Open `path` in `mode` ("r"/"w"/"a", '+' ignored for now). Returns null on error.
|
||||||
|
pub fn fopen(path: []const u8, mode: []const u8) ?*FILE {
|
||||||
|
const fd = unistd.open(path, flagsFor(mode));
|
||||||
|
if (fd < 0) return null;
|
||||||
|
const f = heap.allocator().create(FILE) catch {
|
||||||
|
unistd.close(fd);
|
||||||
|
return null;
|
||||||
|
};
|
||||||
|
f.* = .{ .fd = fd };
|
||||||
|
if (mode.len > 0 and mode[0] == 'a') _ = unistd.lseek(fd, 0, unistd.SEEK_END);
|
||||||
|
return f;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn fclose(f: *FILE) c_int {
|
||||||
|
unistd.close(f.fd);
|
||||||
|
heap.allocator().destroy(f);
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read `size*nmemb` bytes; returns the number of whole items read.
|
||||||
|
pub fn fread(buffer: []u8, size: usize, nmemb: usize, f: *FILE) usize {
|
||||||
|
const total = size * nmemb;
|
||||||
|
if (total == 0) return 0;
|
||||||
|
const n = unistd.read(f.fd, buffer[0..@min(buffer.len, total)]);
|
||||||
|
if (n <= 0) {
|
||||||
|
f.eof = 1;
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
return @as(usize, @intCast(n)) / size;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write `size*nmemb` bytes; returns the number of whole items written.
|
||||||
|
pub fn fwrite(data: []const u8, size: usize, nmemb: usize, f: *FILE) usize {
|
||||||
|
const total = @min(data.len, size * nmemb);
|
||||||
|
if (total == 0) return 0;
|
||||||
|
const n = unistd.write(f.fd, data[0..total]);
|
||||||
|
if (n <= 0) {
|
||||||
|
f.err = 1;
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
return @as(usize, @intCast(n)) / size;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn fseek(f: *FILE, off: i64, whence: u32) c_int {
|
||||||
|
f.eof = 0;
|
||||||
|
return if (unistd.lseek(f.fd, off, whence) < 0) -1 else 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn ftell(f: *FILE) i64 {
|
||||||
|
return unistd.lseek(f.fd, 0, unistd.SEEK_CURRENT);
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn rewind(f: *FILE) void {
|
||||||
|
_ = fseek(f, 0, SEEK_SET);
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn feof(f: *FILE) c_int {
|
||||||
|
return f.eof;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn ferror(f: *FILE) c_int {
|
||||||
|
return f.err;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn fputs(s: []const u8, f: *FILE) c_int {
|
||||||
|
return if (unistd.write(f.fd, s) < 0) -1 else 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn fputc(c: u8, f: *FILE) c_int {
|
||||||
|
const b = [_]u8{c};
|
||||||
|
return if (unistd.write(f.fd, &b) == 1) c else -1;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn fgetc(f: *FILE) c_int {
|
||||||
|
var b: [1]u8 = undefined;
|
||||||
|
const n = unistd.read(f.fd, &b);
|
||||||
|
if (n <= 0) {
|
||||||
|
f.eof = 1;
|
||||||
|
return -1; // EOF
|
||||||
|
}
|
||||||
|
return b[0];
|
||||||
|
}
|
||||||
|
|
||||||
|
// Real `extern "C"` symbols (fopen/fread/fseek/...) — with a C-string signature
|
||||||
|
// distinct from the Zig slice API above — land with the first C program, wired
|
||||||
|
// via @export so they don't collide with these Zig names.
|
||||||
@@ -0,0 +1,148 @@
|
|||||||
|
//! POSIX-style file API for user programs — the low level under C stdio. Files
|
||||||
|
//! are named objects served by the user-space VFS server (system/services/vfs/vfs.zig); each
|
||||||
|
//! call marshals a request, IPC_Calls the VFS, and unmarshals the reply. The
|
||||||
|
//! kernel knows nothing of files or fds — the fd table lives here, per process.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const protocol = @import("vfs-protocol");
|
||||||
|
const ipc = @import("runtime").ipc;
|
||||||
|
|
||||||
|
pub const O_CREAT = protocol.create;
|
||||||
|
pub const SEEK_SET: u32 = 0;
|
||||||
|
pub const SEEK_CURRENT: u32 = 1;
|
||||||
|
pub const SEEK_END: u32 = 2;
|
||||||
|
|
||||||
|
// Resolve (and cache) the VFS server endpoint, looked up by well-known id.
|
||||||
|
var vfs_handle: usize = 0;
|
||||||
|
var vfs_resolved = false;
|
||||||
|
fn vfs() ?usize {
|
||||||
|
if (!vfs_resolved) {
|
||||||
|
vfs_handle = ipc.lookup(.vfs) orelse return null;
|
||||||
|
vfs_resolved = true;
|
||||||
|
}
|
||||||
|
return vfs_handle;
|
||||||
|
}
|
||||||
|
|
||||||
|
const maximum_fds = 32;
|
||||||
|
const Fd = struct { used: bool = false, node: u64 = 0, offset: u64 = 0 };
|
||||||
|
var fds = [_]Fd{.{}} ** maximum_fds;
|
||||||
|
|
||||||
|
fn allocFd() ?usize {
|
||||||
|
for (&fds, 0..) |*f, i| {
|
||||||
|
if (!f.used) {
|
||||||
|
f.* = .{ .used = true };
|
||||||
|
return i;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
const Result = struct { reply: protocol.Reply, payload: []u8 };
|
||||||
|
|
||||||
|
/// One request/reply round trip: [Request header][send payload] -> VFS ->
|
||||||
|
/// [Reply header][receive payload]. The receive payload is written into `out`.
|
||||||
|
fn transact(request: protocol.Request, send: []const u8, out: []u8) ?Result {
|
||||||
|
const h = vfs() orelse return null;
|
||||||
|
var message: [protocol.message_maximum]u8 = undefined;
|
||||||
|
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||||
|
const slen = @min(send.len, protocol.maximum_payload);
|
||||||
|
@memcpy(message[protocol.request_size..][0..slen], send[0..slen]);
|
||||||
|
|
||||||
|
var rbuf: [protocol.message_maximum]u8 = undefined;
|
||||||
|
const n = ipc.call(h, message[0 .. protocol.request_size + slen], &rbuf) catch return null;
|
||||||
|
if (n < protocol.reply_size) return null;
|
||||||
|
const reply = std.mem.bytesToValue(protocol.Reply, rbuf[0..protocol.reply_size]);
|
||||||
|
const rpl = @min(n - protocol.reply_size, out.len);
|
||||||
|
@memcpy(out[0..rpl], rbuf[protocol.reply_size..][0..rpl]);
|
||||||
|
return .{ .reply = reply, .payload = out[0..rpl] };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Open (or create, with O_CREAT) `path`; returns an fd or -1.
|
||||||
|
pub fn open(path: []const u8, flags: u32) i32 {
|
||||||
|
const fd = allocFd() orelse return -1;
|
||||||
|
const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = flags };
|
||||||
|
const r = transact(request, path, &.{}) orelse {
|
||||||
|
fds[fd].used = false;
|
||||||
|
return -1;
|
||||||
|
};
|
||||||
|
if (r.reply.status != 0) {
|
||||||
|
fds[fd].used = false;
|
||||||
|
return -1;
|
||||||
|
}
|
||||||
|
fds[fd] = .{ .used = true, .node = r.reply.node, .offset = 0 };
|
||||||
|
return @intCast(fd);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn fdPtr(fd: i32) ?*Fd {
|
||||||
|
if (fd < 0 or fd >= maximum_fds) return null;
|
||||||
|
const f = &fds[@intCast(fd)];
|
||||||
|
return if (f.used) f else null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read up to `buffer.len` bytes at the current offset; returns the count or -1.
|
||||||
|
pub fn read(fd: i32, buffer: []u8) isize {
|
||||||
|
const f = fdPtr(fd) orelse return -1;
|
||||||
|
const want: u32 = @intCast(@min(buffer.len, protocol.maximum_payload));
|
||||||
|
const request = protocol.Request{ .operation = .read, .node = f.node, .offset = f.offset, .len = want, .flags = 0 };
|
||||||
|
const r = transact(request, &.{}, buffer) orelse return -1;
|
||||||
|
if (r.reply.status != 0) return -1;
|
||||||
|
f.offset += r.reply.len;
|
||||||
|
return @intCast(r.reply.len);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write `data` at the current offset; returns the count or -1.
|
||||||
|
pub fn write(fd: i32, data: []const u8) isize {
|
||||||
|
const f = fdPtr(fd) orelse return -1;
|
||||||
|
const want: u32 = @intCast(@min(data.len, protocol.maximum_payload));
|
||||||
|
const request = protocol.Request{ .operation = .write, .node = f.node, .offset = f.offset, .len = want, .flags = 0 };
|
||||||
|
const r = transact(request, data[0..want], &.{}) orelse return -1;
|
||||||
|
if (r.reply.status != 0) return -1;
|
||||||
|
f.offset += r.reply.len;
|
||||||
|
return @intCast(r.reply.len);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Reposition the fd's offset. Returns the new offset or -1. (SEEK_END needs the
|
||||||
|
/// file size, which `stat` provides; handled by fetching it here.)
|
||||||
|
pub fn lseek(fd: i32, off: i64, whence: u32) i64 {
|
||||||
|
const f = fdPtr(fd) orelse return -1;
|
||||||
|
const base: i64 = switch (whence) {
|
||||||
|
SEEK_SET => 0,
|
||||||
|
SEEK_CURRENT => @intCast(f.offset),
|
||||||
|
SEEK_END => blk: {
|
||||||
|
const request = protocol.Request{ .operation = .status, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
||||||
|
var sbuf: [@sizeOf(protocol.FileStatus)]u8 = undefined;
|
||||||
|
const r = transact(request, &.{}, &sbuf) orelse return -1;
|
||||||
|
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.FileStatus)) return -1;
|
||||||
|
const st = std.mem.bytesToValue(protocol.FileStatus, sbuf[0..@sizeOf(protocol.FileStatus)]);
|
||||||
|
break :blk @intCast(st.size);
|
||||||
|
},
|
||||||
|
else => return -1,
|
||||||
|
};
|
||||||
|
const pos = base + off;
|
||||||
|
if (pos < 0) return -1;
|
||||||
|
f.offset = @intCast(pos);
|
||||||
|
return pos;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Stat `path`. Returns 0 or -1.
|
||||||
|
pub fn stat(path: []const u8, out: *protocol.FileStatus) i32 {
|
||||||
|
// Open, stat by node, close — simple and enough for now.
|
||||||
|
const fd = open(path, 0);
|
||||||
|
if (fd < 0) return -1;
|
||||||
|
defer close(fd);
|
||||||
|
const f = fdPtr(fd).?;
|
||||||
|
const request = protocol.Request{ .operation = .status, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
||||||
|
var sbuf: [@sizeOf(protocol.FileStatus)]u8 = undefined;
|
||||||
|
const r = transact(request, &.{}, &sbuf) orelse return -1;
|
||||||
|
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.FileStatus)) return -1;
|
||||||
|
out.* = std.mem.bytesToValue(protocol.FileStatus, sbuf[0..@sizeOf(protocol.FileStatus)]);
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Close an fd (best effort — tells the VFS to release the open file).
|
||||||
|
pub fn close(fd: i32) void {
|
||||||
|
const f = fdPtr(fd) orelse return;
|
||||||
|
const request = protocol.Request{ .operation = .close, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
||||||
|
_ = transact(request, &.{}, &.{});
|
||||||
|
f.used = false;
|
||||||
|
}
|
||||||
@@ -0,0 +1,131 @@
|
|||||||
|
//! User-space device access: enumerate the kernel's device table, claim a device,
|
||||||
|
//! map its MMIO, and bind its interrupt. A driver uses these to find and take
|
||||||
|
//! ownership of its hardware; the claim is the capability the kernel checks before
|
||||||
|
//! mapping registers or routing an IRQ.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const device_abi = @import("device-abi");
|
||||||
|
const sc = @import("system-call.zig");
|
||||||
|
|
||||||
|
pub const DeviceDescriptor = device_abi.DeviceDescriptor;
|
||||||
|
pub const ResourceDescriptor = device_abi.ResourceDescriptor;
|
||||||
|
pub const DeviceClass = device_abi.DeviceClass;
|
||||||
|
pub const ResourceKind = device_abi.ResourceKind;
|
||||||
|
|
||||||
|
inline fn failed(r: usize) bool {
|
||||||
|
return r > ~@as(usize, 0) - 4095;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Copy up to `buffer.len` device descriptors into `buffer`; returns the total count.
|
||||||
|
pub fn enumerate(buffer: []DeviceDescriptor) usize {
|
||||||
|
return sc.systemCall2(.device_enumerate, @intFromPtr(buffer.ptr), buffer.len);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Take exclusive ownership of device `id`. Returns false if taken or invalid.
|
||||||
|
pub fn claim(id: u64) bool {
|
||||||
|
return !failed(sc.systemCall1(.device_claim, id));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Map resource `resource_index` (which must be an MMIO window) of claimed device
|
||||||
|
/// `device_id` into this address space; returns the register base virtual address.
|
||||||
|
pub fn mmioMap(device_id: u64, resource_index: u64) ?usize {
|
||||||
|
const r = sc.systemCall2(.mmio_map, device_id, resource_index);
|
||||||
|
return if (failed(r)) null else r;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `DeviceDescriptor.parent` for a device with no parent.
|
||||||
|
pub const no_parent = device_abi.no_parent;
|
||||||
|
|
||||||
|
/// `DeviceDescriptor.pci_class` for a device that is not a PCI function. Set this on
|
||||||
|
/// descriptors passed to `register` unless the child really is one.
|
||||||
|
pub const no_pci_class = device_abi.no_pci_class;
|
||||||
|
|
||||||
|
/// Publish `descriptor` as a child of `parent_id`, which this process must have claimed.
|
||||||
|
/// Returns the new device id. The child is left unclaimed, so whichever driver owns
|
||||||
|
/// that class of device can `claim` it — that is how a bus hands off a device.
|
||||||
|
///
|
||||||
|
/// Every resource in `descriptor` must be **contained** in a parent resource of the same
|
||||||
|
/// kind: a sub-window of the parent's MMIO, or one of its IRQs. The kernel refuses
|
||||||
|
/// anything else, because a device descriptor is a licence to map physical memory and
|
||||||
|
/// a bus driver may only subdivide what it already owns. `descriptor.id` and `descriptor.parent`
|
||||||
|
/// are ignored. A device with no resources at all is fine — a USB device is reached
|
||||||
|
/// through its controller, not by MMIO.
|
||||||
|
pub fn register(parent_id: u64, descriptor: *const DeviceDescriptor) ?u64 {
|
||||||
|
const r = sc.systemCall2(.device_register, parent_id, @intFromPtr(descriptor));
|
||||||
|
return if (failed(r)) null else r;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Bind resource `resource_index` (which must be an IRQ) of claimed device `device_id` to
|
||||||
|
/// `endpoint`. From then on the interrupt arrives as an asynchronous notification:
|
||||||
|
/// `ipc.replyWait` on that endpoint returns with the high bit set in `badge` and the
|
||||||
|
/// low bits carrying the GSI. The kernel masks the line before waking you.
|
||||||
|
pub fn irqBind(device_id: u64, resource_index: u64, endpoint: usize) bool {
|
||||||
|
return !failed(sc.systemCall3(.irq_bind, device_id, resource_index, endpoint));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Re-arm a bound IRQ. Call this **after** quieting the device (clearing whatever
|
||||||
|
/// status register holds its line asserted) — the kernel left the line masked
|
||||||
|
/// precisely because it could not do that for you. Skip it and the interrupt never
|
||||||
|
/// fires again; call it before the device is quiet and a level-triggered line storms.
|
||||||
|
pub fn irqAck(device_id: u64, resource_index: u64) bool {
|
||||||
|
return !failed(sc.systemCall2(.irq_ack, device_id, resource_index));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The Message-Signalled Interrupt address/data a driver programs into its device's
|
||||||
|
/// MSI capability. The device raises the interrupt by writing `data` to `address`.
|
||||||
|
pub const Msi = struct { address: u64, data: u32 };
|
||||||
|
|
||||||
|
/// Set up MSI for a claimed device: the kernel allocates a per-device edge-triggered
|
||||||
|
/// vector, binds it to `endpoint` (delivered like `irqBind`, but with no mask and no
|
||||||
|
/// `irqAck` cycle), and returns the (address, data) to write into the device's MSI
|
||||||
|
/// capability — found by mmio_mapping the device's ECAM config space (resource 0) and
|
||||||
|
/// walking its capability list. Returns null on failure. Two return values (address in
|
||||||
|
/// rax, data in rdx), so a hand-written stub.
|
||||||
|
pub fn msiBind(device_id: u64, endpoint: usize) ?Msi {
|
||||||
|
var rax: usize = undefined;
|
||||||
|
var rdx: usize = undefined;
|
||||||
|
asm volatile ("syscall"
|
||||||
|
: [rax] "={rax}" (rax),
|
||||||
|
[rdx] "={rdx}" (rdx),
|
||||||
|
: [n] "{rax}" (@intFromEnum(abi.SystemCall.msi_bind)),
|
||||||
|
[a0] "{rdi}" (device_id),
|
||||||
|
[a1] "{rsi}" (endpoint),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
if (failed(rax)) return null;
|
||||||
|
return .{ .address = rax, .data = @intCast(rdx) };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read `width` bytes (1, 2, or 4) from a port in a claimed device's `io_port`
|
||||||
|
/// resource, at byte `offset` within it. Ring 3 has no direct `in`/`out`, so a legacy
|
||||||
|
/// driver (PS/2, 16550 UART) reaches its ports through this claim-gated call — each
|
||||||
|
/// access is a syscall, which is fine for the low-rate hardware that needs it. Returns
|
||||||
|
/// null if the capability check fails (device not claimed, wrong resource, out of
|
||||||
|
/// range). A device that decodes no data returns all-ones, which is a valid value, not
|
||||||
|
/// a failure.
|
||||||
|
pub fn ioRead(device_id: u64, resource_index: u64, offset: u64, width: u8) ?u32 {
|
||||||
|
const r = sc.systemCall4(.io_read, device_id, resource_index, offset, width);
|
||||||
|
return if (failed(r)) null else @intCast(r);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write `value` (its low `width` bytes, 1/2/4) to a port in a claimed device's
|
||||||
|
/// `io_port` resource, at byte `offset`. Same capability gate as `ioRead`.
|
||||||
|
pub fn ioWrite(device_id: u64, resource_index: u64, offset: u64, width: u8, value: u32) bool {
|
||||||
|
return !failed(sc.systemCall5(.io_write, device_id, resource_index, offset, width, value));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Find DeviceDescription by hid
|
||||||
|
///
|
||||||
|
/// Utility function for driver development
|
||||||
|
pub fn findDeviceDescriptorByHid(buffer: []DeviceDescriptor, hid_needle: []const u8) ?DeviceDescriptor {
|
||||||
|
const total = enumerate(buffer);
|
||||||
|
const n = @min(total, buffer.len);
|
||||||
|
for (@as([]DeviceDescriptor, buffer[0..n])) |d| {
|
||||||
|
const hid_haystack = d.hid[0..@intCast(d.hid_len)];
|
||||||
|
if (std.mem.eql(u8, hid_haystack, hid_needle)) {
|
||||||
|
return d;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return null;
|
||||||
|
}
|
||||||
@@ -0,0 +1,48 @@
|
|||||||
|
//! User-space DMA memory: `dma_alloc` / `dma_free`. A driver that programs a
|
||||||
|
//! bus-mastering engine needs a descriptor ring the device can read — memory that is
|
||||||
|
//! physically contiguous, at a physical address the driver knows, uncacheable, and
|
||||||
|
//! pinned. `mmap` gives none of those; this does. Pair it with the barriers in
|
||||||
|
//! `/lib/mmio` (fill the ring, `wmb()`, ring the doorbell). See docs/driver-model.md.
|
||||||
|
|
||||||
|
const abi = @import("abi");
|
||||||
|
const sc = @import("system-call.zig");
|
||||||
|
|
||||||
|
/// Allocation flags. `coherent` (uncacheable) is the portable default; the rest are
|
||||||
|
/// opt-in for specific hardware — see `abi`.
|
||||||
|
pub const coherent: usize = abi.dma_coherent;
|
||||||
|
pub const write_combining: usize = abi.dma_write_combining;
|
||||||
|
pub const below_4g: usize = abi.dma_below_4g;
|
||||||
|
|
||||||
|
/// A DMA allocation: the `virtual` address the CPU touches, and the `physical` address
|
||||||
|
/// to program into the device's descriptor-ring / base registers.
|
||||||
|
pub const Region = struct {
|
||||||
|
virtual: usize,
|
||||||
|
physical: usize,
|
||||||
|
};
|
||||||
|
|
||||||
|
inline fn failed(r: usize) bool {
|
||||||
|
return r > ~@as(usize, 0) - 4095;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Allocate `len` bytes of DMA-capable memory with `flags` (e.g. `coherent`, or
|
||||||
|
/// `coherent | below_4g`). Returns the virtual/physical pair, or null on failure. Two
|
||||||
|
/// return values — the virtual address in rax, the physical address in rdx — so it
|
||||||
|
/// needs a hand-written stub.
|
||||||
|
pub fn alloc(len: usize, flags: usize) ?Region {
|
||||||
|
var rax: usize = undefined;
|
||||||
|
var rdx: usize = undefined; // out: physical address
|
||||||
|
asm volatile ("syscall"
|
||||||
|
: [rax] "={rax}" (rax),
|
||||||
|
[rdx] "={rdx}" (rdx),
|
||||||
|
: [n] "{rax}" (@intFromEnum(abi.SystemCall.dma_alloc)),
|
||||||
|
[a0] "{rdi}" (len),
|
||||||
|
[a1] "{rsi}" (flags),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
if (failed(rax)) return null;
|
||||||
|
return .{ .virtual = rax, .physical = rdx };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Release a region from a prior `alloc` (`virtual` and the same `len`).
|
||||||
|
pub fn free(virtual: usize, len: usize) void {
|
||||||
|
_ = sc.systemCall2(.dma_free, virtual, len);
|
||||||
|
}
|
||||||
@@ -0,0 +1,193 @@
|
|||||||
|
//! The user-space heap: C-convention dynamic allocation (`malloc`/`free`/…) plus
|
||||||
|
//! a `std.mem.Allocator` adapter over the same free list, so both C-style code
|
||||||
|
//! and Zig `std` containers share one heap.
|
||||||
|
//!
|
||||||
|
//! The algorithm is a straight port of the kernel's first-fit free list
|
||||||
|
//! (system/kernel/heap.zig): an address-ordered singly linked list of free blocks,
|
||||||
|
//! split on allocation and coalesced with neighbours on free. The only thing
|
||||||
|
//! that changes on this side of the system_call boundary is where memory comes from
|
||||||
|
//! — `grow` asks the kernel for pages via `mmap` instead of mapping frames
|
||||||
|
//! itself, and the kernel picks the base address.
|
||||||
|
//!
|
||||||
|
//! Single-threaded and 16-byte maximum alignment, exactly like the kernel heap; a
|
||||||
|
//! lock and larger alignments come when user programs gain threads.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const system_calls = @import("system.zig");
|
||||||
|
|
||||||
|
const page_size = abi.page_size;
|
||||||
|
|
||||||
|
/// A block header, at the start of every block; while free it also links the
|
||||||
|
/// free list via `next`.
|
||||||
|
const Block = extern struct {
|
||||||
|
size: usize, // total block size in bytes, including this header; a multiple of 16
|
||||||
|
next: ?*Block, // free-list link (only meaningful while free)
|
||||||
|
};
|
||||||
|
|
||||||
|
const header_size = @sizeOf(Block); // 16
|
||||||
|
const minimum_block = header_size + 16; // smallest block worth splitting off
|
||||||
|
/// Grow granularity: one `mmap` per 64 KiB amortises the system_call.
|
||||||
|
const chunk = 64 * 1024;
|
||||||
|
|
||||||
|
var free_list: ?*Block = null;
|
||||||
|
|
||||||
|
fn alignUp(value: usize, alignment: usize) usize {
|
||||||
|
return (value + alignment - 1) & ~(alignment - 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn payloadOf(block: *Block) [*]u8 {
|
||||||
|
return @ptrFromInt(@intFromPtr(block) + header_size);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Ask the kernel for more pages and add them as a free block. Because each
|
||||||
|
/// `mmap` is an independent grant, cross-grant coalescing happens only when the
|
||||||
|
/// kernel returns adjacent bases (its arena is a bump allocator, so consecutive
|
||||||
|
/// grants usually are adjacent). Returns false if the kernel is out of memory.
|
||||||
|
fn grow(minimum_bytes: usize) bool {
|
||||||
|
const bytes = alignUp(@max(minimum_bytes, chunk), page_size);
|
||||||
|
const ret = system_calls.mmap(bytes, system_calls.PROT_READ | system_calls.PROT_WRITE);
|
||||||
|
if (system_calls.mmapFailed(ret)) return false;
|
||||||
|
|
||||||
|
const block: *Block = @ptrFromInt(ret);
|
||||||
|
block.size = bytes;
|
||||||
|
insertFree(block); // coalesces if this grant is adjacent to a prior one
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Insert a block into the address-ordered free list, coalescing with the
|
||||||
|
/// physically adjacent free blocks on either side.
|
||||||
|
fn insertFree(block: *Block) void {
|
||||||
|
var previous: ?*Block = null;
|
||||||
|
var current = free_list;
|
||||||
|
while (current) |c| : (current = c.next) {
|
||||||
|
if (@intFromPtr(c) > @intFromPtr(block)) break;
|
||||||
|
previous = c;
|
||||||
|
}
|
||||||
|
|
||||||
|
block.next = current;
|
||||||
|
if (previous) |p| p.next = block else free_list = block;
|
||||||
|
|
||||||
|
// Merge forward into `current` if they're contiguous.
|
||||||
|
if (current) |c| {
|
||||||
|
if (@intFromPtr(block) + block.size == @intFromPtr(c)) {
|
||||||
|
block.size += c.size;
|
||||||
|
block.next = c.next;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Merge `previous` forward into `block` if they're contiguous.
|
||||||
|
if (previous) |p| {
|
||||||
|
if (@intFromPtr(p) + p.size == @intFromPtr(block)) {
|
||||||
|
p.size += block.size;
|
||||||
|
p.next = block.next;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Allocate `len` bytes (16-byte aligned), or null if out of memory.
|
||||||
|
fn rawAlloc(len: usize) ?[*]u8 {
|
||||||
|
const need = alignUp(header_size + len, 16);
|
||||||
|
|
||||||
|
var attempts: u32 = 0;
|
||||||
|
while (attempts < 2) : (attempts += 1) {
|
||||||
|
var previous: ?*Block = null;
|
||||||
|
var current = free_list;
|
||||||
|
while (current) |block| : ({
|
||||||
|
previous = block;
|
||||||
|
current = block.next;
|
||||||
|
}) {
|
||||||
|
if (block.size < need) continue;
|
||||||
|
|
||||||
|
if (block.size >= need + minimum_block) {
|
||||||
|
// Split: carve `need` off the front, leave the rest free.
|
||||||
|
const rest: *Block = @ptrFromInt(@intFromPtr(block) + need);
|
||||||
|
rest.size = block.size - need;
|
||||||
|
rest.next = block.next;
|
||||||
|
if (previous) |p| p.next = rest else free_list = rest;
|
||||||
|
block.size = need;
|
||||||
|
} else {
|
||||||
|
// Take the whole block.
|
||||||
|
if (previous) |p| p.next = block.next else free_list = block.next;
|
||||||
|
}
|
||||||
|
return payloadOf(block);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Nothing fit: grow and try once more.
|
||||||
|
if (!grow(need)) return null;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn rawFree(ptr: [*]u8) void {
|
||||||
|
const block: *Block = @ptrFromInt(@intFromPtr(ptr) - header_size);
|
||||||
|
insertFree(block);
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- C ABI: the global implicit heap ---------------------------------------
|
||||||
|
// `extern "C"` symbols so future C code links the same malloc/free directly.
|
||||||
|
|
||||||
|
export fn malloc(size: usize) callconv(.c) ?*anyopaque {
|
||||||
|
if (size == 0) return null;
|
||||||
|
const p = rawAlloc(size) orelse return null;
|
||||||
|
return @ptrCast(p);
|
||||||
|
}
|
||||||
|
|
||||||
|
export fn free(ptr: ?*anyopaque) callconv(.c) void {
|
||||||
|
const p = ptr orelse return;
|
||||||
|
rawFree(@ptrCast(p));
|
||||||
|
}
|
||||||
|
|
||||||
|
export fn calloc(nmemb: usize, size: usize) callconv(.c) ?*anyopaque {
|
||||||
|
const total = std.math.mul(usize, nmemb, size) catch return null; // overflow-safe
|
||||||
|
if (total == 0) return null;
|
||||||
|
const p = rawAlloc(total) orelse return null;
|
||||||
|
@memset(p[0..total], 0);
|
||||||
|
return @ptrCast(p);
|
||||||
|
}
|
||||||
|
|
||||||
|
export fn realloc(ptr: ?*anyopaque, size: usize) callconv(.c) ?*anyopaque {
|
||||||
|
const p = ptr orelse return malloc(size);
|
||||||
|
if (size == 0) {
|
||||||
|
rawFree(@ptrCast(p));
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
const block: *Block = @ptrFromInt(@intFromPtr(p) - header_size);
|
||||||
|
const old_payload = block.size - header_size;
|
||||||
|
if (size <= old_payload) return p; // shrink/same: keep the block
|
||||||
|
const np = rawAlloc(size) orelse return null; // grow: alloc + copy + free
|
||||||
|
@memcpy(np[0..old_payload], @as([*]u8, @ptrCast(p))[0..old_payload]);
|
||||||
|
rawFree(@ptrCast(p));
|
||||||
|
return @ptrCast(np);
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- std.mem.Allocator interface (same free list) --------------------------
|
||||||
|
|
||||||
|
pub fn allocator() std.mem.Allocator {
|
||||||
|
return .{ .ptr = undefined, .vtable = &vtable };
|
||||||
|
}
|
||||||
|
|
||||||
|
const vtable = std.mem.Allocator.VTable{
|
||||||
|
.alloc = allocImpl,
|
||||||
|
.resize = resizeImpl,
|
||||||
|
.remap = remapImpl,
|
||||||
|
.free = freeImpl,
|
||||||
|
};
|
||||||
|
|
||||||
|
fn allocImpl(_: *anyopaque, len: usize, alignment: std.mem.Alignment, _: usize) ?[*]u8 {
|
||||||
|
if (alignment.toByteUnits() > 16) return null; // blocks are 16-byte aligned
|
||||||
|
return rawAlloc(len);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn resizeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, new_len: usize, _: usize) bool {
|
||||||
|
// In-place iff the new payload still fits the current block.
|
||||||
|
const block: *Block = @ptrFromInt(@intFromPtr(memory.ptr) - header_size);
|
||||||
|
return new_len + header_size <= block.size;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn remapImpl(_: *anyopaque, _: []u8, _: std.mem.Alignment, _: usize, _: usize) ?[*]u8 {
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn freeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, _: usize) void {
|
||||||
|
rawFree(memory.ptr);
|
||||||
|
}
|
||||||
@@ -0,0 +1,220 @@
|
|||||||
|
//! User-space input helpers: the client and publisher sides of the input service, so a
|
||||||
|
//! program listening for input events — or a driver broadcasting them — doesn't hand-roll
|
||||||
|
//! the IPC. Layered over `ipc` (endpoints, capability passing, `send`) and the shared
|
||||||
|
//! `input-protocol` wire format, the way `device.zig` layers over the raw `device_*` calls.
|
||||||
|
//! See system/services/input/input.zig.
|
||||||
|
//!
|
||||||
|
//! The service carries several device classes (keyboard, mouse, joystick/gamepad). A
|
||||||
|
//! **source** publishes its class with the matching method:
|
||||||
|
//! var source = input.connectSource() orelse return;
|
||||||
|
//! _ = source.publishKeyboardEvent(.{ .kind = ..., .keycode = ..., ... });
|
||||||
|
//! _ = source.publishMouseEvent(.{ ... });
|
||||||
|
//! _ = source.publishJoystickEvent(.{ ... });
|
||||||
|
//!
|
||||||
|
//! A **subscriber** either takes one class with a typed helper —
|
||||||
|
//! var keys = input.subscribeKeyboard() orelse return;
|
||||||
|
//! while (true) { const key = keys.next() orelse continue; ... }
|
||||||
|
//! — or takes several at once and inspects the tagged envelope:
|
||||||
|
//! var listener = input.subscribeAll() orelse return;
|
||||||
|
//! while (true) {
|
||||||
|
//! const event = listener.next() orelse continue;
|
||||||
|
//! if (event.asKeyboard()) |k| { ... } else if (event.asMouse()) |m| { ... }
|
||||||
|
//! }
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const ipc = @import("ipc.zig");
|
||||||
|
const system = @import("system.zig");
|
||||||
|
const protocol = @import("input-protocol");
|
||||||
|
|
||||||
|
pub const DeviceKind = protocol.DeviceKind;
|
||||||
|
pub const InputEvent = protocol.InputEvent;
|
||||||
|
pub const KeyEvent = protocol.KeyEvent;
|
||||||
|
pub const MouseEvent = protocol.MouseEvent;
|
||||||
|
pub const JoystickEvent = protocol.JoystickEvent;
|
||||||
|
pub const EventKind = protocol.EventKind;
|
||||||
|
pub const MouseEventKind = protocol.MouseEventKind;
|
||||||
|
pub const JoystickEventKind = protocol.JoystickEventKind;
|
||||||
|
pub const Keycode = protocol.Keycode;
|
||||||
|
|
||||||
|
/// Interest masks re-exported so a caller can `subscribe(input.device_keyboard |
|
||||||
|
/// input.device_mouse)`.
|
||||||
|
pub const device_keyboard = protocol.device_keyboard;
|
||||||
|
pub const device_mouse = protocol.device_mouse;
|
||||||
|
pub const device_joystick = protocol.device_joystick;
|
||||||
|
pub const device_all = protocol.device_all;
|
||||||
|
|
||||||
|
/// Look up the input service, retrying while it is still coming up. Both a subscriber and
|
||||||
|
/// a source race the service's registration at boot, so both wait for it here rather than
|
||||||
|
/// failing. Returns the service endpoint handle, or null if it never appears.
|
||||||
|
fn lookupService() ?ipc.Handle {
|
||||||
|
var attempts: usize = 0;
|
||||||
|
while (attempts < 100) : (attempts += 1) {
|
||||||
|
if (ipc.lookup(.input)) |handle| return handle;
|
||||||
|
system.sleep(50);
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- subscribing ------------------------------------------------------------
|
||||||
|
|
||||||
|
/// A subscription to the input service: our own endpoint, which the service pushes events
|
||||||
|
/// to. `next` returns each event as a tagged `InputEvent`; use `asKeyboard`/`asMouse`/
|
||||||
|
/// `asJoystick` to decode. Created with `subscribe`/`subscribeAll`; for a single device
|
||||||
|
/// class prefer the typed helpers (`subscribeKeyboard`, ...), which return decoded events.
|
||||||
|
pub const Subscriber = struct {
|
||||||
|
/// The endpoint the service delivers events to (created and owned by us; its handle
|
||||||
|
/// was handed to the service as a capability at subscribe time).
|
||||||
|
endpoint: ipc.Handle,
|
||||||
|
receive: [protocol.event_size]u8 = undefined,
|
||||||
|
|
||||||
|
/// Block until the next event is pushed, and return it. Events arrive as asynchronous
|
||||||
|
/// buffered messages (`ipc_send` from the service), so nothing is owed in reply — the
|
||||||
|
/// empty reply this issues is a harmless no-op. Returns null for any non-event wake-up
|
||||||
|
/// (there should be none), so callers can loop.
|
||||||
|
pub fn next(self: *Subscriber) ?InputEvent {
|
||||||
|
const got = ipc.replyWait(self.endpoint, &.{}, &self.receive, null);
|
||||||
|
if (!got.isMessage() or got.len < protocol.event_size) return null;
|
||||||
|
return std.mem.bytesToValue(InputEvent, self.receive[0..protocol.event_size]);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Subscribe to the input classes named in `device_mask` (an OR of `device_*`, or
|
||||||
|
/// `device_all`). Creates an endpoint for the service to push to and hands it over as a
|
||||||
|
/// capability. Returns a `Subscriber` to loop `next` on, or null on failure.
|
||||||
|
pub fn subscribe(device_mask: u32) ?Subscriber {
|
||||||
|
const service = lookupService() orelse return null;
|
||||||
|
const endpoint = ipc.createIpcEndpoint() orelse return null;
|
||||||
|
|
||||||
|
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.subscribe), .device_mask = device_mask };
|
||||||
|
var reply: [protocol.reply_size]u8 = undefined;
|
||||||
|
const result = ipc.callCap(service, std.mem.asBytes(&request), &reply, endpoint) catch return null;
|
||||||
|
if (result.len < protocol.reply_size) return null;
|
||||||
|
if (std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status != 0) return null;
|
||||||
|
return .{ .endpoint = endpoint };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Subscribe to every input class (keyboard, mouse, joystick) on one stream.
|
||||||
|
pub fn subscribeAll() ?Subscriber {
|
||||||
|
return subscribe(device_all);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A subscriber filtered to keyboard events, whose `next` returns a decoded `KeyEvent`.
|
||||||
|
pub const KeyboardSubscriber = struct {
|
||||||
|
inner: Subscriber,
|
||||||
|
pub fn next(self: *KeyboardSubscriber) ?KeyEvent {
|
||||||
|
return (self.inner.next() orelse return null).asKeyboard();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A subscriber filtered to mouse events, whose `next` returns a decoded `MouseEvent`.
|
||||||
|
pub const MouseSubscriber = struct {
|
||||||
|
inner: Subscriber,
|
||||||
|
pub fn next(self: *MouseSubscriber) ?MouseEvent {
|
||||||
|
return (self.inner.next() orelse return null).asMouse();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A subscriber filtered to joystick/gamepad events, whose `next` returns a decoded
|
||||||
|
/// `JoystickEvent`.
|
||||||
|
pub const JoystickSubscriber = struct {
|
||||||
|
inner: Subscriber,
|
||||||
|
pub fn next(self: *JoystickSubscriber) ?JoystickEvent {
|
||||||
|
return (self.inner.next() orelse return null).asJoystick();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Subscribe to keyboard events only; `next` returns decoded `KeyEvent`s.
|
||||||
|
pub fn subscribeKeyboard() ?KeyboardSubscriber {
|
||||||
|
return .{ .inner = subscribe(device_keyboard) orelse return null };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Subscribe to mouse events only; `next` returns decoded `MouseEvent`s.
|
||||||
|
pub fn subscribeMouse() ?MouseSubscriber {
|
||||||
|
return .{ .inner = subscribe(device_mouse) orelse return null };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Subscribe to joystick/gamepad events only; `next` returns decoded `JoystickEvent`s.
|
||||||
|
pub fn subscribeJoystick() ?JoystickSubscriber {
|
||||||
|
return .{ .inner = subscribe(device_joystick) orelse return null };
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- publishing -------------------------------------------------------------
|
||||||
|
|
||||||
|
/// A connection to the input service for a source (a keyboard/mouse/joystick driver) that
|
||||||
|
/// publishes events. Each `publish*Event` is a short synchronous call the service answers
|
||||||
|
/// at once; its own fan-out to subscribers is asynchronous, so publishing never blocks on
|
||||||
|
/// a slow subscriber.
|
||||||
|
pub const Publisher = struct {
|
||||||
|
service: ipc.Handle,
|
||||||
|
|
||||||
|
fn publish(self: Publisher, event: InputEvent) bool {
|
||||||
|
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.publish), .event = event };
|
||||||
|
var reply: [protocol.reply_size]u8 = undefined;
|
||||||
|
const len = ipc.call(self.service, std.mem.asBytes(&request), &reply) catch return false;
|
||||||
|
if (len < protocol.reply_size) return false;
|
||||||
|
return std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Broadcast a keyboard event to every subscriber that took keyboard events.
|
||||||
|
pub fn publishKeyboardEvent(self: Publisher, event: KeyEvent) bool {
|
||||||
|
return self.publish(InputEvent.fromKeyboard(event));
|
||||||
|
}
|
||||||
|
/// Broadcast a mouse event to every subscriber that took mouse events.
|
||||||
|
pub fn publishMouseEvent(self: Publisher, event: MouseEvent) bool {
|
||||||
|
return self.publish(InputEvent.fromMouse(event));
|
||||||
|
}
|
||||||
|
/// Broadcast a joystick/gamepad event to every subscriber that took joystick events.
|
||||||
|
pub fn publishJoystickEvent(self: Publisher, event: JoystickEvent) bool {
|
||||||
|
return self.publish(InputEvent.fromJoystick(event));
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Connect to the input service as an event source, waiting for it to come up. Returns a
|
||||||
|
/// `Publisher`, or null if the service never registered.
|
||||||
|
pub fn connectSource() ?Publisher {
|
||||||
|
return .{ .service = lookupService() orelse return null };
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- synthetic scaffolding --------------------------------------------------
|
||||||
|
|
||||||
|
/// Synthetic key events, shared by the demo source and the keyboard driver's placeholder
|
||||||
|
/// stream while real scancode decoding is still a follow-up. `step` rolls through A..E,
|
||||||
|
/// emitting for each key a `key_down`, then a `key_press` carrying the character, then a
|
||||||
|
/// `key_up`. Scaffolding, not wire protocol — hence it lives with the helpers.
|
||||||
|
pub fn syntheticKeyEvent(step: usize) KeyEvent {
|
||||||
|
const Key = struct { code: Keycode, character: u32 };
|
||||||
|
const keys = [_]Key{
|
||||||
|
.{ .code = .a, .character = 'A' },
|
||||||
|
.{ .code = .b, .character = 'B' },
|
||||||
|
.{ .code = .c, .character = 'C' },
|
||||||
|
.{ .code = .d, .character = 'D' },
|
||||||
|
.{ .code = .e, .character = 'E' },
|
||||||
|
};
|
||||||
|
const key = keys[(step / 3) % keys.len];
|
||||||
|
return switch (step % 3) {
|
||||||
|
0 => .{ .kind = @intFromEnum(EventKind.key_down), .keycode = @intFromEnum(key.code), .character = 0, .modifiers = 0 },
|
||||||
|
1 => .{ .kind = @intFromEnum(EventKind.key_press), .keycode = @intFromEnum(key.code), .character = key.character, .modifiers = 0 },
|
||||||
|
else => .{ .kind = @intFromEnum(EventKind.key_up), .keycode = @intFromEnum(key.code), .character = 0, .modifiers = 0 },
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Synthetic mouse events (placeholder until real PS/2 packet decoding). `step` alternates
|
||||||
|
/// a small diagonal motion with a left-button click.
|
||||||
|
pub fn syntheticMouseEvent(step: usize) MouseEvent {
|
||||||
|
return switch (step % 3) {
|
||||||
|
0 => .{ .kind = @intFromEnum(MouseEventKind.motion), .button = 0, .dx = 1, .dy = 1, .scroll_x = 0, .scroll_y = 0, .buttons = 0 },
|
||||||
|
1 => .{ .kind = @intFromEnum(MouseEventKind.button_down), .button = protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = protocol.mouse_button_left },
|
||||||
|
else => .{ .kind = @intFromEnum(MouseEventKind.button_up), .button = protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = 0 },
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Synthetic joystick/gamepad events (placeholder until a real controller driver). `step`
|
||||||
|
/// sweeps axis 0 and toggles button 0.
|
||||||
|
pub fn syntheticJoystickEvent(step: usize) JoystickEvent {
|
||||||
|
return switch (step % 3) {
|
||||||
|
0 => .{ .kind = @intFromEnum(JoystickEventKind.axis), .control = 0, .value = 16384, .buttons = 0 },
|
||||||
|
1 => .{ .kind = @intFromEnum(JoystickEventKind.button_down), .control = 0, .value = 0, .buttons = 1 },
|
||||||
|
else => .{ .kind = @intFromEnum(JoystickEventKind.button_up), .control = 0, .value = 0, .buttons = 0 },
|
||||||
|
};
|
||||||
|
}
|
||||||
@@ -0,0 +1,194 @@
|
|||||||
|
//! User-space IPC helpers over the kernel's synchronous IPC syscalls. A client
|
||||||
|
//! `call`s an endpoint (send + block for reply); the VFS server and drivers are
|
||||||
|
//! reached this way. The server side (`replyWait`, which returns two values) is
|
||||||
|
//! added with the first server binary.
|
||||||
|
|
||||||
|
const abi = @import("abi");
|
||||||
|
const sc = @import("system-call.zig");
|
||||||
|
|
||||||
|
/// A small-int handle into the calling process's handle table.
|
||||||
|
pub const Handle = usize;
|
||||||
|
|
||||||
|
/// A fixed-size, register-friendly message payload. Server protocols (VFS, driver)
|
||||||
|
/// layer their own wire format on top of the bytes a call carries.
|
||||||
|
pub const Message = extern struct {
|
||||||
|
tag: u64 = 0,
|
||||||
|
a: u64 = 0,
|
||||||
|
b: u64 = 0,
|
||||||
|
c: u64 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Whether a system_call return value is a wrapped -errno (lands in the top page).
|
||||||
|
inline fn failed(r: usize) bool {
|
||||||
|
return r > ~@as(usize, 0) - 4095;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Create a new endpoint owned by this process; returns its handle.
|
||||||
|
pub fn createIpcEndpoint() ?Handle {
|
||||||
|
const r = sc.systemCall0(.create_ipc_endpoint);
|
||||||
|
return if (failed(r)) null else r;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Publish endpoint `h` under a well-known service id so other processes find it.
|
||||||
|
pub fn register(id: abi.ServiceId, h: Handle) bool {
|
||||||
|
return !failed(sc.systemCall2(.ipc_register, @intFromEnum(id), h));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Find the endpoint published under `id`, installing a handle to it in this
|
||||||
|
/// process.
|
||||||
|
pub fn lookup(id: abi.ServiceId) ?Handle {
|
||||||
|
const r = sc.systemCall1(.ipc_lookup, @intFromEnum(id));
|
||||||
|
return if (failed(r)) null else r;
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const CallError = error{Failed};
|
||||||
|
|
||||||
|
/// The result of a capability-passing `callCap`: the reply length, and the handle of
|
||||||
|
/// an endpoint the server sent back (e.g. a per-device channel), or null.
|
||||||
|
pub const Reply = struct {
|
||||||
|
len: usize,
|
||||||
|
cap: ?Handle,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Send `message` to endpoint `h` and block until the server replies into `reply`,
|
||||||
|
/// optionally handing the server a capability (`send_cap`) and receiving one back.
|
||||||
|
/// This is the class-driver "open" primitive: call a bus with `send_cap = null`, get a
|
||||||
|
/// private per-device endpoint back in `.cap`. Two return values (reply length in rax,
|
||||||
|
/// received handle in r8) need a hand-written stub — r8 is read-write (in: reply
|
||||||
|
/// capacity, arg #4; out: the received handle).
|
||||||
|
pub fn callCap(h: Handle, message: []const u8, reply: []u8, send_cap: ?Handle) CallError!Reply {
|
||||||
|
var rax: usize = undefined;
|
||||||
|
var r8: usize = reply.len; // in: reply capacity (arg #4); out: received capability handle
|
||||||
|
asm volatile ("syscall"
|
||||||
|
: [rax] "={rax}" (rax),
|
||||||
|
[r8] "+{r8}" (r8),
|
||||||
|
: [n] "{rax}" (@intFromEnum(abi.SystemCall.ipc_call)),
|
||||||
|
[a0] "{rdi}" (h),
|
||||||
|
[a1] "{rsi}" (@intFromPtr(message.ptr)),
|
||||||
|
[a2] "{rdx}" (message.len),
|
||||||
|
[a3] "{r10}" (@intFromPtr(reply.ptr)),
|
||||||
|
[a5] "{r9}" (send_cap orelse abi.no_cap),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
if (failed(rax)) return error.Failed;
|
||||||
|
return .{ .len = rax, .cap = if (r8 == abi.no_cap) null else r8 };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Send `message` to endpoint `h` and block until the server replies into `reply`.
|
||||||
|
/// Returns the reply length. The common case: no capability passed either way.
|
||||||
|
pub fn call(h: Handle, message: []const u8, reply: []u8) CallError!usize {
|
||||||
|
return (try callCap(h, message, reply, null)).len;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Post `message` to endpoint `h`'s asynchronous queue and return immediately — no
|
||||||
|
/// rendezvous, no reply, no blocking. The receiver picks it up through `replyWait` as a
|
||||||
|
/// buffered message (`Received.isMessage`). Unlike `call`, this **cannot hang on a dead
|
||||||
|
/// or slow peer**, which is why a broadcaster (the input service) delivers events this
|
||||||
|
/// way. The payload must fit an endpoint slot (64 bytes); a full queue drops the oldest
|
||||||
|
/// message. Returns false on failure (bad handle, oversized payload, bad buffer).
|
||||||
|
pub fn send(h: Handle, message: []const u8) bool {
|
||||||
|
return !failed(sc.systemCall3(.ipc_send, h, @intFromPtr(message.ptr), message.len));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Set in `Received.badge` when what arrived is an asynchronous notification — a
|
||||||
|
/// bound device interrupt — rather than a client's message. The low bits carry the
|
||||||
|
/// GSI. See `isNotification`.
|
||||||
|
pub const notify_badge_bit: u64 = abi.notify_badge_bit;
|
||||||
|
|
||||||
|
/// Set alongside `notify_badge_bit` when the notification is a **signal** — the
|
||||||
|
/// lifecycle vocabulary of docs/process-lifecycle.md, delivered to the endpoint
|
||||||
|
/// nominated with `process.bindSignals`. Decode with `process.signalsFrom`.
|
||||||
|
pub const notify_signal_bit: u64 = abi.notify_signal_bit;
|
||||||
|
|
||||||
|
/// Set alongside `notify_badge_bit` when the notification is a **one-shot timer**
|
||||||
|
/// landing (`system.timerOnce`).
|
||||||
|
pub const notify_timer_bit: u64 = abi.notify_timer_bit;
|
||||||
|
|
||||||
|
/// Set alongside `notify_badge_bit` when the notification is a **child-exit
|
||||||
|
/// notice** — a process this one spawned (with an exit endpoint) has ended —
|
||||||
|
/// rather than a device interrupt. The low bits carry the child's process id.
|
||||||
|
pub const notify_exit_bit: u64 = abi.notify_exit_bit;
|
||||||
|
|
||||||
|
/// Set alongside `notify_badge_bit` when the wake-up is a **buffered message** — a payload
|
||||||
|
/// posted with `send` (`ipc_send`) — rather than a bare device interrupt or child-exit
|
||||||
|
/// notice. The payload is in the `replyWait` receive buffer (`Received.len` bytes); the
|
||||||
|
/// low bits of the badge carry the sender's task id. See `Received.isMessage`.
|
||||||
|
pub const notify_message_bit: u64 = abi.notify_message_bit;
|
||||||
|
|
||||||
|
/// The result of a `replyWait`: the request length, the sender's badge (a task id, or
|
||||||
|
/// an IRQ notification if the high bit is set), and any capability the request carried.
|
||||||
|
pub const Received = struct {
|
||||||
|
len: usize,
|
||||||
|
badge: u64,
|
||||||
|
cap: ?Handle,
|
||||||
|
|
||||||
|
/// True if this wake-up was an asynchronous notification (a device interrupt
|
||||||
|
/// or a child-exit notice), not a client request. An event loop branches on
|
||||||
|
/// this; there is no reply owed on the notification path.
|
||||||
|
pub fn isNotification(self: Received) bool {
|
||||||
|
return self.badge & notify_badge_bit != 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// True if this wake-up tells of a supervised child's end — the notification
|
||||||
|
/// requested by passing an exit endpoint to `system.spawnSupervised`.
|
||||||
|
pub fn isChildExit(self: Received) bool {
|
||||||
|
return self.isNotification() and self.badge & notify_exit_bit != 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// True if this wake-up is a **buffered message** posted with `send` (`ipc_send`):
|
||||||
|
/// there is a payload in the receive buffer (`self.len` bytes) and no reply is owed.
|
||||||
|
/// The subscriber side of a broadcast branches on this.
|
||||||
|
pub fn isMessage(self: Received) bool {
|
||||||
|
return self.isNotification() and self.badge & notify_message_bit != 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The task id of whoever posted a buffered message, meaningful only when
|
||||||
|
/// Whether this arrival is a signal notification — decode the set with
|
||||||
|
/// `process.signalsFrom(badge)`.
|
||||||
|
pub fn isSignal(self: Received) bool {
|
||||||
|
return self.isNotification() and self.badge & notify_signal_bit != 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether this arrival is a one-shot timer landing (`system.timerOnce`).
|
||||||
|
pub fn isTimer(self: Received) bool {
|
||||||
|
return self.isNotification() and self.badge & notify_timer_bit != 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `isMessage`. (The badge's low bits, with the three high marker bits masked off.)
|
||||||
|
pub fn senderTaskId(self: Received) u32 {
|
||||||
|
return @intCast(self.badge & ~(notify_badge_bit | notify_exit_bit | notify_message_bit));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The interrupt source (a GSI), meaningful only when `isNotification` and
|
||||||
|
/// not `isChildExit`.
|
||||||
|
pub fn source(self: Received) u64 {
|
||||||
|
return self.badge & ~notify_badge_bit;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The ended child's process id, meaningful only when `isChildExit`.
|
||||||
|
pub fn childProcessId(self: Received) u32 {
|
||||||
|
return @intCast(self.badge & ~(notify_badge_bit | notify_exit_bit));
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Server side of IPC_ReplyWait: deliver `reply` to the client last received (if any,
|
||||||
|
/// optionally handing it `send_cap`), then block until the next request arrives in
|
||||||
|
/// `receive`. Returns its length, the sender badge, and any capability the request
|
||||||
|
/// carried (in `.cap`). Three return values — length in rax, badge in rdx, received
|
||||||
|
/// handle in r8 — so it needs a hand-written stub: rdx is read-write (in: reply length,
|
||||||
|
/// arg #3; out: badge) and r8 is read-write (in: receive capacity, arg #4; out: handle).
|
||||||
|
pub fn replyWait(h: Handle, reply: []const u8, receive: []u8, send_cap: ?Handle) Received {
|
||||||
|
var rax: usize = undefined;
|
||||||
|
var rdx: usize = reply.len; // in: reply_len (arg #3); out: badge
|
||||||
|
var r8: usize = receive.len; // in: receive capacity (arg #4); out: received capability handle
|
||||||
|
asm volatile ("syscall"
|
||||||
|
: [rax] "={rax}" (rax),
|
||||||
|
[rdx] "+{rdx}" (rdx),
|
||||||
|
[r8] "+{r8}" (r8),
|
||||||
|
: [n] "{rax}" (@intFromEnum(abi.SystemCall.ipc_reply_wait)),
|
||||||
|
[a0] "{rdi}" (h),
|
||||||
|
[a1] "{rsi}" (@intFromPtr(reply.ptr)),
|
||||||
|
[a3] "{r10}" (@intFromPtr(receive.ptr)),
|
||||||
|
[a5] "{r9}" (send_cap orelse abi.no_cap),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
return .{ .len = rax, .badge = rdx, .cap = if (r8 == abi.no_cap) null else r8 };
|
||||||
|
}
|
||||||
@@ -0,0 +1,137 @@
|
|||||||
|
//! Process-level runtime types: what a user program receives at entry (`Init`,
|
||||||
|
//! the argv contract) and the process end of the lifecycle
|
||||||
|
//! (docs/process-lifecycle.md) — today the exit reason a supervisor reads to
|
||||||
|
//! decide restart; signals and the stop sequence land here with M17.4. Mirrors
|
||||||
|
//! the spirit of `std.process.Init.Minimal` in danos terms — std's `Args` holds
|
||||||
|
//! no data on freestanding targets, so the type is danos's own.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const sc = @import("system-call.zig");
|
||||||
|
const ipc = @import("ipc.zig");
|
||||||
|
const system = @import("system.zig");
|
||||||
|
|
||||||
|
/// Everything a program receives at entry. Passed to
|
||||||
|
/// `pub fn main(init: runtime.process.Init)`; programs that need nothing keep
|
||||||
|
/// `pub fn main() void`. An `environment` field is added here once the kernel
|
||||||
|
/// passes a non-empty envp (today it is always empty — see docs/sysv.md).
|
||||||
|
pub const Init = struct {
|
||||||
|
arguments: Arguments,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The process arguments (argc/argv), parsed from the kernel-built System V
|
||||||
|
/// entry block. The bytes live in the entry block at the top of the stack page,
|
||||||
|
/// NUL-terminated, valid for the process's lifetime.
|
||||||
|
pub const Arguments = struct {
|
||||||
|
/// argc — at least 1: argument 0 is the path or name this binary was
|
||||||
|
/// spawned as.
|
||||||
|
count: usize,
|
||||||
|
/// The argv pointers in the entry block (NULL-terminated after `count`
|
||||||
|
/// entries).
|
||||||
|
vector: [*]const [*:0]const u8,
|
||||||
|
|
||||||
|
/// Argument `index` (0 = the program's own path/name), or null if out of
|
||||||
|
/// range.
|
||||||
|
pub fn get(arguments: Arguments, index: usize) ?[:0]const u8 {
|
||||||
|
if (index >= arguments.count) return null;
|
||||||
|
return std.mem.span(arguments.vector[index]);
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn iterate(arguments: Arguments) Iterator {
|
||||||
|
return .{ .arguments = arguments };
|
||||||
|
}
|
||||||
|
|
||||||
|
pub const Iterator = struct {
|
||||||
|
arguments: Arguments,
|
||||||
|
index: usize = 0,
|
||||||
|
|
||||||
|
pub fn next(iterator: *Iterator) ?[:0]const u8 {
|
||||||
|
const argument = iterator.arguments.get(iterator.index) orelse return null;
|
||||||
|
iterator.index += 1;
|
||||||
|
return argument;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
/// How a process ended — what a supervisor's restart policy reads: a clean exit
|
||||||
|
/// meant to stop, a fault wants a restart with backoff, killed means the
|
||||||
|
/// supervisor did it itself (docs/process-lifecycle.md).
|
||||||
|
pub const ExitReason = abi.ExitReason;
|
||||||
|
|
||||||
|
/// How dead child `id` ended. Ask after the exit notification arrives — the
|
||||||
|
/// kernel records the reason before it posts the notification, so this never
|
||||||
|
/// races it. Returns null for an id that never lived, is still alive, was
|
||||||
|
/// evicted from the kernel's bounded record, or is not this process's child
|
||||||
|
/// (the same authority gate as `kill`).
|
||||||
|
pub fn exitReason(id: u32) ?ExitReason {
|
||||||
|
const r = sc.systemCall1(.process_exit_reason, id);
|
||||||
|
if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno
|
||||||
|
return @enumFromInt(r);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The signal vocabulary (docs/process-lifecycle.md): POSIX's concepts, danos's
|
||||||
|
/// names, message delivery. A signal is a one-way coalescing statement — never a
|
||||||
|
/// question (liveness is the zero-length ping call) and never kill (that is
|
||||||
|
/// `system.kill`, unhandleable by definition).
|
||||||
|
pub const Signal = abi.Signal;
|
||||||
|
|
||||||
|
/// The coalesced set of signals one notification delivered: two pending
|
||||||
|
/// terminates arrive as one. Decode a received badge with `signalsFrom`.
|
||||||
|
pub const SignalSet = struct {
|
||||||
|
pending: u32,
|
||||||
|
|
||||||
|
pub fn has(set: SignalSet, signal: Signal) bool {
|
||||||
|
return set.pending & (@as(u32, 1) << @intFromEnum(signal)) != 0;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Nominate `endpoint` as this process's signal endpoint. Signals posted while
|
||||||
|
/// unbound have pended; they are delivered immediately on bind, coalesced.
|
||||||
|
pub fn bindSignals(endpoint: usize) bool {
|
||||||
|
return sc.systemCall1(.signal_bind, endpoint) == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode a received badge into the signals it delivered, or null if it is not
|
||||||
|
/// a signal notification.
|
||||||
|
pub fn signalsFrom(badge: u64) ?SignalSet {
|
||||||
|
if (badge & abi.notify_badge_bit == 0 or badge & abi.notify_signal_bit == 0) return null;
|
||||||
|
return .{ .pending = @truncate(badge & ~(abi.notify_badge_bit | abi.notify_signal_bit)) };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Post `signal` to child `id` (or to yourself). Supervisor-gated, like kill;
|
||||||
|
/// non-blocking, always — a statement, not a conversation.
|
||||||
|
pub fn sendSignal(id: u32, signal: Signal) bool {
|
||||||
|
return sc.systemCall2(.process_signal, id, @intFromEnum(signal)) == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The standard stop sequence (docs/process-lifecycle.md): terminate, wait up to
|
||||||
|
/// `deadline_ms` for the exit notification on `exit_endpoint` (the endpoint the
|
||||||
|
/// child was spawned with), then kill. Any *other* notifications arriving on
|
||||||
|
/// that endpoint while stopping are consumed and dropped — a supervisor with
|
||||||
|
/// concurrent traffic implements the same sequence inside its own event loop
|
||||||
|
/// (arm `system.timerOnce`, keep serving) instead of calling this.
|
||||||
|
pub fn stop(id: u32, deadline_ms: u64, exit_endpoint: usize) void {
|
||||||
|
_ = sendSignal(id, .terminate);
|
||||||
|
_ = system.timerOnce(exit_endpoint, deadline_ms);
|
||||||
|
var receive: [8]u8 = undefined;
|
||||||
|
while (true) {
|
||||||
|
const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null);
|
||||||
|
if (got.isChildExit() and got.childProcessId() == id) return;
|
||||||
|
if (got.isTimer()) break; // the deadline passed first — escalate
|
||||||
|
}
|
||||||
|
_ = system.kill(id);
|
||||||
|
while (true) {
|
||||||
|
const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null);
|
||||||
|
if (got.isChildExit() and got.childProcessId() == id) return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Subscribe `endpoint` to published exit events: every process death posts an
|
||||||
|
/// asynchronous notification with the same badge encoding as a supervisor's exit
|
||||||
|
/// notice (decode with `ipc.Received.isChildExit`/`childProcessId`). For stateful
|
||||||
|
/// services: release what the dead client held — file handles, subscriptions —
|
||||||
|
/// because a service must never depend on clients cleaning up after themselves
|
||||||
|
/// (docs/process-lifecycle.md). Ungated, like `system.processes`.
|
||||||
|
pub fn subscribeExits(endpoint: usize) bool {
|
||||||
|
return sc.systemCall1(.process_subscribe, endpoint) == 0;
|
||||||
|
}
|
||||||
@@ -0,0 +1,46 @@
|
|||||||
|
//! danos user-space runtime library — a nascent libc. Every user binary (init,
|
||||||
|
//! and later the VFS server + device drivers) imports this as `@import("runtime")`:
|
||||||
|
//! system_call wrappers, the C-convention heap, IPC helpers, and the process start
|
||||||
|
//! shim. It is compiled into each binary (inheriting its `.large` code model and
|
||||||
|
//! freestanding target), so all user programs share one implementation.
|
||||||
|
//!
|
||||||
|
//! A user binary needs three lines:
|
||||||
|
//! const runtime = @import("runtime");
|
||||||
|
//! pub const panic = runtime.panic;
|
||||||
|
//! comptime { _ = &runtime.start._start; } // pull the entry shim in
|
||||||
|
//! and a `pub fn main() void` or `pub fn main(init: runtime.process.Init) void`
|
||||||
|
//! (arguments arrive via `init`).
|
||||||
|
|
||||||
|
pub const system = @import("system.zig");
|
||||||
|
pub const heap = @import("heap.zig");
|
||||||
|
pub const ipc = @import("ipc.zig");
|
||||||
|
pub const start = @import("start.zig");
|
||||||
|
/// The VFS wire protocol (shared with the VFS server).
|
||||||
|
pub const vfs_protocol = @import("vfs-protocol");
|
||||||
|
|
||||||
|
/// The device-manager protocol: hello + tree reports (docs/device-manager.md).
|
||||||
|
pub const device_manager_protocol = @import("device-manager-protocol");
|
||||||
|
/// Keyboard-event listening (subscribe/next) and broadcasting (publish), over the input
|
||||||
|
/// service. See library/runtime/input.zig and system/services/input/.
|
||||||
|
pub const input = @import("input.zig");
|
||||||
|
/// The input wire protocol (shared with the input service and its clients).
|
||||||
|
pub const input_protocol = @import("input-protocol");
|
||||||
|
/// POSIX-style file API: open/read/write/lseek/stat/close.
|
||||||
|
/// C stdio: fopen/fread/fwrite/fseek/ftell/fclose over unistd.
|
||||||
|
/// Device access for drivers: enumerate/claim/mmioMap.
|
||||||
|
pub const device = @import("device.zig");
|
||||||
|
/// DMA-capable memory for drivers: contiguous, pinned, uncacheable buffers.
|
||||||
|
pub const dma = @import("dma.zig");
|
||||||
|
|
||||||
|
/// Re-exported so a user binary can `pub const panic = runtime.panic;`.
|
||||||
|
pub const panic = start.panic;
|
||||||
|
|
||||||
|
/// Process entry types: the `Init` handed to `main`, and its `Arguments`.
|
||||||
|
pub const process = @import("process.zig");
|
||||||
|
|
||||||
|
/// The service harness: one replyWait loop folding requests, signals, and
|
||||||
|
/// notifications into callbacks (docs/process-lifecycle.md).
|
||||||
|
pub const service = @import("service.zig");
|
||||||
|
|
||||||
|
/// The heap as a `std.mem.Allocator`, for Zig `std` containers in user code.
|
||||||
|
pub const allocator = heap.allocator;
|
||||||
@@ -0,0 +1,83 @@
|
|||||||
|
//! The service harness (docs/process-lifecycle.md): one replyWait loop that
|
||||||
|
//! folds protocol requests, signals, and subscribed notifications into
|
||||||
|
//! callbacks — so the lifecycle contract ("answers ping, exits on terminate")
|
||||||
|
//! is satisfied by construction and a service author writes domain logic only.
|
||||||
|
//! Nothing is asynchronous inside the process: a callback runs at a point the
|
||||||
|
//! loop chose, never on a hijacked stack — the whole reason signals are
|
||||||
|
//! messages.
|
||||||
|
//!
|
||||||
|
//! The liveness probe: a **zero-length request is the universal ping**, answered
|
||||||
|
//! with a zero-length reply by the harness itself. No protocol's requests start
|
||||||
|
//! at length zero, so the encoding cannot collide, and there is nothing for a
|
||||||
|
//! service author to implement — a wedged service simply fails to answer, which
|
||||||
|
//! is the diagnosis (see docs/ipc.md).
|
||||||
|
|
||||||
|
const abi = @import("abi");
|
||||||
|
const ipc = @import("ipc.zig");
|
||||||
|
const process = @import("process.zig");
|
||||||
|
|
||||||
|
pub const Callbacks = struct {
|
||||||
|
/// Called once with the service's endpoint before the loop starts — the
|
||||||
|
/// place to subscribe to exit events, bind IRQs, or announce readiness.
|
||||||
|
/// Return false to abort startup (the process exits).
|
||||||
|
init: ?*const fn (endpoint: ipc.Handle) bool = null,
|
||||||
|
/// One protocol request from `sender` (a task id): write the reply into
|
||||||
|
/// `reply`, return its length. `capability` is the handle the request
|
||||||
|
/// carried, if any (M13 cap passing — how a subscriber hands over its
|
||||||
|
/// endpoint). The zero-length ping never reaches this.
|
||||||
|
on_message: *const fn (message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize,
|
||||||
|
/// A notification that is not a signal — a subscribed exit event, a bound
|
||||||
|
/// IRQ, a timer landing. The raw badge; decode with the ipc helpers.
|
||||||
|
on_notification: ?*const fn (badge: u64) void = null,
|
||||||
|
/// The reload signal. Default: ignored.
|
||||||
|
on_reload: ?*const fn () void = null,
|
||||||
|
/// The terminate signal, called before the loop returns. The clean exit is
|
||||||
|
/// the return itself — never put *necessary* work here (iron rule 1: a kill
|
||||||
|
/// arrives with no warning; this is for graceful extras only).
|
||||||
|
on_terminate: ?*const fn () void = null,
|
||||||
|
/// Publish the endpoint under a well-known service id at startup.
|
||||||
|
service: ?abi.ServiceId = null,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Run the service: create and (optionally) register the endpoint, bind signals
|
||||||
|
/// to it, call `init`, then serve until `terminate` arrives — at which point the
|
||||||
|
/// loop returns and main's return is the clean exit the supervisor reads as
|
||||||
|
/// `ExitReason.exited`. `maximum_message` sizes the receive and reply buffers
|
||||||
|
/// (a service passes its protocol's message maximum).
|
||||||
|
pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
|
||||||
|
const endpoint = ipc.createIpcEndpoint() orelse return;
|
||||||
|
if (callbacks.service) |id| {
|
||||||
|
if (!ipc.register(id, endpoint)) return;
|
||||||
|
}
|
||||||
|
_ = process.bindSignals(endpoint);
|
||||||
|
if (callbacks.init) |initialise| {
|
||||||
|
if (!initialise(endpoint)) return;
|
||||||
|
}
|
||||||
|
|
||||||
|
var reply_buffer: [maximum_message]u8 = undefined;
|
||||||
|
var reply_len: usize = 0;
|
||||||
|
var receive: [maximum_message]u8 = undefined;
|
||||||
|
while (true) {
|
||||||
|
const got = ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive, null);
|
||||||
|
if (got.isNotification()) {
|
||||||
|
reply_len = 0; // nothing owed for a notification
|
||||||
|
if (process.signalsFrom(got.badge)) |signals| {
|
||||||
|
if (signals.has(.reload)) {
|
||||||
|
if (callbacks.on_reload) |onReload| onReload();
|
||||||
|
}
|
||||||
|
if (signals.has(.terminate)) {
|
||||||
|
if (callbacks.on_terminate) |onTerminate| onTerminate();
|
||||||
|
return; // the loop's return IS the clean exit
|
||||||
|
}
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (callbacks.on_notification) |onNotification| onNotification(got.badge);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (got.len == 0) {
|
||||||
|
reply_len = 0; // the universal ping: a zero-length reply, from the harness
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
reply_len = callbacks.on_message(receive[0..got.len], &reply_buffer, got.senderTaskId(), got.cap);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,86 @@
|
|||||||
|
//! The user-space process entry shim. Every user binary roots `_start` here (via
|
||||||
|
//! `entry = _start` in build.zig) and forces this file to be analysed with
|
||||||
|
//! `comptime { _ = &runtime.start._start; }`, so the whole runtime is linked in.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const system = @import("system.zig");
|
||||||
|
const process = @import("process.zig");
|
||||||
|
|
||||||
|
/// The kernel enters at `_start` with rsp 16-aligned, pointing at the System V
|
||||||
|
/// process-entry block it built: argc, argv pointers, NULL, envp terminator, the
|
||||||
|
/// auxiliary vector, then the strings (see system/kernel/process.zig,
|
||||||
|
/// `buildEntryStack`). Capture that address in rdi — the first SysV argument —
|
||||||
|
/// before `call` disturbs the stack; the call's pushed return address also puts
|
||||||
|
/// rsp ≡ 8 (mod 16), satisfying the ABI before any Zig frame runs. The `ud2` is a
|
||||||
|
/// safety net if `rt_start` ever returns.
|
||||||
|
pub export fn _start() callconv(.naked) noreturn {
|
||||||
|
asm volatile (
|
||||||
|
\\mov %%rsp, %%rdi
|
||||||
|
\\call rt_start
|
||||||
|
\\ud2
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The first Zig frame, entered with `stack` pointing at the kernel-built entry
|
||||||
|
/// block. Build the `process.Init` from it and dispatch to the program's `main`,
|
||||||
|
/// whose signature is inspected at comptime. The heap is lazy (first alloc grows
|
||||||
|
/// it), so there is no other runtime init to order here.
|
||||||
|
export fn rt_start(stack: [*]const u64) callconv(.c) noreturn {
|
||||||
|
const init: process.Init = .{ .arguments = .{
|
||||||
|
.count = stack[0],
|
||||||
|
.vector = @ptrCast(stack + 1),
|
||||||
|
} };
|
||||||
|
system.exit(callMain(init));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Comptime-dispatch on root.main's signature, in the spirit of std's start.zig:
|
||||||
|
/// zero parameters or one `process.Init`; returns void, noreturn, u8, !void, or !u8.
|
||||||
|
fn callMain(init: process.Init) u8 {
|
||||||
|
const root = @import("root"); // the user binary's root source file
|
||||||
|
const main_information = @typeInfo(@TypeOf(root.main)).@"fn";
|
||||||
|
|
||||||
|
const call_arguments = switch (main_information.params.len) {
|
||||||
|
0 => .{},
|
||||||
|
1 => arguments: {
|
||||||
|
const Parameter = main_information.params[0].type orelse
|
||||||
|
@compileError("main's parameter must be runtime.process.Init (not anytype)");
|
||||||
|
if (Parameter != process.Init)
|
||||||
|
@compileError("main's parameter must be runtime.process.Init, found " ++ @typeName(Parameter));
|
||||||
|
break :arguments .{init};
|
||||||
|
},
|
||||||
|
else => @compileError("main takes no parameters or a single runtime.process.Init"),
|
||||||
|
};
|
||||||
|
|
||||||
|
const ReturnType = main_information.return_type.?;
|
||||||
|
switch (@typeInfo(ReturnType)) {
|
||||||
|
.noreturn => @call(.auto, root.main, call_arguments),
|
||||||
|
.void => {
|
||||||
|
@call(.auto, root.main, call_arguments);
|
||||||
|
return 0;
|
||||||
|
},
|
||||||
|
.int => {
|
||||||
|
if (ReturnType != u8)
|
||||||
|
@compileError("main's integer return type must be u8, found " ++ @typeName(ReturnType));
|
||||||
|
return @call(.auto, root.main, call_arguments);
|
||||||
|
},
|
||||||
|
.error_union => {
|
||||||
|
const payload = @call(.auto, root.main, call_arguments) catch |err| {
|
||||||
|
var buffer: [128]u8 = undefined;
|
||||||
|
const line = std.fmt.bufPrint(&buffer, "main returned error: {s}\n", .{@errorName(err)}) catch "main returned an error\n";
|
||||||
|
_ = system.write(line);
|
||||||
|
return 1; // distinct from panic's 127
|
||||||
|
};
|
||||||
|
if (@TypeOf(payload) == void) return 0;
|
||||||
|
if (@TypeOf(payload) == u8) return payload;
|
||||||
|
@compileError("main's error-union payload must be void or u8, found " ++ @typeName(@TypeOf(payload)));
|
||||||
|
},
|
||||||
|
else => @compileError("main must return void, noreturn, u8, !void, or !u8, found " ++ @typeName(ReturnType)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// No runtime to unwind into — report a panic as a nonzero exit code.
|
||||||
|
pub const panic = std.debug.FullPanic(struct {
|
||||||
|
fn panic(_: []const u8, _: ?usize) noreturn {
|
||||||
|
system.exit(127);
|
||||||
|
}
|
||||||
|
}.panic);
|
||||||
@@ -0,0 +1,67 @@
|
|||||||
|
//! Raw `system_call` instruction wrappers for user space — one per arity.
|
||||||
|
//!
|
||||||
|
//! ABI: number in rax, arguments in rdi, rsi, rdx, r10, r8, r9, result in rax.
|
||||||
|
//! The `system_call` instruction itself clobbers rcx (it holds the return rip) and
|
||||||
|
//! r11 (the saved rflags); the kernel entry stub preserves everything else.
|
||||||
|
//! Note argument #3 goes in **r10, not rcx** — rcx is unavailable across the
|
||||||
|
//! instruction, so the kernel reads the 4th argument from r10.
|
||||||
|
|
||||||
|
const abi = @import("abi");
|
||||||
|
const SystemCall = abi.SystemCall;
|
||||||
|
|
||||||
|
pub inline fn systemCall0(n: SystemCall) usize {
|
||||||
|
return asm volatile ("syscall"
|
||||||
|
: [ret] "={rax}" (-> usize),
|
||||||
|
: [n] "{rax}" (@intFromEnum(n)),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
}
|
||||||
|
|
||||||
|
pub inline fn systemCall1(n: SystemCall, a0: usize) usize {
|
||||||
|
return asm volatile ("syscall"
|
||||||
|
: [ret] "={rax}" (-> usize),
|
||||||
|
: [n] "{rax}" (@intFromEnum(n)),
|
||||||
|
[a0] "{rdi}" (a0),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
}
|
||||||
|
|
||||||
|
pub inline fn systemCall2(n: SystemCall, a0: usize, a1: usize) usize {
|
||||||
|
return asm volatile ("syscall"
|
||||||
|
: [ret] "={rax}" (-> usize),
|
||||||
|
: [n] "{rax}" (@intFromEnum(n)),
|
||||||
|
[a0] "{rdi}" (a0),
|
||||||
|
[a1] "{rsi}" (a1),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
}
|
||||||
|
|
||||||
|
pub inline fn systemCall3(n: SystemCall, a0: usize, a1: usize, a2: usize) usize {
|
||||||
|
return asm volatile ("syscall"
|
||||||
|
: [ret] "={rax}" (-> usize),
|
||||||
|
: [n] "{rax}" (@intFromEnum(n)),
|
||||||
|
[a0] "{rdi}" (a0),
|
||||||
|
[a1] "{rsi}" (a1),
|
||||||
|
[a2] "{rdx}" (a2),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
}
|
||||||
|
|
||||||
|
pub inline fn systemCall4(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: usize) usize {
|
||||||
|
return asm volatile ("syscall"
|
||||||
|
: [ret] "={rax}" (-> usize),
|
||||||
|
: [n] "{rax}" (@intFromEnum(n)),
|
||||||
|
[a0] "{rdi}" (a0),
|
||||||
|
[a1] "{rsi}" (a1),
|
||||||
|
[a2] "{rdx}" (a2),
|
||||||
|
[a3] "{r10}" (a3),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
}
|
||||||
|
|
||||||
|
pub inline fn systemCall5(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: usize, a4: usize) usize {
|
||||||
|
return asm volatile ("syscall"
|
||||||
|
: [ret] "={rax}" (-> usize),
|
||||||
|
: [n] "{rax}" (@intFromEnum(n)),
|
||||||
|
[a0] "{rdi}" (a0),
|
||||||
|
[a1] "{rsi}" (a1),
|
||||||
|
[a2] "{rdx}" (a2),
|
||||||
|
[a3] "{r10}" (a3),
|
||||||
|
[a4] "{r8}" (a4),
|
||||||
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
|
}
|
||||||
@@ -0,0 +1,147 @@
|
|||||||
|
//! Typed system_call surface for user space — thin wrappers over the raw `system_call`
|
||||||
|
//! stubs, one per kernel call. Numbers come from `abi.SystemCall`, the single
|
||||||
|
//! source of truth shared with the kernel dispatcher.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const sc = @import("system-call.zig");
|
||||||
|
|
||||||
|
/// `mmap` protection flags (matching the usual C bit values). Grants are always
|
||||||
|
/// readable+writable today; the kernel does not yet honour finer prot.
|
||||||
|
pub const PROT_READ: usize = abi.prot_read;
|
||||||
|
pub const PROT_WRITE: usize = abi.prot_write;
|
||||||
|
pub const PROT_EXEC: usize = abi.prot_exec;
|
||||||
|
|
||||||
|
/// One `processes` entry — re-exported from the shared ABI so a user program can
|
||||||
|
/// declare its snapshot buffer without importing `abi` itself.
|
||||||
|
pub const ProcessDescriptor = abi.ProcessDescriptor;
|
||||||
|
|
||||||
|
/// Give up the rest of this quantum.
|
||||||
|
pub fn yield() void {
|
||||||
|
_ = sc.systemCall0(.yield);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write raw bytes to the kernel log (a bring-up diagnostic; real output goes
|
||||||
|
/// through the console/VFS later). Returns the byte count, or a wrapped -1.
|
||||||
|
pub fn write(message: []const u8) usize {
|
||||||
|
return sc.systemCall2(.debug_write, @intFromPtr(message.ptr), message.len);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Block the caller for `ms` milliseconds.
|
||||||
|
pub fn sleep(ms: usize) void {
|
||||||
|
_ = sc.systemCall1(.sleep, ms);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Arm a one-shot timer: after `ms` milliseconds the kernel posts a timer
|
||||||
|
/// notification (`ipc.Received.isTimer`) to `endpoint`. The timed wait of
|
||||||
|
/// docs/process-lifecycle.md — a service arms a deadline and keeps serving,
|
||||||
|
/// instead of blocking in sleep; what stop-sequence escalation, hello deadlines,
|
||||||
|
/// and restart backoff are built from.
|
||||||
|
pub fn timerOnce(endpoint: usize, ms: u64) bool {
|
||||||
|
return sc.systemCall2(.timer_bind, endpoint, ms) == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Monotonic nanoseconds since boot — a time source for timeouts and short delays. It
|
||||||
|
/// only ever moves forward. This is *not* wall-clock time (no date, no timezone — that
|
||||||
|
/// is a user-space service layered on top). Deadline pattern for a bounded poll loop:
|
||||||
|
///
|
||||||
|
/// const deadline = clock() + timeout_ns;
|
||||||
|
/// while (clock() < deadline) { ... }
|
||||||
|
pub fn clock() u64 {
|
||||||
|
return @intCast(sc.systemCall0(.clock));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// End the process. Never returns.
|
||||||
|
pub fn exit(code: usize) noreturn {
|
||||||
|
_ = sc.systemCall1(.exit, code);
|
||||||
|
unreachable; // the kernel never returns from exit
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Start the binary bundled in the initial-ramdisk under `name` as a new ring-3
|
||||||
|
/// process, returning the child's process id (or null on failure). The child's
|
||||||
|
/// argv[0] is `name`, and the caller becomes its **supervisor** — the only process
|
||||||
|
/// allowed to `kill` it. This is how a supervisor (the device manager) launches a
|
||||||
|
/// driver it matched — danos-native, not POSIX (a spawn/exec family comes with the
|
||||||
|
/// POSIX layer later).
|
||||||
|
pub fn spawn(name: []const u8) ?u32 {
|
||||||
|
return spawnSupervised(name, &.{}, null);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Like `spawn`, but hands the child command-line arguments: they arrive as
|
||||||
|
/// argv[1..] on its System V entry stack (argv[0] is still `name`).
|
||||||
|
pub fn spawnWithArguments(name: []const u8, arguments: []const []const u8) ?u32 {
|
||||||
|
return spawnSupervised(name, arguments, null);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The full spawn: command-line arguments for the child, and an optional endpoint
|
||||||
|
/// (a handle from `ipc.createIpcEndpoint`) the kernel notifies when the child ends
|
||||||
|
/// — any way it ends: clean exit, fault, or `kill`. The notification arrives via
|
||||||
|
/// `ipc.replyWait` as a badge with the child-exit bit set and the child's id in
|
||||||
|
/// the low bits (`ipc.Received.isChildExit`/`childProcessId`), so one endpoint can
|
||||||
|
/// supervise many children. Arguments are marshalled to the kernel as one
|
||||||
|
/// NUL-separated blob; the combined arguments must fit `blob` (the kernel caps the
|
||||||
|
/// blob at 256 bytes and argc at 8 anyway). Returns the child's process id, or
|
||||||
|
/// null on failure.
|
||||||
|
pub fn spawnSupervised(name: []const u8, arguments: []const []const u8, exit_endpoint: ?usize) ?u32 {
|
||||||
|
var blob: [256]u8 = undefined;
|
||||||
|
var len: usize = 0;
|
||||||
|
for (arguments, 0..) |argument, i| {
|
||||||
|
if (i != 0) {
|
||||||
|
if (len >= blob.len) return null;
|
||||||
|
blob[len] = 0;
|
||||||
|
len += 1;
|
||||||
|
}
|
||||||
|
if (len + argument.len > blob.len) return null;
|
||||||
|
@memcpy(blob[len..][0..argument.len], argument);
|
||||||
|
len += argument.len;
|
||||||
|
}
|
||||||
|
const r = sc.systemCall5(.system_spawn, @intFromPtr(name.ptr), name.len, if (len == 0) 0 else @intFromPtr(&blob), len, exit_endpoint orelse abi.no_cap);
|
||||||
|
if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno
|
||||||
|
return @intCast(r);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Snapshot the process table into `out` (up to its length) and return the total
|
||||||
|
/// number of live processes — which may exceed `out.len`; call again with a larger
|
||||||
|
/// buffer for the full listing. Kernel tasks are included, with an empty name.
|
||||||
|
/// The primitive `ps` is built on.
|
||||||
|
pub fn processes(out: []abi.ProcessDescriptor) usize {
|
||||||
|
return sc.systemCall2(.process_enumerate, @intFromPtr(out.ptr), out.len);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether a process spawned under `name` (its argv[0]) is currently alive.
|
||||||
|
pub fn isProcessRunning(name: []const u8) bool {
|
||||||
|
var table: [32]ProcessDescriptor = undefined;
|
||||||
|
const total = processes(&table);
|
||||||
|
for (table[0..@min(total, table.len)]) |descriptor| {
|
||||||
|
if (std.mem.eql(u8, descriptor.name[0..descriptor.name_length], name)) return true;
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// End process `id`. Only its supervisor — the process that spawned it — may;
|
||||||
|
/// anyone else gets false, as does a stale or unknown id (ids are never reused).
|
||||||
|
/// Delivery is prompt but asynchronous, like a signal: a target caught running on
|
||||||
|
/// another core dies at its next system call or timer tick. True means the kill
|
||||||
|
/// is accepted and irrevocable; the exit notification (if an endpoint was given
|
||||||
|
/// at spawn) confirms completion.
|
||||||
|
pub fn kill(id: u32) bool {
|
||||||
|
return sc.systemCall1(.process_kill, id) == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable
|
||||||
|
/// memory and return the base virtual address. On failure returns a value in the
|
||||||
|
/// top page (see `mmapFailed`). The user heap grows through this call.
|
||||||
|
pub fn mmap(len: usize, prot: usize) usize {
|
||||||
|
return sc.systemCall2(.mmap, len, prot);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Release a range previously handed out by `mmap`.
|
||||||
|
pub fn munmap(base: usize, len: usize) usize {
|
||||||
|
return sc.systemCall2(.munmap, base, len);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether an `mmap` return value is an error (the kernel returns a wrapped
|
||||||
|
/// -errno, which lands in the top page — no real grant base is ever that high).
|
||||||
|
pub inline fn mmapFailed(ret: usize) bool {
|
||||||
|
return ret > ~@as(usize, 0) - 4095;
|
||||||
|
}
|
||||||
@@ -0,0 +1,52 @@
|
|||||||
|
/* Shared link layout for every user binary (init, servers, drivers).
|
||||||
|
*
|
||||||
|
* Linked at a fixed user-space virtual base (set by `image_base` in build.zig,
|
||||||
|
* inside the kernel's user region). Same discipline as the kernel's script:
|
||||||
|
* one PT_LOAD per permission set, every section page-aligned, so the kernel's
|
||||||
|
* user-ELF loader can map each segment with exact W^X permissions. Note the
|
||||||
|
* linker also emits a read-only PT_LOAD covering the ELF headers at the image
|
||||||
|
* base, so the entry point comes from e_entry, not the base address.
|
||||||
|
*/
|
||||||
|
|
||||||
|
ENTRY(_start)
|
||||||
|
|
||||||
|
/* FLAGS bits: 1=X, 2=W, 4=R. */
|
||||||
|
PHDRS {
|
||||||
|
text PT_LOAD FLAGS(5); /* R + X */
|
||||||
|
rodata PT_LOAD FLAGS(4); /* R */
|
||||||
|
data PT_LOAD FLAGS(6); /* R + W */
|
||||||
|
}
|
||||||
|
|
||||||
|
SECTIONS {
|
||||||
|
/* The `.large` code model (needed for the >4 GiB image base) emits code and
|
||||||
|
* data into .ltext/.lrodata/.ldata/.lbss; fold those into the matching
|
||||||
|
* permission segment alongside the normal names. */
|
||||||
|
.text ALIGN(4K) : {
|
||||||
|
*(.text .text.*)
|
||||||
|
*(.ltext .ltext.*)
|
||||||
|
} :text
|
||||||
|
|
||||||
|
.rodata ALIGN(4K) : {
|
||||||
|
*(.rodata .rodata.*)
|
||||||
|
*(.lrodata .lrodata.*)
|
||||||
|
} :rodata
|
||||||
|
|
||||||
|
.data ALIGN(4K) : {
|
||||||
|
*(.data .data.*)
|
||||||
|
*(.ldata .ldata.*)
|
||||||
|
} :data
|
||||||
|
|
||||||
|
/* .bss occupies memory but not file space; the loader zeroes the
|
||||||
|
* filesz..memsz gap. */
|
||||||
|
.bss ALIGN(4K) : {
|
||||||
|
*(.bss .bss.*)
|
||||||
|
*(.lbss .lbss.*)
|
||||||
|
*(COMMON)
|
||||||
|
} :data
|
||||||
|
|
||||||
|
/DISCARD/ : {
|
||||||
|
*(.comment)
|
||||||
|
*(.note .note.*)
|
||||||
|
*(.eh_frame .eh_frame_hdr)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,65 @@
|
|||||||
|
# xkeyboard-config — X11 keyboard layouts, compiled to Zig
|
||||||
|
|
||||||
|
This module turns a physical key (a **USB HID usage**, as the [input module](../../docs/input.md)
|
||||||
|
delivers in `KeyEvent.keycode`) plus a modifier state into a **keysym** and, when the key
|
||||||
|
produces one, a **character** (a Unicode scalar). It is what lets a `keycode` become a
|
||||||
|
`character` — a keymap — without danos shipping an X11 runtime.
|
||||||
|
|
||||||
|
The layout data comes from the X11 [xkeyboard-config](https://gitlab.freedesktop.org/xkeyboard-config/xkeyboard-config)
|
||||||
|
database, but it is **compiled to native Zig at build time** rather than parsed at runtime.
|
||||||
|
`tools/make-xkeyboard-config.py` reads the vendored xkb data and emits pure-data tables into
|
||||||
|
`generated/layouts.zig`; `xkeyboard-config.zig` is the hand-written API over them. This is
|
||||||
|
the same build-time-codegen pattern as `tools/make-initial-ramdisk.py`.
|
||||||
|
|
||||||
|
## Using it
|
||||||
|
|
||||||
|
```zig
|
||||||
|
const xkb = @import("xkeyboard-config");
|
||||||
|
|
||||||
|
const m = xkb.map(xkb.us, key_event.keycode, .{ .shift = shift_held, .caps_lock = caps });
|
||||||
|
if (m.character) |ch| { /* a printable Unicode scalar */ }
|
||||||
|
// m.keysym is always set (e.g. an X11 keysym for Return / F1 / a dead key).
|
||||||
|
|
||||||
|
const layout = xkb.byName("gb") orelse xkb.us; // choose a layout by name
|
||||||
|
for (xkb.all) |l| { /* enumerate available layouts */ }
|
||||||
|
```
|
||||||
|
|
||||||
|
`Modifiers` carries `shift`, `caps_lock`, `level3` (AltGr), and `control`. `map` selects the
|
||||||
|
level from the key's XKB *type* (the generated data) and those modifiers (the policy, in
|
||||||
|
`xkeyboard-config.zig`), so data and semantics stay separable.
|
||||||
|
|
||||||
|
Layouts: **us, gb, de, fr, es, dvorak**.
|
||||||
|
|
||||||
|
## Regenerating
|
||||||
|
|
||||||
|
```sh
|
||||||
|
python3 tools/make-xkeyboard-config.py fetch # network: download + vendor the data subset
|
||||||
|
python3 tools/make-xkeyboard-config.py generate # offline: emit generated/layouts.zig
|
||||||
|
# or, from the build:
|
||||||
|
zig build gen-xkeyboard-config
|
||||||
|
```
|
||||||
|
|
||||||
|
- **`fetch`** downloads the pinned xkeyboard-config release (version + sha256 in the script),
|
||||||
|
resolves the `include` graph for the configured layouts, and vendors *only* the symbols
|
||||||
|
files actually reached (plus `keysymdef.h`, `COPYING`, and `PROVENANCE.md`) into `vendor/`.
|
||||||
|
Run it when bumping the version or adding a layout.
|
||||||
|
- **`generate`** is deterministic and offline — same vendored input produces byte-identical
|
||||||
|
output. To add a layout, extend `TARGETS` (and `HID_TO_NAME` if a new physical key is
|
||||||
|
involved), then re-run `fetch` (to vendor any new includes) and `generate`.
|
||||||
|
|
||||||
|
## Scope
|
||||||
|
|
||||||
|
A pragmatic subset, enough for real Latin-script typing:
|
||||||
|
|
||||||
|
- **Group 1 only** — no multi-layout group switching.
|
||||||
|
- **No dead-key / compose composition** — a dead key returns its keysym with no `character`
|
||||||
|
(composing `´` + `e` → `é` is a higher layer's job).
|
||||||
|
- **Curated key types** — the common XKB types (one/two-level, alphabetic, four-level, …);
|
||||||
|
unmapped keys and unknown types fall back to level-by-shift.
|
||||||
|
- **6 layouts** — extend via `TARGETS` as above.
|
||||||
|
|
||||||
|
## Licensing
|
||||||
|
|
||||||
|
xkeyboard-config and `keysymdef.h` (xorgproto) are MIT/X11 licensed. The vendored data
|
||||||
|
subset carries the upstream `vendor/COPYING`, and `vendor/PROVENANCE.md` records the exact
|
||||||
|
version, source URL, and sha256. The generated tables are a derived work under the same terms.
|
||||||
File diff suppressed because it is too large
Load Diff
+190
@@ -0,0 +1,190 @@
|
|||||||
|
Copyright 1996 by Joseph Moss
|
||||||
|
Copyright (C) 2002-2007 Free Software Foundation, Inc.
|
||||||
|
Copyright (C) Dmitry Golubev <lastguru@mail.ru>, 2003-2004
|
||||||
|
Copyright (C) 2004, Gregory Mokhin <mokhin@bog.msu.ru>
|
||||||
|
Copyright (C) 2006 Erdal Ronahî
|
||||||
|
|
||||||
|
Permission to use, copy, modify, distribute, and sell this software and its
|
||||||
|
documentation for any purpose is hereby granted without fee, provided that
|
||||||
|
the above copyright notice appear in all copies and that both that
|
||||||
|
copyright notice and this permission notice appear in supporting
|
||||||
|
documentation, and that the name of the copyright holder(s) not be used in
|
||||||
|
advertising or publicity pertaining to distribution of the software without
|
||||||
|
specific, written prior permission. The copyright holder(s) makes no
|
||||||
|
representations about the suitability of this software for any purpose. It
|
||||||
|
is provided "as is" without express or implied warranty.
|
||||||
|
|
||||||
|
THE COPYRIGHT HOLDER(S) DISCLAIMS ALL WARRANTIES WITH REGARD TO THIS SOFTWARE,
|
||||||
|
INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS, IN NO
|
||||||
|
EVENT SHALL THE COPYRIGHT HOLDER(S) BE LIABLE FOR ANY SPECIAL, INDIRECT OR
|
||||||
|
CONSEQUENTIAL DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE,
|
||||||
|
DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER
|
||||||
|
TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR
|
||||||
|
PERFORMANCE OF THIS SOFTWARE.
|
||||||
|
|
||||||
|
|
||||||
|
Copyright (c) 1996 Digital Equipment Corporation
|
||||||
|
|
||||||
|
Permission is hereby granted, free of charge, to any person obtaining
|
||||||
|
a copy of this software and associated documentation files (the
|
||||||
|
"Software"), to deal in the Software without restriction, including
|
||||||
|
without limitation the rights to use, copy, modify, merge, publish,
|
||||||
|
distribute, sublicense, and sell copies of the Software, and to
|
||||||
|
permit persons to whom the Software is furnished to do so, subject to
|
||||||
|
the following conditions:
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice shall be included
|
||||||
|
in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS
|
||||||
|
OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||||
|
IN NO EVENT SHALL DIGITAL EQUIPMENT CORPORATION BE LIABLE FOR ANY CLAIM,
|
||||||
|
DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR
|
||||||
|
OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR
|
||||||
|
THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||||
|
|
||||||
|
Except as contained in this notice, the name of the Digital Equipment
|
||||||
|
Corporation shall not be used in advertising or otherwise to promote
|
||||||
|
the sale, use or other dealings in this Software without prior written
|
||||||
|
authorization from Digital Equipment Corporation.
|
||||||
|
|
||||||
|
|
||||||
|
Copyright 1996, 1998 The Open Group
|
||||||
|
|
||||||
|
Permission to use, copy, modify, distribute, and sell this software and its
|
||||||
|
documentation for any purpose is hereby granted without fee, provided that
|
||||||
|
the above copyright notice appear in all copies and that both that
|
||||||
|
copyright notice and this permission notice appear in supporting
|
||||||
|
documentation.
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice shall be
|
||||||
|
included in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||||
|
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||||
|
IN NO EVENT SHALL THE OPEN GROUP BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||||
|
OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||||
|
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||||
|
OTHER DEALINGS IN THE SOFTWARE.
|
||||||
|
|
||||||
|
Except as contained in this notice, the name of The Open Group shall
|
||||||
|
not be used in advertising or otherwise to promote the sale, use or
|
||||||
|
other dealings in this Software without prior written authorization
|
||||||
|
from The Open Group.
|
||||||
|
|
||||||
|
|
||||||
|
Copyright 2004-2005 Sun Microsystems, Inc. All rights reserved.
|
||||||
|
|
||||||
|
Permission is hereby granted, free of charge, to any person obtaining a
|
||||||
|
copy of this software and associated documentation files (the "Software"),
|
||||||
|
to deal in the Software without restriction, including without limitation
|
||||||
|
the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||||
|
and/or sell copies of the Software, and to permit persons to whom the
|
||||||
|
Software is furnished to do so, subject to the following conditions:
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice (including the next
|
||||||
|
paragraph) shall be included in all copies or substantial portions of the
|
||||||
|
Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||||
|
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||||
|
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||||
|
THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||||
|
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
|
||||||
|
FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||||
|
DEALINGS IN THE SOFTWARE.
|
||||||
|
|
||||||
|
|
||||||
|
Copyright (c) 1996 by Silicon Graphics Computer Systems, Inc.
|
||||||
|
|
||||||
|
Permission to use, copy, modify, and distribute this
|
||||||
|
software and its documentation for any purpose and without
|
||||||
|
fee is hereby granted, provided that the above copyright
|
||||||
|
notice appear in all copies and that both that copyright
|
||||||
|
notice and this permission notice appear in supporting
|
||||||
|
documentation, and that the name of Silicon Graphics not be
|
||||||
|
used in advertising or publicity pertaining to distribution
|
||||||
|
of the software without specific prior written permission.
|
||||||
|
Silicon Graphics makes no representation about the suitability
|
||||||
|
of this software for any purpose. It is provided "as is"
|
||||||
|
without any express or implied warranty.
|
||||||
|
|
||||||
|
SILICON GRAPHICS DISCLAIMS ALL WARRANTIES WITH REGARD TO THIS
|
||||||
|
SOFTWARE, INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY
|
||||||
|
AND FITNESS FOR A PARTICULAR PURPOSE. IN NO EVENT SHALL SILICON
|
||||||
|
GRAPHICS BE LIABLE FOR ANY SPECIAL, INDIRECT OR CONSEQUENTIAL
|
||||||
|
DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE,
|
||||||
|
DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE
|
||||||
|
OR OTHER TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH
|
||||||
|
THE USE OR PERFORMANCE OF THIS SOFTWARE.
|
||||||
|
|
||||||
|
|
||||||
|
Copyright (c) 1996 X Consortium
|
||||||
|
|
||||||
|
Permission is hereby granted, free of charge, to any person obtaining
|
||||||
|
a copy of this software and associated documentation files (the
|
||||||
|
"Software"), to deal in the Software without restriction, including
|
||||||
|
without limitation the rights to use, copy, modify, merge, publish,
|
||||||
|
distribute, sublicense, and/or sell copies of the Software, and to
|
||||||
|
permit persons to whom the Software is furnished to do so, subject to
|
||||||
|
the following conditions:
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice shall be
|
||||||
|
included in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||||
|
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||||
|
IN NO EVENT SHALL THE X CONSORTIUM BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||||
|
OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||||
|
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||||
|
OTHER DEALINGS IN THE SOFTWARE.
|
||||||
|
|
||||||
|
Except as contained in this notice, the name of the X Consortium shall
|
||||||
|
not be used in advertising or otherwise to promote the sale, use or
|
||||||
|
other dealings in this Software without prior written authorization
|
||||||
|
from the X Consortium.
|
||||||
|
|
||||||
|
|
||||||
|
Copyright (C) 2004, 2006 Ævar Arnfjörð Bjarmason <avarab@gmail.com>
|
||||||
|
|
||||||
|
Permission to use, copy, modify, distribute, and sell this software and its
|
||||||
|
documentation for any purpose is hereby granted without fee, provided that
|
||||||
|
the above copyright notice appear in all copies and that both that
|
||||||
|
copyright notice and this permission notice appear in supporting
|
||||||
|
documentation.
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice shall be
|
||||||
|
included in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||||
|
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||||
|
IN NO EVENT SHALL THE OPEN GROUP BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||||
|
OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||||
|
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||||
|
OTHER DEALINGS IN THE SOFTWARE.
|
||||||
|
|
||||||
|
Except as contained in this notice, the name of a copyright holder shall
|
||||||
|
not be used in advertising or otherwise to promote the sale, use or
|
||||||
|
other dealings in this Software without prior written authorization of
|
||||||
|
the copyright holder.
|
||||||
|
|
||||||
|
|
||||||
|
Copyright (C) 1999, 2000 by Anton Zinoviev <anton@lml.bas.bg>
|
||||||
|
|
||||||
|
This software may be used, modified, copied, distributed, and sold,
|
||||||
|
in both source and binary form provided that the above copyright
|
||||||
|
and these terms are retained. Under no circumstances is the author
|
||||||
|
responsible for the proper functioning of this software, nor does
|
||||||
|
the author assume any responsibility for damages incurred with its
|
||||||
|
use.
|
||||||
|
|
||||||
|
Permission is granted to anyone to use, distribute and modify
|
||||||
|
this file in any way, provided that the above copyright notice
|
||||||
|
is left intact and the author of the modification summarizes
|
||||||
|
the changes in this header.
|
||||||
|
|
||||||
|
This file is distributed without any expressed or implied warranty.
|
||||||
+21
@@ -0,0 +1,21 @@
|
|||||||
|
# Vendored xkeyboard-config subset
|
||||||
|
|
||||||
|
- **Package**: xkeyboard-config 2.44
|
||||||
|
- **Source**: https://gitlab.freedesktop.org/xkeyboard-config/xkeyboard-config/-/archive/xkeyboard-config-2.44/xkeyboard-config-2.44.tar.gz
|
||||||
|
- **sha256**: `35e34edeaf4e8da8d0696ff6b241ee11ddb1b8c6730bac7252d4d0a88ea5f05b`
|
||||||
|
- **keysymdef.h**: xorgproto, copied from `/opt/homebrew/include/X11/keysymdef.h`
|
||||||
|
- **License**: MIT/X11 (see COPYING)
|
||||||
|
|
||||||
|
Only the symbols files reachable from the generated layouts (tools/make-xkeyboard-config.py `TARGETS`) are vendored; regenerate with
|
||||||
|
`python3 tools/make-xkeyboard-config.py fetch` then `... generate`.
|
||||||
|
|
||||||
|
Vendored symbols files:
|
||||||
|
|
||||||
|
- `symbols/de`
|
||||||
|
- `symbols/es`
|
||||||
|
- `symbols/fr`
|
||||||
|
- `symbols/gb`
|
||||||
|
- `symbols/kpdl`
|
||||||
|
- `symbols/latin`
|
||||||
|
- `symbols/level3`
|
||||||
|
- `symbols/us`
|
||||||
+2584
File diff suppressed because it is too large
Load Diff
+1232
File diff suppressed because it is too large
Load Diff
+250
@@ -0,0 +1,250 @@
|
|||||||
|
// Keyboard layouts for Spain.
|
||||||
|
|
||||||
|
// Modified for a real Spanish keyboard by Jon Tombs.
|
||||||
|
default partial alphanumeric_keys
|
||||||
|
xkb_symbols "basic" {
|
||||||
|
|
||||||
|
include "latin(type4)"
|
||||||
|
|
||||||
|
name[Group1]="Spanish";
|
||||||
|
|
||||||
|
key <TLDE> { [ masculine, ordfeminine, backslash, backslash ] };
|
||||||
|
key <AE01> { [ 1, exclam, bar, exclamdown ] };
|
||||||
|
key <AE03> { [ 3, periodcentered, numbersign, sterling ] };
|
||||||
|
key <AE04> { [ 4, dollar, asciitilde, dollar ] };
|
||||||
|
key <AE11> { [apostrophe, question, backslash, questiondown ] };
|
||||||
|
key <AE12> { [exclamdown, questiondown, dead_cedilla, dead_ogonek] };
|
||||||
|
|
||||||
|
key <AD11> { [dead_grave, dead_circumflex, bracketleft, dead_abovering ] };
|
||||||
|
key <AD12> { [ plus, asterisk, bracketright, dead_macron ] };
|
||||||
|
|
||||||
|
key <AC10> { [ ntilde, Ntilde, dead_tilde, dead_doubleacute ] };
|
||||||
|
key <AC11> { [dead_acute, dead_diaeresis, braceleft, dead_caron ] };
|
||||||
|
key <BKSL> { [ ccedilla, Ccedilla, braceright, dead_breve ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "winkeys" {
|
||||||
|
|
||||||
|
include "es(basic)"
|
||||||
|
name[Group1]="Spanish (Windows)";
|
||||||
|
include "eurosign(5)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "nodeadkeys" {
|
||||||
|
|
||||||
|
include "es(basic)"
|
||||||
|
|
||||||
|
name[Group1]="Spanish (no dead keys)";
|
||||||
|
|
||||||
|
key <AE12> { [exclamdown, questiondown, cedilla, ogonek ] };
|
||||||
|
key <AD11> { [ grave, asciicircum, bracketleft, degree ] };
|
||||||
|
key <AD12> { [ plus, asterisk, bracketright, macron ] };
|
||||||
|
key <AC07> { [ j, J, ezh, EZH ] };
|
||||||
|
key <AC10> { [ ntilde, Ntilde, asciitilde, doubleacute ] };
|
||||||
|
key <AC11> { [ acute, diaeresis, braceleft, caron ] };
|
||||||
|
key <BKSL> { [ ccedilla, Ccedilla, braceright, breve ] };
|
||||||
|
key <AB10> { [ minus, underscore, ellipsis, abovedot ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Spanish Dvorak mapping (note R-H exchange)
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "dvorak" {
|
||||||
|
|
||||||
|
name[Group1]="Spanish (Dvorak)";
|
||||||
|
|
||||||
|
key <TLDE> {[ masculine, ordfeminine, backslash, degree ]};
|
||||||
|
key <AE01> {[ 1, exclam, bar, onesuperior ]};
|
||||||
|
key <AE02> {[ 2, quotedbl, at, twosuperior ]};
|
||||||
|
key <AE03> {[ 3, periodcentered, numbersign, threesuperior ]};
|
||||||
|
key <AE04> {[ 4, dollar, asciitilde, onequarter ]};
|
||||||
|
key <AE05> {[ 5, percent, brokenbar, fiveeighths ]};
|
||||||
|
key <AE06> {[ 6, ampersand, notsign, threequarters ]};
|
||||||
|
key <AE07> {[ 7, slash, onehalf, seveneighths ]};
|
||||||
|
key <AE08> {[ 8, parenleft, oneeighth, threeeighths ]};
|
||||||
|
key <AE09> {[ 9, parenright, asciicircum ]};
|
||||||
|
key <AE10> {[ 0, equal, grave, dead_doubleacute ]};
|
||||||
|
key <AE11> {[ apostrophe, question, dead_macron, dead_ogonek ]};
|
||||||
|
key <AE12> {[ exclamdown, questiondown, dead_breve, dead_abovedot ]};
|
||||||
|
|
||||||
|
key <AD01> {[ period, colon, less, guillemotleft ]};
|
||||||
|
key <AD02> {[ comma, semicolon, greater, guillemotright ]};
|
||||||
|
key <AD03> {[ ntilde, Ntilde, lstroke, Lstroke ]};
|
||||||
|
key <AD04> {[ p, P, paragraph ]};
|
||||||
|
key <AD05> {[ y, Y, yen ]};
|
||||||
|
key <AD06> {[ f, F, tslash, Tslash ]};
|
||||||
|
key <AD07> {[ g, G, dstroke, Dstroke ]};
|
||||||
|
key <AD08> {[ c, C, cent, copyright ]};
|
||||||
|
key <AD09> {[ h, H, hstroke, Hstroke ]};
|
||||||
|
key <AD10> {[ l, L, sterling ]};
|
||||||
|
key <AD11> {[ dead_grave, dead_circumflex, bracketleft, dead_caron ]};
|
||||||
|
key <AD12> {[ plus, asterisk, bracketright, plusminus ]};
|
||||||
|
|
||||||
|
key <AC01> {[ a, A, ae, AE ]};
|
||||||
|
key <AC02> {[ o, O, oslash, Oslash ]};
|
||||||
|
key <AC03> {[ e, E, EuroSign ]};
|
||||||
|
key <AC04> {[ u, U, aring, Aring ]};
|
||||||
|
key <AC05> {[ i, I, oe, OE ]};
|
||||||
|
key <AC06> {[ d, D, eth, ETH ]};
|
||||||
|
key <AC07> {[ r, R, registered, trademark ]};
|
||||||
|
key <AC08> {[ t, T, thorn, THORN ]};
|
||||||
|
key <AC09> {[ n, N, eng, ENG ]};
|
||||||
|
key <AC10> {[ s, S, ssharp, section ]};
|
||||||
|
key <AC11> {[ dead_acute, dead_diaeresis, braceleft, dead_tilde ]};
|
||||||
|
key <BKSL> {[ ccedilla, Ccedilla, braceright, dead_cedilla ]};
|
||||||
|
|
||||||
|
key <LSGT> {[ less, greater, guillemotleft, guillemotright ]};
|
||||||
|
key <AB01> {[ minus, underscore, hyphen, macron ]};
|
||||||
|
key <AB02> {[ q, Q, currency ]};
|
||||||
|
key <AB03> {[ j, J ]};
|
||||||
|
key <AB04> {[ k, K, kra ]};
|
||||||
|
key <AB05> {[ x, X, multiply, division ]};
|
||||||
|
key <AB06> {[ b, B ]};
|
||||||
|
key <AB07> {[ m, M, mu ]};
|
||||||
|
key <AB08> {[ w, W ]};
|
||||||
|
key <AB09> {[ v, V ]};
|
||||||
|
key <AB10> {[ z, Z ]};
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "cat" {
|
||||||
|
|
||||||
|
include "es(basic)"
|
||||||
|
|
||||||
|
name[Group1]="Catalan (Spain, with middle-dot L)";
|
||||||
|
|
||||||
|
key <AC09> { [ l, L, 0x1000140, 0x100013F ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "ast" {
|
||||||
|
|
||||||
|
include "es(basic)"
|
||||||
|
|
||||||
|
name[Group1]="Asturian (Spain, with bottom-dot H and L)";
|
||||||
|
|
||||||
|
key <AC06> { [ h, H, 0x1001E25, 0x1001E24 ] };
|
||||||
|
key <AC09> { [ l, L, 0x1001E37, 0x1001E36 ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "olpc" {
|
||||||
|
|
||||||
|
// #HW-SPECIFIC
|
||||||
|
|
||||||
|
// http://wiki.laptop.org/go/OLPC_Spanish_Keyboard
|
||||||
|
|
||||||
|
include "us(basic)"
|
||||||
|
name[Group1]="Spanish";
|
||||||
|
|
||||||
|
key <AE00> { [ masculine, ordfeminine ] };
|
||||||
|
key <AE01> { [ 1, exclam, bar ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, at ] };
|
||||||
|
key <AE03> { [ 3, dead_grave, numbersign, grave ] };
|
||||||
|
key <AE05> { [ 5, percent, asciicircum, dead_circumflex ] };
|
||||||
|
key <AE06> { [ 6, ampersand, notsign ] };
|
||||||
|
key <AE07> { [ 7, slash, backslash ] };
|
||||||
|
key <AE08> { [ 8, parenleft ] };
|
||||||
|
key <AE09> { [ 9, parenright ] };
|
||||||
|
key <AE10> { [ 0, equal ] };
|
||||||
|
key <AE11> { [ apostrophe, question ] };
|
||||||
|
key <AE12> { [ exclamdown, questiondown ] };
|
||||||
|
|
||||||
|
key <AD03> { [ e, E, EuroSign ] };
|
||||||
|
key <AD11> { [ dead_acute, dead_diaeresis, acute, dead_abovering ] };
|
||||||
|
key <AD12> { [ bracketleft, braceleft ] };
|
||||||
|
|
||||||
|
key <AC10> { [ ntilde, Ntilde ] };
|
||||||
|
key <AC11> { [ plus, asterisk, dead_tilde ] };
|
||||||
|
key <AC12> { [ bracketright, braceright, section ] };
|
||||||
|
|
||||||
|
key <AB08> { [ comma, semicolon ] };
|
||||||
|
key <AB09> { [ period, colon ] };
|
||||||
|
key <AB10> { [ minus, underscore ] };
|
||||||
|
|
||||||
|
key <I219> { [ less, greater, ISO_Next_Group ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "olpcm" {
|
||||||
|
|
||||||
|
// #HW-SPECIFIC
|
||||||
|
|
||||||
|
// Mechanical (non-membrane) OLPC Spanish keyboard layout.
|
||||||
|
// See: http://wiki.laptop.org/go/OLPC_Spanish_Non-membrane_Keyboard
|
||||||
|
|
||||||
|
include "us(basic)"
|
||||||
|
name[Group1]="Spanish";
|
||||||
|
|
||||||
|
key <AE00> { [ questiondown, exclamdown, backslash ] };
|
||||||
|
key <AE01> { [ 1, exclam, bar ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, at ] };
|
||||||
|
key <AE03> { [ 3, dead_grave, numbersign, grave ] };
|
||||||
|
key <AE04> { [ 4, dollar, asciitilde, dead_tilde ] };
|
||||||
|
key <AE05> { [ 5, percent, asciicircum, dead_circumflex ] };
|
||||||
|
key <AE06> { [ 6, ampersand, notsign ] };
|
||||||
|
key <AE07> { [ 7, slash, backslash ] }; // no '\' label on olpcm, leave for compatibility
|
||||||
|
key <AE08> { [ 8, parenleft, masculine ] };
|
||||||
|
key <AE09> { [ 9, parenright, ordfeminine ] };
|
||||||
|
key <AE10> { [ 0, equal ] };
|
||||||
|
key <AE11> { [ apostrophe, question ] };
|
||||||
|
|
||||||
|
key <AD03> { [ e, E, EuroSign ] };
|
||||||
|
key <AD11> { [ dead_acute, dead_diaeresis, dead_abovering, acute ] };
|
||||||
|
key <AD12> { [ plus, asterisk ] };
|
||||||
|
|
||||||
|
key <AC10> { [ ntilde, Ntilde ] };
|
||||||
|
// no AC11 or AC12 on olpcm
|
||||||
|
|
||||||
|
key <AB08> { [ comma, semicolon ] };
|
||||||
|
key <AB09> { [ period, colon ] };
|
||||||
|
key <AB10> { [ minus, underscore ] };
|
||||||
|
|
||||||
|
key <AA02> { [ less, greater ] };
|
||||||
|
key <AA06> { [ bracketleft, braceleft, ccedilla, Ccedilla ] };
|
||||||
|
key <AA07> { [ bracketright, braceright ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "deadtilde" {
|
||||||
|
|
||||||
|
include "es(basic)"
|
||||||
|
|
||||||
|
name[Group1]="Spanish (dead tilde)";
|
||||||
|
|
||||||
|
key <AE04> { [ 4, dollar, dead_tilde, dollar ] };
|
||||||
|
key <AC10> { [ ntilde, Ntilde, asciitilde, dead_doubleacute ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "olpc2" {
|
||||||
|
// #HW-SPECIFIC
|
||||||
|
|
||||||
|
// Modified variant of US International layout, specifically for Peru
|
||||||
|
// Contact: Sayamindu Dasgupta <sayamindu@laptop.org>
|
||||||
|
|
||||||
|
include "us(olpc)"
|
||||||
|
name[Group1]="Spanish";
|
||||||
|
|
||||||
|
key <AE03> { [ 3, numbersign, dead_grave, dead_grave] }; // combining grave
|
||||||
|
key <I236> { [ XF86Start ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
// EXTRAS:
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "sun_type6" {
|
||||||
|
include "sun_vndr/es(sun_type6)"
|
||||||
|
};
|
||||||
+1404
File diff suppressed because it is too large
Load Diff
+249
@@ -0,0 +1,249 @@
|
|||||||
|
// Keyboard layouts for Great Britain.
|
||||||
|
|
||||||
|
default partial alphanumeric_keys
|
||||||
|
xkb_symbols "basic" {
|
||||||
|
|
||||||
|
// The basic UK layout, also known as the IBM 166 layout,
|
||||||
|
// but with the useless brokenbar pushed two levels up.
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
name[Group1]="English (UK)";
|
||||||
|
|
||||||
|
key <TLDE> { [ grave, notsign, bar, bar ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, twosuperior, oneeighth ] };
|
||||||
|
key <AE03> { [ 3, sterling, threesuperior, sterling ] };
|
||||||
|
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||||
|
|
||||||
|
key <AC11> { [apostrophe, at, dead_circumflex, dead_caron] };
|
||||||
|
key <BKSL> { [numbersign, asciitilde, dead_grave, dead_breve ] };
|
||||||
|
|
||||||
|
key <LSGT> { [ backslash, bar, bar, brokenbar ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "intl" {
|
||||||
|
|
||||||
|
// A UK layout but with five accents made into dead keys:
|
||||||
|
// grave, diaeresis, circumflex, acute, and tilde.
|
||||||
|
// By Phil Jones <philjones1 at blueyonder.co.uk>.
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
name[Group1]="English (UK, intl., with dead keys)";
|
||||||
|
|
||||||
|
key <TLDE> { [ dead_grave, notsign, bar, bar ] };
|
||||||
|
key <AE02> { [ 2, dead_diaeresis, twosuperior, onehalf ] };
|
||||||
|
key <AE03> { [ 3, sterling, threesuperior, onethird ] };
|
||||||
|
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||||
|
key <AE06> { [ 6, dead_circumflex, threequarters, onesixth ] };
|
||||||
|
|
||||||
|
key <AC11> { [ dead_acute, at, apostrophe, bar ] };
|
||||||
|
key <BKSL> { [ numbersign, dead_tilde, bar, bar ] };
|
||||||
|
|
||||||
|
key <LSGT> { [ backslash, bar, bar, bar ] };
|
||||||
|
key <AB08> { [ comma, less, ccedilla, Ccedilla ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "extd" {
|
||||||
|
// Clone of the Microsoft "United Kingdom Extended" layout, which
|
||||||
|
// includes dead keys for: grave; diaeresis; circumflex; tilde; and
|
||||||
|
// accute. It also enables direct access to accute characters using
|
||||||
|
// the Multi_key (Alt Gr).
|
||||||
|
//
|
||||||
|
// Taken from...
|
||||||
|
// "Windows Keyboard Layouts"
|
||||||
|
// https://docs.microsoft.com/en-gb/globalization/windows-keyboard-layouts#U
|
||||||
|
//
|
||||||
|
// -- Jonathan Miles <jon@cybah.co.uk>
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
name[Group1]="English (UK, extended, Windows)";
|
||||||
|
|
||||||
|
key <TLDE> { [ dead_grave, notsign, brokenbar, NoSymbol ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, dead_diaeresis, onehalf ] };
|
||||||
|
key <AE03> { [ 3, sterling, threesuperior, onethird ] };
|
||||||
|
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||||
|
key <AE06> { [ 6, asciicircum, dead_circumflex, NoSymbol ] };
|
||||||
|
|
||||||
|
key <AD02> { [ w, W, wacute, Wacute ] };
|
||||||
|
key <AD03> { [ e, E, eacute, Eacute ] };
|
||||||
|
key <AD06> { [ y, Y, yacute, Yacute ] };
|
||||||
|
key <AD07> { [ u, U, uacute, Uacute ] };
|
||||||
|
key <AD08> { [ i, I, iacute, Iacute ] };
|
||||||
|
key <AD09> { [ o, O, oacute, Oacute ] };
|
||||||
|
key <AD12> { [ bracketright, braceright, NoSymbol, bar ] };
|
||||||
|
|
||||||
|
key <AC01> { [ a, A, aacute, Aacute ] };
|
||||||
|
key <AC11> { [ apostrophe, at, dead_acute, grave ] };
|
||||||
|
key <BKSL> { [ numbersign, asciitilde, dead_tilde, backslash ] };
|
||||||
|
|
||||||
|
key <LSGT> { [ backslash, bar, NoSymbol, NoSymbol ] };
|
||||||
|
key <AB03> { [ c, C, ccedilla, Ccedilla ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
// Describe the differences between the US Colemak layout
|
||||||
|
// and a UK variant. By Andy Buckley (andy@insectnation.org)
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "colemak" {
|
||||||
|
include "us(colemak)"
|
||||||
|
|
||||||
|
name[Group1]="English (UK, Colemak)";
|
||||||
|
|
||||||
|
key <TLDE> { [ grave, notsign, bar, asciitilde ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, twosuperior, oneeighth ] };
|
||||||
|
key <AE03> { [ 3, sterling, threesuperior, sterling ] };
|
||||||
|
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||||
|
|
||||||
|
key <AC11> { [apostrophe, at, dead_circumflex, dead_caron] };
|
||||||
|
key <BKSL> { [numbersign, asciitilde, dead_grave, dead_breve ] };
|
||||||
|
|
||||||
|
key <LSGT> { [ backslash, bar, asciitilde, brokenbar ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Colemak-DH (ISO) layout, UK Variant, https://colemakmods.github.io/mod-dh/
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "colemak_dh" {
|
||||||
|
include "us(colemak_dh)"
|
||||||
|
|
||||||
|
name[Group1]="English (UK, Colemak-DH)";
|
||||||
|
|
||||||
|
key <TLDE> { [ grave, notsign, bar, asciitilde ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, twosuperior, oneeighth ] };
|
||||||
|
key <AE03> { [ 3, sterling, threesuperior, sterling ] };
|
||||||
|
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||||
|
|
||||||
|
key <AC11> { [apostrophe, at, dead_circumflex, dead_caron] };
|
||||||
|
key <BKSL> { [numbersign, asciitilde, dead_grave, dead_breve ] };
|
||||||
|
|
||||||
|
key <AB05> { [ backslash, bar, asciitilde, brokenbar ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
// Dvorak (UK) keymap (by odaen) allowing the usage of
|
||||||
|
// the £ and ? key and swapping the @ and " keys.
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "dvorak" {
|
||||||
|
include "us(dvorak-alt-intl)"
|
||||||
|
|
||||||
|
name[Group1]="English (UK, Dvorak)";
|
||||||
|
|
||||||
|
key <TLDE> { [ grave, notsign, bar, bar ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, twosuperior, NoSymbol ] };
|
||||||
|
key <AE03> { [ 3, sterling, threesuperior, NoSymbol ] };
|
||||||
|
key <AD01> { [ apostrophe, at ] };
|
||||||
|
key <BKSL> { [ numbersign, asciitilde ] };
|
||||||
|
key <LSGT> { [ backslash, bar ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Dvorak letter positions, but punctuation all in the normal UK positions.
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "dvorakukp" {
|
||||||
|
include "gb(dvorak)"
|
||||||
|
|
||||||
|
name[Group1]="English (UK, Dvorak, with UK punctuation)";
|
||||||
|
|
||||||
|
key <AE11> { [ minus, underscore ] };
|
||||||
|
key <AE12> { [ equal, plus ] };
|
||||||
|
key <AD11> { [ bracketleft, braceleft ] };
|
||||||
|
key <AD12> { [ bracketright, braceright ] };
|
||||||
|
key <AD01> { [ slash, question ] };
|
||||||
|
key <AC11> { [apostrophe, at, dead_circumflex, dead_caron] };
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "mac" {
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
name[Group1]= "English (UK, Macintosh)";
|
||||||
|
|
||||||
|
key <TLDE> { [ section, plusminus ] };
|
||||||
|
key <AE02> { [ 2, at, EuroSign ] };
|
||||||
|
key <AE03> { [ 3, sterling, numbersign ] };
|
||||||
|
key <LSGT> { [ grave, asciitilde ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
include "level3(enter_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "mac_intl" {
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
name[Group1]="English (UK, Macintosh, intl.)";
|
||||||
|
|
||||||
|
key <TLDE> { [ section, plusminus, notsign, notsign ] }; //dead_grave
|
||||||
|
key <AE02> { [ 2, at, EuroSign, onehalf ] };
|
||||||
|
key <AE03> { [ 3, sterling, twosuperior, onethird ] };
|
||||||
|
key <AE04> { [ 4, dollar, threesuperior, onequarter ] };
|
||||||
|
key <AE06> { [ 6, dead_circumflex, NoSymbol, onesixth ] };
|
||||||
|
key <AD09> { [ o, O, oe, OE ] };
|
||||||
|
|
||||||
|
key <AC11> { [ dead_acute, dead_diaeresis, dead_diaeresis, bar ] }; //dead_doubleacute
|
||||||
|
key <BKSL> { [ backslash, bar, numbersign, bar ] };
|
||||||
|
|
||||||
|
key <LSGT> { [ dead_grave, dead_tilde, brokenbar, bar ] };
|
||||||
|
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "pl" {
|
||||||
|
|
||||||
|
// Polish accented letters on upper levels of corresponding base letters.
|
||||||
|
// Idea from Wawrzyniec Niewodniczański, adapted by Aleksander Kowalski.
|
||||||
|
|
||||||
|
include "gb(basic)"
|
||||||
|
|
||||||
|
name[Group1]="Polish (British keyboard)";
|
||||||
|
|
||||||
|
key <AD03> { [ e, E, eogonek, Eogonek ] };
|
||||||
|
key <AD09> { [ o, O, oacute, Oacute ] };
|
||||||
|
|
||||||
|
key <AC01> { [ a, A, aogonek, Aogonek ] };
|
||||||
|
key <AC02> { [ s, S, sacute, Sacute ] };
|
||||||
|
|
||||||
|
key <AB01> { [ z, Z, zabovedot, Zabovedot ] };
|
||||||
|
key <AB02> { [ x, X, zacute, Zacute ] };
|
||||||
|
key <AB03> { [ c, C, cacute, Cacute ] };
|
||||||
|
key <AB06> { [ n, N, nacute, Nacute ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "gla" {
|
||||||
|
|
||||||
|
// Grave-accented letters on the upper levels of the relevant vowels.
|
||||||
|
|
||||||
|
include "gb(basic)"
|
||||||
|
|
||||||
|
name[Group1]="Scottish Gaelic";
|
||||||
|
|
||||||
|
key <AD03> { [ e, E, egrave, Egrave ] };
|
||||||
|
key <AD07> { [ u, U, ugrave, Ugrave ] };
|
||||||
|
key <AD08> { [ i, I, igrave, Igrave ] };
|
||||||
|
key <AD09> { [ o, O, ograve, Ograve ] };
|
||||||
|
|
||||||
|
key <AC01> { [ a, A, agrave, Agrave ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// EXTRAS:
|
||||||
|
|
||||||
|
partial alphanumeric_keys
|
||||||
|
xkb_symbols "sun_type6" {
|
||||||
|
include "sun_vndr/gb(sun_type6)"
|
||||||
|
};
|
||||||
+102
@@ -0,0 +1,102 @@
|
|||||||
|
// The <KPDL> key is a mess.
|
||||||
|
// It was probably originally meant to be a decimal separator.
|
||||||
|
// Except since it was declared by USA people it didn't use the original
|
||||||
|
// SI separator "," but a "." (since then the USA managed to f-up the SI
|
||||||
|
// by making "." an accepted alternative, but standards still use "," as
|
||||||
|
// default)
|
||||||
|
// As a result users of SI-abiding countries expect either a "." or a ","
|
||||||
|
// or a "decimal_separator" which may or may not be translated in one of the
|
||||||
|
// above depending on applications.
|
||||||
|
// It's not possible to define a default per-country since user expectations
|
||||||
|
// depend on the conflicting choices of their most-used applications,
|
||||||
|
// operating system, etc. Therefore it needs to be a configuration setting
|
||||||
|
// Copyright © 2007 Nicolas Mailhot <nicolas.mailhot @ laposte.net>
|
||||||
|
|
||||||
|
|
||||||
|
// Legacy <KPDL> #1
|
||||||
|
// This assumes KP_Decimal will be translated in a dot
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "dot" {
|
||||||
|
|
||||||
|
key.type[Group1]="KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ KP_Delete, KP_Decimal ] }; // <delete> <separator>
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
// Legacy <KPDL> #2
|
||||||
|
// This assumes KP_Separator will be translated in a comma
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "comma" {
|
||||||
|
|
||||||
|
key.type[Group1]="KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ KP_Delete, KP_Separator ] }; // <delete> <separator>
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
// Period <KPDL>, usual keyboard serigraphy in most countries
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "dotoss" {
|
||||||
|
|
||||||
|
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ KP_Delete, period, comma, 0x100202F ] }; // <delete> . , ⍽ (narrow no-break space)
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
// Period <KPDL>, usual keyboard serigraphy in most countries, latin-9 restriction
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "dotoss_latin9" {
|
||||||
|
|
||||||
|
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ KP_Delete, period, comma, nobreakspace ] }; // <delete> . , ⍽ (no-break space)
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
// Comma <KPDL>, what most non anglo-saxon people consider the real separator
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "commaoss" {
|
||||||
|
|
||||||
|
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ KP_Delete, comma, period, 0x100202F ] }; // <delete> , . ⍽ (narrow no-break space)
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
// Momayyez <KPDL>: Bahrain, Iran, Iraq, Kuwait, Oman, Qatar, Saudi Arabia, Syria, UAE
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "momayyezoss" {
|
||||||
|
|
||||||
|
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ KP_Delete, 0x100066B, comma, 0x100202F ] }; // <delete> ? , ⍽ (narrow no-break space)
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
// Abstracted <KPDL>, pray everything will work out (it usually does not)
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "kposs" {
|
||||||
|
|
||||||
|
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ KP_Delete, KP_Decimal, KP_Separator, 0x100202F ] }; // <delete> ? ? ⍽ (narrow no-break space)
|
||||||
|
};
|
||||||
|
|
||||||
|
// Spreadsheets may be configured to use the dot as decimal
|
||||||
|
// punctuation, comma as a thousands separator and then semi-colon as
|
||||||
|
// the list separator. Of these, dot and semi-colon is most important
|
||||||
|
// when entering data by the keyboard; the comma can then be inferred
|
||||||
|
// and added to the presentation afterwards. Using semi-colon as a
|
||||||
|
// general separator may in fact be preferred to avoid ambiguities
|
||||||
|
// in data files. Most times a decimal separator is hard-coded, it
|
||||||
|
// seems to be period, probably since this is the syntax used in
|
||||||
|
// (most) programming languages.
|
||||||
|
partial keypad_keys
|
||||||
|
xkb_symbols "semi" {
|
||||||
|
|
||||||
|
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||||
|
|
||||||
|
key <KPDL> { [ NoSymbol, NoSymbol, semicolon ] };
|
||||||
|
};
|
||||||
+255
@@ -0,0 +1,255 @@
|
|||||||
|
// Common Latin alphabet layout
|
||||||
|
|
||||||
|
default partial
|
||||||
|
xkb_symbols "basic" {
|
||||||
|
|
||||||
|
key <AE01> { [ 1, exclam, onesuperior, exclamdown ] };
|
||||||
|
key <AE02> { [ 2, at, twosuperior, oneeighth ] };
|
||||||
|
key <AE03> { [ 3, numbersign, threesuperior, sterling ] };
|
||||||
|
key <AE04> { [ 4, dollar, onequarter, dollar ] };
|
||||||
|
key <AE05> { [ 5, percent, onehalf, threeeighths ] };
|
||||||
|
key <AE06> { [ 6, asciicircum, threequarters, fiveeighths ] };
|
||||||
|
key <AE07> { [ 7, ampersand, braceleft, seveneighths ] };
|
||||||
|
key <AE08> { [ 8, asterisk, bracketleft, trademark ] };
|
||||||
|
key <AE09> { [ 9, parenleft, bracketright, plusminus ] };
|
||||||
|
key <AE10> { [ 0, parenright, braceright, degree ] };
|
||||||
|
key <AE11> { [ minus, underscore, backslash, questiondown ] };
|
||||||
|
key <AE12> { [ equal, plus, dead_cedilla, dead_ogonek ] };
|
||||||
|
|
||||||
|
key <AD01> { [ q, Q, at, Greek_OMEGA ] };
|
||||||
|
key <AD02> { [ w, W, U017F, section ] };
|
||||||
|
key <AD03> { [ e, E, e, E ] };
|
||||||
|
key <AD04> { [ r, R, paragraph, registered ] };
|
||||||
|
key <AD05> { [ t, T, tslash, Tslash ] };
|
||||||
|
key <AD06> { [ y, Y, leftarrow, yen ] };
|
||||||
|
key <AD07> { [ u, U, downarrow, uparrow ] };
|
||||||
|
key <AD08> { [ i, I, rightarrow, idotless ] };
|
||||||
|
key <AD09> { [ o, O, oslash, Oslash ] };
|
||||||
|
key <AD10> { [ p, P, thorn, THORN ] };
|
||||||
|
key <AD11> { [bracketleft, braceleft, dead_diaeresis, dead_abovering ] };
|
||||||
|
key <AD12> { [bracketright, braceright, dead_tilde, dead_macron ] };
|
||||||
|
|
||||||
|
key <AC01> { [ a, A, ae, AE ] };
|
||||||
|
key <AC02> { [ s, S, ssharp, U1E9E ] };
|
||||||
|
key <AC03> { [ d, D, eth, ETH ] };
|
||||||
|
key <AC04> { [ f, F, dstroke, ordfeminine ] };
|
||||||
|
key <AC05> { [ g, G, eng, ENG ] };
|
||||||
|
key <AC06> { [ h, H, hstroke, Hstroke ] };
|
||||||
|
key <AC07> { [ j, J, dead_hook, dead_horn ] };
|
||||||
|
key <AC08> { [ k, K, kra, ampersand ] };
|
||||||
|
key <AC09> { [ l, L, lstroke, Lstroke ] };
|
||||||
|
key <AC10> { [ semicolon, colon, dead_acute, dead_doubleacute ] };
|
||||||
|
key <AC11> { [apostrophe, quotedbl, dead_circumflex, dead_caron ] };
|
||||||
|
key <TLDE> { [ grave, asciitilde, notsign, notsign ] };
|
||||||
|
|
||||||
|
key <BKSL> { [ backslash, bar, dead_grave, dead_breve ] };
|
||||||
|
key <AB01> { [ z, Z, guillemotleft, less ] };
|
||||||
|
key <AB02> { [ x, X, guillemotright, greater ] };
|
||||||
|
key <AB03> { [ c, C, cent, copyright ] };
|
||||||
|
key <AB04> { [ v, V, doublelowquotemark, singlelowquotemark ] };
|
||||||
|
key <AB05> { [ b, B, leftdoublequotemark, leftsinglequotemark ] };
|
||||||
|
key <AB06> { [ n, N, rightdoublequotemark, rightsinglequotemark ] };
|
||||||
|
key <AB07> { [ m, M, mu, masculine ] };
|
||||||
|
key <AB08> { [ comma, less, U2022, multiply ] }; // bullet
|
||||||
|
key <AB09> { [ period, greater, periodcentered, division ] };
|
||||||
|
key <AB10> { [ slash, question, dead_belowdot, dead_abovedot ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Northern Europe ( Danish, Finnish, Norwegian, Swedish) common layout
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "type2" {
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
key <AE01> { [ 1, exclam, exclamdown, onesuperior ] };
|
||||||
|
key <AE02> { [ 2, quotedbl, at, twosuperior ] };
|
||||||
|
key <AE03> { [ 3, numbersign, sterling, threesuperior] };
|
||||||
|
key <AE04> { [ 4, currency, dollar, onequarter ] };
|
||||||
|
key <AE05> { [ 5, percent, onehalf, cent ] };
|
||||||
|
key <AE06> { [ 6, ampersand, yen, fiveeighths ] };
|
||||||
|
key <AE07> { [ 7, slash, braceleft, division ] };
|
||||||
|
key <AE08> { [ 8, parenleft, bracketleft, guillemotleft] };
|
||||||
|
key <AE09> { [ 9, parenright, bracketright, guillemotright] };
|
||||||
|
key <AE10> { [ 0, equal, braceright, degree ] };
|
||||||
|
|
||||||
|
key <AD03> { [ e, E, EuroSign, cent ] };
|
||||||
|
key <AD04> { [ r, R, registered, registered ] };
|
||||||
|
key <AD05> { [ t, T, thorn, THORN ] };
|
||||||
|
key <AD09> { [ o, O, oe, OE ] };
|
||||||
|
key <AD11> { [ aring, Aring, dead_diaeresis, dead_abovering ] };
|
||||||
|
key <AD12> { [dead_diaeresis, dead_circumflex, dead_tilde, dead_caron ] };
|
||||||
|
|
||||||
|
key <AC01> { [ a, A, ordfeminine, masculine ] };
|
||||||
|
|
||||||
|
key <AB03> { [ c, C, copyright, copyright ] };
|
||||||
|
key <AB08> { [ comma, semicolon, dead_cedilla, dead_ogonek ] };
|
||||||
|
key <AB09> { [ period, colon, periodcentered, dead_abovedot ] };
|
||||||
|
key <AB10> { [ minus, underscore, dead_belowdot, dead_abovedot ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Slavic Latin ( Albanian, Croatian, Polish, Slovene, Yugoslav)
|
||||||
|
// common layout
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "type3" {
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
key <AD01> { [ q, Q, backslash, Greek_OMEGA ] };
|
||||||
|
key <AD02> { [ w, W, bar, section ] };
|
||||||
|
key <AD06> { [ z, Z, leftarrow, yen ] };
|
||||||
|
|
||||||
|
key <AC04> { [ f, F, bracketleft, ordfeminine ] };
|
||||||
|
key <AC05> { [ g, G, bracketright, ENG ] };
|
||||||
|
key <AC08> { [ k, K, lstroke, ampersand ] };
|
||||||
|
|
||||||
|
key <AB01> { [ y, Y, guillemotleft, less ] };
|
||||||
|
key <AB04> { [ v, V, at, grave ] };
|
||||||
|
key <AB05> { [ b, B, braceleft, apostrophe ] };
|
||||||
|
key <AB06> { [ n, N, braceright, acute ] };
|
||||||
|
key <AB07> { [ m, M, section, masculine ] };
|
||||||
|
key <AB08> { [ comma, semicolon, less, multiply ] };
|
||||||
|
key <AB09> { [ period, colon, greater, division ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Another common Latin layout
|
||||||
|
// (German, Estonian, Spanish, Icelandic, Italian, Latin American, Portuguese)
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "type4" {
|
||||||
|
|
||||||
|
include "latin"
|
||||||
|
|
||||||
|
key <AE02> { [ 2, quotedbl, at, oneeighth ] };
|
||||||
|
key <AE06> { [ 6, ampersand, notsign, fiveeighths ] };
|
||||||
|
key <AE07> { [ 7, slash, braceleft, seveneighths ] };
|
||||||
|
key <AE08> { [ 8, parenleft, bracketleft, trademark ] };
|
||||||
|
key <AE09> { [ 9, parenright, bracketright, plusminus ] };
|
||||||
|
key <AE10> { [ 0, equal, braceright, degree ] };
|
||||||
|
|
||||||
|
key <AD03> { [ e, E, EuroSign, cent ] };
|
||||||
|
|
||||||
|
key <AB08> { [ comma, semicolon, U2022, multiply ] }; // bullet
|
||||||
|
key <AB09> { [ period, colon, periodcentered, division ] };
|
||||||
|
key <AB10> { [ minus, underscore, dead_belowdot, dead_abovedot ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "nodeadkeys" {
|
||||||
|
|
||||||
|
key <AE12> { [ equal, plus, cedilla, ogonek ] };
|
||||||
|
key <AD11> { [bracketleft, braceleft, diaeresis, degree ] };
|
||||||
|
key <AD12> { [bracketright, braceright, asciitilde, macron ] };
|
||||||
|
key <AC07> { [ j, J, ezh, EZH ] };
|
||||||
|
key <AC10> { [ semicolon, colon, acute, doubleacute ] };
|
||||||
|
key <AC11> { [apostrophe, quotedbl, asciicircum, caron ] };
|
||||||
|
key <BKSL> { [ backslash, bar, grave, breve ] };
|
||||||
|
key <AB10> { [ slash, question, ellipsis, abovedot ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "type2_nodeadkeys" {
|
||||||
|
|
||||||
|
include "latin(nodeadkeys)"
|
||||||
|
|
||||||
|
key <AD11> { [ aring, Aring, diaeresis, degree ] };
|
||||||
|
key <AD12> { [ diaeresis, asciicircum, asciitilde, caron ] };
|
||||||
|
key <AB08> { [ comma, semicolon, cedilla, ogonek ] };
|
||||||
|
key <AB09> { [ period, colon, periodcentered, abovedot ] };
|
||||||
|
key <AB10> { [ minus, underscore, ellipsis, abovedot ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "type3_nodeadkeys" {
|
||||||
|
|
||||||
|
include "latin(nodeadkeys)"
|
||||||
|
};
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "type4_nodeadkeys" {
|
||||||
|
|
||||||
|
include "latin(nodeadkeys)"
|
||||||
|
|
||||||
|
key <AB10> { [ minus, underscore, ellipsis, abovedot ] };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Added 2008.03.05 by Marcin Woliński
|
||||||
|
// See http://marcinwolinski.pl/keyboard/ for a description.
|
||||||
|
// Used by pl(intl)
|
||||||
|
//
|
||||||
|
// ┌─────┐
|
||||||
|
// │ 2 4 │ 2 = Shift, 4 = Level3 + Shift
|
||||||
|
// │ 1 3 │ 1 = Normal, 3 = Level3
|
||||||
|
// └─────┘
|
||||||
|
// ┌─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┲━━━━━━━━━┓
|
||||||
|
// │ ~ ~ │ ! ' │ @ " │ # ˝ │ $ ¸ │ % ˇ │ ^ ^ │ & ˘ │ * ̇ │ ( ̣ │ ) ° │ _ ¯ │ + ˛ ┃ ⌫ Back- ┃
|
||||||
|
// │ ` ` │ 1 ¡ │ 2 © │ 3 • │ 4 § │ 5 € │ 6 ¢ │ 7 − │ 8 × │ 9 ÷ │ 0 ° │ - – │ = — ┃ space ┃
|
||||||
|
// ┢━━━━━┷━┱───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┺━┳━━━━━━━┫
|
||||||
|
// ┃ ┃ Q │ W │ E │ R │ T │ Y │ U │ I │ O │ P │ { « │ } » ┃ Enter ┃
|
||||||
|
// ┃Tab ↹ ┃ q │ w │ e │ r │ t │ y │ u │ i │ o │ p │ [ ‹ │ ] › ┃ ⏎ ┃
|
||||||
|
// ┣━━━━━━━┻┱────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┺┓ ┃
|
||||||
|
// ┃ ┃ A │ S │ D │ F │ G │ H │ J │ K │ L │ : “ │ " ” │ | ¶ ┃ ┃
|
||||||
|
// ┃Caps ⇬ ┃ a │ s │ d │ f │ g │ h │ j │ k │ l │ ; ‘ │ ' ’ │ \ ┃ ┃
|
||||||
|
// ┣━━━━━━━━┹────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┲┷━━━━━┻━━━━━━┫
|
||||||
|
// ┃ │ Z │ X │ C │ V │ B │ N │ M │ < „ │ > · │ ? ¿ ┃ ┃
|
||||||
|
// ┃Shift ⇧ │ z │ x │ c │ v │ b │ n │ m │ , ‚ │ . … │ / ⁄ ┃Shift ⇧ ┃
|
||||||
|
// ┣━━━━━━━┳━━━━━┷━┳━━━┷━━━┱─┴─────┴─────┴─────┴─────┴─────┴───┲━┷━━━━━╈━━━━━┻━┳━━━━━━━┳━━━┛
|
||||||
|
// ┃ ┃ ┃ ┃ ␣ ⍽ ┃ ┃ ┃ ┃
|
||||||
|
// ┃Ctrl ┃Meta ┃Alt ┃ ␣ Space ⍽ ┃AltGr ⇮┃Menu ┃Ctrl ┃
|
||||||
|
// ┗━━━━━━━┻━━━━━━━┻━━━━━━━┹───────────────────────────────────┺━━━━━━━┻━━━━━━━┻━━━━━━━┛
|
||||||
|
|
||||||
|
partial
|
||||||
|
xkb_symbols "intl" {
|
||||||
|
|
||||||
|
key <TLDE> { [ grave, asciitilde, dead_grave, dead_tilde ] };
|
||||||
|
key <AE01> { [ 1, exclam, exclamdown, dead_acute ] };
|
||||||
|
key <AE02> { [ 2, at, copyright, dead_diaeresis ] };
|
||||||
|
key <AE03> { [ 3, numbersign, U2022, dead_doubleacute ] }; // U+2022 is bullet (the name bullet does not work)
|
||||||
|
key <AE04> { [ 4, dollar, section, dead_cedilla ] };
|
||||||
|
key <AE05> { [ 5, percent, EuroSign, dead_caron ] };
|
||||||
|
key <AE06> { [ 6, asciicircum, cent, dead_circumflex ] };
|
||||||
|
key <AE07> { [ 7, ampersand, U2212, dead_breve ] }; // U+2212 is MINUS SIGN
|
||||||
|
key <AE08> { [ 8, asterisk, multiply, dead_abovedot ] };
|
||||||
|
key <AE09> { [ 9, parenleft, division, dead_belowdot ] };
|
||||||
|
key <AE10> { [ 0, parenright, degree, dead_abovering ] };
|
||||||
|
key <AE11> { [ minus, underscore, endash, dead_macron ] };
|
||||||
|
key <AE12> { [ equal, plus, emdash, dead_ogonek ] };
|
||||||
|
|
||||||
|
key <AD01> { [ q, Q ] };
|
||||||
|
key <AD02> { [ w, W ] };
|
||||||
|
key <AD03> { [ e, E ] };
|
||||||
|
key <AD04> { [ r, R ] };
|
||||||
|
key <AD05> { [ t, T ] };
|
||||||
|
key <AD06> { [ y, Y ] };
|
||||||
|
key <AD07> { [ u, U ] };
|
||||||
|
key <AD08> { [ i, I ] };
|
||||||
|
key <AD09> { [ o, O ] };
|
||||||
|
key <AD10> { [ p, P ] };
|
||||||
|
key <AD11> { [bracketleft, braceleft, U2039, guillemotleft ] };
|
||||||
|
key <AD12> { [bracketright, braceright, U203A, guillemotright ] };
|
||||||
|
|
||||||
|
key <AC01> { [ a, A ] };
|
||||||
|
key <AC02> { [ s, S ] };
|
||||||
|
key <AC03> { [ d, D ] };
|
||||||
|
key <AC04> { [ f, F ] };
|
||||||
|
key <AC05> { [ g, G ] };
|
||||||
|
key <AC06> { [ h, H ] };
|
||||||
|
key <AC07> { [ j, J ] };
|
||||||
|
key <AC08> { [ k, K ] };
|
||||||
|
key <AC09> { [ l, L ] };
|
||||||
|
key <AC10> { [ semicolon, colon, leftsinglequotemark, leftdoublequotemark ] };
|
||||||
|
key <AC11> { [apostrophe, quotedbl, rightsinglequotemark, rightdoublequotemark ] };
|
||||||
|
|
||||||
|
key <BKSL> { [ backslash, bar, NoSymbol, paragraph ] };
|
||||||
|
key <AB01> { [ z, Z ] };
|
||||||
|
key <AB02> { [ x, X ] };
|
||||||
|
key <AB03> { [ c, C ] };
|
||||||
|
key <AB04> { [ v, V ] };
|
||||||
|
key <AB05> { [ b, B ] };
|
||||||
|
key <AB06> { [ n, N ] };
|
||||||
|
key <AB07> { [ m, M ] };
|
||||||
|
key <AB08> { [ comma, less, singlelowquotemark, doublelowquotemark ] };
|
||||||
|
key <AB09> { [ period, greater, ellipsis, periodcentered ] };
|
||||||
|
key <AB10> { [ slash, question, U2044, questiondown ] }; // U+2044 is FRACTION SLASH
|
||||||
|
};
|
||||||
+156
@@ -0,0 +1,156 @@
|
|||||||
|
// These variants assign ISO_Level3_Shift to various keys
|
||||||
|
// so that levels 3 and 4 can be reached.
|
||||||
|
|
||||||
|
// The default behaviour:
|
||||||
|
// the right Alt key (AltGr) chooses the third symbol engraved on a key.
|
||||||
|
default partial modifier_keys
|
||||||
|
xkb_symbols "ralt_switch" {
|
||||||
|
key <RALT> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The right Alt key never chooses the third level.
|
||||||
|
// This option attempts to undo the effect of a layout's inclusion of
|
||||||
|
// 'ralt_switch'. You may want to also select another level3 option
|
||||||
|
// to map the level3 shift to some other key.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "ralt_alt" {
|
||||||
|
key <RALT> {[ Alt_R, Meta_R ], type[group1]="TWO_LEVEL" };
|
||||||
|
modifier_map Mod1 { <RALT> };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The right Alt key (while pressed) chooses the third shift level,
|
||||||
|
// and Compose is mapped to its second level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "ralt_switch_multikey" {
|
||||||
|
key <RALT> {[ ISO_Level3_Shift, Multi_key ], type[group1]="TWO_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Either Alt key (while pressed) chooses the third shift level.
|
||||||
|
// (To be used mostly to imitate Mac OS functionality.)
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "alt_switch" {
|
||||||
|
include "level3(lalt_switch)"
|
||||||
|
include "level3(ralt_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
// The left Alt key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "lalt_switch" {
|
||||||
|
key <LALT> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The right Ctrl key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "switch" {
|
||||||
|
key <RCTL> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The Menu key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "menu_switch" {
|
||||||
|
key <MENU> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Either Win key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "win_switch" {
|
||||||
|
include "level3(lwin_switch)"
|
||||||
|
include "level3(rwin_switch)"
|
||||||
|
};
|
||||||
|
|
||||||
|
// The left Win key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "lwin_switch" {
|
||||||
|
key <LWIN> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The right Win key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "rwin_switch" {
|
||||||
|
key <RWIN> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The Enter key on the kepypad (while pressed) chooses the third shift level.
|
||||||
|
// (This is especially useful for Mac laptops which miss the right Alt key.)
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "enter_switch" {
|
||||||
|
key <KPEN> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The CapsLock key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "caps_switch" {
|
||||||
|
key <CAPS> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The CapsLock key (while pressed) chooses the third shift level and
|
||||||
|
// Ctrl + CapsLock has the original CapsLock function.
|
||||||
|
// The 2023 DIN standard for German keyboards recommends it as an option:
|
||||||
|
// - https://de.wikipedia.org/wiki/E1_(Tastaturbelegung)#Feststelltaste/Umschaltsperre
|
||||||
|
// - https://en.wikipedia.org/wiki/Caps_Lock#Abolition
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "caps_switch_capslock_with_ctrl" {
|
||||||
|
virtual_modifiers LevelThree;
|
||||||
|
|
||||||
|
key <CAPS> {
|
||||||
|
type[Group1] = "PC_CONTROL_LEVEL2",
|
||||||
|
symbols[Group1] = [ ISO_Level3_Shift, Caps_Lock ],
|
||||||
|
// Explicit actions are preferred over modMap None/Mod5 { Caps_Lock }
|
||||||
|
// because they have no side effect
|
||||||
|
actions[Group1] = [ SetMods(modifiers = LevelThree), LockMods(modifiers = Lock) ]
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
// The Backslash key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "bksl_switch" {
|
||||||
|
key <BKSL> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The AC11 key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "ac11_switch" {
|
||||||
|
key <AC11> {[ ISO_Level3_Shift ], type[Group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The Less/Greater key (while pressed) chooses the third shift level.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "lsgt_switch" {
|
||||||
|
key <LSGT> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The CapsLock key (while pressed) chooses the third shift level,
|
||||||
|
// and latches when pressed together with another third-level chooser.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "caps_switch_latch" {
|
||||||
|
key <CAPS> {[ ISO_Level3_Shift, ISO_Level3_Shift, ISO_Level3_Latch ],
|
||||||
|
type[group1]="THREE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The Backslash key (while pressed) chooses the third shift level,
|
||||||
|
// and latches when pressed together with another third-level chooser.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "bksl_switch_latch" {
|
||||||
|
key <BKSL> {[ ISO_Level3_Shift, ISO_Level3_Shift, ISO_Level3_Latch ],
|
||||||
|
type[group1]="THREE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// The Less/Greater key (while pressed) chooses the third shift level,
|
||||||
|
// and latches when pressed together with another third-level chooser.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "lsgt_switch_latch" {
|
||||||
|
key <LSGT> {[ ISO_Level3_Shift, ISO_Level3_Shift, ISO_Level3_Latch ],
|
||||||
|
type[group1]="THREE_LEVEL" };
|
||||||
|
};
|
||||||
|
|
||||||
|
// Top-row digit key 4 chooses third shift level when pressed alone.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "4_switch_isolated" {
|
||||||
|
override key <AE04> {[ ISO_Level3_Shift ]};
|
||||||
|
};
|
||||||
|
|
||||||
|
// Top-row digit key 9 chooses third shift level when pressed alone.
|
||||||
|
partial modifier_keys
|
||||||
|
xkb_symbols "9_switch_isolated" {
|
||||||
|
override key <AE09> {[ ISO_Level3_Shift ]};
|
||||||
|
};
|
||||||
+2238
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,153 @@
|
|||||||
|
//! xkeyboard-config — keyboard layouts, compiled from the X11 xkeyboard-config database
|
||||||
|
//! into native Zig. It turns a physical key (a USB HID usage, as the input module delivers)
|
||||||
|
//! plus a modifier state into a **keysym** and, when the key produces one, a **character**
|
||||||
|
//! (a Unicode scalar). This is the piece that lets a `KeyEvent.keycode` become a
|
||||||
|
//! `KeyEvent.character`, without shipping an X11 runtime.
|
||||||
|
//!
|
||||||
|
//! The layout tables in `generated/layouts.zig` are produced by
|
||||||
|
//! `tools/make-xkeyboard-config.py` (see ./README.md to regenerate). Those tables are
|
||||||
|
//! deliberately pure data — each key carries its up-to-four levels and an XKB *type*. The
|
||||||
|
//! type -> level selection semantics (which modifier picks which level) live here, so the
|
||||||
|
//! data and the policy are separable.
|
||||||
|
//!
|
||||||
|
//! Scope (documented in README.md): group 1 only, no dead-key/compose composition (a dead
|
||||||
|
//! key returns its keysym with no character), and a curated set of key types. Layouts:
|
||||||
|
//! us, gb, de, fr, es, dvorak.
|
||||||
|
//!
|
||||||
|
//! Upstream xkeyboard-config and keysymdef.h are MIT/X11 licensed; see vendor/COPYING and
|
||||||
|
//! vendor/PROVENANCE.md.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const generated = @import("layouts");
|
||||||
|
|
||||||
|
pub const Level = generated.Level;
|
||||||
|
pub const KeyType = generated.KeyType;
|
||||||
|
pub const Key = generated.Key;
|
||||||
|
pub const Layout = generated.Layout;
|
||||||
|
|
||||||
|
/// The generated layouts, by name — as pointers, so they share identity with `all` and
|
||||||
|
/// `byName` (and match the `*const Layout` that `map` takes).
|
||||||
|
pub const us: *const Layout = &generated.us;
|
||||||
|
pub const gb: *const Layout = &generated.gb;
|
||||||
|
pub const de: *const Layout = &generated.de;
|
||||||
|
pub const fr: *const Layout = &generated.fr;
|
||||||
|
pub const es: *const Layout = &generated.es;
|
||||||
|
pub const dvorak: *const Layout = &generated.dvorak;
|
||||||
|
|
||||||
|
/// Every generated layout, for enumeration (e.g. a settings UI).
|
||||||
|
pub const all = generated.all;
|
||||||
|
|
||||||
|
/// The modifier state that selects a key's level. `level3` is AltGr (ISO Level3 Shift);
|
||||||
|
/// `control` is accepted for completeness but does not affect level selection here.
|
||||||
|
pub const Modifiers = struct {
|
||||||
|
shift: bool = false,
|
||||||
|
caps_lock: bool = false,
|
||||||
|
level3: bool = false,
|
||||||
|
control: bool = false,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The result of a lookup: the X11 `keysym`, and the `character` it produces (a Unicode
|
||||||
|
/// scalar) when it is a printable key — null for keys that produce none (Return, F1, a
|
||||||
|
/// bare dead key, an unmapped key).
|
||||||
|
pub const Mapping = struct {
|
||||||
|
keysym: u32,
|
||||||
|
character: ?u21,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Which level (0..3) a key of `kind` selects under `mods`. XKB's canonical semantics:
|
||||||
|
/// Shift picks the odd level, AltGr (level3) adds 2, and Caps acts like Shift for the
|
||||||
|
/// alphabetic types. See the XKB "key types" — this covers the ones the vendored layouts
|
||||||
|
/// use; anything else falls back to shift-or-not.
|
||||||
|
fn selectLevel(kind: KeyType, mods: Modifiers) usize {
|
||||||
|
const shift_or_caps = mods.shift != mods.caps_lock; // XOR: Caps behaves like Shift
|
||||||
|
const low: usize = if (mods.shift) 1 else 0;
|
||||||
|
const high: usize = if (mods.level3) 2 else 0;
|
||||||
|
return switch (kind) {
|
||||||
|
.one_level => 0,
|
||||||
|
.two_level, .keypad, .other => low,
|
||||||
|
.alphabetic => if (shift_or_caps) 1 else 0,
|
||||||
|
.four_level => low + high,
|
||||||
|
.four_level_alphabetic => (if (shift_or_caps) @as(usize, 1) else 0) + high,
|
||||||
|
// Caps affects only the base pair, not the AltGr pair.
|
||||||
|
.four_level_semialphabetic => if (mods.level3) 2 + low else (if (shift_or_caps) @as(usize, 1) else 0),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Map a physical key (`hid_usage`, a USB HID keyboard-page usage) under `mods` on
|
||||||
|
/// `layout` to its keysym and character. Falls back gracefully when the selected level is
|
||||||
|
/// undefined for the key: it drops the AltGr component, then the shift component, so a key
|
||||||
|
/// with only a base/shift pair still yields something sensible under AltGr.
|
||||||
|
pub fn map(layout: *const Layout, hid_usage: u8, mods: Modifiers) Mapping {
|
||||||
|
const key = &layout.keys[hid_usage];
|
||||||
|
var level = selectLevel(key.kind, mods);
|
||||||
|
// Fall back to a defined level: full -> without AltGr -> base.
|
||||||
|
if (key.levels[level].keysym == 0 and key.levels[level].unicode == 0) {
|
||||||
|
const candidates = [_]usize{ level & 1, 0 };
|
||||||
|
for (candidates) |candidate| {
|
||||||
|
if (key.levels[candidate].keysym != 0 or key.levels[candidate].unicode != 0) {
|
||||||
|
level = candidate;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const chosen = key.levels[level];
|
||||||
|
return .{
|
||||||
|
.keysym = chosen.keysym,
|
||||||
|
.character = if (chosen.unicode != 0) @intCast(chosen.unicode) else null,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Look up a layout by its name (`"us"`, `"gb"`, ...), or null if unknown.
|
||||||
|
pub fn byName(name: []const u8) ?*const Layout {
|
||||||
|
for (all) |layout| {
|
||||||
|
if (std.mem.eql(u8, layout.name, name)) return layout;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- tests (host-run via `zig build test`) ---------------------------------
|
||||||
|
|
||||||
|
const testing = std.testing;
|
||||||
|
|
||||||
|
// USB HID usages used in the tests (keyboard page 0x07).
|
||||||
|
const hid_a: u8 = 0x04;
|
||||||
|
const hid_1: u8 = 0x1e;
|
||||||
|
const hid_3: u8 = 0x20;
|
||||||
|
|
||||||
|
test "us: letters obey shift and caps" {
|
||||||
|
try testing.expectEqual(@as(?u21, 'a'), map(us, hid_a, .{}).character);
|
||||||
|
try testing.expectEqual(@as(?u21, 'A'), map(us, hid_a, .{ .shift = true }).character);
|
||||||
|
try testing.expectEqual(@as(?u21, 'A'), map(us, hid_a, .{ .caps_lock = true }).character);
|
||||||
|
// Shift + Caps cancels for an alphabetic key.
|
||||||
|
try testing.expectEqual(@as(?u21, 'a'), map(us, hid_a, .{ .shift = true, .caps_lock = true }).character);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "us: digits and their shifted symbols" {
|
||||||
|
try testing.expectEqual(@as(?u21, '1'), map(us, hid_1, .{}).character);
|
||||||
|
try testing.expectEqual(@as(?u21, '!'), map(us, hid_1, .{ .shift = true }).character);
|
||||||
|
try testing.expectEqual(@as(?u21, '3'), map(us, hid_3, .{}).character);
|
||||||
|
try testing.expectEqual(@as(?u21, '#'), map(us, hid_3, .{ .shift = true }).character);
|
||||||
|
// A digit is not alphabetic: Caps alone must not shift it.
|
||||||
|
try testing.expectEqual(@as(?u21, '3'), map(us, hid_3, .{ .caps_lock = true }).character);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "layouts differ: GB pound vs US hash on shift+3" {
|
||||||
|
try testing.expectEqual(@as(?u21, '#'), map(us, hid_3, .{ .shift = true }).character);
|
||||||
|
try testing.expectEqual(@as(?u21, '£'), map(gb, hid_3, .{ .shift = true }).character);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "french azerty places q where us has a" {
|
||||||
|
try testing.expectEqual(@as(?u21, 'q'), map(fr, hid_a, .{}).character);
|
||||||
|
try testing.expectEqual(@as(?u21, 'Q'), map(fr, hid_a, .{ .shift = true }).character);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "byName resolves and rejects" {
|
||||||
|
try testing.expect(byName("us") == us);
|
||||||
|
try testing.expect(byName("gb") == gb);
|
||||||
|
try testing.expect(byName("nonsense") == null);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "unmapped key yields no character" {
|
||||||
|
// HID 0x00 is not a key; every level is empty.
|
||||||
|
try testing.expectEqual(@as(?u21, null), map(us, 0x00, .{}).character);
|
||||||
|
}
|
||||||
@@ -1,737 +0,0 @@
|
|||||||
//! A tree-walking AML interpreter — the evaluation stage on top of the parser's
|
|
||||||
//! structural namespace. It executes control methods (their bodies captured by
|
|
||||||
//! the parser) far enough to serve device discovery: device status (`_STA`, is a
|
|
||||||
//! device present), current resource settings (`_CRS`), and the operators, control
|
|
||||||
//! flow, locals/args, and
|
|
||||||
//! OperationRegion field access those methods reach for.
|
|
||||||
//!
|
|
||||||
//! Scope: integers, buffers, strings, packages, and references; If/Else/While/
|
|
||||||
//! Return; the arithmetic/logic operators; method invocation; Name/Local/Arg
|
|
||||||
//! access; CreateField buffer patching (the common current-resource-settings
|
|
||||||
//! (`_CRS`) idiom); and field
|
|
||||||
//! reads/writes against SystemMemory and SystemIO regions. Opcodes outside this
|
|
||||||
//! set return `error.Unsupported`, which callers treat as "couldn't evaluate" and
|
|
||||||
//! fall back — never a hard failure.
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
const op = @import("opcodes.zig");
|
|
||||||
const nsp = @import("namespace.zig");
|
|
||||||
const Node = nsp.Node;
|
|
||||||
const Namespace = nsp.Namespace;
|
|
||||||
|
|
||||||
/// Injected hardware access for OperationRegion reads/writes (the arch VMM + pio).
|
|
||||||
pub const Hal = struct {
|
|
||||||
mapMmio: *const fn (virt: u64, phys: u64, writable: bool) void,
|
|
||||||
pioRead: *const fn (width: u8, port: u16) u32,
|
|
||||||
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
|
||||||
};
|
|
||||||
|
|
||||||
pub const Error = error{ Unsupported, Truncated, DivByZero } || std.mem.Allocator.Error;
|
|
||||||
|
|
||||||
/// A runtime AML value.
|
|
||||||
pub const Object = union(enum) {
|
|
||||||
uninitialized,
|
|
||||||
integer: u64,
|
|
||||||
buffer: []u8,
|
|
||||||
string: []u8,
|
|
||||||
package: []Object,
|
|
||||||
reference: *Node,
|
|
||||||
|
|
||||||
pub fn asInt(self: Object) Error!u64 {
|
|
||||||
return switch (self) {
|
|
||||||
.integer => |v| v,
|
|
||||||
.buffer => |b| blk: {
|
|
||||||
var v: u64 = 0;
|
|
||||||
for (b, 0..) |byte, i| {
|
|
||||||
if (i >= 8) break;
|
|
||||||
v |= @as(u64, byte) << @intCast(i * 8);
|
|
||||||
}
|
|
||||||
break :blk v;
|
|
||||||
},
|
|
||||||
else => error.Unsupported,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
const max_segs = 16;
|
|
||||||
const NamePath = struct {
|
|
||||||
rooted: bool = false,
|
|
||||||
parents: u8 = 0,
|
|
||||||
segs: [max_segs][4]u8 = undefined,
|
|
||||||
count: usize = 0,
|
|
||||||
fn slice(self: *const NamePath) []const [4]u8 {
|
|
||||||
return self.segs[0..self.count];
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
const Cursor = struct {
|
|
||||||
b: []const u8,
|
|
||||||
i: usize = 0,
|
|
||||||
|
|
||||||
fn eof(self: *Cursor) bool {
|
|
||||||
return self.i >= self.b.len;
|
|
||||||
}
|
|
||||||
fn peek(self: *Cursor) ?u8 {
|
|
||||||
return if (self.eof()) null else self.b[self.i];
|
|
||||||
}
|
|
||||||
fn byte(self: *Cursor) Error!u8 {
|
|
||||||
if (self.eof()) return error.Truncated;
|
|
||||||
const v = self.b[self.i];
|
|
||||||
self.i += 1;
|
|
||||||
return v;
|
|
||||||
}
|
|
||||||
fn take(self: *Cursor, n: usize) Error![]const u8 {
|
|
||||||
if (self.i + n > self.b.len) return error.Truncated;
|
|
||||||
const s = self.b[self.i .. self.i + n];
|
|
||||||
self.i += n;
|
|
||||||
return s;
|
|
||||||
}
|
|
||||||
fn pkgLen(self: *Cursor) Error!usize {
|
|
||||||
const lead = try self.byte();
|
|
||||||
const follow: usize = lead >> 6;
|
|
||||||
if (follow == 0) return lead & 0x3F;
|
|
||||||
var value: usize = lead & 0x0F;
|
|
||||||
var k: usize = 0;
|
|
||||||
while (k < follow) : (k += 1) value |= @as(usize, try self.byte()) << @intCast(4 + k * 8);
|
|
||||||
return value;
|
|
||||||
}
|
|
||||||
fn nameString(self: *Cursor) Error!NamePath {
|
|
||||||
var np = NamePath{};
|
|
||||||
if (self.peek() == op.root_char) {
|
|
||||||
np.rooted = true;
|
|
||||||
self.i += 1;
|
|
||||||
} else {
|
|
||||||
while (self.peek() == op.parent_prefix_char) : (self.i += 1) np.parents += 1;
|
|
||||||
}
|
|
||||||
const lead = self.peek() orelse return np;
|
|
||||||
switch (lead) {
|
|
||||||
0x00 => self.i += 1,
|
|
||||||
op.dual_name_prefix => {
|
|
||||||
self.i += 1;
|
|
||||||
try self.seg(&np);
|
|
||||||
try self.seg(&np);
|
|
||||||
},
|
|
||||||
op.multi_name_prefix => {
|
|
||||||
self.i += 1;
|
|
||||||
const cnt = try self.byte();
|
|
||||||
var k: usize = 0;
|
|
||||||
while (k < cnt) : (k += 1) try self.seg(&np);
|
|
||||||
},
|
|
||||||
else => try self.seg(&np),
|
|
||||||
}
|
|
||||||
return np;
|
|
||||||
}
|
|
||||||
fn seg(self: *Cursor, np: *NamePath) Error!void {
|
|
||||||
const s = try self.take(4);
|
|
||||||
if (np.count < max_segs) {
|
|
||||||
np.segs[np.count] = s[0..4].*;
|
|
||||||
np.count += 1;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
const Frame = struct {
|
|
||||||
args: [7]Object = .{.uninitialized} ** 7,
|
|
||||||
locals: [8]Object = .{.uninitialized} ** 8,
|
|
||||||
scope: *Node,
|
|
||||||
ret: Object = .uninitialized,
|
|
||||||
returned: bool = false,
|
|
||||||
broke: bool = false,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// A CreateField binding: a name that indexes into a buffer object.
|
|
||||||
const BufField = struct { buf: *Node, byte_off: usize, bit_width: u32 };
|
|
||||||
|
|
||||||
pub const Interp = struct {
|
|
||||||
ns: *Namespace,
|
|
||||||
hal: Hal,
|
|
||||||
arena: std.mem.Allocator,
|
|
||||||
/// Runtime object overrides for Name nodes (Store targets, patched buffers).
|
|
||||||
dyn: std.AutoHashMapUnmanaged(*Node, Object) = .{},
|
|
||||||
/// CreateField bindings active for the current evaluation.
|
|
||||||
fields: std.AutoHashMapUnmanaged(*Node, BufField) = .{},
|
|
||||||
|
|
||||||
pub fn init(ns: *Namespace, hal: Hal, arena: std.mem.Allocator) Interp {
|
|
||||||
return .{ .ns = ns, .hal = hal, .arena = arena };
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Evaluate a namespace object: invoke a Method, read a Name's value, or read a
|
|
||||||
/// Field. Resets per-evaluation runtime state first.
|
|
||||||
pub fn evaluate(self: *Interp, node: *Node, args: []const Object) Error!Object {
|
|
||||||
self.dyn.clearRetainingCapacity();
|
|
||||||
self.fields.clearRetainingCapacity();
|
|
||||||
return self.invoke(node, args);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn invoke(self: *Interp, node: *Node, args: []const Object) Error!Object {
|
|
||||||
switch (node.kind) {
|
|
||||||
.method => {
|
|
||||||
var frame = Frame{ .scope = node };
|
|
||||||
for (args, 0..) |a, i| {
|
|
||||||
if (i < frame.args.len) frame.args[i] = a;
|
|
||||||
}
|
|
||||||
var cur = Cursor{ .b = node.value };
|
|
||||||
try self.execList(&cur, &frame);
|
|
||||||
return frame.ret;
|
|
||||||
},
|
|
||||||
.name => {
|
|
||||||
if (self.dyn.get(node)) |o| return o;
|
|
||||||
var cur = Cursor{ .b = node.value };
|
|
||||||
var frame = Frame{ .scope = node.parent orelse self.ns.root };
|
|
||||||
return self.term(&cur, &frame);
|
|
||||||
},
|
|
||||||
.field => return .{ .integer = try self.readField(node) },
|
|
||||||
else => return .{ .reference = node },
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Execute a TermList until it ends or the frame returns/breaks.
|
|
||||||
fn execList(self: *Interp, cur: *Cursor, frame: *Frame) Error!void {
|
|
||||||
while (!cur.eof() and !frame.returned and !frame.broke) {
|
|
||||||
_ = try self.term(cur, frame);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Evaluate/execute one term, returning its value (`.uninitialized` for pure
|
|
||||||
/// statements).
|
|
||||||
fn term(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const lead = cur.peek() orelse return error.Truncated;
|
|
||||||
if (isNameStart(lead)) return self.nameRef(cur, frame);
|
|
||||||
_ = try cur.byte();
|
|
||||||
|
|
||||||
return switch (lead) {
|
|
||||||
op.zero_op => Object{ .integer = 0 },
|
|
||||||
op.one_op => Object{ .integer = 1 },
|
|
||||||
op.ones_op => Object{ .integer = ~@as(u64, 0) },
|
|
||||||
op.byte_prefix => Object{ .integer = try self.readConst(cur, 1) },
|
|
||||||
op.word_prefix => Object{ .integer = try self.readConst(cur, 2) },
|
|
||||||
op.dword_prefix => Object{ .integer = try self.readConst(cur, 4) },
|
|
||||||
op.qword_prefix => Object{ .integer = try self.readConst(cur, 8) },
|
|
||||||
op.string_prefix => try self.readString(cur),
|
|
||||||
op.buffer_op => try self.buffer(cur, frame),
|
|
||||||
op.package_op, op.var_package_op => try self.package(cur, frame, lead == op.var_package_op),
|
|
||||||
|
|
||||||
op.local0_op...op.local7_op => frame.locals[lead - op.local0_op],
|
|
||||||
op.arg0_op...op.arg6_op => frame.args[lead - op.arg0_op],
|
|
||||||
|
|
||||||
op.return_op => blk: {
|
|
||||||
frame.ret = try self.term(cur, frame);
|
|
||||||
frame.returned = true;
|
|
||||||
break :blk .uninitialized;
|
|
||||||
},
|
|
||||||
op.break_op => blk: {
|
|
||||||
frame.broke = true;
|
|
||||||
break :blk .uninitialized;
|
|
||||||
},
|
|
||||||
op.continue_op, op.noop_op => .uninitialized,
|
|
||||||
|
|
||||||
op.if_op => try self.ifElse(cur, frame),
|
|
||||||
op.while_op => try self.whileLoop(cur, frame),
|
|
||||||
op.store_op => try self.store(cur, frame),
|
|
||||||
op.increment_op => try self.incDec(cur, frame, 1),
|
|
||||||
op.decrement_op => try self.incDec(cur, frame, -1),
|
|
||||||
|
|
||||||
op.add_op => try self.binary(cur, frame, .add),
|
|
||||||
op.subtract_op => try self.binary(cur, frame, .sub),
|
|
||||||
op.multiply_op => try self.binary(cur, frame, .mul),
|
|
||||||
op.mod_op => try self.binary(cur, frame, .mod),
|
|
||||||
op.and_op => try self.binary(cur, frame, .band),
|
|
||||||
op.or_op => try self.binary(cur, frame, .bor),
|
|
||||||
op.xor_op => try self.binary(cur, frame, .bxor),
|
|
||||||
op.nand_op => try self.binary(cur, frame, .nand),
|
|
||||||
op.nor_op => try self.binary(cur, frame, .nor),
|
|
||||||
op.shift_left_op => try self.binary(cur, frame, .shl),
|
|
||||||
op.shift_right_op => try self.binary(cur, frame, .shr),
|
|
||||||
op.divide_op => try self.divide(cur, frame),
|
|
||||||
|
|
||||||
op.land_op => try self.logic2(cur, frame, .land),
|
|
||||||
op.lor_op => try self.logic2(cur, frame, .lor),
|
|
||||||
op.lequal_op => try self.logic2(cur, frame, .eq),
|
|
||||||
op.lgreater_op => try self.logic2(cur, frame, .gt),
|
|
||||||
op.lless_op => try self.logic2(cur, frame, .lt),
|
|
||||||
op.lnot_op => try self.lnot(cur, frame),
|
|
||||||
|
|
||||||
op.not_op => blk: {
|
|
||||||
const v = try self.evalInt(cur, frame);
|
|
||||||
const r = ~v;
|
|
||||||
try self.storeTarget(cur, frame, .{ .integer = r });
|
|
||||||
break :blk .{ .integer = r };
|
|
||||||
},
|
|
||||||
|
|
||||||
op.size_of_op => try self.sizeOf(cur, frame),
|
|
||||||
op.index_op => try self.index(cur, frame),
|
|
||||||
op.deref_of_op => try self.derefOf(cur, frame),
|
|
||||||
op.to_integer_op => blk: {
|
|
||||||
const v = try self.evalInt(cur, frame);
|
|
||||||
try self.storeTarget(cur, frame, .{ .integer = v });
|
|
||||||
break :blk .{ .integer = v };
|
|
||||||
},
|
|
||||||
op.to_buffer_op => try self.passThroughUnary(cur, frame),
|
|
||||||
|
|
||||||
op.ext_op_prefix => try self.ext(cur, frame),
|
|
||||||
|
|
||||||
// CreateXField: source, index, name (bit widths differ by op)
|
|
||||||
op.create_bit_field_op => try self.createField(cur, frame, 1),
|
|
||||||
op.create_byte_field_op => try self.createField(cur, frame, 8),
|
|
||||||
op.create_word_field_op => try self.createField(cur, frame, 16),
|
|
||||||
op.create_dword_field_op => try self.createField(cur, frame, 32),
|
|
||||||
op.create_qword_field_op => try self.createField(cur, frame, 64),
|
|
||||||
|
|
||||||
else => error.Unsupported,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- name references ----------------------------------------------------
|
|
||||||
|
|
||||||
fn nameRef(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const np = try cur.nameString();
|
|
||||||
const node = self.ns.resolve(frame.scope, np.rooted, np.parents, np.slice()) orelse
|
|
||||||
return .uninitialized; // unknown name -> treat as uninitialised
|
|
||||||
switch (node.kind) {
|
|
||||||
.method => {
|
|
||||||
var argbuf: [7]Object = undefined;
|
|
||||||
var i: usize = 0;
|
|
||||||
while (i < node.arg_count and i < argbuf.len) : (i += 1) argbuf[i] = try self.term(cur, frame);
|
|
||||||
return self.invoke(node, argbuf[0..@min(node.arg_count, argbuf.len)]);
|
|
||||||
},
|
|
||||||
.field => return .{ .integer = try self.readField(node) },
|
|
||||||
.name => return self.invoke(node, &.{}),
|
|
||||||
else => return .{ .reference = node },
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- data objects -------------------------------------------------------
|
|
||||||
|
|
||||||
fn readConst(self: *Interp, cur: *Cursor, n: usize) Error!u64 {
|
|
||||||
_ = self;
|
|
||||||
const bytes = try cur.take(n);
|
|
||||||
var v: u64 = 0;
|
|
||||||
for (bytes, 0..) |b, i| v |= @as(u64, b) << @intCast(i * 8);
|
|
||||||
return v;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn readString(self: *Interp, cur: *Cursor) Error!Object {
|
|
||||||
const start = cur.i;
|
|
||||||
while (cur.peek()) |c| {
|
|
||||||
cur.i += 1;
|
|
||||||
if (c == 0) break;
|
|
||||||
}
|
|
||||||
const raw = cur.b[start .. cur.i - 1];
|
|
||||||
const s = try self.arena.dupe(u8, raw);
|
|
||||||
return .{ .string = s };
|
|
||||||
}
|
|
||||||
|
|
||||||
fn buffer(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const start = cur.i;
|
|
||||||
const len = try cur.pkgLen();
|
|
||||||
const end = @min(start + len, cur.b.len);
|
|
||||||
const size = try self.evalInt(cur, frame);
|
|
||||||
const data = cur.b[@min(cur.i, end)..end];
|
|
||||||
const buf = try self.arena.alloc(u8, @intCast(size));
|
|
||||||
@memset(buf, 0);
|
|
||||||
@memcpy(buf[0..@min(buf.len, data.len)], data[0..@min(buf.len, data.len)]);
|
|
||||||
cur.i = end;
|
|
||||||
return .{ .buffer = buf };
|
|
||||||
}
|
|
||||||
|
|
||||||
fn package(self: *Interp, cur: *Cursor, frame: *Frame, variable: bool) Error!Object {
|
|
||||||
const start = cur.i;
|
|
||||||
const len = try cur.pkgLen();
|
|
||||||
const end = @min(start + len, cur.b.len);
|
|
||||||
const count: usize = if (variable) @intCast(try self.evalInt(cur, frame)) else try cur.byte();
|
|
||||||
const elems = try self.arena.alloc(Object, count);
|
|
||||||
var i: usize = 0;
|
|
||||||
while (i < count and cur.i < end) : (i += 1) elems[i] = try self.term(cur, frame);
|
|
||||||
while (i < count) : (i += 1) elems[i] = .uninitialized;
|
|
||||||
cur.i = end;
|
|
||||||
return .{ .package = elems };
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- operators ----------------------------------------------------------
|
|
||||||
|
|
||||||
const BinOp = enum { add, sub, mul, mod, band, bor, bxor, nand, nor, shl, shr };
|
|
||||||
|
|
||||||
fn binary(self: *Interp, cur: *Cursor, frame: *Frame, kind: BinOp) Error!Object {
|
|
||||||
const a = try self.evalInt(cur, frame);
|
|
||||||
const b = try self.evalInt(cur, frame);
|
|
||||||
const r: u64 = switch (kind) {
|
|
||||||
.add => a +% b,
|
|
||||||
.sub => a -% b,
|
|
||||||
.mul => a *% b,
|
|
||||||
.mod => if (b == 0) return error.DivByZero else a % b,
|
|
||||||
.band => a & b,
|
|
||||||
.bor => a | b,
|
|
||||||
.bxor => a ^ b,
|
|
||||||
.nand => ~(a & b),
|
|
||||||
.nor => ~(a | b),
|
|
||||||
.shl => if (b >= 64) 0 else a << @intCast(b),
|
|
||||||
.shr => if (b >= 64) 0 else a >> @intCast(b),
|
|
||||||
};
|
|
||||||
try self.storeTarget(cur, frame, .{ .integer = r });
|
|
||||||
return .{ .integer = r };
|
|
||||||
}
|
|
||||||
|
|
||||||
fn divide(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const a = try self.evalInt(cur, frame);
|
|
||||||
const b = try self.evalInt(cur, frame);
|
|
||||||
if (b == 0) return error.DivByZero;
|
|
||||||
try self.storeTarget(cur, frame, .{ .integer = a % b }); // remainder target
|
|
||||||
try self.storeTarget(cur, frame, .{ .integer = a / b }); // quotient target
|
|
||||||
return .{ .integer = a / b };
|
|
||||||
}
|
|
||||||
|
|
||||||
const LogicOp = enum { land, lor, eq, gt, lt };
|
|
||||||
|
|
||||||
fn logic2(self: *Interp, cur: *Cursor, frame: *Frame, kind: LogicOp) Error!Object {
|
|
||||||
const a = try self.evalInt(cur, frame);
|
|
||||||
const b = try self.evalInt(cur, frame);
|
|
||||||
const r = switch (kind) {
|
|
||||||
.land => a != 0 and b != 0,
|
|
||||||
.lor => a != 0 or b != 0,
|
|
||||||
.eq => a == b,
|
|
||||||
.gt => a > b,
|
|
||||||
.lt => a < b,
|
|
||||||
};
|
|
||||||
return .{ .integer = if (r) ~@as(u64, 0) else 0 };
|
|
||||||
}
|
|
||||||
|
|
||||||
fn lnot(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
// 0x92 0x93/94/95 are the compound comparisons.
|
|
||||||
const b = cur.peek() orelse return error.Truncated;
|
|
||||||
switch (b) {
|
|
||||||
op.lnot.not_equal => {
|
|
||||||
cur.i += 1;
|
|
||||||
const x = try self.evalInt(cur, frame);
|
|
||||||
const y = try self.evalInt(cur, frame);
|
|
||||||
return .{ .integer = if (x != y) ~@as(u64, 0) else 0 };
|
|
||||||
},
|
|
||||||
op.lnot.less_equal => {
|
|
||||||
cur.i += 1;
|
|
||||||
const x = try self.evalInt(cur, frame);
|
|
||||||
const y = try self.evalInt(cur, frame);
|
|
||||||
return .{ .integer = if (x <= y) ~@as(u64, 0) else 0 };
|
|
||||||
},
|
|
||||||
op.lnot.greater_equal => {
|
|
||||||
cur.i += 1;
|
|
||||||
const x = try self.evalInt(cur, frame);
|
|
||||||
const y = try self.evalInt(cur, frame);
|
|
||||||
return .{ .integer = if (x >= y) ~@as(u64, 0) else 0 };
|
|
||||||
},
|
|
||||||
else => {
|
|
||||||
const x = try self.evalInt(cur, frame);
|
|
||||||
return .{ .integer = if (x == 0) ~@as(u64, 0) else 0 };
|
|
||||||
},
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn incDec(self: *Interp, cur: *Cursor, frame: *Frame, delta: i64) Error!Object {
|
|
||||||
// Operand is a SuperName that is both read and written.
|
|
||||||
const save = cur.i;
|
|
||||||
const cur_val = try self.term(cur, frame);
|
|
||||||
const v = try cur_val.asInt();
|
|
||||||
const r = if (delta > 0) v +% 1 else v -% 1;
|
|
||||||
var tcur = Cursor{ .b = cur.b, .i = save };
|
|
||||||
try self.storeInto(&tcur, frame, .{ .integer = r });
|
|
||||||
return .{ .integer = r };
|
|
||||||
}
|
|
||||||
|
|
||||||
fn sizeOf(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const o = try self.term(cur, frame);
|
|
||||||
return .{ .integer = switch (o) {
|
|
||||||
.buffer => |b| b.len,
|
|
||||||
.string => |s| s.len,
|
|
||||||
.package => |p| p.len,
|
|
||||||
else => 0,
|
|
||||||
} };
|
|
||||||
}
|
|
||||||
|
|
||||||
fn passThroughUnary(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const o = try self.term(cur, frame);
|
|
||||||
try self.storeTarget(cur, frame, o);
|
|
||||||
return o;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn index(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const src = try self.term(cur, frame);
|
|
||||||
const idx: usize = @intCast(try self.evalInt(cur, frame));
|
|
||||||
// Optional target (a reference); we don't materialise references, so store
|
|
||||||
// the indexed value if a target is present.
|
|
||||||
const val: Object = switch (src) {
|
|
||||||
.buffer => |b| .{ .integer = if (idx < b.len) b[idx] else 0 },
|
|
||||||
.package => |p| if (idx < p.len) p[idx] else .uninitialized,
|
|
||||||
.string => |s| .{ .integer = if (idx < s.len) s[idx] else 0 },
|
|
||||||
else => .uninitialized,
|
|
||||||
};
|
|
||||||
try self.storeTarget(cur, frame, val);
|
|
||||||
return val;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn derefOf(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const o = try self.term(cur, frame);
|
|
||||||
return switch (o) {
|
|
||||||
.reference => |n| self.invoke(n, &.{}),
|
|
||||||
else => o,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- control flow -------------------------------------------------------
|
|
||||||
|
|
||||||
fn ifElse(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const start = cur.i;
|
|
||||||
const end = @min(start + try cur.pkgLen(), cur.b.len);
|
|
||||||
const cond = try self.evalInt(cur, frame);
|
|
||||||
if (cond != 0) {
|
|
||||||
var body = Cursor{ .b = cur.b[0..end], .i = cur.i };
|
|
||||||
try self.execList(&body, frame);
|
|
||||||
cur.i = end;
|
|
||||||
// Skip a trailing Else.
|
|
||||||
if (cur.peek() == op.else_op) {
|
|
||||||
cur.i += 1;
|
|
||||||
const es = cur.i;
|
|
||||||
const ee = @min(es + try cur.pkgLen(), cur.b.len);
|
|
||||||
cur.i = ee;
|
|
||||||
}
|
|
||||||
} else {
|
|
||||||
cur.i = end;
|
|
||||||
if (cur.peek() == op.else_op) {
|
|
||||||
cur.i += 1;
|
|
||||||
const es = cur.i;
|
|
||||||
const ee = @min(es + try cur.pkgLen(), cur.b.len);
|
|
||||||
var body = Cursor{ .b = cur.b[0..ee], .i = cur.i };
|
|
||||||
try self.execList(&body, frame);
|
|
||||||
cur.i = ee;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return .uninitialized;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn whileLoop(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const start = cur.i;
|
|
||||||
const end = @min(start + try cur.pkgLen(), cur.b.len);
|
|
||||||
const pred_at = cur.i;
|
|
||||||
var guard: usize = 0;
|
|
||||||
while (guard < 100_000) : (guard += 1) {
|
|
||||||
var pc = Cursor{ .b = cur.b[0..end], .i = pred_at };
|
|
||||||
const cond = try self.evalInt(&pc, frame);
|
|
||||||
if (cond == 0) break;
|
|
||||||
var body = Cursor{ .b = cur.b[0..end], .i = pc.i };
|
|
||||||
try self.execList(&body, frame);
|
|
||||||
if (frame.returned) break;
|
|
||||||
if (frame.broke) {
|
|
||||||
frame.broke = false;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
cur.i = end;
|
|
||||||
return .uninitialized;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- store --------------------------------------------------------------
|
|
||||||
|
|
||||||
fn store(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const value = try self.term(cur, frame);
|
|
||||||
try self.storeInto(cur, frame, value);
|
|
||||||
return value;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// A Store *target* that may be NullName (no store).
|
|
||||||
fn storeTarget(self: *Interp, cur: *Cursor, frame: *Frame, value: Object) Error!void {
|
|
||||||
if (cur.peek() == 0x00) {
|
|
||||||
cur.i += 1; // NullName
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
try self.storeInto(cur, frame, value);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn storeInto(self: *Interp, cur: *Cursor, frame: *Frame, value: Object) Error!void {
|
|
||||||
const lead = cur.peek() orelse return error.Truncated;
|
|
||||||
if (isNameStart(lead)) {
|
|
||||||
const np = try cur.nameString();
|
|
||||||
const node = self.ns.resolve(frame.scope, np.rooted, np.parents, np.slice()) orelse return;
|
|
||||||
if (self.fields.get(node)) |bf| {
|
|
||||||
try self.writeBufField(bf, try value.asInt());
|
|
||||||
} else if (node.kind == .field) {
|
|
||||||
try self.writeField(node, try value.asInt());
|
|
||||||
} else {
|
|
||||||
try self.dyn.put(self.arena, node, value);
|
|
||||||
}
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
_ = try cur.byte();
|
|
||||||
switch (lead) {
|
|
||||||
0x00 => {}, // NullName
|
|
||||||
op.local0_op...op.local7_op => frame.locals[lead - op.local0_op] = value,
|
|
||||||
op.arg0_op...op.arg6_op => frame.args[lead - op.arg0_op] = value,
|
|
||||||
op.index_op => {
|
|
||||||
const src = try self.term(cur, frame);
|
|
||||||
const idx: usize = @intCast(try self.evalInt(cur, frame));
|
|
||||||
switch (src) {
|
|
||||||
.buffer => |b| if (idx < b.len) {
|
|
||||||
b[idx] = @truncate(try value.asInt());
|
|
||||||
},
|
|
||||||
.package => |p| if (idx < p.len) {
|
|
||||||
p[idx] = value;
|
|
||||||
},
|
|
||||||
else => {},
|
|
||||||
}
|
|
||||||
},
|
|
||||||
else => return error.Unsupported,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- CreateField (buffer patching) --------------------------------------
|
|
||||||
|
|
||||||
fn createField(self: *Interp, cur: *Cursor, frame: *Frame, bit_width: u32) Error!Object {
|
|
||||||
const src = try self.term(cur, frame); // source buffer (as a reference or value)
|
|
||||||
const bit_index = try self.evalInt(cur, frame);
|
|
||||||
const np = try cur.nameString();
|
|
||||||
const node = self.ns.resolve(frame.scope, np.rooted, np.parents, np.slice()) orelse return .uninitialized;
|
|
||||||
|
|
||||||
// Bind the new name to the source buffer's node so stores land in it.
|
|
||||||
const buf_node: *Node = switch (src) {
|
|
||||||
.reference => |n| n,
|
|
||||||
else => return .uninitialized,
|
|
||||||
};
|
|
||||||
// Materialise the buffer into `dyn` so patches persist and are returned.
|
|
||||||
if (self.dyn.get(buf_node) == null) {
|
|
||||||
const val = try self.invoke(buf_node, &.{});
|
|
||||||
try self.dyn.put(self.arena, buf_node, val);
|
|
||||||
}
|
|
||||||
const byte_off: usize = @intCast(bit_index / 8);
|
|
||||||
try self.fields.put(self.arena, node, .{ .buf = buf_node, .byte_off = byte_off, .bit_width = bit_width });
|
|
||||||
return .uninitialized;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn writeBufField(self: *Interp, bf: BufField, value: u64) Error!void {
|
|
||||||
const obj = self.dyn.get(bf.buf) orelse return;
|
|
||||||
const buf = switch (obj) {
|
|
||||||
.buffer => |b| b,
|
|
||||||
else => return,
|
|
||||||
};
|
|
||||||
const nbytes = (bf.bit_width + 7) / 8;
|
|
||||||
var k: usize = 0;
|
|
||||||
while (k < nbytes and bf.byte_off + k < buf.len) : (k += 1) {
|
|
||||||
buf[bf.byte_off + k] = @truncate(value >> @intCast(k * 8));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- OperationRegion field access ---------------------------------------
|
|
||||||
|
|
||||||
fn readField(self: *Interp, field: *Node) Error!u64 {
|
|
||||||
const region = field.region orelse return error.Unsupported;
|
|
||||||
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
|
||||||
const base = try self.regionBase(region);
|
|
||||||
const start_byte = base + field.bit_offset / 8;
|
|
||||||
const shift: u7 = @intCast(field.bit_offset % 8);
|
|
||||||
const total = @as(usize, shift) + field.bit_width;
|
|
||||||
const nbytes = (total + 7) / 8;
|
|
||||||
var raw: u128 = 0;
|
|
||||||
var k: usize = 0;
|
|
||||||
while (k < nbytes) : (k += 1) {
|
|
||||||
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
|
||||||
}
|
|
||||||
const masked = (raw >> shift) & bitMask(field.bit_width);
|
|
||||||
return @truncate(masked);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn writeField(self: *Interp, field: *Node, value: u64) Error!void {
|
|
||||||
const region = field.region orelse return error.Unsupported;
|
|
||||||
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
|
||||||
const base = try self.regionBase(region);
|
|
||||||
const start_byte = base + field.bit_offset / 8;
|
|
||||||
const shift: u7 = @intCast(field.bit_offset % 8);
|
|
||||||
const total = @as(usize, shift) + field.bit_width;
|
|
||||||
const nbytes = (total + 7) / 8;
|
|
||||||
// Read-modify-write byte by byte.
|
|
||||||
var raw: u128 = 0;
|
|
||||||
var k: usize = 0;
|
|
||||||
while (k < nbytes) : (k += 1) {
|
|
||||||
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
|
||||||
}
|
|
||||||
const mask = bitMask(field.bit_width) << shift;
|
|
||||||
raw = (raw & ~mask) | ((@as(u128, value) << shift) & mask);
|
|
||||||
k = 0;
|
|
||||||
while (k < nbytes) : (k += 1) {
|
|
||||||
try self.writeRegionByte(region.region_space, start_byte + k, @truncate(raw >> @intCast(k * 8)));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn regionBase(self: *Interp, region: *Node) Error!u64 {
|
|
||||||
var cur = Cursor{ .b = region.region_offset_aml };
|
|
||||||
var frame = Frame{ .scope = region.parent orelse self.ns.root };
|
|
||||||
return (try self.term(&cur, &frame)).asInt();
|
|
||||||
}
|
|
||||||
|
|
||||||
fn readRegionByte(self: *Interp, space: u8, addr: u64) Error!u8 {
|
|
||||||
switch (space) {
|
|
||||||
0 => { // SystemMemory
|
|
||||||
self.hal.mapMmio(addr & ~@as(u64, 0xFFF), addr & ~@as(u64, 0xFFF), true);
|
|
||||||
const p: *align(1) const volatile u8 = @ptrFromInt(addr);
|
|
||||||
return p.*;
|
|
||||||
},
|
|
||||||
1 => return @truncate(self.hal.pioRead(1, @intCast(addr & 0xFFFF))), // SystemIO
|
|
||||||
else => return error.Unsupported,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn writeRegionByte(self: *Interp, space: u8, addr: u64, value: u8) Error!void {
|
|
||||||
switch (space) {
|
|
||||||
0 => {
|
|
||||||
self.hal.mapMmio(addr & ~@as(u64, 0xFFF), addr & ~@as(u64, 0xFFF), true);
|
|
||||||
const p: *align(1) volatile u8 = @ptrFromInt(addr);
|
|
||||||
p.* = value;
|
|
||||||
},
|
|
||||||
1 => self.hal.pioWrite(1, @intCast(addr & 0xFFFF), value),
|
|
||||||
else => return error.Unsupported,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- extended opcodes ---------------------------------------------------
|
|
||||||
|
|
||||||
fn ext(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
|
||||||
const e = try cur.byte();
|
|
||||||
switch (e) {
|
|
||||||
op.ext.debug => return .uninitialized,
|
|
||||||
op.ext.revision => return .{ .integer = 2 },
|
|
||||||
op.ext.timer => return .{ .integer = 0 },
|
|
||||||
// Mutex/Event ops are no-ops in this single-threaded evaluator.
|
|
||||||
op.ext.acquire => {
|
|
||||||
_ = try self.term(cur, frame); // mutex SuperName
|
|
||||||
_ = try cur.take(2); // timeout
|
|
||||||
return .{ .integer = 0 }; // acquired
|
|
||||||
},
|
|
||||||
op.ext.release, op.ext.reset, op.ext.signal => {
|
|
||||||
_ = try self.term(cur, frame);
|
|
||||||
return .uninitialized;
|
|
||||||
},
|
|
||||||
op.ext.wait => {
|
|
||||||
_ = try self.term(cur, frame);
|
|
||||||
_ = try self.term(cur, frame);
|
|
||||||
return .{ .integer = 0 };
|
|
||||||
},
|
|
||||||
op.ext.sleep, op.ext.stall => {
|
|
||||||
_ = try self.term(cur, frame);
|
|
||||||
return .uninitialized;
|
|
||||||
},
|
|
||||||
else => return error.Unsupported,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn evalInt(self: *Interp, cur: *Cursor, frame: *Frame) Error!u64 {
|
|
||||||
return (try self.term(cur, frame)).asInt();
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
fn bitMask(width: u32) u128 {
|
|
||||||
if (width >= 128) return ~@as(u128, 0);
|
|
||||||
return (@as(u128, 1) << @intCast(width)) - 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn isNameStart(b: u8) bool {
|
|
||||||
return (b >= op.name_char_start and b <= op.name_char_end) or
|
|
||||||
b == op.name_char_underscore or
|
|
||||||
b == op.root_char or
|
|
||||||
b == op.parent_prefix_char or
|
|
||||||
b == op.dual_name_prefix or
|
|
||||||
b == op.multi_name_prefix;
|
|
||||||
}
|
|
||||||
@@ -1,137 +0,0 @@
|
|||||||
//! AML opcode constants — the full ACPI Machine Language opcode table.
|
|
||||||
//!
|
|
||||||
//! Single-byte opcodes are plain values. Extended opcodes are a two-byte sequence
|
|
||||||
//! `ext_prefix` (0x5B) followed by a byte listed under `ext`. A few comparison
|
|
||||||
//! opcodes are `lnot_op` (0x92) followed by a second byte (see `lnot`).
|
|
||||||
|
|
||||||
// --- name / path characters -------------------------------------------------
|
|
||||||
pub const zero_op = 0x00;
|
|
||||||
pub const one_op = 0x01;
|
|
||||||
pub const alias_op = 0x06;
|
|
||||||
pub const name_op = 0x08;
|
|
||||||
pub const byte_prefix = 0x0A;
|
|
||||||
pub const word_prefix = 0x0B;
|
|
||||||
pub const dword_prefix = 0x0C;
|
|
||||||
pub const string_prefix = 0x0D;
|
|
||||||
pub const qword_prefix = 0x0E;
|
|
||||||
pub const scope_op = 0x10;
|
|
||||||
pub const buffer_op = 0x11;
|
|
||||||
pub const package_op = 0x12;
|
|
||||||
pub const var_package_op = 0x13;
|
|
||||||
pub const method_op = 0x14;
|
|
||||||
pub const external_op = 0x15;
|
|
||||||
|
|
||||||
pub const dual_name_prefix = 0x2E;
|
|
||||||
pub const multi_name_prefix = 0x2F;
|
|
||||||
pub const ext_op_prefix = 0x5B;
|
|
||||||
pub const root_char = 0x5C;
|
|
||||||
pub const parent_prefix_char = 0x5E;
|
|
||||||
pub const name_char_underscore = 0x5F;
|
|
||||||
|
|
||||||
pub const digit_char_start = 0x30;
|
|
||||||
pub const digit_char_end = 0x39;
|
|
||||||
pub const name_char_start = 0x41; // 'A'
|
|
||||||
pub const name_char_end = 0x5A; // 'Z'
|
|
||||||
|
|
||||||
// --- locals / args ----------------------------------------------------------
|
|
||||||
pub const local0_op = 0x60;
|
|
||||||
pub const local7_op = 0x67;
|
|
||||||
pub const arg0_op = 0x68;
|
|
||||||
pub const arg6_op = 0x6E;
|
|
||||||
|
|
||||||
// --- store / references / arithmetic ---------------------------------------
|
|
||||||
pub const store_op = 0x70;
|
|
||||||
pub const ref_of_op = 0x71;
|
|
||||||
pub const add_op = 0x72;
|
|
||||||
pub const concat_op = 0x73;
|
|
||||||
pub const subtract_op = 0x74;
|
|
||||||
pub const increment_op = 0x75;
|
|
||||||
pub const decrement_op = 0x76;
|
|
||||||
pub const multiply_op = 0x77;
|
|
||||||
pub const divide_op = 0x78;
|
|
||||||
pub const shift_left_op = 0x79;
|
|
||||||
pub const shift_right_op = 0x7A;
|
|
||||||
pub const and_op = 0x7B;
|
|
||||||
pub const nand_op = 0x7C;
|
|
||||||
pub const or_op = 0x7D;
|
|
||||||
pub const nor_op = 0x7E;
|
|
||||||
pub const xor_op = 0x7F;
|
|
||||||
pub const not_op = 0x80;
|
|
||||||
pub const find_set_left_bit_op = 0x81;
|
|
||||||
pub const find_set_right_bit_op = 0x82;
|
|
||||||
pub const deref_of_op = 0x83;
|
|
||||||
pub const concat_res_op = 0x84;
|
|
||||||
pub const mod_op = 0x85;
|
|
||||||
pub const notify_op = 0x86;
|
|
||||||
pub const size_of_op = 0x87;
|
|
||||||
pub const index_op = 0x88;
|
|
||||||
pub const match_op = 0x89;
|
|
||||||
pub const create_dword_field_op = 0x8A;
|
|
||||||
pub const create_word_field_op = 0x8B;
|
|
||||||
pub const create_byte_field_op = 0x8C;
|
|
||||||
pub const create_bit_field_op = 0x8D;
|
|
||||||
pub const object_type_op = 0x8E;
|
|
||||||
pub const create_qword_field_op = 0x8F;
|
|
||||||
|
|
||||||
pub const land_op = 0x90;
|
|
||||||
pub const lor_op = 0x91;
|
|
||||||
pub const lnot_op = 0x92; // may be followed by a second byte (see `lnot`)
|
|
||||||
pub const lequal_op = 0x93;
|
|
||||||
pub const lgreater_op = 0x94;
|
|
||||||
pub const lless_op = 0x95;
|
|
||||||
pub const to_buffer_op = 0x96;
|
|
||||||
pub const to_decimal_string_op = 0x97;
|
|
||||||
pub const to_hex_string_op = 0x98;
|
|
||||||
pub const to_integer_op = 0x99;
|
|
||||||
pub const to_string_op = 0x9C;
|
|
||||||
pub const copy_object_op = 0x9D;
|
|
||||||
pub const mid_op = 0x9E;
|
|
||||||
pub const continue_op = 0x9F;
|
|
||||||
pub const if_op = 0xA0;
|
|
||||||
pub const else_op = 0xA1;
|
|
||||||
pub const while_op = 0xA2;
|
|
||||||
pub const noop_op = 0xA3;
|
|
||||||
pub const return_op = 0xA4;
|
|
||||||
pub const break_op = 0xA5;
|
|
||||||
pub const break_point_op = 0xCC;
|
|
||||||
pub const ones_op = 0xFF;
|
|
||||||
|
|
||||||
/// Second bytes of the `lnot_op` (0x92) compound comparison opcodes.
|
|
||||||
pub const lnot = struct {
|
|
||||||
pub const not_equal = 0x93; // LNotEqualOp: 0x92 0x93
|
|
||||||
pub const less_equal = 0x94; // LLessEqualOp: 0x92 0x94
|
|
||||||
pub const greater_equal = 0x95; // LGreaterEqualOp: 0x92 0x95
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Second bytes of extended opcodes (prefixed by `ext_op_prefix`, 0x5B).
|
|
||||||
pub const ext = struct {
|
|
||||||
pub const mutex = 0x01;
|
|
||||||
pub const event = 0x02;
|
|
||||||
pub const cond_ref_of = 0x12;
|
|
||||||
pub const create_field = 0x13;
|
|
||||||
pub const load_table = 0x1F;
|
|
||||||
pub const load = 0x20;
|
|
||||||
pub const stall = 0x21;
|
|
||||||
pub const sleep = 0x22;
|
|
||||||
pub const acquire = 0x23;
|
|
||||||
pub const signal = 0x24;
|
|
||||||
pub const wait = 0x25;
|
|
||||||
pub const reset = 0x26;
|
|
||||||
pub const release = 0x27;
|
|
||||||
pub const from_bcd = 0x28;
|
|
||||||
pub const to_bcd = 0x29;
|
|
||||||
pub const unload = 0x2A;
|
|
||||||
pub const revision = 0x30;
|
|
||||||
pub const debug = 0x31;
|
|
||||||
pub const fatal = 0x32;
|
|
||||||
pub const timer = 0x33;
|
|
||||||
pub const op_region = 0x80;
|
|
||||||
pub const field = 0x81;
|
|
||||||
pub const device = 0x82;
|
|
||||||
pub const processor = 0x83;
|
|
||||||
pub const power_res = 0x84;
|
|
||||||
pub const thermal_zone = 0x85;
|
|
||||||
pub const index_field = 0x86;
|
|
||||||
pub const bank_field = 0x87;
|
|
||||||
pub const data_region = 0x88;
|
|
||||||
};
|
|
||||||
@@ -1,517 +0,0 @@
|
|||||||
//! Recursive-descent AML parser. Walks the entire byte stream — including method
|
|
||||||
//! bodies — building the ACPI namespace as it goes. It does not *evaluate*
|
|
||||||
//! anything (no OperationRegion reads, no arithmetic); it parses structure so the
|
|
||||||
//! cursor stays aligned and every named object is recorded.
|
|
||||||
//!
|
|
||||||
//! The one genuine ambiguity in AML is method invocation: a bare NameString in an
|
|
||||||
//! operand position is a call whose argument count is only known from the method's
|
|
||||||
//! (earlier) declaration. Because we build the namespace in the same in-order pass,
|
|
||||||
//! `resolve` finds that declaration and tells us how many operands to consume.
|
|
||||||
//!
|
|
||||||
//! Safety net: every object delimited by a PkgLength (Scope/Device/Method/If/While/
|
|
||||||
//! Field/Buffer/Package/…) is parsed within its known extent, and the cursor is
|
|
||||||
//! snapped to that extent afterwards. So a mis-resolved invocation can only desync
|
|
||||||
//! *within* one such object; the enclosing walk realigns at the boundary.
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
const op = @import("opcodes.zig");
|
|
||||||
const ns = @import("namespace.zig");
|
|
||||||
const Namespace = ns.Namespace;
|
|
||||||
const Node = ns.Node;
|
|
||||||
|
|
||||||
pub const Error = error{ Truncated, Malformed } || std.mem.Allocator.Error;
|
|
||||||
|
|
||||||
const max_segs = 64;
|
|
||||||
|
|
||||||
/// A parsed NameString: an optional root anchor or some parent hops, then a list
|
|
||||||
/// of 4-byte segments.
|
|
||||||
const NamePath = struct {
|
|
||||||
rooted: bool = false,
|
|
||||||
parents: u8 = 0,
|
|
||||||
segs: [max_segs][4]u8 = undefined,
|
|
||||||
count: usize = 0,
|
|
||||||
|
|
||||||
fn slice(self: *const NamePath) []const [4]u8 {
|
|
||||||
return self.segs[0..self.count];
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
pub const Parser = struct {
|
|
||||||
aml: []const u8,
|
|
||||||
pos: usize = 0,
|
|
||||||
namespace: *Namespace,
|
|
||||||
|
|
||||||
pub fn init(aml: []const u8, namespace: *Namespace) Parser {
|
|
||||||
return .{ .aml = aml, .namespace = namespace };
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Parse the whole block as a TermList under the namespace root. Returns the
|
|
||||||
/// number of bytes consumed — equal to `aml.len` for a clean full traversal.
|
|
||||||
pub fn parseAll(self: *Parser) usize {
|
|
||||||
self.termList(self.aml.len, self.namespace.root);
|
|
||||||
return self.pos;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- cursor primitives --------------------------------------------------
|
|
||||||
|
|
||||||
fn eof(self: *Parser) bool {
|
|
||||||
return self.pos >= self.aml.len;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn peek(self: *Parser) ?u8 {
|
|
||||||
return if (self.eof()) null else self.aml[self.pos];
|
|
||||||
}
|
|
||||||
|
|
||||||
fn readByte(self: *Parser) Error!u8 {
|
|
||||||
if (self.eof()) return error.Truncated;
|
|
||||||
const b = self.aml[self.pos];
|
|
||||||
self.pos += 1;
|
|
||||||
return b;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn skip(self: *Parser, n: usize) Error!void {
|
|
||||||
if (self.pos + n > self.aml.len) return error.Truncated;
|
|
||||||
self.pos += n;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn skipCString(self: *Parser) Error!void {
|
|
||||||
while (true) {
|
|
||||||
const b = try self.readByte();
|
|
||||||
if (b == 0) return;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// AML PkgLength: the lead byte's top two bits give how many extra bytes
|
|
||||||
/// follow; the value counts from the start of the PkgLength field.
|
|
||||||
fn readPkgLength(self: *Parser) Error!usize {
|
|
||||||
const lead = try self.readByte();
|
|
||||||
const follow: usize = lead >> 6;
|
|
||||||
if (follow == 0) return lead & 0x3F;
|
|
||||||
var value: usize = lead & 0x0F;
|
|
||||||
var i: usize = 0;
|
|
||||||
while (i < follow) : (i += 1) {
|
|
||||||
const b = try self.readByte();
|
|
||||||
value |= @as(usize, b) << @intCast(4 + i * 8);
|
|
||||||
}
|
|
||||||
return value;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn readNameSeg(self: *Parser) Error![4]u8 {
|
|
||||||
if (self.pos + 4 > self.aml.len) return error.Truncated;
|
|
||||||
const seg = self.aml[self.pos..][0..4].*;
|
|
||||||
self.pos += 4;
|
|
||||||
return seg;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn readNameString(self: *Parser) Error!NamePath {
|
|
||||||
var np = NamePath{};
|
|
||||||
// A NameString is either root-anchored or parent-relative, not both.
|
|
||||||
if (self.peek() == op.root_char) {
|
|
||||||
np.rooted = true;
|
|
||||||
self.pos += 1;
|
|
||||||
} else {
|
|
||||||
while (self.peek() == op.parent_prefix_char) : (self.pos += 1) np.parents += 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
const lead = self.peek() orelse return np;
|
|
||||||
switch (lead) {
|
|
||||||
0x00 => self.pos += 1, // NullName
|
|
||||||
op.dual_name_prefix => {
|
|
||||||
self.pos += 1;
|
|
||||||
try self.appendSeg(&np);
|
|
||||||
try self.appendSeg(&np);
|
|
||||||
},
|
|
||||||
op.multi_name_prefix => {
|
|
||||||
self.pos += 1;
|
|
||||||
const cnt = try self.readByte();
|
|
||||||
var i: usize = 0;
|
|
||||||
while (i < cnt) : (i += 1) try self.appendSeg(&np);
|
|
||||||
},
|
|
||||||
else => {
|
|
||||||
if (isNameStart(lead)) try self.appendSeg(&np);
|
|
||||||
},
|
|
||||||
}
|
|
||||||
return np;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn appendSeg(self: *Parser, np: *NamePath) Error!void {
|
|
||||||
const seg = try self.readNameSeg();
|
|
||||||
if (np.count < max_segs) {
|
|
||||||
np.segs[np.count] = seg;
|
|
||||||
np.count += 1;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- term list / object -------------------------------------------------
|
|
||||||
|
|
||||||
/// Parse objects until `end`, then snap to `end`. Any parse error resyncs to
|
|
||||||
/// the boundary rather than propagating — containment for the rare desync.
|
|
||||||
fn termList(self: *Parser, end: usize, scope: *Node) void {
|
|
||||||
while (self.pos < end) {
|
|
||||||
self.object(scope) catch break;
|
|
||||||
}
|
|
||||||
self.pos = end;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Parse exactly one object/term at the cursor. Used for both TermObjs and
|
|
||||||
/// operands (TermArg / SuperName / Target all reduce to "one object" for the
|
|
||||||
/// purpose of advancing the cursor).
|
|
||||||
fn object(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const lead = self.peek() orelse return error.Truncated;
|
|
||||||
if (isNameStart(lead)) return self.nameInvocation(scope);
|
|
||||||
|
|
||||||
_ = try self.readByte();
|
|
||||||
switch (lead) {
|
|
||||||
// constants and no-operand statements
|
|
||||||
op.zero_op, op.one_op, op.ones_op => {},
|
|
||||||
op.noop_op, op.continue_op, op.break_op, op.break_point_op => {},
|
|
||||||
op.local0_op...op.local7_op => {},
|
|
||||||
op.arg0_op...op.arg6_op => {},
|
|
||||||
|
|
||||||
// literal data
|
|
||||||
op.byte_prefix => try self.skip(1),
|
|
||||||
op.word_prefix => try self.skip(2),
|
|
||||||
op.dword_prefix => try self.skip(4),
|
|
||||||
op.qword_prefix => try self.skip(8),
|
|
||||||
op.string_prefix => try self.skipCString(),
|
|
||||||
|
|
||||||
// data containers (contents skipped via their PkgLength)
|
|
||||||
op.buffer_op, op.package_op, op.var_package_op => try self.skipPkg(),
|
|
||||||
|
|
||||||
// namespace modifiers / named objects
|
|
||||||
op.name_op => try self.opName(scope),
|
|
||||||
op.alias_op => try self.opAlias(scope),
|
|
||||||
op.scope_op => try self.opScopeLike(scope, .scope),
|
|
||||||
op.method_op => try self.opMethod(scope),
|
|
||||||
op.external_op => try self.opExternal(scope),
|
|
||||||
op.ext_op_prefix => try self.opExt(scope),
|
|
||||||
|
|
||||||
// control flow
|
|
||||||
op.if_op => try self.opIf(scope),
|
|
||||||
op.else_op => try self.opElse(scope),
|
|
||||||
op.while_op => try self.opWhile(scope),
|
|
||||||
op.return_op => try self.object(scope),
|
|
||||||
op.notify_op => try self.args(scope, 2),
|
|
||||||
|
|
||||||
// stores / references / unary+target
|
|
||||||
op.store_op => try self.args(scope, 2),
|
|
||||||
op.ref_of_op, op.deref_of_op, op.size_of_op, op.object_type_op => try self.args(scope, 1),
|
|
||||||
op.increment_op, op.decrement_op => try self.args(scope, 1),
|
|
||||||
op.not_op, op.find_set_left_bit_op, op.find_set_right_bit_op => try self.args(scope, 2),
|
|
||||||
op.to_buffer_op, op.to_decimal_string_op, op.to_hex_string_op, op.to_integer_op => try self.args(scope, 2),
|
|
||||||
op.copy_object_op => try self.args(scope, 2),
|
|
||||||
|
|
||||||
// binary + target
|
|
||||||
op.add_op, op.subtract_op, op.multiply_op, op.mod_op => try self.args(scope, 3),
|
|
||||||
op.and_op, op.nand_op, op.or_op, op.nor_op, op.xor_op => try self.args(scope, 3),
|
|
||||||
op.shift_left_op, op.shift_right_op, op.concat_op, op.concat_res_op, op.index_op => try self.args(scope, 3),
|
|
||||||
op.divide_op => try self.args(scope, 4),
|
|
||||||
op.to_string_op => try self.args(scope, 3),
|
|
||||||
op.mid_op => try self.args(scope, 4),
|
|
||||||
|
|
||||||
// logical
|
|
||||||
op.land_op, op.lor_op => try self.args(scope, 2),
|
|
||||||
op.lequal_op, op.lgreater_op, op.lless_op => try self.args(scope, 2),
|
|
||||||
op.lnot_op => try self.opLnot(scope),
|
|
||||||
|
|
||||||
op.match_op => try self.opMatch(scope),
|
|
||||||
|
|
||||||
// CreateXField: <source> <index> NameString
|
|
||||||
op.create_dword_field_op,
|
|
||||||
op.create_word_field_op,
|
|
||||||
op.create_byte_field_op,
|
|
||||||
op.create_bit_field_op,
|
|
||||||
op.create_qword_field_op,
|
|
||||||
=> try self.opCreateField(scope, 2),
|
|
||||||
|
|
||||||
else => return error.Malformed,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Parse `n` operands.
|
|
||||||
fn args(self: *Parser, scope: *Node, n: usize) Error!void {
|
|
||||||
var i: usize = 0;
|
|
||||||
while (i < n) : (i += 1) try self.object(scope);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// A NameString in operand/statement position: a method invocation (consuming
|
|
||||||
/// the callee's declared argument count) or a plain name reference.
|
|
||||||
fn nameInvocation(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
if (self.namespace.resolve(scope, np.rooted, np.parents, np.slice())) |node| {
|
|
||||||
if ((node.kind == .method or node.kind == .external) and node.arg_count > 0) {
|
|
||||||
try self.args(scope, node.arg_count);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Skip a PkgLength-delimited body wholesale (Buffer / Package / VarPackage):
|
|
||||||
/// the contents are pure data, never namespace declarations.
|
|
||||||
fn skipPkg(self: *Parser) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const len = try self.readPkgLength();
|
|
||||||
const end = start + len;
|
|
||||||
if (end > self.aml.len) return error.Truncated;
|
|
||||||
self.pos = end;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- namespace objects --------------------------------------------------
|
|
||||||
|
|
||||||
fn opName(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
const val_start = self.pos;
|
|
||||||
try self.object(scope); // the DataRefObject value
|
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .name);
|
|
||||||
node.value = self.aml[val_start..self.pos];
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opAlias(self: *Parser, scope: *Node) Error!void {
|
|
||||||
_ = try self.readNameString(); // source
|
|
||||||
const np = try self.readNameString(); // the alias name
|
|
||||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .alias);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opMethod(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
const np = try self.readNameString();
|
|
||||||
const flags = try self.readByte();
|
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .method);
|
|
||||||
node.arg_count = flags & 0x7;
|
|
||||||
// Capture the body for on-demand evaluation and skip it — objects declared
|
|
||||||
// inside a method are created at *runtime*, not at load, so they must not
|
|
||||||
// become permanent namespace nodes.
|
|
||||||
node.value = self.aml[self.pos..@min(end, self.aml.len)];
|
|
||||||
self.pos = end;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opExternal(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
_ = try self.readByte(); // object type
|
|
||||||
const arg_count = try self.readByte();
|
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .external);
|
|
||||||
node.arg_count = arg_count;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Scope / Device / ThermalZone: PkgLength, NameString, then a nested TermList.
|
|
||||||
fn opScopeLike(self: *Parser, scope: *Node, kind: ns.NodeKind) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
const np = try self.readNameString();
|
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), kind);
|
|
||||||
self.termList(end, node);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opProcessor(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
const np = try self.readNameString();
|
|
||||||
try self.skip(6); // ProcID(byte) + PblkAddr(dword) + PblkLen(byte)
|
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .processor);
|
|
||||||
self.termList(end, node);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opPowerRes(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
const np = try self.readNameString();
|
|
||||||
try self.skip(3); // SystemLevel(byte) + ResourceOrder(word)
|
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .power_res);
|
|
||||||
self.termList(end, node);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// OperationRegion: NameString, RegionSpace(byte), Offset(TermArg), Len(TermArg).
|
|
||||||
/// The offset/length expressions are kept as AML for lazy evaluation.
|
|
||||||
fn opRegion(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
const space = try self.readByte();
|
|
||||||
const off_start = self.pos;
|
|
||||||
try self.object(scope);
|
|
||||||
const off_end = self.pos;
|
|
||||||
try self.object(scope);
|
|
||||||
const len_end = self.pos;
|
|
||||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .region);
|
|
||||||
node.region_space = space;
|
|
||||||
node.region_offset_aml = self.aml[off_start..off_end];
|
|
||||||
node.region_len_aml = self.aml[off_end..len_end];
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opDataRegion(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
try self.args(scope, 3); // signature, oem id, oem table id (TermArgs)
|
|
||||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .region);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opMutex(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
try self.skip(1); // sync flags
|
|
||||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .mutex);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opEvent(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .event);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// CreateXField: `count` TermArgs then the new field's NameString.
|
|
||||||
fn opCreateField(self: *Parser, scope: *Node, count: usize) Error!void {
|
|
||||||
try self.args(scope, count);
|
|
||||||
const np = try self.readNameString();
|
|
||||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .name);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Field / IndexField / BankField: a region/bank reference, flags, then a
|
|
||||||
/// FieldList whose NamedFields become nodes in the current scope. For a plain
|
|
||||||
/// Field, the first NameString is the backing region — captured so field units
|
|
||||||
/// carry a region + bit position the evaluator can read/write.
|
|
||||||
fn opField(self: *Parser, scope: *Node, name_strings: u8, bank: bool) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
var region: ?*Node = null;
|
|
||||||
var i: u8 = 0;
|
|
||||||
while (i < name_strings) : (i += 1) {
|
|
||||||
const np = try self.readNameString();
|
|
||||||
// Only a plain Field's single NameString denotes an OperationRegion.
|
|
||||||
if (name_strings == 1) region = self.namespace.resolve(scope, np.rooted, np.parents, np.slice());
|
|
||||||
}
|
|
||||||
if (bank) try self.object(scope); // bank value TermArg
|
|
||||||
const flags = try self.readByte();
|
|
||||||
self.fieldList(end, scope, region, flags & 0x0F);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn fieldList(self: *Parser, end: usize, scope: *Node, region: ?*Node, initial_access: u8) void {
|
|
||||||
var bit_offset: u32 = 0;
|
|
||||||
var access = initial_access;
|
|
||||||
while (self.pos < end) {
|
|
||||||
const lead = self.peek() orelse break;
|
|
||||||
switch (lead) {
|
|
||||||
0x00 => { // ReservedField: advances the bit position
|
|
||||||
self.pos += 1;
|
|
||||||
const width = self.readPkgLength() catch break;
|
|
||||||
bit_offset += @intCast(width);
|
|
||||||
},
|
|
||||||
0x01 => { // AccessField: AccessType (low nibble) + AccessAttrib
|
|
||||||
self.pos += 1;
|
|
||||||
const at = self.readByte() catch break;
|
|
||||||
self.skip(1) catch break;
|
|
||||||
access = at & 0x0F;
|
|
||||||
},
|
|
||||||
0x02 => { // ConnectField: NameString | BufferData
|
|
||||||
self.pos += 1;
|
|
||||||
self.object(scope) catch break;
|
|
||||||
},
|
|
||||||
0x03 => { // ExtendedAccessField: type + attrib + length
|
|
||||||
self.pos += 1;
|
|
||||||
self.skip(3) catch break;
|
|
||||||
},
|
|
||||||
else => { // NamedField: NameSeg + PkgLength (bit width)
|
|
||||||
const seg = self.readNameSeg() catch break;
|
|
||||||
const width = self.readPkgLength() catch break;
|
|
||||||
const unit = self.namespace.newFieldUnit(scope, seg) catch break;
|
|
||||||
unit.region = region;
|
|
||||||
unit.bit_offset = bit_offset;
|
|
||||||
unit.bit_width = @intCast(width);
|
|
||||||
unit.access_type = access;
|
|
||||||
bit_offset += @intCast(width);
|
|
||||||
},
|
|
||||||
}
|
|
||||||
}
|
|
||||||
self.pos = end;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- control flow -------------------------------------------------------
|
|
||||||
|
|
||||||
fn opIf(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
try self.object(scope); // predicate
|
|
||||||
self.termList(end, scope);
|
|
||||||
if (self.peek() == op.else_op) {
|
|
||||||
self.pos += 1;
|
|
||||||
try self.opElse(scope);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opElse(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
self.termList(end, scope);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opWhile(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const start = self.pos;
|
|
||||||
const end = start + try self.readPkgLength();
|
|
||||||
try self.object(scope); // predicate
|
|
||||||
self.termList(end, scope);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opLnot(self: *Parser, scope: *Node) Error!void {
|
|
||||||
// 0x92 followed by 0x93/94/95 is a compound comparison (two operands);
|
|
||||||
// otherwise it is a plain LNot of one operand.
|
|
||||||
const b = self.peek() orelse return error.Truncated;
|
|
||||||
switch (b) {
|
|
||||||
op.lnot.not_equal, op.lnot.less_equal, op.lnot.greater_equal => {
|
|
||||||
self.pos += 1;
|
|
||||||
try self.args(scope, 2);
|
|
||||||
},
|
|
||||||
else => try self.object(scope),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn opMatch(self: *Parser, scope: *Node) Error!void {
|
|
||||||
try self.object(scope); // search package
|
|
||||||
try self.skip(1); // match opcode 1
|
|
||||||
try self.object(scope); // operand 1
|
|
||||||
try self.skip(1); // match opcode 2
|
|
||||||
try self.object(scope); // operand 2
|
|
||||||
try self.object(scope); // start index
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- extended opcodes (0x5B xx) -----------------------------------------
|
|
||||||
|
|
||||||
fn opExt(self: *Parser, scope: *Node) Error!void {
|
|
||||||
const e = try self.readByte();
|
|
||||||
switch (e) {
|
|
||||||
op.ext.mutex => try self.opMutex(scope),
|
|
||||||
op.ext.event => try self.opEvent(scope),
|
|
||||||
op.ext.op_region => try self.opRegion(scope),
|
|
||||||
op.ext.data_region => try self.opDataRegion(scope),
|
|
||||||
op.ext.field => try self.opField(scope, 1, false),
|
|
||||||
op.ext.index_field => try self.opField(scope, 2, false),
|
|
||||||
op.ext.bank_field => try self.opField(scope, 2, true),
|
|
||||||
op.ext.device => try self.opScopeLike(scope, .device),
|
|
||||||
op.ext.thermal_zone => try self.opScopeLike(scope, .thermal_zone),
|
|
||||||
op.ext.processor => try self.opProcessor(scope),
|
|
||||||
op.ext.power_res => try self.opPowerRes(scope),
|
|
||||||
|
|
||||||
op.ext.cond_ref_of => try self.args(scope, 2), // SuperName, Target
|
|
||||||
op.ext.create_field => try self.opCreateField(scope, 3),
|
|
||||||
op.ext.load_table => try self.args(scope, 6),
|
|
||||||
op.ext.load => try self.args(scope, 2), // NameString, Target
|
|
||||||
op.ext.stall, op.ext.sleep => try self.args(scope, 1),
|
|
||||||
op.ext.acquire => {
|
|
||||||
try self.object(scope); // mutex SuperName
|
|
||||||
try self.skip(2); // timeout WordData
|
|
||||||
},
|
|
||||||
op.ext.signal, op.ext.reset, op.ext.release, op.ext.unload => try self.args(scope, 1),
|
|
||||||
op.ext.wait => try self.args(scope, 2),
|
|
||||||
op.ext.from_bcd, op.ext.to_bcd => try self.args(scope, 2),
|
|
||||||
op.ext.fatal => {
|
|
||||||
try self.skip(5); // Type(byte) + Code(dword)
|
|
||||||
try self.object(scope); // Arg TermArg
|
|
||||||
},
|
|
||||||
op.ext.revision, op.ext.debug, op.ext.timer => {},
|
|
||||||
|
|
||||||
else => return error.Malformed,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
fn isNameStart(b: u8) bool {
|
|
||||||
return (b >= op.name_char_start and b <= op.name_char_end) or
|
|
||||||
b == op.name_char_underscore or
|
|
||||||
b == op.root_char or
|
|
||||||
b == op.parent_prefix_char or
|
|
||||||
b == op.dual_name_prefix or
|
|
||||||
b == op.multi_name_prefix;
|
|
||||||
}
|
|
||||||
@@ -1,76 +0,0 @@
|
|||||||
//! The firmware-agnostic discovery facade.
|
|
||||||
//!
|
|
||||||
//! The kernel calls `platform.discover()` and gets back a generic `DeviceTree`
|
|
||||||
//! without ever naming ACPI or device-tree — the same way it imports `arch`
|
|
||||||
//! without naming x86_64. Which backend runs is decided *at runtime* from what
|
|
||||||
//! the bootloader handed us (an ACPI RSDP today, a device-tree blob later),
|
|
||||||
//! because a single image — a future ARM kernel especially — may boot under
|
|
||||||
//! either firmware. That's a deliberate divergence from `arch`, which is a
|
|
||||||
//! compile-time choice.
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
const danos = @import("danos");
|
|
||||||
const device = @import("device.zig");
|
|
||||||
const acpi = @import("acpi.zig");
|
|
||||||
const power = @import("power.zig");
|
|
||||||
const devicetree = @import("devicetree.zig");
|
|
||||||
|
|
||||||
pub const DeviceTree = device.DeviceTree;
|
|
||||||
pub const Device = device.Device;
|
|
||||||
pub const DeviceClass = device.DeviceClass;
|
|
||||||
pub const Hal = device.Hal;
|
|
||||||
pub const PowerInfo = acpi.PowerInfo;
|
|
||||||
pub const AmlStats = acpi.AmlStats;
|
|
||||||
pub const PlatformInfo = acpi.PlatformInfo;
|
|
||||||
pub const RegAccess = acpi.RegAccess;
|
|
||||||
pub const IsoEntry = acpi.IsoEntry;
|
|
||||||
|
|
||||||
/// The register map + sleep types discovery extracted, for logging/diagnostics.
|
|
||||||
pub fn powerInfo() PowerInfo {
|
|
||||||
return acpi.power_info;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The scalar firmware facts the arch layer needs to avoid legacy assumptions
|
|
||||||
/// (8259 presence, LAPIC base, PM timer, SPCR UART, IRQ overrides).
|
|
||||||
pub fn platformInfo() PlatformInfo {
|
|
||||||
return acpi.platform_info;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// AML parse integrity/diagnostics (namespace node count, bytes consumed).
|
|
||||||
pub fn amlStats() AmlStats {
|
|
||||||
return acpi.aml_stats;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Enumerate hardware into a fresh device tree. `hal` supplies the hardware
|
|
||||||
/// primitives the backend needs (MMIO mapping for PCIe config space, port I/O for
|
|
||||||
/// ACPI registers); pass the arch implementation. Errors leave nothing to clean up
|
|
||||||
/// beyond the tree's own allocations.
|
|
||||||
pub fn discover(
|
|
||||||
boot_info: *const danos.BootInfo,
|
|
||||||
allocator: std.mem.Allocator,
|
|
||||||
hal: Hal,
|
|
||||||
) !DeviceTree {
|
|
||||||
var dt = try DeviceTree.init(allocator);
|
|
||||||
|
|
||||||
if (boot_info.acpi_rsdp != 0) {
|
|
||||||
try acpi.discover(boot_info.acpi_rsdp, &dt, hal);
|
|
||||||
} else {
|
|
||||||
// No ACPI RSDP. A device-tree boot would parse its blob here; today that
|
|
||||||
// path is a stub, so this reports the machine described itself no way we
|
|
||||||
// understand yet.
|
|
||||||
try devicetree.discover(&dt);
|
|
||||||
}
|
|
||||||
|
|
||||||
return dt;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Restart the machine. Never returns on success; returns only if no reset method
|
|
||||||
/// worked (extremely unlikely). Backend-agnostic entry the kernel calls.
|
|
||||||
pub fn reboot(hal: Hal) void {
|
|
||||||
power.reboot(hal);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Power the machine off (ACPI S5). Never returns on success.
|
|
||||||
pub fn shutdown(hal: Hal) void {
|
|
||||||
power.shutdown(hal);
|
|
||||||
}
|
|
||||||
@@ -1,284 +0,0 @@
|
|||||||
//! x86_64 CPU operations. This is the "arch" module: the generic kernel imports
|
|
||||||
//! it as `@import("arch")` and never names x86_64 directly, so a second
|
|
||||||
//! architecture is added by pointing that module at a different directory in
|
|
||||||
//! build.zig — no change to the generic code. Keep everything CPU-specific here
|
|
||||||
//! (halt, the descriptor tables, later paging), and nothing generic.
|
|
||||||
|
|
||||||
const danos = @import("danos");
|
|
||||||
const gdt = @import("gdt.zig");
|
|
||||||
const tss = @import("tss.zig");
|
|
||||||
const idt = @import("idt.zig");
|
|
||||||
const paging = @import("paging.zig");
|
|
||||||
const serial = @import("serial.zig");
|
|
||||||
const apic = @import("apic.zig");
|
|
||||||
const ioapic = @import("ioapic.zig");
|
|
||||||
const io = @import("io.zig");
|
|
||||||
|
|
||||||
/// The saved register/trap frame passed to a fault handler.
|
|
||||||
pub const CpuState = idt.CpuState;
|
|
||||||
|
|
||||||
/// Bring up the serial port (the kernel's machine-readable log). No dependencies,
|
|
||||||
/// so it can be the very first thing called.
|
|
||||||
pub fn serialInit() void {
|
|
||||||
serial.init();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Write bytes to the serial port.
|
|
||||||
pub fn serialWrite(bytes: []const u8) void {
|
|
||||||
serial.write(bytes);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Emit a one-byte checkpoint to the POST diagnostic port (0x80). A POST card or
|
|
||||||
/// BMC displays it; it's the last-resort progress signal when there's no text
|
|
||||||
/// output at all. Writing 0x80 is universally safe (it's the legacy I/O-delay port).
|
|
||||||
pub fn postCode(code: u8) void {
|
|
||||||
io.outb(0x80, code);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Whether a Bochs/QEMU-style debug console is on port 0xE9 (it returns 0xE9 when
|
|
||||||
/// read). On real hardware the port reads back 0xFF, so this stays false — a safe
|
|
||||||
/// probe before we write to it.
|
|
||||||
pub fn debugconPresent() bool {
|
|
||||||
return io.inb(0xE9) == 0xE9;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Output sink: write bytes to the 0xE9 debug console (see `debugconPresent`).
|
|
||||||
pub fn debugconWrite(bytes: []const u8) void {
|
|
||||||
for (bytes) |b| io.outb(0xE9, b);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Set up the CPU's descriptor tables: our own GDT, the TSS (with an interrupt
|
|
||||||
/// stack for double faults), then the IDT with exception handlers. After this a
|
|
||||||
/// CPU fault is reported instead of triple-faulting. Install the fault handler
|
|
||||||
/// (setFaultHandler) first so early faults are caught.
|
|
||||||
pub fn init() void {
|
|
||||||
gdt.init();
|
|
||||||
tss.init();
|
|
||||||
idt.init();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Build the kernel's own page tables (with real permissions) and switch onto
|
|
||||||
/// them. Needs the frame allocator and the boot info (for the memory map and the
|
|
||||||
/// kernel's segment layout). Call once the frame allocator is up.
|
|
||||||
pub fn enablePaging(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void {
|
|
||||||
paging.init(allocFrame, boot_info);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Map a page into the kernel address space (non-executable). For the heap, etc.
|
|
||||||
pub fn mapPage(virt: u64, phys: u64, writable: bool) void {
|
|
||||||
paging.map(virt, phys, writable);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Remove a kernel mapping.
|
|
||||||
pub fn unmapPage(virt: u64) void {
|
|
||||||
paging.unmap(virt);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// CR3 holds the physical address of the active top-level page table.
|
|
||||||
pub fn readCr3() u64 {
|
|
||||||
return asm volatile ("mov %%cr3, %[out]"
|
|
||||||
: [out] "=r" (-> u64),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Kernel tick rate: 1000 Hz (1 ms), the scheduler's time quantum.
|
|
||||||
pub const timer_hz = 1000;
|
|
||||||
|
|
||||||
/// The ACPI PM timer, as a calibration reference (re-exported for the config).
|
|
||||||
pub const PmTimer = apic.PmTimer;
|
|
||||||
/// A MADT interrupt-source override (re-exported for the config).
|
|
||||||
pub const IsoEntry = ioapic.IsoEntry;
|
|
||||||
|
|
||||||
/// Discovered platform facts the arch layer needs so it makes no legacy
|
|
||||||
/// assumptions — sourced from the device tree + ACPI, passed in by the kernel.
|
|
||||||
pub const PlatformConfig = struct {
|
|
||||||
/// Whether the legacy 8259 PIC is present (skip programming it if not).
|
|
||||||
pic_present: bool = true,
|
|
||||||
/// HPET MMIO base (0 = none) — a calibration reference for the timer.
|
|
||||||
hpet_base: u64 = 0,
|
|
||||||
/// The ACPI PM timer, another calibration reference.
|
|
||||||
pm_timer: ?PmTimer = null,
|
|
||||||
/// I/O APIC MMIO base + its first global system interrupt (0 = none).
|
|
||||||
ioapic_base: u64 = 0,
|
|
||||||
ioapic_gsi_base: u32 = 0,
|
|
||||||
/// MADT ISA-IRQ overrides, for I/O APIC routing.
|
|
||||||
overrides: []const IsoEntry = &.{},
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Apply the discovered platform config. Must run before `startTimer` (the timer
|
|
||||||
/// calibration reads `hpet_base`/`pm_timer`) and before any interrupt routing.
|
|
||||||
/// Maps + masks the I/O APIC immediately.
|
|
||||||
pub fn configurePlatform(cfg: PlatformConfig) void {
|
|
||||||
apic.configure(cfg.pic_present, cfg.hpet_base, cfg.pm_timer);
|
|
||||||
ioapic.configure(cfg.ioapic_base, cfg.ioapic_gsi_base, cfg.overrides);
|
|
||||||
ioapic.init();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Point the serial console at the UART ACPI's SPCR table named (MMIO or I/O port).
|
|
||||||
pub fn serialReconfigure(is_mmio: bool, addr: u64) void {
|
|
||||||
serial.reconfigure(is_mmio, addr);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The reference clock the timer was calibrated against ("cpuid"/"hpet"/…).
|
|
||||||
pub fn timerCalibrationSource() []const u8 {
|
|
||||||
return apic.calibrationSource();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// I/O APIC diagnostics (for boot logging / verification).
|
|
||||||
pub fn ioapicEntryCount() u32 {
|
|
||||||
return ioapic.entryCount();
|
|
||||||
}
|
|
||||||
pub fn ioapicEntryLow(n: u32) u32 {
|
|
||||||
return ioapic.entryLow(n);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Enable the Local APIC, calibrate its timer against the best available reference
|
|
||||||
/// (see apic.calibrate — no longer the PIT by default), and start it firing at
|
|
||||||
/// `timer_hz` — the kernel's real-time heartbeat. Interrupts still have to be
|
|
||||||
/// unmasked with enableInterrupts() to be delivered. Run `configurePlatform` first.
|
|
||||||
pub fn startTimer() void {
|
|
||||||
apic.init();
|
|
||||||
apic.calibrate();
|
|
||||||
idt.setHandler(apic.timer_vector, apic.timerTick);
|
|
||||||
apic.initTimer(timer_hz);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Number of timer ticks since startTimer().
|
|
||||||
pub fn ticks() u64 {
|
|
||||||
return apic.ticks();
|
|
||||||
}
|
|
||||||
|
|
||||||
// Monotonic high-resolution clock (from the TSC), one function per resolution.
|
|
||||||
pub fn nanos() u64 {
|
|
||||||
return apic.nanos();
|
|
||||||
}
|
|
||||||
pub fn micros() u64 {
|
|
||||||
return apic.micros();
|
|
||||||
}
|
|
||||||
pub fn millis() u64 {
|
|
||||||
return apic.millis();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Measured LAPIC timer / TSC frequencies in Hz (from calibration).
|
|
||||||
pub fn lapicHz() u64 {
|
|
||||||
return apic.lapicHz();
|
|
||||||
}
|
|
||||||
pub fn tscHz() u64 {
|
|
||||||
return apic.tscHz();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Unmask maskable interrupts (`sti`) so device interrupts get delivered.
|
|
||||||
pub fn enableInterrupts() void {
|
|
||||||
asm volatile ("sti");
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Mask maskable interrupts (`cli`).
|
|
||||||
pub fn disableInterrupts() void {
|
|
||||||
asm volatile ("cli");
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Disable interrupts and return the previous flags, so a nested critical section
|
|
||||||
/// can restore the caller's state rather than blindly re-enabling. Pairs with
|
|
||||||
/// restoreInterrupts.
|
|
||||||
pub fn saveInterrupts() u64 {
|
|
||||||
var flags: u64 = undefined;
|
|
||||||
asm volatile (
|
|
||||||
\\pushfq
|
|
||||||
\\pop %[f]
|
|
||||||
\\cli
|
|
||||||
: [f] "=r" (flags),
|
|
||||||
:
|
|
||||||
: .{ .memory = true }
|
|
||||||
);
|
|
||||||
return flags;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Re-enable interrupts only if they were enabled when `flags` was captured.
|
|
||||||
pub fn restoreInterrupts(flags: u64) void {
|
|
||||||
if (flags & 0x200 != 0) asm volatile ("sti" ::: .{ .memory = true }); // bit 9 = IF
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Register a callback the timer interrupt invokes each tick (e.g. the scheduler).
|
|
||||||
pub fn setTickHook(hook: *const fn () void) void {
|
|
||||||
apic.setTickHook(hook);
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- context switching (for the scheduler) -------------------------------
|
|
||||||
|
|
||||||
/// Save the current task's registers/stack and resume `new_rsp`; the old stack
|
|
||||||
/// pointer is written to `old_rsp`. Defined in isr.s.
|
|
||||||
extern fn switch_context(old_rsp: *usize, new_rsp: usize) callconv(.c) void;
|
|
||||||
|
|
||||||
pub fn switchContext(old_rsp: *usize, new_rsp: usize) void {
|
|
||||||
switch_context(old_rsp, new_rsp);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Build the initial stack for a new task so that switching to it lands in
|
|
||||||
/// `task_trampoline`, which then calls `entry`. Returns the saved stack pointer.
|
|
||||||
/// The layout must match switch_context's push order (callee-saved, then the
|
|
||||||
/// return address on top); `entry` is smuggled in via the r15 slot.
|
|
||||||
pub fn initTaskStack(stack_top: usize, entry: usize) usize {
|
|
||||||
const trampoline = @extern(*const anyopaque, .{ .name = "task_trampoline" });
|
|
||||||
var sp = stack_top;
|
|
||||||
const push = struct {
|
|
||||||
fn f(p: *usize, value: usize) void {
|
|
||||||
p.* -= @sizeOf(usize);
|
|
||||||
@as(*usize, @ptrFromInt(p.*)).* = value;
|
|
||||||
}
|
|
||||||
}.f;
|
|
||||||
push(&sp, @intFromPtr(trampoline)); // return address for switch_context's `ret`
|
|
||||||
push(&sp, 0); // rbx
|
|
||||||
push(&sp, 0); // rbp
|
|
||||||
push(&sp, 0); // r12
|
|
||||||
push(&sp, 0); // r13
|
|
||||||
push(&sp, 0); // r14
|
|
||||||
push(&sp, entry); // r15 -> task entry, read by task_trampoline
|
|
||||||
return sp;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Route CPU exceptions to `handler`, which receives the trap frame and does not
|
|
||||||
/// return. Until set, faults just halt the core.
|
|
||||||
pub fn setFaultHandler(handler: *const fn (*const CpuState) noreturn) void {
|
|
||||||
idt.on_fault = handler;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// A human-readable name for a CPU exception vector.
|
|
||||||
pub fn vectorName(vector: u64) []const u8 {
|
|
||||||
return idt.vectorName(vector);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Read `width` bytes (1/2/4) from an I/O port. The generic device layer drives
|
|
||||||
/// ACPI registers through this rather than naming x86 port instructions; on an
|
|
||||||
/// MMIO-only architecture this would be implemented differently.
|
|
||||||
pub fn pioRead(width: u8, port: u16) u32 {
|
|
||||||
return switch (width) {
|
|
||||||
1 => io.inb(port),
|
|
||||||
2 => io.inw(port),
|
|
||||||
4 => io.inl(port),
|
|
||||||
else => 0,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Write `width` bytes (1/2/4) to an I/O port.
|
|
||||||
pub fn pioWrite(width: u8, port: u16, value: u32) void {
|
|
||||||
switch (width) {
|
|
||||||
1 => io.outb(port, @truncate(value)),
|
|
||||||
2 => io.outw(port, @truncate(value)),
|
|
||||||
4 => io.outl(port, value),
|
|
||||||
else => {},
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// CR2 holds the faulting linear address after a page fault (#PF, vector 14).
|
|
||||||
pub fn readCr2() u64 {
|
|
||||||
return asm volatile ("mov %%cr2, %[out]"
|
|
||||||
: [out] "=r" (-> u64),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Park the core forever. `hlt` drops it into a low-power idle until the next
|
|
||||||
/// interrupt; the loop re-halts on every wake so the stop is permanent. See
|
|
||||||
/// docs/halting.md for the full reasoning.
|
|
||||||
pub fn halt() noreturn {
|
|
||||||
while (true) asm volatile ("hlt");
|
|
||||||
}
|
|
||||||
@@ -1,54 +0,0 @@
|
|||||||
//! Global Descriptor Table. In long mode segmentation is mostly vestigial, but
|
|
||||||
//! the CPU still needs valid code/data segment descriptors, and the IDT's gates
|
|
||||||
//! reference a code selector — so we install our own flat GDT with known
|
|
||||||
//! selectors (0x08 kernel code, 0x10 kernel data) rather than trusting whatever
|
|
||||||
//! the firmware left in place.
|
|
||||||
|
|
||||||
/// Selectors into the table below (index * 8).
|
|
||||||
pub const kernel_code = 0x08;
|
|
||||||
pub const kernel_data = 0x10;
|
|
||||||
pub const tss_selector = 0x18;
|
|
||||||
|
|
||||||
/// Flat 64-bit descriptors. Base/limit are ignored in long mode; what matters is
|
|
||||||
/// the access byte and, for code, the long-mode (L) flag.
|
|
||||||
/// code: present, ring 0, executable, readable, L=1 -> 0x00AF9A00_0000FFFF
|
|
||||||
/// data: present, ring 0, writable -> 0x00CF9200_0000FFFF
|
|
||||||
/// The last two slots hold one 16-byte TSS descriptor, filled in by setTss.
|
|
||||||
var table = [_]u64{
|
|
||||||
0, // null descriptor (required)
|
|
||||||
0x00AF9A000000FFFF, // kernel code (0x08)
|
|
||||||
0x00CF92000000FFFF, // kernel data (0x10)
|
|
||||||
0, // TSS descriptor low (0x18)
|
|
||||||
0, // TSS descriptor high
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Fill the 64-bit TSS system descriptor (two GDT slots) so the task register can
|
|
||||||
/// point at our TSS. Type 0x89 = present, ring 0, available 64-bit TSS.
|
|
||||||
pub fn setTss(base: u64, limit: u64) void {
|
|
||||||
table[3] = (limit & 0xFFFF) |
|
|
||||||
((base & 0xFFFF) << 16) |
|
|
||||||
(((base >> 16) & 0xFF) << 32) |
|
|
||||||
(@as(u64, 0x89) << 40) |
|
|
||||||
(((limit >> 16) & 0xF) << 48) |
|
|
||||||
(((base >> 24) & 0xFF) << 56);
|
|
||||||
table[4] = (base >> 32) & 0xFFFFFFFF;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The operand `lgdt` wants: table byte-length minus one, then its address.
|
|
||||||
const Descriptor = packed struct {
|
|
||||||
limit: u16,
|
|
||||||
base: u64,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Loads the GDT and reloads the segment registers (including CS). Defined in
|
|
||||||
/// isr.s — it uses the selectors 0x08 (code) and 0x10 (data) that match `table`.
|
|
||||||
extern fn gdt_flush(descriptor: *const Descriptor) callconv(.c) void;
|
|
||||||
|
|
||||||
/// Install our GDT and switch onto its segments.
|
|
||||||
pub fn init() void {
|
|
||||||
const descriptor = Descriptor{
|
|
||||||
.limit = @sizeOf(@TypeOf(table)) - 1,
|
|
||||||
.base = @intFromPtr(&table),
|
|
||||||
};
|
|
||||||
gdt_flush(&descriptor);
|
|
||||||
}
|
|
||||||
@@ -1,97 +0,0 @@
|
|||||||
//! I/O APIC — routes external device interrupts (a device's line) to a LAPIC
|
|
||||||
//! vector on a chosen CPU. Its address and the ISA-IRQ-to-GSI remappings come from
|
|
||||||
//! ACPI's MADT (via discovery), never assumed.
|
|
||||||
//!
|
|
||||||
//! Status: groundwork. The only interrupt danos handles today is the LAPIC's own
|
|
||||||
//! timer, which needs no I/O APIC — so nothing calls `routeIrq` yet. What runs now
|
|
||||||
//! is `init`, which maps the I/O APIC and **masks every input**, the correct
|
|
||||||
//! quiescent state on a legacy-free machine. `routeIrq` is ready for the first real
|
|
||||||
//! device driver (a keyboard, say).
|
|
||||||
|
|
||||||
const paging = @import("paging.zig");
|
|
||||||
|
|
||||||
/// A MADT Interrupt Source Override: an ISA IRQ that appears at a different global
|
|
||||||
/// system interrupt, with its own polarity/trigger (MPS INTI `flags`).
|
|
||||||
pub const IsoEntry = struct { source: u8, gsi: u32, flags: u16 };
|
|
||||||
|
|
||||||
var base: u64 = 0; // 0 = no I/O APIC discovered
|
|
||||||
var gsi_base: u32 = 0;
|
|
||||||
var max_entries: u32 = 0;
|
|
||||||
var overrides: [16]IsoEntry = undefined;
|
|
||||||
var override_count: usize = 0;
|
|
||||||
|
|
||||||
// The I/O APIC exposes an index register (IOREGSEL) and a data window (IOWIN).
|
|
||||||
const reg_ioregsel = 0x00;
|
|
||||||
const reg_iowin = 0x10;
|
|
||||||
const reg_version = 0x01;
|
|
||||||
const redir_base = 0x10; // redirection table: two 32-bit regs per entry
|
|
||||||
const redir_mask = 1 << 16; // mask bit in the low dword
|
|
||||||
|
|
||||||
/// Supply the discovered I/O APIC location + the MADT IRQ overrides. Call before `init`.
|
|
||||||
pub fn configure(ioapic_base: u64, ioapic_gsi_base: u32, isos: []const IsoEntry) void {
|
|
||||||
base = ioapic_base;
|
|
||||||
gsi_base = ioapic_gsi_base;
|
|
||||||
override_count = @min(isos.len, overrides.len);
|
|
||||||
for (isos[0..override_count], 0..) |iso, i| overrides[i] = iso;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn regRead(index: u32) u32 {
|
|
||||||
@as(*volatile u32, @ptrFromInt(base + reg_ioregsel)).* = index;
|
|
||||||
return @as(*volatile u32, @ptrFromInt(base + reg_iowin)).*;
|
|
||||||
}
|
|
||||||
fn regWrite(index: u32, value: u32) void {
|
|
||||||
@as(*volatile u32, @ptrFromInt(base + reg_ioregsel)).* = index;
|
|
||||||
@as(*volatile u32, @ptrFromInt(base + reg_iowin)).* = value;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn writeEntry(n: u32, low: u32, high: u32) void {
|
|
||||||
regWrite(redir_base + 2 * n, low);
|
|
||||||
regWrite(redir_base + 2 * n + 1, high);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Map the I/O APIC and mask every redirection entry — the safe quiescent state.
|
|
||||||
pub fn init() void {
|
|
||||||
if (base == 0) return;
|
|
||||||
paging.map(base & ~@as(u64, 0xFFF), base & ~@as(u64, 0xFFF), true);
|
|
||||||
max_entries = ((regRead(reg_version) >> 16) & 0xFF) + 1;
|
|
||||||
var n: u32 = 0;
|
|
||||||
while (n < max_entries) : (n += 1) writeEntry(n, redir_mask, 0);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Route ISA `irq` to `vector` on the LAPIC `apic_id`, honouring a MADT override
|
|
||||||
/// for its GSI/polarity/trigger, and unmask it. No caller yet — groundwork for the
|
|
||||||
/// first device driver.
|
|
||||||
pub fn routeIrq(irq: u8, vector: u8, apic_id: u8) void {
|
|
||||||
if (base == 0) return;
|
|
||||||
|
|
||||||
var gsi: u32 = irq;
|
|
||||||
var flags: u16 = 0;
|
|
||||||
for (overrides[0..override_count]) |o| {
|
|
||||||
if (o.source == irq) {
|
|
||||||
gsi = o.gsi;
|
|
||||||
flags = o.flags;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (gsi < gsi_base) return;
|
|
||||||
const n = gsi - gsi_base;
|
|
||||||
if (n >= max_entries) return;
|
|
||||||
|
|
||||||
// Low dword: vector + delivery mode fixed(0) + physical dest(0), unmasked.
|
|
||||||
// MPS INTI flags: bits [1:0] polarity (3 = active low), [3:2] trigger (3 = level).
|
|
||||||
var low: u32 = vector;
|
|
||||||
if (flags & 0x3 == 3) low |= (1 << 13);
|
|
||||||
if ((flags >> 2) & 0x3 == 3) low |= (1 << 15);
|
|
||||||
const high: u32 = @as(u32, apic_id) << 24; // destination APIC ID
|
|
||||||
writeEntry(n, low, high);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Number of redirection entries the I/O APIC advertises (0 until `init`).
|
|
||||||
pub fn entryCount() u32 {
|
|
||||||
return max_entries;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The low dword of redirection entry `n` — for diagnostics/read-back.
|
|
||||||
pub fn entryLow(n: u32) u32 {
|
|
||||||
if (base == 0) return 0;
|
|
||||||
return regRead(redir_base + 2 * n);
|
|
||||||
}
|
|
||||||
@@ -1,180 +0,0 @@
|
|||||||
# x86_64 low-level entry code: the CPU-exception stubs, plus the GDT/IDT load
|
|
||||||
# helpers. Kept in a dedicated assembly file rather than inline asm because these
|
|
||||||
# need real labels and cross-symbol jumps/calls (isr_common, exceptionHandler),
|
|
||||||
# and because `lgdt`/`lidt` memory operands aren't expressible in Zig inline asm.
|
|
||||||
#
|
|
||||||
# Each exception vector normalises the stack to a uniform trap frame — a dummy
|
|
||||||
# error code where the CPU pushes none, then the vector number — and jumps to the
|
|
||||||
# shared tail, which saves the general registers and calls the Zig handler with a
|
|
||||||
# pointer to the frame (matching src/arch/x86_64/idt.zig's CpuState).
|
|
||||||
|
|
||||||
.text
|
|
||||||
|
|
||||||
# gdt_flush(rdi = *GDT descriptor): load the GDT, reload the data segment
|
|
||||||
# registers to the data selector, and reload CS to the code selector. CS can't be
|
|
||||||
# set with mov, so we far-return through the caller's own return address.
|
|
||||||
.global gdt_flush
|
|
||||||
gdt_flush:
|
|
||||||
lgdt (%rdi)
|
|
||||||
mov $0x10, %ax # kernel data selector
|
|
||||||
mov %ax, %ds
|
|
||||||
mov %ax, %es
|
|
||||||
mov %ax, %ss
|
|
||||||
mov %ax, %fs
|
|
||||||
mov %ax, %gs
|
|
||||||
pop %rax # caller's return address
|
|
||||||
push $0x08 # kernel code selector (new CS)
|
|
||||||
push %rax # return address (new RIP)
|
|
||||||
lretq
|
|
||||||
|
|
||||||
# idt_flush(rdi = *IDT descriptor): load the IDT.
|
|
||||||
.global idt_flush
|
|
||||||
idt_flush:
|
|
||||||
lidt (%rdi)
|
|
||||||
ret
|
|
||||||
|
|
||||||
# load_tr(di = TSS selector): load the task register.
|
|
||||||
.global load_tr
|
|
||||||
load_tr:
|
|
||||||
ltr %di
|
|
||||||
ret
|
|
||||||
|
|
||||||
# switch_context(rdi = &old_task.rsp, rsi = new_task.rsp)
|
|
||||||
# Cooperative context switch: save the callee-saved registers on the current
|
|
||||||
# stack, stash the stack pointer in the old task, load the new task's stack
|
|
||||||
# pointer, restore its callee-saved registers, and return into it. Caller-saved
|
|
||||||
# registers are the compiler's responsibility (this looks like a normal call).
|
|
||||||
.global switch_context
|
|
||||||
switch_context:
|
|
||||||
push %rbx
|
|
||||||
push %rbp
|
|
||||||
push %r12
|
|
||||||
push %r13
|
|
||||||
push %r14
|
|
||||||
push %r15
|
|
||||||
mov %rsp, (%rdi) # save old stack pointer into old_task.rsp
|
|
||||||
mov %rsi, %rsp # switch to the new task's stack
|
|
||||||
pop %r15
|
|
||||||
pop %r14
|
|
||||||
pop %r13
|
|
||||||
pop %r12
|
|
||||||
pop %rbp
|
|
||||||
pop %rbx
|
|
||||||
ret # return into the new task's saved instruction pointer
|
|
||||||
|
|
||||||
# task_trampoline: the first thing a freshly-spawned task runs. init_task_stack
|
|
||||||
# leaves its entry function in r15. New tasks start with interrupts enabled.
|
|
||||||
.global task_trampoline
|
|
||||||
task_trampoline:
|
|
||||||
sti
|
|
||||||
call *%r15 # call the task entry (fn() void)
|
|
||||||
1: hlt # if the entry returns, idle (still preemptible)
|
|
||||||
jmp 1b
|
|
||||||
|
|
||||||
# Stub for a vector the CPU does NOT push an error code for: push a dummy 0.
|
|
||||||
.macro STUB_NOERR vec
|
|
||||||
.global isr\vec
|
|
||||||
isr\vec:
|
|
||||||
pushq $0
|
|
||||||
pushq $\vec
|
|
||||||
jmp isr_common
|
|
||||||
.endm
|
|
||||||
|
|
||||||
# Stub for a vector the CPU DOES push an error code for: leave it in place.
|
|
||||||
.macro STUB_ERR vec
|
|
||||||
.global isr\vec
|
|
||||||
isr\vec:
|
|
||||||
pushq $\vec
|
|
||||||
jmp isr_common
|
|
||||||
.endm
|
|
||||||
|
|
||||||
STUB_NOERR 0
|
|
||||||
STUB_NOERR 1
|
|
||||||
STUB_NOERR 2
|
|
||||||
STUB_NOERR 3
|
|
||||||
STUB_NOERR 4
|
|
||||||
STUB_NOERR 5
|
|
||||||
STUB_NOERR 6
|
|
||||||
STUB_NOERR 7
|
|
||||||
STUB_ERR 8
|
|
||||||
STUB_NOERR 9
|
|
||||||
STUB_ERR 10
|
|
||||||
STUB_ERR 11
|
|
||||||
STUB_ERR 12
|
|
||||||
STUB_ERR 13
|
|
||||||
STUB_ERR 14
|
|
||||||
STUB_NOERR 15
|
|
||||||
STUB_NOERR 16
|
|
||||||
STUB_ERR 17
|
|
||||||
STUB_NOERR 18
|
|
||||||
STUB_NOERR 19
|
|
||||||
STUB_NOERR 20
|
|
||||||
STUB_ERR 21
|
|
||||||
STUB_NOERR 22
|
|
||||||
STUB_NOERR 23
|
|
||||||
STUB_NOERR 24
|
|
||||||
STUB_NOERR 25
|
|
||||||
STUB_NOERR 26
|
|
||||||
STUB_NOERR 27
|
|
||||||
STUB_NOERR 28
|
|
||||||
STUB_NOERR 29
|
|
||||||
STUB_NOERR 30
|
|
||||||
STUB_NOERR 31
|
|
||||||
|
|
||||||
# Device-interrupt vectors (timer, spurious, room for more). None push an error
|
|
||||||
# code, so they all use the dummy-zero form.
|
|
||||||
STUB_NOERR 32
|
|
||||||
STUB_NOERR 33
|
|
||||||
STUB_NOERR 34
|
|
||||||
STUB_NOERR 35
|
|
||||||
STUB_NOERR 36
|
|
||||||
STUB_NOERR 37
|
|
||||||
STUB_NOERR 38
|
|
||||||
STUB_NOERR 39
|
|
||||||
STUB_NOERR 40
|
|
||||||
STUB_NOERR 41
|
|
||||||
STUB_NOERR 42
|
|
||||||
STUB_NOERR 43
|
|
||||||
STUB_NOERR 44
|
|
||||||
STUB_NOERR 45
|
|
||||||
STUB_NOERR 46
|
|
||||||
STUB_NOERR 47
|
|
||||||
|
|
||||||
.extern interruptDispatch
|
|
||||||
|
|
||||||
# Shared tail. Register push order here defines the CpuState field order.
|
|
||||||
isr_common:
|
|
||||||
push %rax
|
|
||||||
push %rbx
|
|
||||||
push %rcx
|
|
||||||
push %rdx
|
|
||||||
push %rsi
|
|
||||||
push %rdi
|
|
||||||
push %rbp
|
|
||||||
push %r8
|
|
||||||
push %r9
|
|
||||||
push %r10
|
|
||||||
push %r11
|
|
||||||
push %r12
|
|
||||||
push %r13
|
|
||||||
push %r14
|
|
||||||
push %r15
|
|
||||||
mov %rsp, %rdi # first argument: pointer to the trap frame
|
|
||||||
call interruptDispatch
|
|
||||||
pop %r15
|
|
||||||
pop %r14
|
|
||||||
pop %r13
|
|
||||||
pop %r12
|
|
||||||
pop %r11
|
|
||||||
pop %r10
|
|
||||||
pop %r9
|
|
||||||
pop %r8
|
|
||||||
pop %rbp
|
|
||||||
pop %rdi
|
|
||||||
pop %rsi
|
|
||||||
pop %rdx
|
|
||||||
pop %rcx
|
|
||||||
pop %rbx
|
|
||||||
pop %rax
|
|
||||||
add $16, %rsp # drop the vector and error code
|
|
||||||
iretq
|
|
||||||
@@ -1,47 +0,0 @@
|
|||||||
/* Kernel link layout.
|
|
||||||
*
|
|
||||||
* The kernel is linked at a fixed low physical address (set by `image_base` in
|
|
||||||
* build.zig). UEFI runs with memory identity-mapped, so the bootloader can load
|
|
||||||
* each PT_LOAD segment to the physical address matching its virtual address and
|
|
||||||
* jump straight to _start — no page tables to build yet. (Moving to a
|
|
||||||
* higher-half virtual base is a later step, once the bootloader sets up paging.)
|
|
||||||
*/
|
|
||||||
|
|
||||||
ENTRY(_start)
|
|
||||||
|
|
||||||
/* One loadable segment per permission set, so the loader can map .text as R+X,
|
|
||||||
* .rodata as R, and .data/.bss as R+W. FLAGS bits: 1=X, 2=W, 4=R. */
|
|
||||||
PHDRS {
|
|
||||||
text PT_LOAD FLAGS(5); /* R + X */
|
|
||||||
rodata PT_LOAD FLAGS(4); /* R */
|
|
||||||
data PT_LOAD FLAGS(6); /* R + W */
|
|
||||||
}
|
|
||||||
|
|
||||||
SECTIONS {
|
|
||||||
.text ALIGN(4K) : {
|
|
||||||
*(.text .text.*)
|
|
||||||
} :text
|
|
||||||
|
|
||||||
.rodata ALIGN(4K) : {
|
|
||||||
*(.rodata .rodata.*)
|
|
||||||
} :rodata
|
|
||||||
|
|
||||||
.data ALIGN(4K) : {
|
|
||||||
*(.data .data.*)
|
|
||||||
} :data
|
|
||||||
|
|
||||||
/* .bss occupies memory but not file space. The loader zeroes it via the
|
|
||||||
* gap between each PT_LOAD segment's file size and memory size, so no
|
|
||||||
* boundary symbols are needed here. (Zig's self-hosted linker also does not
|
|
||||||
* yet honour linker-script symbol assignments.) */
|
|
||||||
.bss ALIGN(4K) : {
|
|
||||||
*(.bss .bss.*)
|
|
||||||
*(COMMON)
|
|
||||||
} :data
|
|
||||||
|
|
||||||
/DISCARD/ : {
|
|
||||||
*(.comment)
|
|
||||||
*(.note .note.*)
|
|
||||||
*(.eh_frame .eh_frame_hdr)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,153 +0,0 @@
|
|||||||
//! The kernel's page tables and virtual memory manager.
|
|
||||||
//!
|
|
||||||
//! Builds our own 4-level page tables and switches CR3 onto them, replacing the
|
|
||||||
//! firmware's. Unlike the earlier bootstrap this maps with real permissions:
|
|
||||||
//! RAM is identity-mapped read-write + no-execute, the kernel's own segments get
|
|
||||||
//! their ELF permissions (code R+X, rodata R, data R+W+NX), and page 0 is left
|
|
||||||
//! unmapped as a null guard. It also exposes map/unmap for on-demand mapping,
|
|
||||||
//! which the kernel heap will build on.
|
|
||||||
//!
|
|
||||||
//! Everything is 4 KiB pages — precise and simple; the extra table memory is
|
|
||||||
//! negligible against available RAM.
|
|
||||||
|
|
||||||
const danos = @import("danos");
|
|
||||||
const io = @import("io.zig");
|
|
||||||
|
|
||||||
const page_size = danos.page_size;
|
|
||||||
|
|
||||||
// Page-table entry bits.
|
|
||||||
const present: u64 = 1 << 0;
|
|
||||||
const writable: u64 = 1 << 1;
|
|
||||||
const no_execute: u64 = 1 << 63;
|
|
||||||
const addr_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
|
||||||
|
|
||||||
// ELF segment flags (p_flags).
|
|
||||||
const pf_x: u32 = 1;
|
|
||||||
const pf_w: u32 = 2;
|
|
||||||
|
|
||||||
// State kept after init so map()/unmap() can serve later callers (e.g. the heap).
|
|
||||||
var kernel_pml4: u64 = 0;
|
|
||||||
var alloc_frame: *const fn () ?u64 = undefined;
|
|
||||||
|
|
||||||
fn tableAt(phys: u64) *[512]u64 {
|
|
||||||
return @ptrFromInt(phys);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn allocTable() u64 {
|
|
||||||
const frame = alloc_frame() orelse @panic("paging: out of memory building page tables");
|
|
||||||
@memset(tableAt(frame)[0..], 0);
|
|
||||||
return frame;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Return the table an entry points at, creating it if empty. Intermediate
|
|
||||||
/// entries are writable and executable so the leaf's bits govern (a page is
|
|
||||||
/// writable only if every level is; non-executable if any level is).
|
|
||||||
fn descend(entry: *u64) u64 {
|
|
||||||
if (entry.* & present != 0) return entry.* & addr_mask;
|
|
||||||
const frame = allocTable();
|
|
||||||
entry.* = frame | present | writable;
|
|
||||||
return frame;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Map one 4 KiB page `virt` -> `phys` with `flags` (present is added).
|
|
||||||
fn mapPage(pml4: u64, virt: u64, phys: u64, flags: u64) void {
|
|
||||||
const pml4e = &tableAt(pml4)[(virt >> 39) & 0x1FF];
|
|
||||||
const pdpt = descend(pml4e);
|
|
||||||
const pdpte = &tableAt(pdpt)[(virt >> 30) & 0x1FF];
|
|
||||||
const pd = descend(pdpte);
|
|
||||||
const pde = &tableAt(pd)[(virt >> 21) & 0x1FF];
|
|
||||||
const pt = descend(pde);
|
|
||||||
tableAt(pt)[(virt >> 12) & 0x1FF] = (phys & addr_mask) | flags | present;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Identity-map [base, base+len) with `flags`, rounded out to whole pages.
|
|
||||||
fn mapRangeIdentity(pml4: u64, base: u64, len: u64, flags: u64) void {
|
|
||||||
var addr = base & ~@as(u64, page_size - 1);
|
|
||||||
const end = base + len;
|
|
||||||
while (addr < end) : (addr += page_size) {
|
|
||||||
if (addr == 0) continue; // leave page 0 unmapped: the null guard
|
|
||||||
mapPage(pml4, addr, addr, flags);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn regions(mm: danos.MemoryMap) []const danos.MemoryRegion {
|
|
||||||
return @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Enable the NX bit in the page-table format (EFER.NXE). Must happen before we
|
|
||||||
/// load a CR3 whose entries set the NX bit, or those bits are reserved and fault.
|
|
||||||
fn enableNx() void {
|
|
||||||
const efer_msr = 0xC0000080;
|
|
||||||
io.wrmsr(efer_msr, io.rdmsr(efer_msr) | (1 << 11));
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Build the address space and switch onto it.
|
|
||||||
pub fn init(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void {
|
|
||||||
alloc_frame = allocFrame;
|
|
||||||
enableNx();
|
|
||||||
const pml4 = allocTable();
|
|
||||||
|
|
||||||
// 1. All RAM identity-mapped RW + NX. Non-RAM (MMIO) is skipped and stays
|
|
||||||
// unmapped unless mapped explicitly below.
|
|
||||||
for (regions(boot_info.memory_map)) |r| {
|
|
||||||
if (r.kind == .mmio) continue;
|
|
||||||
mapRangeIdentity(pml4, r.base, r.pages * page_size, present | writable | no_execute);
|
|
||||||
}
|
|
||||||
|
|
||||||
// 2. The framebuffer and the Local APIC (device memory we need), RW + NX.
|
|
||||||
const fb = boot_info.framebuffer;
|
|
||||||
mapRangeIdentity(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute);
|
|
||||||
mapPage(pml4, 0xFEE00000, 0xFEE00000, present | writable | no_execute);
|
|
||||||
|
|
||||||
// 3. Overlay the kernel's own segments with their real ELF permissions,
|
|
||||||
// replacing the blanket RW+NX from step 1: code becomes R+X, rodata R,
|
|
||||||
// data R+W+NX. This is the W^X guarantee.
|
|
||||||
for (boot_info.kernel_segments[0..boot_info.kernel_segment_count]) |seg| {
|
|
||||||
var flags: u64 = present;
|
|
||||||
if (seg.flags & pf_w != 0) flags |= writable;
|
|
||||||
if (seg.flags & pf_x == 0) flags |= no_execute;
|
|
||||||
var addr = seg.virt;
|
|
||||||
const end = seg.virt + seg.pages * page_size;
|
|
||||||
while (addr < end) : (addr += page_size) mapPage(pml4, addr, addr, flags);
|
|
||||||
}
|
|
||||||
|
|
||||||
kernel_pml4 = pml4;
|
|
||||||
asm volatile ("mov %[pml4], %%cr3"
|
|
||||||
:
|
|
||||||
: [pml4] "r" (pml4),
|
|
||||||
: .{ .memory = true }
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Map a page into the kernel address space on demand (for the heap, etc.).
|
|
||||||
/// `writable_page` controls W; pages are always mapped non-executable.
|
|
||||||
pub fn map(virt: u64, phys: u64, writable_page: bool) void {
|
|
||||||
var flags: u64 = present | no_execute;
|
|
||||||
if (writable_page) flags |= writable;
|
|
||||||
mapPage(kernel_pml4, virt, phys, flags);
|
|
||||||
invalidate(virt);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Remove a mapping and flush it from the TLB.
|
|
||||||
pub fn unmap(virt: u64) void {
|
|
||||||
const pml4e = tableAt(kernel_pml4)[(virt >> 39) & 0x1FF];
|
|
||||||
if (pml4e & present == 0) return;
|
|
||||||
const pdpte = tableAt(pml4e & addr_mask)[(virt >> 30) & 0x1FF];
|
|
||||||
if (pdpte & present == 0) return;
|
|
||||||
const pde = tableAt(pdpte & addr_mask)[(virt >> 21) & 0x1FF];
|
|
||||||
if (pde & present == 0) return;
|
|
||||||
tableAt(pde & addr_mask)[(virt >> 12) & 0x1FF] = 0;
|
|
||||||
invalidate(virt);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn invalidate(virt: u64) void {
|
|
||||||
// invlpg needs its operand via a register-indirect memory reference that Zig
|
|
||||||
// inline asm won't form directly, so stage the address in a register first.
|
|
||||||
asm volatile (
|
|
||||||
\\mov %[v], %%rax
|
|
||||||
\\invlpg (%%rax)
|
|
||||||
:
|
|
||||||
: [v] "r" (virt),
|
|
||||||
: .{ .rax = true, .memory = true }
|
|
||||||
);
|
|
||||||
}
|
|
||||||
@@ -1,48 +0,0 @@
|
|||||||
//! Task State Segment and its interrupt stack. In long mode the TSS's main job
|
|
||||||
//! is the Interrupt Stack Table: an IDT gate can name an IST entry, and the CPU
|
|
||||||
//! switches to that stack when the exception fires — no matter how broken the
|
|
||||||
//! interrupted stack was. We use IST1 for the double-fault handler, so a fault
|
|
||||||
//! that happens *because* the current stack is unusable still lands on solid
|
|
||||||
//! ground instead of triple-faulting.
|
|
||||||
|
|
||||||
const gdt = @import("gdt.zig");
|
|
||||||
|
|
||||||
/// x86_64 TSS. `packed` because several 64-bit fields sit at 4-byte-unaligned
|
|
||||||
/// offsets (rsp0 at byte 4), which a normal struct would pad away.
|
|
||||||
const Tss = packed struct {
|
|
||||||
reserved0: u32 = 0,
|
|
||||||
rsp0: u64 = 0,
|
|
||||||
rsp1: u64 = 0,
|
|
||||||
rsp2: u64 = 0,
|
|
||||||
reserved1: u64 = 0,
|
|
||||||
ist1: u64 = 0,
|
|
||||||
ist2: u64 = 0,
|
|
||||||
ist3: u64 = 0,
|
|
||||||
ist4: u64 = 0,
|
|
||||||
ist5: u64 = 0,
|
|
||||||
ist6: u64 = 0,
|
|
||||||
ist7: u64 = 0,
|
|
||||||
reserved2: u64 = 0,
|
|
||||||
reserved3: u16 = 0,
|
|
||||||
iomap_base: u16 = 0,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// The IST slot (1-based, as the IDT gate encodes it) used for critical faults.
|
|
||||||
pub const double_fault_ist = 1;
|
|
||||||
|
|
||||||
var tss: Tss align(16) = .{};
|
|
||||||
|
|
||||||
/// Dedicated stack for IST1. Static so it needs no allocator and is always valid.
|
|
||||||
var ist1_stack: [16 * 1024]u8 align(16) = undefined;
|
|
||||||
|
|
||||||
/// Loads the task register with the TSS selector. Defined in isr.s.
|
|
||||||
extern fn load_tr(selector: u16) callconv(.c) void;
|
|
||||||
|
|
||||||
/// Point IST1 at its stack, publish the TSS through the GDT, and load it into the
|
|
||||||
/// task register. Requires the GDT to already be loaded (gdt.init first).
|
|
||||||
pub fn init() void {
|
|
||||||
tss.ist1 = @intFromPtr(&ist1_stack) + ist1_stack.len; // stacks grow down
|
|
||||||
tss.iomap_base = @sizeOf(Tss); // == limit: no I/O permission bitmap
|
|
||||||
gdt.setTss(@intFromPtr(&tss), @sizeOf(Tss) - 1);
|
|
||||||
load_tr(gdt.tss_selector);
|
|
||||||
}
|
|
||||||
@@ -1,290 +0,0 @@
|
|||||||
const std = @import("std");
|
|
||||||
const danos = @import("danos");
|
|
||||||
const arch = @import("arch");
|
|
||||||
const console = @import("console.zig");
|
|
||||||
const log = @import("log.zig");
|
|
||||||
const pmm = @import("pmm.zig");
|
|
||||||
const heap = @import("heap.zig");
|
|
||||||
const scheduler = @import("scheduler.zig");
|
|
||||||
const platform = @import("platform");
|
|
||||||
const tests = @import("tests.zig");
|
|
||||||
const build_options = @import("build_options");
|
|
||||||
const BootInfo = danos.BootInfo;
|
|
||||||
|
|
||||||
/// The calling convention used to enter the kernel. Pinned to SysV explicitly:
|
|
||||||
/// the bootloader is built for the UEFI target, whose C convention is Microsoft
|
|
||||||
/// x64 (first argument in RCX), while the kernel is SysV (first argument in
|
|
||||||
/// RDI). Both sides reference this so the `boot_info` pointer lands in the
|
|
||||||
/// register the other expects. `danos.kernel_abi` re-exports it to the loader.
|
|
||||||
pub const kernel_abi = danos.kernel_abi;
|
|
||||||
|
|
||||||
// POST/checkpoint codes emitted to I/O port 0x80 at boot milestones — the
|
|
||||||
// last-resort progress signal on a machine with no text output at all.
|
|
||||||
const cp_entry = 0x10;
|
|
||||||
const cp_paging = 0x20;
|
|
||||||
const cp_heap = 0x30;
|
|
||||||
const cp_discovery = 0x40;
|
|
||||||
const cp_scheduler = 0x50;
|
|
||||||
const cp_timer = 0x60;
|
|
||||||
const cp_running = 0x70;
|
|
||||||
const cp_exception = 0xE0;
|
|
||||||
const cp_panic = 0xEE;
|
|
||||||
|
|
||||||
/// Kernel entry point. The bootloader jumps here after `ExitBootServices` with a
|
|
||||||
/// pointer to the handoff data. There is no runtime, no stack unwinding, and no
|
|
||||||
/// caller to return to, so this never returns.
|
|
||||||
export fn _start(boot_info: *const BootInfo) callconv(kernel_abi) noreturn {
|
|
||||||
kmain(boot_info);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn kmain(boot_info: *const BootInfo) noreturn {
|
|
||||||
// The **log** is the machine-readable diagnostic stream: it fans out to every
|
|
||||||
// *diagnostic* channel that exists (serial, the 0xE9 debug console, and later a
|
|
||||||
// file on a ramdisk/USB/SSD), so a message survives as long as any is present.
|
|
||||||
// A headless, serial-less machine still boots correctly — it just goes quiet,
|
|
||||||
// with port-0x80 checkpoints as the only progress signal.
|
|
||||||
arch.serialInit();
|
|
||||||
log.addSink(arch.serialWrite);
|
|
||||||
if (arch.debugconPresent()) log.addSink(arch.debugconWrite);
|
|
||||||
|
|
||||||
// The **framebuffer** is deliberately *not* a log sink. It's a separate output
|
|
||||||
// surface — a bootstrap text console today, a graphics device driver later — so
|
|
||||||
// we never assume the OS is text-based. Only a few user-facing status lines
|
|
||||||
// (via `status`) and panics are mirrored to it; the verbose log stays out.
|
|
||||||
const fb = boot_info.framebuffer;
|
|
||||||
console.init(fb);
|
|
||||||
|
|
||||||
log.checkpoint(cp_entry);
|
|
||||||
|
|
||||||
// Catch CPU exceptions before doing anything that might fault: install our
|
|
||||||
// reporter, then bring up the GDT + IDT.
|
|
||||||
arch.setFaultHandler(onException);
|
|
||||||
arch.init();
|
|
||||||
|
|
||||||
status("danos: initialising kernel...\n");
|
|
||||||
log.write(if (console.present())
|
|
||||||
"danos: framebuffer console online (bootstrap; graphics driver later)\n"
|
|
||||||
else
|
|
||||||
"danos: no framebuffer (headless) -> logging to serial/debugcon only\n");
|
|
||||||
log.write("danos: cpu tables online (GDT, IDT, TSS)\n");
|
|
||||||
log.print(" resolution : {d}x{d}\n", .{ fb.width, fb.height });
|
|
||||||
log.print(" pitch : {d} bytes\n", .{fb.pitch});
|
|
||||||
log.print(" format : {s}\n", .{@tagName(fb.format)});
|
|
||||||
log.print(" framebuffer: 0x{x:0>16}\n", .{fb.base});
|
|
||||||
log.print (" footprint : {d} MiB\n", .{(fb.pitch * fb.height) / (1024 * 1024)});
|
|
||||||
|
|
||||||
// Summarise the physical memory the loader handed us. The array is danos's
|
|
||||||
// own MemoryRegion, so this is a plain slice — no firmware layout in sight.
|
|
||||||
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(boot_info.memory_map.regions))[0..boot_info.memory_map.len];
|
|
||||||
var usable_pages: u64 = 0;
|
|
||||||
var reserved_pages: u64 = 0; // reserved RAM only — MMIO is device space, not RAM
|
|
||||||
for (regions) |r| {
|
|
||||||
switch (r.kind) {
|
|
||||||
.usable => usable_pages += r.pages,
|
|
||||||
.reserved, .acpi_tables, .acpi_nvs => reserved_pages += r.pages,
|
|
||||||
.mmio => {},
|
|
||||||
}
|
|
||||||
}
|
|
||||||
const total_pages = usable_pages + reserved_pages;
|
|
||||||
const total_bytes = total_pages * danos.page_size;
|
|
||||||
const gib = 1 << 30;
|
|
||||||
|
|
||||||
log.write("\ndanos: physical memory\n");
|
|
||||||
log.print(" total RAM : {d}.{d:0>2} GiB ({d} MiB) - RAM the firmware reported\n", .{ total_bytes / gib, (total_bytes % gib) * 100 / gib, mib(total_pages) });
|
|
||||||
log.print(" usable : {d} MiB - free RAM (incl. reclaimed boot-services memory)\n", .{mib(usable_pages)});
|
|
||||||
log.print(" reserved : {d} MiB - kernel image, boot stack, ACPI, runtime services\n", .{mib(reserved_pages)});
|
|
||||||
log.print(" regions : {d} - entries in the firmware memory map\n", .{regions.len});
|
|
||||||
|
|
||||||
// Bring up the physical frame allocator over that map, and prove it works:
|
|
||||||
// allocate three frames, then hand them back.
|
|
||||||
pmm.init(boot_info.memory_map);
|
|
||||||
const s1 = pmm.stats();
|
|
||||||
log.print("\ndanos: frame allocator online\n", .{});
|
|
||||||
log.print(" free frames: {d} ({d} MiB)\n", .{ s1.free_frames, mib(s1.free_frames) });
|
|
||||||
const f0 = pmm.alloc();
|
|
||||||
const f1 = pmm.alloc();
|
|
||||||
const f2 = pmm.alloc();
|
|
||||||
log.print(" alloc x3 : 0x{x} 0x{x} 0x{x}\n", .{ f0 orelse 0, f1 orelse 0, f2 orelse 0 });
|
|
||||||
if (f0) |p| pmm.free(p);
|
|
||||||
if (f1) |p| pmm.free(p);
|
|
||||||
if (f2) |p| pmm.free(p);
|
|
||||||
log.print(" after free : {d} frames free\n", .{pmm.stats().free_frames});
|
|
||||||
|
|
||||||
// Switch off the firmware's page tables onto our own (with real permissions).
|
|
||||||
arch.enablePaging(pmm.alloc, boot_info);
|
|
||||||
log.checkpoint(cp_paging);
|
|
||||||
log.print("\ndanos: paging enabled\n", .{});
|
|
||||||
log.print(" page tables: CR3 = 0x{x:0>16}\n", .{arch.readCr3()});
|
|
||||||
log.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_info.kernel_segment_count});
|
|
||||||
|
|
||||||
// Bring up the kernel heap (dynamic allocation), built on the VMM.
|
|
||||||
heap.init();
|
|
||||||
log.checkpoint(cp_heap);
|
|
||||||
log.write("\ndanos: kernel heap online\n");
|
|
||||||
// Measure the amount of resources the kernel is actually using
|
|
||||||
const s2 = pmm.stats();
|
|
||||||
log.print(" Kernel footprint: {d} KiB\n", .{kib(s1.free_frames - s2.free_frames)});
|
|
||||||
|
|
||||||
// Enumerate hardware from the firmware tables (ACPI here) into a generic
|
|
||||||
// device tree, then list it. Discovery walks ACPI memory directly (identity-
|
|
||||||
// mapped) and maps PCIe config space on demand via the VMM. A failure here is
|
|
||||||
// not fatal yet — log it and carry on.
|
|
||||||
const hal = platform.Hal{
|
|
||||||
.mapMmio = arch.mapPage,
|
|
||||||
.pioRead = arch.pioRead,
|
|
||||||
.pioWrite = arch.pioWrite,
|
|
||||||
};
|
|
||||||
if (platform.discover(boot_info, heap.allocator(), hal)) |devtree| {
|
|
||||||
var dt = devtree;
|
|
||||||
log.write("\ndanos: device discovery online\n");
|
|
||||||
dt.dump(log.write);
|
|
||||||
|
|
||||||
// Power register map extracted from the FADT + AML, for confidence it parsed.
|
|
||||||
const pw = platform.powerInfo();
|
|
||||||
log.write("danos: power\n");
|
|
||||||
log.print(" pm1a_cnt : {s} 0x{x} (width {d})\n", .{ if (pw.pm1a_cnt.mmio) "mmio" else "io", pw.pm1a_cnt.address, pw.pm1a_cnt.width });
|
|
||||||
if (pw.s5) |s| {
|
|
||||||
log.print(" S5 slp_typ : a={d} b={d}\n", .{ s.slp_typ_a, s.slp_typ_b });
|
|
||||||
} else {
|
|
||||||
log.write(" S5 slp_typ : (not found)\n");
|
|
||||||
}
|
|
||||||
log.print(" reset : supported={} {s} 0x{x} val 0x{x}\n", .{ pw.reset_supported, if (pw.reset.mmio) "mmio" else "io", pw.reset.address, pw.reset_value });
|
|
||||||
|
|
||||||
// AML namespace parse integrity: consumed should equal total.
|
|
||||||
const am = platform.amlStats();
|
|
||||||
log.print(" aml : {d} namespace nodes, parsed {d}/{d} bytes\n", .{ am.nodes, am.consumed, am.total });
|
|
||||||
|
|
||||||
// Feed the arch layer the discovered addresses/facts so it makes no legacy
|
|
||||||
// assumptions — the point of all this on UEFI Class 3 firmware. MMIO bases
|
|
||||||
// (HPET, I/O APIC) come from the device tree; scalar facts from ACPI.
|
|
||||||
const pinfo = platform.platformInfo();
|
|
||||||
const hpet_base: u64 = if (dt.firstOfClass(.timer)) |t|
|
|
||||||
(if (t.firstResource(.memory)) |r| r.start else 0)
|
|
||||||
else
|
|
||||||
0;
|
|
||||||
var ioapic_base: u64 = 0;
|
|
||||||
var ioapic_gsi: u32 = 0;
|
|
||||||
if (dt.firstOfClass(.interrupt_controller)) |ic| {
|
|
||||||
if (ic.firstResource(.memory)) |r| ioapic_base = r.start;
|
|
||||||
if (ic.firstResource(.irq)) |r| ioapic_gsi = @intCast(r.start);
|
|
||||||
}
|
|
||||||
var isos: [16]arch.IsoEntry = undefined;
|
|
||||||
const iso_n = @min(pinfo.override_count, isos.len);
|
|
||||||
for (0..iso_n) |i| isos[i] = .{
|
|
||||||
.source = pinfo.overrides[i].source,
|
|
||||||
.gsi = pinfo.overrides[i].gsi,
|
|
||||||
.flags = pinfo.overrides[i].flags,
|
|
||||||
};
|
|
||||||
const pm_timer: ?arch.PmTimer = if (pinfo.pm_timer.present())
|
|
||||||
.{ .mmio = pinfo.pm_timer.mmio, .address = pinfo.pm_timer.address, .is_32bit = pinfo.pm_timer_32bit }
|
|
||||||
else
|
|
||||||
null;
|
|
||||||
arch.configurePlatform(.{
|
|
||||||
.pic_present = pinfo.pic_present,
|
|
||||||
.hpet_base = hpet_base,
|
|
||||||
.pm_timer = pm_timer,
|
|
||||||
.ioapic_base = ioapic_base,
|
|
||||||
.ioapic_gsi_base = ioapic_gsi,
|
|
||||||
.overrides = isos[0..iso_n],
|
|
||||||
});
|
|
||||||
if (pinfo.spcr_uart) |u| arch.serialReconfigure(u.mmio, u.address);
|
|
||||||
|
|
||||||
log.write("danos: platform\n");
|
|
||||||
log.print(" 8259 PIC : {s}\n", .{if (pinfo.pic_present) "present" else "absent"});
|
|
||||||
log.print(" lapic base : 0x{x}\n", .{pinfo.lapic_base});
|
|
||||||
log.print(" hpet base : 0x{x}\n", .{hpet_base});
|
|
||||||
log.print(" pm timer : {s} 0x{x} ({s})\n", .{ if (pinfo.pm_timer.mmio) "mmio" else "io", pinfo.pm_timer.address, if (pinfo.pm_timer_32bit) "32-bit" else "24-bit" });
|
|
||||||
if (pinfo.spcr_uart) |u| {
|
|
||||||
log.print(" console UART: {s} 0x{x} (SPCR type {d})\n", .{ if (u.mmio) "mmio" else "io", u.address, pinfo.spcr_kind });
|
|
||||||
} else {
|
|
||||||
log.write(" console UART: none in SPCR -> legacy COM1\n");
|
|
||||||
}
|
|
||||||
log.print(" ioapic : base 0x{x}, {d} inputs (masked); entry0 low 0x{x}\n", .{ ioapic_base, arch.ioapicEntryCount(), arch.ioapicEntryLow(0) });
|
|
||||||
} else |err| {
|
|
||||||
log.print("\ndanos: device discovery failed: {s}\n", .{@errorName(err)});
|
|
||||||
}
|
|
||||||
log.checkpoint(cp_discovery);
|
|
||||||
|
|
||||||
// Register the current context as the first task before enabling preemption.
|
|
||||||
scheduler.init(4);
|
|
||||||
log.checkpoint(cp_scheduler);
|
|
||||||
log.write("\ndanos: scheduler online\n");
|
|
||||||
|
|
||||||
// Start the timer and unmask interrupts — the kernel now has a heartbeat, and
|
|
||||||
// the timer preempts among tasks.
|
|
||||||
arch.startTimer();
|
|
||||||
arch.enableInterrupts();
|
|
||||||
log.checkpoint(cp_timer);
|
|
||||||
log.print("danos: timer online ({d} Hz tick; LAPIC {d} MHz, TSC {d} MHz; calibrated via {s})\n", .{ arch.timer_hz, arch.lapicHz() / 1_000_000, arch.tscHz() / 1_000_000, arch.timerCalibrationSource() });
|
|
||||||
|
|
||||||
// In a test build (`zig build -Dtest-case=<name>`), run that case and stop.
|
|
||||||
// Normal builds fall through to the idle halt.
|
|
||||||
if (build_options.test_case) |case| {
|
|
||||||
tests.run(case, boot_info);
|
|
||||||
arch.halt();
|
|
||||||
}
|
|
||||||
|
|
||||||
log.checkpoint(cp_running);
|
|
||||||
status("kernel initialised.\n");
|
|
||||||
|
|
||||||
// TODO: init process
|
|
||||||
|
|
||||||
status("\nnothing left to do; halting CPU.\n");
|
|
||||||
|
|
||||||
arch.halt();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// A user-facing status line: to the diagnostic `log` *and* the on-screen console
|
|
||||||
/// (if a framebuffer is present). The verbose log uses `log.*` directly and never
|
|
||||||
/// touches the framebuffer.
|
|
||||||
fn status(msg: []const u8) void {
|
|
||||||
log.write(msg);
|
|
||||||
console.write(msg);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn statusPrint(comptime fmt: []const u8, args: anytype) void {
|
|
||||||
var buf: [256]u8 = undefined;
|
|
||||||
status(std.fmt.bufPrint(&buf, fmt, args) catch return);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Frames (4 KiB pages) to whole MiB.
|
|
||||||
fn mib(pages: u64) u64 {
|
|
||||||
return pages * danos.page_size / (1024 * 1024);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn kib(frames: u64) u64 {
|
|
||||||
return frames * danos.page_size / (1024);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Report a CPU exception and halt. There's no fault recovery yet, so any
|
|
||||||
/// exception is terminal — but it reports what and where (to every output sink,
|
|
||||||
/// plus a POST code and a persistent breadcrumb) instead of silently resetting.
|
|
||||||
fn onException(state: *const arch.CpuState) noreturn {
|
|
||||||
log.checkpoint(cp_exception);
|
|
||||||
// A fault is user-facing enough to paint on screen too (via statusPrint), on
|
|
||||||
// top of the diagnostic log.
|
|
||||||
statusPrint("\nCPU EXCEPTION: {s} (vector {d})\n", .{ arch.vectorName(state.vector), state.vector });
|
|
||||||
statusPrint(" error code : 0x{x}\n", .{state.error_code});
|
|
||||||
statusPrint(" RIP : 0x{x:0>16}\n", .{state.rip});
|
|
||||||
statusPrint(" RSP : 0x{x:0>16}\n", .{state.rsp});
|
|
||||||
if (state.vector == 14) statusPrint(" CR2 (addr) : 0x{x:0>16}\n", .{arch.readCr2()});
|
|
||||||
|
|
||||||
var buf: [128]u8 = undefined;
|
|
||||||
log.recordPanic(std.fmt.bufPrint(&buf, "CPU exception {s} (vector {d}) at RIP 0x{x}", .{ arch.vectorName(state.vector), state.vector, state.rip }) catch "cpu exception");
|
|
||||||
arch.halt();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Freestanding has no OS to receive a panic. Emit it to every output sink, drop a
|
|
||||||
/// POST code + a persistent breadcrumb (so a post-mortem can recover it even with
|
|
||||||
/// no live console), then halt. Assumes no console — the sinks self-guard.
|
|
||||||
pub const panic = std.debug.FullPanic(struct {
|
|
||||||
fn panic(msg: []const u8, first_trace_addr: ?usize) noreturn {
|
|
||||||
_ = first_trace_addr;
|
|
||||||
log.checkpoint(cp_panic);
|
|
||||||
log.recordPanic(msg);
|
|
||||||
status("\nKERNEL PANIC: ");
|
|
||||||
status(msg);
|
|
||||||
status("\n");
|
|
||||||
arch.halt();
|
|
||||||
}
|
|
||||||
}.panic);
|
|
||||||
@@ -1,251 +0,0 @@
|
|||||||
//! The scheduler: fixed-priority preemptive multitasking.
|
|
||||||
//!
|
|
||||||
//! Tasks are kernel threads (ring 0, each with its own stack). The **highest-
|
|
||||||
//! priority ready task always runs**; within a priority level, tasks round-robin.
|
|
||||||
//! Selection is O(1) — a bitmap of non-empty priority levels plus a FIFO queue per
|
|
||||||
//! level — which keeps scheduling deterministic, as a real-time kernel needs (see
|
|
||||||
//! docs/vision.md).
|
|
||||||
//!
|
|
||||||
//! Switching happens both cooperatively (`yield`) and preemptively (the timer
|
|
||||||
//! calls `tick`). See docs/scheduling.md for the interrupt-flag discipline that
|
|
||||||
//! makes those two paths coexist.
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
const arch = @import("arch");
|
|
||||||
const heap = @import("heap.zig");
|
|
||||||
|
|
||||||
/// Priority level: 0 (lowest) .. 7 (highest). 8 levels total.
|
|
||||||
pub const Priority = u3;
|
|
||||||
const num_priorities = 8;
|
|
||||||
|
|
||||||
const stack_size = 16 * 1024; // each task's kernel stack is 16 KiB
|
|
||||||
const max_tasks = 16; // the maximum number of tasks alive at once is 16 in a static sized pool
|
|
||||||
|
|
||||||
const State = enum { free, ready, running, blocked };
|
|
||||||
|
|
||||||
const Task = struct {
|
|
||||||
id: u32 = 0,
|
|
||||||
state: State = .free,
|
|
||||||
priority: Priority = 0,
|
|
||||||
rsp: usize = 0, // saved stack pointer, valid while not running
|
|
||||||
stack: []u8 = &.{},
|
|
||||||
wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping
|
|
||||||
next: ?*Task = null, // ready-queue link
|
|
||||||
};
|
|
||||||
|
|
||||||
var tasks = [_]Task{.{}} ** max_tasks;
|
|
||||||
var current: *Task = undefined;
|
|
||||||
var next_id: u32 = 1;
|
|
||||||
|
|
||||||
// Per-priority FIFO ready queues, and a bitmap of which levels are non-empty.
|
|
||||||
var ready_head: [num_priorities]?*Task = .{null} ** num_priorities;
|
|
||||||
var ready_tail: [num_priorities]?*Task = .{null} ** num_priorities;
|
|
||||||
var ready_bitmap: u8 = 0;
|
|
||||||
|
|
||||||
var preemption_enabled = true;
|
|
||||||
|
|
||||||
/// Register the currently-running kernel context as the first task, spawn the
|
|
||||||
/// idle task, and hook the timer for preemption.
|
|
||||||
pub fn init(boot_priority: Priority) void {
|
|
||||||
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
|
|
||||||
current = &tasks[0];
|
|
||||||
spawn(idle, 0); // lowest priority, always runnable — runs when nothing else is
|
|
||||||
arch.setTickHook(tick);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The idle task: run when every other task is blocked or sleeping. `hlt` waits
|
|
||||||
/// for the next interrupt at near-zero power (see docs/halting.md).
|
|
||||||
fn idle() void {
|
|
||||||
while (true) asm volatile ("hlt");
|
|
||||||
}
|
|
||||||
|
|
||||||
fn enqueue(t: *Task) void {
|
|
||||||
t.next = null;
|
|
||||||
const p: usize = t.priority;
|
|
||||||
if (ready_tail[p]) |tail| tail.next = t else ready_head[p] = t;
|
|
||||||
ready_tail[p] = t;
|
|
||||||
ready_bitmap |= levelBit(t.priority);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn dequeueHighest() ?*Task {
|
|
||||||
if (ready_bitmap == 0) return null;
|
|
||||||
const level: Priority = @intCast(num_priorities - 1 - @clz(ready_bitmap));
|
|
||||||
const t = ready_head[level].?;
|
|
||||||
ready_head[level] = t.next;
|
|
||||||
if (ready_head[level] == null) {
|
|
||||||
ready_tail[level] = null;
|
|
||||||
ready_bitmap &= ~levelBit(level);
|
|
||||||
}
|
|
||||||
t.next = null;
|
|
||||||
return t;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn levelBit(p: Priority) u8 {
|
|
||||||
return @as(u8, 1) << p;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Create a task that runs `entry` at `priority`. It becomes ready immediately.
|
|
||||||
pub fn spawn(entry: *const fn () void, priority: Priority) void {
|
|
||||||
const t = freeSlot() orelse @panic("sched: task table full");
|
|
||||||
const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack");
|
|
||||||
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack };
|
|
||||||
next_id += 1;
|
|
||||||
const top = @intFromPtr(stack.ptr) + stack.len;
|
|
||||||
t.rsp = arch.initTaskStack(top, @intFromPtr(entry));
|
|
||||||
enqueue(t);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn freeSlot() ?*Task {
|
|
||||||
for (&tasks) |*t| {
|
|
||||||
if (t.state == .free) return t;
|
|
||||||
}
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Pick the highest-priority ready task and switch to it. Interrupts must be
|
|
||||||
/// disabled by the caller.
|
|
||||||
fn schedule() void {
|
|
||||||
const prev = current;
|
|
||||||
if (prev.state == .running) {
|
|
||||||
prev.state = .ready;
|
|
||||||
enqueue(prev); // back of its level's queue (round-robin)
|
|
||||||
}
|
|
||||||
const next = dequeueHighest() orelse {
|
|
||||||
prev.state = .running; // nothing else ready — keep running
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
next.state = .running;
|
|
||||||
current = next;
|
|
||||||
if (next != prev) arch.switchContext(&prev.rsp, next.rsp);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Voluntarily give up the CPU to the next ready task.
|
|
||||||
pub fn yield() void {
|
|
||||||
const flags = arch.saveInterrupts();
|
|
||||||
schedule();
|
|
||||||
arch.restoreInterrupts(flags);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Block the current task for `ms` milliseconds, then let it become runnable
|
|
||||||
/// again. The idle task (or other work) runs in the meantime.
|
|
||||||
pub fn sleep(ms: u64) void {
|
|
||||||
const flags = arch.saveInterrupts();
|
|
||||||
current.wake_at = arch.millis() + ms;
|
|
||||||
current.state = .blocked;
|
|
||||||
schedule(); // current is blocked, so schedule() won't re-enqueue it
|
|
||||||
arch.restoreInterrupts(flags);
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- event-based blocking -------------------------------------------------
|
|
||||||
//
|
|
||||||
// A WaitQueue is a set of tasks blocked waiting for something (a resource, a
|
|
||||||
// message). Tasks link into it through the same `next` field the ready queues
|
|
||||||
// use — a task is in exactly one queue at a time. These are the primitive locks,
|
|
||||||
// semaphores and IPC channels are built on.
|
|
||||||
|
|
||||||
pub const WaitQueue = struct {
|
|
||||||
head: ?*Task = null,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Block the current task on `wq` and switch away. Precondition: interrupts are
|
|
||||||
/// disabled (the caller holds them, so a condition can be checked and the block
|
|
||||||
/// committed atomically). On return — when woken — interrupts are still disabled.
|
|
||||||
pub fn waitLocked(wq: *WaitQueue) void {
|
|
||||||
current.state = .blocked;
|
|
||||||
current.next = wq.head;
|
|
||||||
wq.head = current;
|
|
||||||
schedule();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Move the highest-priority waiter on `wq` (if any) to the ready queue.
|
|
||||||
/// Precondition: interrupts disabled. Does not preempt — the caller decides.
|
|
||||||
pub fn wakeLocked(wq: *WaitQueue) void {
|
|
||||||
// Find the highest-priority waiter (bounded scan) and unlink it.
|
|
||||||
var best_prev: ?*Task = null;
|
|
||||||
var best: ?*Task = null;
|
|
||||||
var prev: ?*Task = null;
|
|
||||||
var cur = wq.head;
|
|
||||||
while (cur) |t| : ({
|
|
||||||
prev = t;
|
|
||||||
cur = t.next;
|
|
||||||
}) {
|
|
||||||
if (best == null or t.priority > best.?.priority) {
|
|
||||||
best = t;
|
|
||||||
best_prev = prev;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
const t = best orelse return;
|
|
||||||
if (best_prev) |p| p.next = t.next else wq.head = t.next;
|
|
||||||
t.state = .ready;
|
|
||||||
enqueue(t);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Block on `wq` (a self-contained critical section).
|
|
||||||
pub fn wait(wq: *WaitQueue) void {
|
|
||||||
const flags = arch.saveInterrupts();
|
|
||||||
waitLocked(wq);
|
|
||||||
arch.restoreInterrupts(flags);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Wake the highest-priority waiter on `wq`, preempting if it outranks us.
|
|
||||||
pub fn wake(wq: *WaitQueue) void {
|
|
||||||
const flags = arch.saveInterrupts();
|
|
||||||
wakeLocked(wq);
|
|
||||||
// If a higher-priority task is now ready, run it immediately.
|
|
||||||
if (highestReadyPriority()) |p| {
|
|
||||||
if (p > current.priority) schedule();
|
|
||||||
}
|
|
||||||
arch.restoreInterrupts(flags);
|
|
||||||
}
|
|
||||||
|
|
||||||
fn highestReadyPriority() ?Priority {
|
|
||||||
if (ready_bitmap == 0) return null;
|
|
||||||
return @intCast(num_priorities - 1 - @clz(ready_bitmap));
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Wake any sleeping task whose deadline has passed. Bounded by the task count,
|
|
||||||
/// so it stays deterministic. Called from the timer tick (interrupts disabled).
|
|
||||||
fn wakeExpired() void {
|
|
||||||
const now = arch.millis();
|
|
||||||
for (&tasks) |*t| {
|
|
||||||
if (t.state == .blocked and t.wake_at != 0 and now >= t.wake_at) {
|
|
||||||
t.wake_at = 0;
|
|
||||||
t.state = .ready;
|
|
||||||
enqueue(t);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Called from the timer interrupt (interrupts already disabled): wake due
|
|
||||||
/// sleepers, then preempt.
|
|
||||||
pub fn tick() void {
|
|
||||||
wakeExpired();
|
|
||||||
if (preemption_enabled) schedule();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Enable or disable timer-driven preemption (cooperative-only when off).
|
|
||||||
pub fn setPreemption(enabled: bool) void {
|
|
||||||
preemption_enabled = enabled;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// End the current task and switch away for good; never returns. The task's stack
|
|
||||||
/// is leaked for now (no reaper yet).
|
|
||||||
pub fn exit() noreturn {
|
|
||||||
arch.disableInterrupts();
|
|
||||||
current.state = .free;
|
|
||||||
const next = dequeueHighest() orelse @panic("sched: no task left to run");
|
|
||||||
next.state = .running;
|
|
||||||
current = next;
|
|
||||||
var discard: usize = 0;
|
|
||||||
arch.switchContext(&discard, next.rsp);
|
|
||||||
unreachable;
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn currentId() u32 {
|
|
||||||
return current.id;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Change the running task's priority (takes effect next time it's enqueued).
|
|
||||||
pub fn setPriority(p: Priority) void {
|
|
||||||
current.priority = p;
|
|
||||||
}
|
|
||||||
@@ -1,491 +0,0 @@
|
|||||||
//! In-kernel test cases, run at the end of bring-up when the kernel is built with
|
|
||||||
//! `-Dtest-case=<name>`. Each case writes structured markers to the serial port
|
|
||||||
//! that the QEMU harness (test/qemu_test.py) asserts on:
|
|
||||||
//!
|
|
||||||
//! [PASS]/[FAIL] <check> per assertion
|
|
||||||
//! DANOS-TEST-RESULT: PASS|FAIL overall, for non-faulting cases
|
|
||||||
//!
|
|
||||||
//! Faulting cases (fault-ud, fault-pf, fault-df) deliberately don't return a
|
|
||||||
//! result line — they trigger a CPU exception, and the harness asserts on the
|
|
||||||
//! exception report the handler prints (which also reaches serial).
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
const danos = @import("danos");
|
|
||||||
const arch = @import("arch");
|
|
||||||
const platform = @import("platform");
|
|
||||||
const pmm = @import("pmm.zig");
|
|
||||||
const heap = @import("heap.zig");
|
|
||||||
const sched = @import("scheduler.zig");
|
|
||||||
const ipc = @import("ipc.zig");
|
|
||||||
|
|
||||||
/// Formatted write straight to serial, independent of the framebuffer console.
|
|
||||||
fn log(comptime fmt: []const u8, args: anytype) void {
|
|
||||||
var buf: [128]u8 = undefined;
|
|
||||||
arch.serialWrite(std.fmt.bufPrint(&buf, fmt, args) catch return);
|
|
||||||
}
|
|
||||||
|
|
||||||
var passed: u32 = 0;
|
|
||||||
var failed: u32 = 0;
|
|
||||||
|
|
||||||
fn check(name: []const u8, ok: bool) void {
|
|
||||||
if (ok) {
|
|
||||||
passed += 1;
|
|
||||||
log("[PASS] {s}\n", .{name});
|
|
||||||
} else {
|
|
||||||
failed += 1;
|
|
||||||
log("[FAIL] {s}\n", .{name});
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Emit the overall result line the harness matches, then the done sentinel.
|
|
||||||
fn result() void {
|
|
||||||
log("DANOS-TEST-RESULT: {s} ({d} passed, {d} failed)\n", .{
|
|
||||||
if (failed == 0) "PASS" else "FAIL",
|
|
||||||
passed,
|
|
||||||
failed,
|
|
||||||
});
|
|
||||||
log("DANOS-TEST-DONE\n", .{});
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn run(case: []const u8, boot_info: *const BootInfo) void {
|
|
||||||
if (eql(case, "smoke")) {
|
|
||||||
smoke(boot_info);
|
|
||||||
} else if (eql(case, "timer")) {
|
|
||||||
timer();
|
|
||||||
} else if (eql(case, "clock")) {
|
|
||||||
clock();
|
|
||||||
} else if (eql(case, "vmm")) {
|
|
||||||
vmm();
|
|
||||||
} else if (eql(case, "heap")) {
|
|
||||||
heapTest();
|
|
||||||
} else if (eql(case, "sched")) {
|
|
||||||
schedTest();
|
|
||||||
} else if (eql(case, "priority")) {
|
|
||||||
priorityTest();
|
|
||||||
} else if (eql(case, "sleep")) {
|
|
||||||
sleepTest();
|
|
||||||
} else if (eql(case, "event")) {
|
|
||||||
eventTest();
|
|
||||||
} else if (eql(case, "ipc")) {
|
|
||||||
ipcTest();
|
|
||||||
} else if (eql(case, "fault-ud")) {
|
|
||||||
faultInvalidOpcode();
|
|
||||||
} else if (eql(case, "fault-pf")) {
|
|
||||||
faultPageFault();
|
|
||||||
} else if (eql(case, "fault-df")) {
|
|
||||||
faultDoubleFault();
|
|
||||||
} else if (eql(case, "fault-nx")) {
|
|
||||||
faultNoExecute();
|
|
||||||
} else if (eql(case, "fault-null")) {
|
|
||||||
faultNull();
|
|
||||||
} else if (eql(case, "poweroff")) {
|
|
||||||
powerTest(.off);
|
|
||||||
} else if (eql(case, "reboot")) {
|
|
||||||
powerTest(.reboot);
|
|
||||||
} else {
|
|
||||||
log("DANOS-TEST-RESULT: FAIL (unknown case '{s}')\n", .{case});
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn platformHal() platform.Hal {
|
|
||||||
return .{
|
|
||||||
.mapMmio = arch.mapPage,
|
|
||||||
.pioRead = arch.pioRead,
|
|
||||||
.pioWrite = arch.pioWrite,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Drive an ACPI power transition. On success the machine powers off or resets,
|
|
||||||
/// so QEMU exits — the harness observes the process exit. If control returns, the
|
|
||||||
/// transition failed and we emit a FAIL result.
|
|
||||||
fn powerTest(comptime action: enum { off, reboot }) void {
|
|
||||||
const name = if (action == .off) "poweroff" else "reboot";
|
|
||||||
log("DANOS-TEST-BEGIN: {s}\n", .{name});
|
|
||||||
const hal = platformHal();
|
|
||||||
log("DANOS-POWER: attempting {s}\n", .{name});
|
|
||||||
switch (action) {
|
|
||||||
.off => platform.shutdown(hal),
|
|
||||||
.reboot => platform.reboot(hal),
|
|
||||||
}
|
|
||||||
check("power transition took effect", false);
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
const BootInfo = danos.BootInfo;
|
|
||||||
|
|
||||||
fn eql(a: []const u8, b: []const u8) bool {
|
|
||||||
return std.mem.eql(u8, a, b);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Non-destructive checks of the memory map and frame allocator.
|
|
||||||
fn smoke(boot_info: *const BootInfo) void {
|
|
||||||
log("DANOS-TEST-BEGIN: smoke\n", .{});
|
|
||||||
|
|
||||||
// The memory map has some usable RAM.
|
|
||||||
const mm = boot_info.memory_map;
|
|
||||||
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
|
||||||
var usable: u64 = 0;
|
|
||||||
for (regions) |r| {
|
|
||||||
if (r.kind == .usable) usable += r.pages;
|
|
||||||
}
|
|
||||||
check("memory map reports usable RAM", usable > 0);
|
|
||||||
|
|
||||||
// The frame allocator hands out distinct, page-aligned frames.
|
|
||||||
const a = pmm.alloc();
|
|
||||||
const b = pmm.alloc();
|
|
||||||
check("alloc returns a frame", a != null);
|
|
||||||
check("alloc returns distinct frames", a != null and b != null and a.? != b.?);
|
|
||||||
check("frames are page-aligned", (a orelse 1) % danos.page_size == 0);
|
|
||||||
|
|
||||||
// Freeing restores the count.
|
|
||||||
const before = pmm.stats().free_frames;
|
|
||||||
if (a) |p| pmm.free(p);
|
|
||||||
if (b) |p| pmm.free(p);
|
|
||||||
check("free returns frames to the pool", pmm.stats().free_frames == before + 2);
|
|
||||||
|
|
||||||
// Paging is active on our own tables (CR3 is non-zero and page-aligned).
|
|
||||||
const cr3 = arch.readCr3();
|
|
||||||
check("paging active (CR3 set)", cr3 != 0 and cr3 % danos.page_size == 0);
|
|
||||||
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Verify device interrupts fire and return: the timer tick counter must advance
|
|
||||||
/// on its own. Interrupts are already enabled by kmain before tests run.
|
|
||||||
fn timer() void {
|
|
||||||
log("DANOS-TEST-BEGIN: timer\n", .{});
|
|
||||||
const start = arch.ticks();
|
|
||||||
// Busy-wait for the counter to advance. arch.ticks() is a volatile load, so
|
|
||||||
// the compiler re-reads it each iteration and sees the interrupt's update.
|
|
||||||
// The cap is only a safety net; the harness timeout is the real backstop.
|
|
||||||
var spins: u64 = 0;
|
|
||||||
while (arch.ticks() == start and spins < 5_000_000_000) spins +%= 1;
|
|
||||||
check("timer interrupts advance the tick count", arch.ticks() > start);
|
|
||||||
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Verify the on-demand VMM: map a fresh frame at an unused virtual address, and
|
|
||||||
/// check it's writable and reads back.
|
|
||||||
fn vmm() void {
|
|
||||||
log("DANOS-TEST-BEGIN: vmm\n", .{});
|
|
||||||
const frame = pmm.alloc();
|
|
||||||
check("frame available to map", frame != null);
|
|
||||||
if (frame) |phys| {
|
|
||||||
var virt: u64 = 0x0000_4000_0000_0000; // canonical, well clear of everything mapped
|
|
||||||
arch.mapPage(virt, phys, true);
|
|
||||||
const p: *volatile u64 = @ptrFromInt(virt);
|
|
||||||
p.* = 0xdead_c0de_cafe_babe;
|
|
||||||
check("mapped page is writable and reads back", p.* == 0xdead_c0de_cafe_babe);
|
|
||||||
arch.unmapPage(virt);
|
|
||||||
pmm.free(phys);
|
|
||||||
virt += 0;
|
|
||||||
}
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Exercise the kernel heap: basic alloc/write/free, reuse, growth beyond the
|
|
||||||
/// initial region, and a std container backed by it.
|
|
||||||
fn heapTest() void {
|
|
||||||
log("DANOS-TEST-BEGIN: heap\n", .{});
|
|
||||||
const a = heap.allocator();
|
|
||||||
|
|
||||||
// Allocate, write a pattern, read it back, free.
|
|
||||||
const buf = a.alloc(u8, 4096) catch null;
|
|
||||||
check("alloc 4096 bytes", buf != null);
|
|
||||||
if (buf) |b| {
|
|
||||||
@memset(b, 0xAB);
|
|
||||||
check("heap memory is writable and reads back", b[0] == 0xAB and b[4095] == 0xAB);
|
|
||||||
a.free(b);
|
|
||||||
}
|
|
||||||
|
|
||||||
// Freeing then re-allocating the same size should reuse the block.
|
|
||||||
const p1 = a.alloc(u64, 8) catch null;
|
|
||||||
const addr1 = if (p1) |p| @intFromPtr(p.ptr) else 0;
|
|
||||||
if (p1) |p| a.free(p);
|
|
||||||
const p2 = a.alloc(u64, 8) catch null;
|
|
||||||
const addr2 = if (p2) |p| @intFromPtr(p.ptr) else 0;
|
|
||||||
check("freed block is reused", addr1 != 0 and addr1 == addr2);
|
|
||||||
if (p2) |p| a.free(p);
|
|
||||||
|
|
||||||
// Force growth past the initial page and check every block is usable.
|
|
||||||
var blocks: [64]?[]u8 = .{null} ** 64;
|
|
||||||
var ok = true;
|
|
||||||
for (&blocks, 0..) |*slot, i| {
|
|
||||||
const b = a.alloc(u8, 4096) catch null;
|
|
||||||
slot.* = b;
|
|
||||||
if (b) |bb| @memset(bb, @intCast(i & 0xff)) else {
|
|
||||||
ok = false;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for (blocks, 0..) |slot, i| {
|
|
||||||
if (slot) |bb| {
|
|
||||||
if (bb[0] != @as(u8, @intCast(i & 0xff)) or bb[4095] != @as(u8, @intCast(i & 0xff))) ok = false;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
check("many allocations (heap growth) stay valid", ok);
|
|
||||||
for (blocks) |slot| {
|
|
||||||
if (slot) |bb| a.free(bb);
|
|
||||||
}
|
|
||||||
|
|
||||||
// A std container backed by the kernel heap.
|
|
||||||
var list: std.ArrayList(u32) = .empty;
|
|
||||||
var sum: u64 = 0;
|
|
||||||
var expected: u64 = 0;
|
|
||||||
var i: u32 = 0;
|
|
||||||
var list_ok = true;
|
|
||||||
while (i < 1000) : (i += 1) {
|
|
||||||
list.append(a, i) catch {
|
|
||||||
list_ok = false;
|
|
||||||
};
|
|
||||||
expected += i;
|
|
||||||
}
|
|
||||||
for (list.items) |v| sum += v;
|
|
||||||
list.deinit(a);
|
|
||||||
check("std.ArrayList on the kernel heap", list_ok and sum == expected);
|
|
||||||
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Verify the calibrated clocks: sane measured frequencies, monotonic uptime that
|
|
||||||
/// advances with real ticks, and — the point of the TSC clock — nanosecond
|
|
||||||
/// resolution far finer than the 1 ms tick, with the unit functions consistent.
|
|
||||||
fn clock() void {
|
|
||||||
log("DANOS-TEST-BEGIN: clock\n", .{});
|
|
||||||
|
|
||||||
const lapic = arch.lapicHz();
|
|
||||||
check("LAPIC frequency measured", lapic > 1_000_000 and lapic < 100_000_000_000);
|
|
||||||
const tsc = arch.tscHz();
|
|
||||||
check("TSC frequency measured", tsc > 100_000_000 and tsc < 100_000_000_000);
|
|
||||||
|
|
||||||
// Uptime advances over ~5 real ticks (1000 Hz => 1 tick == 1 ms).
|
|
||||||
const start_ticks = arch.ticks();
|
|
||||||
const start_ms = arch.millis();
|
|
||||||
var spins: u64 = 0;
|
|
||||||
while (arch.ticks() < start_ticks + 5 and spins < 5_000_000_000) spins +%= 1;
|
|
||||||
const elapsed_ms = arch.millis() - start_ms;
|
|
||||||
check("uptime advances with ticks", elapsed_ms >= 5 and elapsed_ms < 100);
|
|
||||||
|
|
||||||
// Sub-millisecond resolution: spin until nanos() first advances, then confirm
|
|
||||||
// that first step happened within a millisecond — so nanos() resolves finer
|
|
||||||
// than the 1 ms tick (a tick clock's smallest step *is* 1 ms). Spinning to the
|
|
||||||
// first change is robust to QEMU's coarse TSC update granularity.
|
|
||||||
const n1 = arch.nanos();
|
|
||||||
var s2: u64 = 0;
|
|
||||||
while (arch.nanos() == n1 and s2 < 10_000_000) s2 +%= 1;
|
|
||||||
const n2 = arch.nanos();
|
|
||||||
check("nanos() has sub-millisecond resolution", n2 > n1 and (n2 - n1) < 1_000_000);
|
|
||||||
|
|
||||||
// The unit functions agree (within rounding).
|
|
||||||
const ns = arch.nanos();
|
|
||||||
check("nanos/micros/millis are consistent", diffWithin(arch.micros(), ns / 1000, 1000) and diffWithin(arch.millis(), ns / 1_000_000, 2));
|
|
||||||
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
fn diffWithin(a: u64, b: u64, tol: u64) bool {
|
|
||||||
return if (a > b) a - b <= tol else b - a <= tol;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- scheduler tests ------------------------------------------------------
|
|
||||||
|
|
||||||
var counters = [_]u64{0} ** 3;
|
|
||||||
|
|
||||||
fn spin0() void {
|
|
||||||
const p: *volatile u64 = &counters[0];
|
|
||||||
while (true) p.* = p.* +% 1;
|
|
||||||
}
|
|
||||||
fn spin1() void {
|
|
||||||
const p: *volatile u64 = &counters[1];
|
|
||||||
while (true) p.* = p.* +% 1;
|
|
||||||
}
|
|
||||||
fn spin2() void {
|
|
||||||
const p: *volatile u64 = &counters[2];
|
|
||||||
while (true) p.* = p.* +% 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Preemption: spawn three tasks that busy-loop *without* yielding. If they all
|
|
||||||
/// make progress, the timer must be preempting between them (and the context
|
|
||||||
/// switch works) — because nothing yields voluntarily.
|
|
||||||
fn schedTest() void {
|
|
||||||
log("DANOS-TEST-BEGIN: sched\n", .{});
|
|
||||||
counters = .{ 0, 0, 0 };
|
|
||||||
sched.spawn(spin0, 4);
|
|
||||||
sched.spawn(spin1, 4);
|
|
||||||
sched.spawn(spin2, 4);
|
|
||||||
|
|
||||||
const c0: *volatile u64 = &counters[0];
|
|
||||||
const c1: *volatile u64 = &counters[1];
|
|
||||||
const c2: *volatile u64 = &counters[2];
|
|
||||||
var spins: u64 = 0;
|
|
||||||
while ((c0.* == 0 or c1.* == 0 or c2.* == 0) and spins < 5_000_000_000) spins +%= 1;
|
|
||||||
|
|
||||||
check("all three non-yielding tasks made progress (preemption)", c0.* > 0 and c1.* > 0 and c2.* > 0);
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
var run_order = [_]u8{0} ** 4;
|
|
||||||
var run_n: usize = 0;
|
|
||||||
|
|
||||||
fn recordExit(priority: u8) void {
|
|
||||||
run_order[run_n] = priority;
|
|
||||||
run_n += 1;
|
|
||||||
sched.exit();
|
|
||||||
}
|
|
||||||
fn taskHigh() void {
|
|
||||||
recordExit(6);
|
|
||||||
}
|
|
||||||
fn taskMid() void {
|
|
||||||
recordExit(4);
|
|
||||||
}
|
|
||||||
fn taskLow() void {
|
|
||||||
recordExit(2);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Fixed priority: with preemption off (deterministic), spawn tasks at three
|
|
||||||
/// priorities and let them run cooperatively. They must run highest-first.
|
|
||||||
fn priorityTest() void {
|
|
||||||
log("DANOS-TEST-BEGIN: priority\n", .{});
|
|
||||||
sched.setPreemption(false);
|
|
||||||
sched.setPriority(1); // above the idle task (0), below the workers — runs last
|
|
||||||
run_n = 0;
|
|
||||||
|
|
||||||
sched.spawn(taskLow, 2);
|
|
||||||
sched.spawn(taskMid, 4);
|
|
||||||
sched.spawn(taskHigh, 6);
|
|
||||||
|
|
||||||
while (run_n < 3) sched.yield(); // regain control only once the workers are done
|
|
||||||
|
|
||||||
check("tasks ran highest-priority first", run_order[0] == 6 and run_order[1] == 4 and run_order[2] == 2);
|
|
||||||
|
|
||||||
sched.setPriority(4);
|
|
||||||
sched.setPreemption(true);
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
var event_wq: sched.WaitQueue = .{};
|
|
||||||
var event_stage: u32 = 0;
|
|
||||||
|
|
||||||
fn eventWaiter() void {
|
|
||||||
event_stage = 1; // reached the wait
|
|
||||||
sched.wait(&event_wq); // block until woken
|
|
||||||
event_stage = 3; // woken and resumed
|
|
||||||
sched.exit();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Event-based blocking: a task blocks on a wait queue and is woken. The waiter is
|
|
||||||
/// higher priority, so waking it preempts us and it runs to completion at once.
|
|
||||||
fn eventTest() void {
|
|
||||||
log("DANOS-TEST-BEGIN: event\n", .{});
|
|
||||||
event_stage = 0;
|
|
||||||
sched.spawn(eventWaiter, 6); // higher priority than this task (4)
|
|
||||||
|
|
||||||
var spins: u64 = 0;
|
|
||||||
while (event_stage != 1 and spins < 1_000_000_000) : (spins += 1) sched.yield();
|
|
||||||
check("waiter reached the wait and blocked", event_stage == 1);
|
|
||||||
|
|
||||||
sched.wake(&event_wq);
|
|
||||||
check("wake resumed the blocked waiter (preempting)", event_stage == 3);
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
var channel: ipc.Channel(u64, 4) = .{};
|
|
||||||
var recv_sum: u64 = 0;
|
|
||||||
var recv_count: u64 = 0;
|
|
||||||
|
|
||||||
fn producer() void {
|
|
||||||
var i: u64 = 1;
|
|
||||||
while (i <= 100) : (i += 1) channel.send(i);
|
|
||||||
sched.exit();
|
|
||||||
}
|
|
||||||
fn consumer() void {
|
|
||||||
var n: u64 = 0;
|
|
||||||
while (n < 100) : (n += 1) {
|
|
||||||
recv_sum += channel.recv();
|
|
||||||
recv_count += 1;
|
|
||||||
}
|
|
||||||
sched.exit();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// IPC: a producer and consumer pass 100 messages through a 4-slot channel. The
|
|
||||||
/// small buffer forces the channel full and empty repeatedly, exercising both the
|
|
||||||
/// blocking-send and blocking-recv paths. The messages must arrive intact.
|
|
||||||
fn ipcTest() void {
|
|
||||||
log("DANOS-TEST-BEGIN: ipc\n", .{});
|
|
||||||
channel = .{};
|
|
||||||
recv_sum = 0;
|
|
||||||
recv_count = 0;
|
|
||||||
sched.spawn(consumer, 5); // above this task (4) so they run and we observe after
|
|
||||||
sched.spawn(producer, 5);
|
|
||||||
|
|
||||||
var spins: u64 = 0;
|
|
||||||
while (recv_count < 100 and spins < 2_000_000_000) : (spins += 1) sched.yield();
|
|
||||||
|
|
||||||
check("all 100 messages received", recv_count == 100);
|
|
||||||
check("messages arrived intact (sum 1..100 == 5050)", recv_sum == 5050);
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Blocking: sleep(50) should block this task for about 50 ms (measured on the
|
|
||||||
/// calibrated clock) — not busy-wait — while the idle task runs.
|
|
||||||
fn sleepTest() void {
|
|
||||||
log("DANOS-TEST-BEGIN: sleep\n", .{});
|
|
||||||
const t0 = arch.millis();
|
|
||||||
sched.sleep(50);
|
|
||||||
const elapsed = arch.millis() - t0;
|
|
||||||
check("sleep(50) blocked for ~50 ms", elapsed >= 50 and elapsed <= 70);
|
|
||||||
result();
|
|
||||||
}
|
|
||||||
|
|
||||||
fn faultInvalidOpcode() void {
|
|
||||||
log("DANOS-TEST-BEGIN: fault-ud\n", .{});
|
|
||||||
asm volatile ("ud2");
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Verify NX: fetching an instruction from a data page (mapped no-execute) faults.
|
|
||||||
fn faultNoExecute() void {
|
|
||||||
log("DANOS-TEST-BEGIN: fault-nx\n", .{});
|
|
||||||
var scratch: u64 = 0xC3; // a lone `ret` — harmless if NX somehow let it run
|
|
||||||
const f: *const fn () void = @ptrFromInt(@intFromPtr(&scratch));
|
|
||||||
f(); // instruction fetch from an NX page -> #PF before it executes
|
|
||||||
log("DANOS-TEST-RESULT: FAIL (NX not enforced)\n", .{});
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Verify the null guard: dereferencing address 0 (page 0 left unmapped) faults.
|
|
||||||
fn faultNull() void {
|
|
||||||
log("DANOS-TEST-BEGIN: fault-null\n", .{});
|
|
||||||
// Launder the address through empty asm so the compiler no longer knows it's
|
|
||||||
// 0 (otherwise it folds a null-pointer safety panic instead of doing the real
|
|
||||||
// access). `allowzero` skips the same null check on the cast. The write then
|
|
||||||
// hits the unmapped page 0 and takes a real hardware #PF.
|
|
||||||
var addr: u64 = 0;
|
|
||||||
addr = asm ("" : [ret] "=r" (-> u64) : [in] "0" (addr));
|
|
||||||
const p: *allowzero volatile u64 = @ptrFromInt(addr);
|
|
||||||
p.* = 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn faultPageFault() void {
|
|
||||||
log("DANOS-TEST-BEGIN: fault-pf\n", .{});
|
|
||||||
// Runtime address so the backend emits a register store (not a `mov moffs`,
|
|
||||||
// which the self-hosted x86_64 backend can't encode).
|
|
||||||
var addr: u64 = 0xdeadbeef000; // well above all mapped RAM
|
|
||||||
const p: *volatile u64 = @ptrFromInt(addr);
|
|
||||||
p.* = 1;
|
|
||||||
addr += 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn faultDoubleFault() void {
|
|
||||||
log("DANOS-TEST-BEGIN: fault-df\n", .{});
|
|
||||||
arch.disableInterrupts(); // so only the ud2 delivery (not a timer tick) triggers the #DF
|
|
||||||
// Point RSP at unmapped memory, then fault: the CPU can't push the fault
|
|
||||||
// frame, which escalates to #DF — survivable only because #DF runs on IST1.
|
|
||||||
var bad_sp: u64 = 0x5000000000;
|
|
||||||
asm volatile (
|
|
||||||
\\mov %[sp], %%rsp
|
|
||||||
\\ud2
|
|
||||||
:
|
|
||||||
: [sp] "r" (bad_sp),
|
|
||||||
: .{ .memory = true }
|
|
||||||
);
|
|
||||||
bad_sp += 0;
|
|
||||||
}
|
|
||||||
-112
@@ -1,112 +0,0 @@
|
|||||||
//! Shared definitions that form the contract between a bootloader
|
|
||||||
//! (src/boot/, e.g. efi.zig built as BOOTX64.efi) and the kernel (src/kernel/main.zig).
|
|
||||||
//!
|
|
||||||
//! Both binaries import this as the "danos" module, so the handoff layout is
|
|
||||||
//! defined in exactly one place.
|
|
||||||
|
|
||||||
const std = @import("std");
|
|
||||||
|
|
||||||
/// Calling convention for the bootloader→kernel jump. Pinned to SysV so it does
|
|
||||||
/// not depend on each binary's target default: the UEFI bootloader's C
|
|
||||||
/// convention is Microsoft x64 (first arg in RCX), the freestanding kernel's is
|
|
||||||
/// SysV (first arg in RDI). Both reference this to agree on where `*BootInfo`
|
|
||||||
/// is passed.
|
|
||||||
pub const kernel_abi: std.builtin.CallingConvention = .{ .x86_64_sysv = .{} };
|
|
||||||
|
|
||||||
/// Pixel byte order of the linear framebuffer the firmware handed us.
|
|
||||||
pub const PixelFormat = enum(u32) {
|
|
||||||
/// Byte 0 = Red, 1 = Green, 2 = Blue, 3 = reserved.
|
|
||||||
rgbx,
|
|
||||||
/// Byte 0 = Blue, 1 = Green, 2 = Red, 3 = reserved.
|
|
||||||
bgrx,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// A linear framebuffer: `width`x`height` pixels, each a 32-bit value, with
|
|
||||||
/// `pitch` bytes between the start of one row and the next (which may be larger
|
|
||||||
/// than `width * 4` due to hardware padding).
|
|
||||||
///
|
|
||||||
/// A `base` of 0 means **no framebuffer** — the firmware exposed no Graphics
|
|
||||||
/// Output Protocol (a headless server, say). The kernel must treat on-screen
|
|
||||||
/// output as optional and never assume a framebuffer exists.
|
|
||||||
pub const Framebuffer = extern struct {
|
|
||||||
base: usize, // the memory address where pixel data starts (0 = none)
|
|
||||||
width: u32, // visible pixels per row (e.g. 1920)
|
|
||||||
height: u32, // visible rows (e.g. 1080)
|
|
||||||
pitch: u32, // bytes from the start of one row to the start of the next
|
|
||||||
format: PixelFormat,
|
|
||||||
|
|
||||||
/// Whether a usable framebuffer was handed over.
|
|
||||||
pub fn present(self: Framebuffer) bool {
|
|
||||||
return self.base != 0 and self.width != 0 and self.height != 0;
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Page size the memory map is measured in. 4 KiB on every architecture danos
|
|
||||||
/// targets so far.
|
|
||||||
pub const page_size = 4096;
|
|
||||||
|
|
||||||
/// danos's own classification of a span of physical memory — deliberately not
|
|
||||||
/// UEFI's vocabulary. Each boot path (UEFI now, device tree later) translates its
|
|
||||||
/// native memory description into these kinds, so the kernel never learns what
|
|
||||||
/// booted it. [[arch]] keeps the same discipline for CPU code.
|
|
||||||
pub const MemoryKind = enum(u32) {
|
|
||||||
/// Free RAM the kernel may allocate. Each boot path folds its own transient
|
|
||||||
/// memory into this once it's genuinely free (e.g. the UEFI loader classifies
|
|
||||||
/// boot-services memory as usable after ExitBootServices), so the kernel never
|
|
||||||
/// has to know about boot-protocol-specific "reclaimable" states.
|
|
||||||
usable,
|
|
||||||
/// Firmware, MMIO, the kernel image, our own boot buffers, the boot stack —
|
|
||||||
/// never hand out.
|
|
||||||
reserved,
|
|
||||||
/// ACPI tables: parse, then reclaim.
|
|
||||||
acpi_tables,
|
|
||||||
/// ACPI non-volatile storage: preserve across sleep, do not allocate.
|
|
||||||
acpi_nvs,
|
|
||||||
/// Not backed by RAM: memory-mapped device registers or a reserved
|
|
||||||
/// address-space window (e.g. PCIe config space). Kept distinct from
|
|
||||||
/// `reserved` so RAM accounting doesn't count device address space.
|
|
||||||
mmio,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// One contiguous span of physical memory. Because danos defines this layout
|
|
||||||
/// itself (unlike the UEFI descriptor it's built from), `@sizeOf` is
|
|
||||||
/// authoritative — the kernel walks a plain `[]MemoryRegion`, with none of the
|
|
||||||
/// firmware's variable descriptor-stride to worry about.
|
|
||||||
pub const MemoryRegion = extern struct {
|
|
||||||
base: u64, // physical start address
|
|
||||||
pages: u64, // length in `page_size` units
|
|
||||||
kind: MemoryKind,
|
|
||||||
_pad: u32 = 0,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// The physical memory layout handed to the kernel: a pointer to an array of
|
|
||||||
/// `len` `MemoryRegion`s, in a buffer that outlives the loader.
|
|
||||||
pub const MemoryMap = extern struct {
|
|
||||||
regions: usize, // address of a `[len]MemoryRegion`
|
|
||||||
len: usize,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// One PT_LOAD segment of the kernel image, so the kernel can re-map itself with
|
|
||||||
/// correct permissions (code R+X, rodata R, data R+W+NX). `flags` are raw ELF
|
|
||||||
/// segment flags: PF_X=1, PF_W=2, PF_R=4.
|
|
||||||
pub const KernelSegment = extern struct {
|
|
||||||
virt: u64,
|
|
||||||
pages: u64,
|
|
||||||
flags: u32,
|
|
||||||
_pad: u32 = 0,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Handoff structure the bootloader fills in and passes to the kernel's
|
|
||||||
/// `_start` in RDI (the first argument under the SysV AMD64 C ABI).
|
|
||||||
pub const BootInfo = extern struct {
|
|
||||||
framebuffer: Framebuffer,
|
|
||||||
memory_map: MemoryMap,
|
|
||||||
/// The kernel's own PT_LOAD segments (it has three: text, rodata, data).
|
|
||||||
kernel_segments: [8]KernelSegment,
|
|
||||||
kernel_segment_count: u32,
|
|
||||||
/// Physical address of the ACPI RSDP the firmware exposed, or 0 if none. The
|
|
||||||
/// kernel's device layer parses the ACPI tables from here to discover hardware.
|
|
||||||
/// A device-tree boot path leaves this 0 and (later) fills a `device_tree_blob`
|
|
||||||
/// field instead, so the kernel discovers devices without knowing what booted it.
|
|
||||||
acpi_rsdp: u64 = 0,
|
|
||||||
};
|
|
||||||
+191
@@ -0,0 +1,191 @@
|
|||||||
|
//! The **private kernel ↔ runtime** ABI: the raw system_call contract — the call
|
||||||
|
//! numbers, `mmap` protection flags, the page size those calls work in, and the IPC
|
||||||
|
//! name-registry ids and notification bit. Shared by the kernel dispatcher
|
||||||
|
//! (system/kernel/process.zig) and the user-space runtime library (library/runtime/),
|
||||||
|
//! so the two can never drift.
|
||||||
|
//!
|
||||||
|
//! **Application code does not speak this.** danos programs call the `runtime` library —
|
||||||
|
//! the stable, danos-native ABI — and the runtime is the one thing that issues the
|
||||||
|
//! actual system calls (POSIX code layers over the runtime, never on this directly). It
|
||||||
|
//! is the same split as libSystem on macOS or win32 over the NT syscalls: the numbers
|
||||||
|
//! here are an implementation detail the runtime hides and may renumber, not a public
|
||||||
|
//! interface. See docs/coding-standards.md and library/runtime/.
|
||||||
|
//!
|
||||||
|
//! This is the *core* contract; the device half — `DeviceDescriptor` and friends, which
|
||||||
|
//! also cross this boundary — lives with the device sub-project as [[device-abi]]
|
||||||
|
//! (system/devices/device-abi.zig). The loader↔kernel handoff is [[boot-handoff]].
|
||||||
|
|
||||||
|
/// Page size every `mmap`/`munmap` grant and the boot memory map are measured in.
|
||||||
|
/// 4 KiB on every architecture danos targets so far. Part of the ABI because the
|
||||||
|
/// runtime aligns to it (grants are page-granular) and the kernel guarantees it.
|
||||||
|
pub const page_size = 4096;
|
||||||
|
|
||||||
|
/// The kernel system_call numbers — the single source of truth shared by the kernel
|
||||||
|
/// dispatcher (system/kernel/process.zig) and the user runtime library, so the two
|
||||||
|
/// can never drift. The set is deliberately microkernel-minimal: file/device I/O
|
||||||
|
/// is not here — it lives in user-space servers reached through the IPC calls.
|
||||||
|
/// The table grows one milestone at a time; see docs/syscall.md.
|
||||||
|
pub const SystemCall = enum(u64) {
|
||||||
|
exit = 0, // exit(code): end the calling process
|
||||||
|
yield = 1, // yield(): give up the rest of this quantum
|
||||||
|
debug_write = 2, // debug_write(ptr, len): raw bytes to the kernel log (bring-up only)
|
||||||
|
sleep = 3, // sleep(ms): block the caller for ms milliseconds
|
||||||
|
mmap = 4, // mmap(len, prot) -> base: grant zeroed, page-aligned user pages
|
||||||
|
munmap = 5, // munmap(base, len): release pages from a prior mmap
|
||||||
|
create_ipc_endpoint = 6, // create_ipc_endpoint() -> handle: a new IPC endpoint
|
||||||
|
ipc_register = 7, // ipc_register(service_id, handle): publish an endpoint by well-known id
|
||||||
|
ipc_lookup = 8, // ipc_lookup(service_id) -> handle: find a published endpoint
|
||||||
|
ipc_call = 9, // ipc_call(h, message, len, reply, cap) -> reply_len: send + block for reply
|
||||||
|
ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, receive, cap) -> receive_len (+badge in rdx)
|
||||||
|
device_enumerate = 11, // device_enumerate(buffer, maximum) -> count: snapshot the device table
|
||||||
|
device_claim = 12, // device_claim(id) -> ok: take exclusive ownership of a device
|
||||||
|
mmio_map = 13, // mmio_map(id, resource_index) -> vaddr: map a claimed device's MMIO into this AS
|
||||||
|
irq_bind = 14, // irq_bind(id, resource_index, endpoint): deliver a device IRQ as an IPC notification
|
||||||
|
irq_ack = 15, // irq_ack(id, resource_index): re-arm a bound IRQ after servicing it
|
||||||
|
device_register = 16, // device_register(parent_id, descriptor) -> id: publish a child of a device you claimed
|
||||||
|
system_spawn = 17, // system_spawn(name_ptr, name_len, arguments_ptr, arguments_len, exit_endpoint) -> child process id: start a named initial-ramdisk binary as a new ring-3 process
|
||||||
|
dma_alloc = 18, // dma_alloc(len, flags) -> vaddr (rax), paddr (rdx): contiguous, pinned, uncacheable DMA memory
|
||||||
|
dma_free = 19, // dma_free(vaddr, len) -> 0: release a prior dma_alloc
|
||||||
|
msi_bind = 20, // msi_bind(device_id, endpoint) -> address (rax), data (rdx): a per-device MSI vector for a claimed device
|
||||||
|
io_read = 21, // io_read(device_id, resource_index, offset, width) -> value: read a port in a claimed device's io_port resource
|
||||||
|
io_write = 22, // io_write(device_id, resource_index, offset, width, value) -> 0: write a port in a claimed device's io_port resource
|
||||||
|
clock = 23, // clock() -> nanoseconds since boot: a monotonic time source (for timeouts/delays)
|
||||||
|
process_enumerate = 24, // process_enumerate(buffer, maximum) -> total: snapshot the task table
|
||||||
|
process_kill = 25, // process_kill(id) -> 0/-errno: end a process this process spawned
|
||||||
|
ipc_send = 26, // ipc_send(handle, message_ptr, message_len) -> 0/-errno: post a payload to an endpoint's async queue without blocking
|
||||||
|
process_exit_reason = 27, // process_exit_reason(id) -> ExitReason/-errno: how a dead child ended (its supervisor only)
|
||||||
|
process_subscribe = 28, // process_subscribe(endpoint) -> 0/-errno: subscribe to published exit events — every death posts a notification
|
||||||
|
signal_bind = 29, // signal_bind(endpoint) -> 0/-errno: nominate the endpoint this process's signals arrive on
|
||||||
|
process_signal = 30, // process_signal(id, signal) -> 0/-errno: post a signal to a child (or to yourself)
|
||||||
|
timer_bind = 31, // timer_bind(endpoint, ms) -> 0/-errno: one-shot timer — posts a notification when ms elapse
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// How a process ended — recorded by the kernel at death, queried by the
|
||||||
|
/// supervisor with `process_exit_reason`, and the input to its restart decision
|
||||||
|
/// (docs/process-lifecycle.md): a clean exit meant to stop, a fault wants a
|
||||||
|
/// restart with backoff, killed means the supervisor did it itself. The faults
|
||||||
|
/// mirror the CPU exceptions a ring-3 process can die of; they are exit reasons,
|
||||||
|
/// never delivered to the faulting process (recovery is restart, not a handler).
|
||||||
|
pub const ExitReason = enum(u8) {
|
||||||
|
exited = 0, // returned from main / called exit
|
||||||
|
aborted = 1, // deliberate self-termination (reserved: no abort path yet)
|
||||||
|
segmentation_fault = 2, // page fault
|
||||||
|
illegal_instruction = 3, // invalid opcode
|
||||||
|
arithmetic_fault = 4, // divide error, x87 or SIMD fault
|
||||||
|
protection_fault = 5, // general protection fault
|
||||||
|
fault = 6, // any other CPU exception
|
||||||
|
killed = 7, // process_kill
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The x86 MSI message address base (`0xFEE0_0000`): a device raises an MSI by writing
|
||||||
|
/// `data` to this address, which the Local APIC turns into an interrupt at the vector
|
||||||
|
/// in `data`. The kernel returns the concrete (address, data) from `msi_bind`; this is
|
||||||
|
/// the fixed prefix, exposed so a driver's config-space programming reads clearly.
|
||||||
|
pub const msi_address_base: u64 = 0xFEE0_0000;
|
||||||
|
|
||||||
|
/// `dma_alloc` flags. `coherent` (uncacheable) is the portable default; the others are
|
||||||
|
/// opt-in for specific hardware. `write_combining` needs PAT programming (not yet — it
|
||||||
|
/// currently falls back to coherent); see docs/driver-model.md (M14).
|
||||||
|
pub const dma_coherent: u64 = 1; // strong-uncacheable — the default, the only portable one
|
||||||
|
pub const dma_write_combining: u64 = 2; // write-combining (framebuffers); needs PAT
|
||||||
|
pub const dma_below_4g: u64 = 4; // physical address must fit 32 bits (legacy DMA engines)
|
||||||
|
|
||||||
|
/// Set in the badge returned by `ipc_reply_wait` when what arrived is an
|
||||||
|
/// **asynchronous notification** (a device interrupt bound with `irq_bind`, or a
|
||||||
|
/// child-exit notice — see `notify_exit_bit`) rather than a message from a client.
|
||||||
|
/// There is no payload and no reply owed; the low bits carry the source. Shared so
|
||||||
|
/// the kernel's ISR and the driver's event loop can't disagree about which bit
|
||||||
|
/// means "the hardware spoke".
|
||||||
|
pub const notify_badge_bit: u64 = 1 << 63;
|
||||||
|
|
||||||
|
/// Set (alongside `notify_badge_bit`) in the badge of a **child-exit notification**:
|
||||||
|
/// posted to the endpoint a supervisor passed to `system_spawn` when that child ends
|
||||||
|
/// — by clean exit, by a fault, or by `process_kill`. The low bits carry the child's
|
||||||
|
/// process id, so one endpoint can supervise many children (and even share with IRQ
|
||||||
|
/// notifications, which never set this bit). The microkernel's SIGCHLD.
|
||||||
|
pub const notify_exit_bit: u64 = 1 << 62;
|
||||||
|
|
||||||
|
/// Set (alongside `notify_badge_bit`) in the badge of a **buffered message** — a payload
|
||||||
|
/// posted to an endpoint's async queue by `ipc_send`, delivered through `ipc_reply_wait`
|
||||||
|
/// like a notification (no reply owed) but carrying bytes in the receive buffer, not just
|
||||||
|
/// a badge. This is what distinguishes a payload-bearing async message from a bare IRQ /
|
||||||
|
/// child-exit notification (which sets neither this nor `notify_exit_bit`). The low bits
|
||||||
|
/// carry the sender's task id. The async counterpart of the synchronous `ipc_call`, for
|
||||||
|
/// broadcasts where a rendezvous is the wrong shape (the input service is the first user).
|
||||||
|
pub const notify_message_bit: u64 = 1 << 61;
|
||||||
|
|
||||||
|
/// Set (alongside `notify_badge_bit`) in the badge of a **signal notification** —
|
||||||
|
/// the process-lifecycle vocabulary of docs/process-lifecycle.md, delivered to the
|
||||||
|
/// endpoint the process nominated with `signal_bind`. The low bits carry the
|
||||||
|
/// coalesced pending mask (bit positions = `Signal` values): signals are
|
||||||
|
/// statements, not questions, and two pending terminates are one terminate.
|
||||||
|
pub const notify_signal_bit: u64 = 1 << 60;
|
||||||
|
|
||||||
|
/// Set (alongside `notify_badge_bit`) in the badge of a **timer notification** —
|
||||||
|
/// a one-shot `timer_bind` deadline landing. No payload bits: what to do when the
|
||||||
|
/// deadline fires is whatever the receiver armed it for (a stop-sequence
|
||||||
|
/// escalation, a restart backoff, an alarm).
|
||||||
|
pub const notify_timer_bit: u64 = 1 << 59;
|
||||||
|
|
||||||
|
/// The signal vocabulary (docs/process-lifecycle.md): POSIX's concepts, danos's
|
||||||
|
/// names, message delivery. The value is the bit position in the pending mask — a
|
||||||
|
/// private kernel/runtime detail, free to change while they ship together. Kill
|
||||||
|
/// is not here (it is `process_kill`, unhandleable by definition); faults are not
|
||||||
|
/// here (they are `ExitReason`s — recovery is restart, not a handler); liveness is
|
||||||
|
/// not here (a question, asked as the zero-length ping call, not a statement).
|
||||||
|
pub const Signal = enum(u5) {
|
||||||
|
terminate = 0, // finish up and exit (the polite half of the stop sequence)
|
||||||
|
reload = 1, // re-read configuration / re-scan
|
||||||
|
interrupt = 2, // interactive interrupt (no sender until a console exists)
|
||||||
|
quit = 3, // as interrupt, by convention more final
|
||||||
|
alarm = 4, // a timer the process armed for itself (unbuilt: no consumer yet)
|
||||||
|
user_1 = 5, // service-defined
|
||||||
|
user_2 = 6, // service-defined
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Capacity of `ProcessDescriptor.name` — matches the longest name `system_spawn`
|
||||||
|
/// accepts, so a process's recorded name (its argv[0]) is never truncated.
|
||||||
|
pub const maximum_process_name = 64;
|
||||||
|
|
||||||
|
/// What a process is doing right now, as reported by `process_enumerate`. Crosses
|
||||||
|
/// the system_call boundary as `ProcessDescriptor.state`.
|
||||||
|
pub const ProcessState = enum(u32) {
|
||||||
|
ready = 0, // runnable, waiting for a core
|
||||||
|
running = 1, // executing on a core right now
|
||||||
|
blocked = 2, // waiting (sleeping, or blocked in IPC)
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One `process_enumerate` entry — the kernel's view of a live task, kernel tasks
|
||||||
|
/// included (they carry an empty name and id 0 is the boot task). Fixed layout
|
||||||
|
/// (extern) because it crosses the kernel↔user boundary by memory copy, like
|
||||||
|
/// `DeviceDescriptor` in the device ABI.
|
||||||
|
pub const ProcessDescriptor = extern struct {
|
||||||
|
id: u32, // kernel-assigned process id; never reused (monotonic)
|
||||||
|
supervisor: u32, // id of the process that spawned it (0 = the kernel)
|
||||||
|
state: u32, // a ProcessState value
|
||||||
|
priority: u32,
|
||||||
|
name_length: u32,
|
||||||
|
name: [maximum_process_name]u8, // argv[0] at spawn; empty for kernel tasks
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Well-known IPC service ids for the bootstrap name registry (create_ipc_endpoint +
|
||||||
|
/// ipc_register/ipc_lookup). Small integers, so no string interning is needed
|
||||||
|
/// during bring-up. The VFS server registers under `vfs`; clients look it up.
|
||||||
|
pub const ServiceId = enum(u32) {
|
||||||
|
vfs = 1,
|
||||||
|
input = 2,
|
||||||
|
ps2_bus = 3, // the 8042 owner; child device drivers attach here for raw bytes
|
||||||
|
device_manager = 4, // the tree, the matcher, the supervisor (docs/device-manager.md)
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Protection flags for `mmap` (matching the usual C bit values).
|
||||||
|
pub const prot_read: u64 = 1;
|
||||||
|
pub const prot_write: u64 = 2;
|
||||||
|
pub const prot_exec: u64 = 4;
|
||||||
|
|
||||||
|
/// `send_cap` / `received_cap` sentinel meaning "no capability" on the `ipc_call` /
|
||||||
|
/// `ipc_reply_wait` cap-passing path (M13). `~0`, like `no_parent` — a real handle is
|
||||||
|
/// a small index, so it can never collide.
|
||||||
|
pub const no_cap: u64 = ~@as(u64, 0);
|
||||||
@@ -0,0 +1,156 @@
|
|||||||
|
//! The **loader ↔ kernel** contract: everything a bootloader (boot/, e.g. efi.zig
|
||||||
|
//! built as BOOTX64.efi) and the kernel (system/kernel/kernel.zig) must agree on to
|
||||||
|
//! hand control over — the handoff structures the loader fills in, plus the kernel's
|
||||||
|
//! virtual-memory layout and the physical↔virtual addressing both sides use.
|
||||||
|
//!
|
||||||
|
//! Both binaries import this as the `boot-handoff` module, so the layout is defined
|
||||||
|
//! in exactly one place. **User space never sees this** — the kernel↔user contract is
|
||||||
|
//! [[abi]] (system/abi.zig); device types are [[device-abi]] (system/devices/device-abi.zig).
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
/// Calling convention for the bootloader→kernel jump. Pinned to SystemV so it does
|
||||||
|
/// not depend on each binary's target default: the UEFI bootloader's C
|
||||||
|
/// convention is Microsoft x64 (first arg in RCX), the freestanding kernel's is
|
||||||
|
/// SystemV (first arg in RDI). Both reference this to agree on where `*BootInformation`
|
||||||
|
/// is passed.
|
||||||
|
pub const kernel_abi: std.builtin.CallingConvention = .{ .x86_64_sysv = .{} };
|
||||||
|
|
||||||
|
/// Pixel byte order of the linear framebuffer the firmware handed us.
|
||||||
|
pub const PixelFormat = enum(u32) {
|
||||||
|
/// Byte 0 = Red, 1 = Green, 2 = Blue, 3 = reserved.
|
||||||
|
rgbx,
|
||||||
|
/// Byte 0 = Blue, 1 = Green, 2 = Red, 3 = reserved.
|
||||||
|
bgrx,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A linear framebuffer: `width`x`height` pixels, each a 32-bit value, with
|
||||||
|
/// `pitch` bytes between the start of one row and the next (which may be larger
|
||||||
|
/// than `width * 4` due to hardware padding).
|
||||||
|
///
|
||||||
|
/// A `base` of 0 means **no framebuffer** — the firmware exposed no Graphics
|
||||||
|
/// Output Protocol (a headless server, say). The kernel must treat on-screen
|
||||||
|
/// output as optional and never assume a framebuffer exists.
|
||||||
|
pub const Framebuffer = extern struct {
|
||||||
|
base: usize, // the memory address where pixel data starts (0 = none)
|
||||||
|
width: u32, // visible pixels per row (e.g. 1920)
|
||||||
|
height: u32, // visible rows (e.g. 1080)
|
||||||
|
pitch: u32, // bytes from the start of one row to the start of the next
|
||||||
|
format: PixelFormat,
|
||||||
|
|
||||||
|
/// Whether a usable framebuffer was handed over.
|
||||||
|
pub fn present(self: Framebuffer) bool {
|
||||||
|
return self.base != 0 and self.width != 0 and self.height != 0;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The kernel's virtual-memory layout (higher-half). The kernel is linked at
|
||||||
|
/// `kernel_virt_base` but loaded at a low physical address; all of RAM (and the
|
||||||
|
/// device MMIO windows) is also mapped at `physmap_base + physical`, so the kernel
|
||||||
|
/// can reach any physical address by adding a constant. The low half is left
|
||||||
|
/// entirely to user space.
|
||||||
|
///
|
||||||
|
/// user image + stack : 0x0000_7000_0000_0000 (PML4[224], low half)
|
||||||
|
/// kernel heap : 0xFFFF_8000_0000_0000 (PML4[256])
|
||||||
|
/// physmap : 0xFFFF_8800_0000_0000 (PML4[272]) + physical
|
||||||
|
/// kernel image : 0xFFFF_FFFF_8000_0000 (PML4[511])
|
||||||
|
pub const physmap_base: u64 = 0xFFFF_8800_0000_0000;
|
||||||
|
pub const kernel_virt_base: u64 = 0xFFFF_FFFF_8000_0000;
|
||||||
|
|
||||||
|
/// Physical address -> its virtual address in the physmap. The single way the
|
||||||
|
/// kernel dereferences a physical address once paging is up.
|
||||||
|
///
|
||||||
|
/// **Hazard:** valid only once the (bootstrap or final) page tables are live.
|
||||||
|
/// The bootloader may use the *constant* `physmap_base` to build those tables,
|
||||||
|
/// but must not call this to dereference memory before its own CR3 is loaded —
|
||||||
|
/// it runs under the firmware's identity map, where these addresses are unmapped.
|
||||||
|
pub inline fn physicalToVirtual(physical: u64) u64 {
|
||||||
|
return physical + physmap_base;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Physmap virtual address -> physical. Inverse of `physicalToVirtual`; for producing
|
||||||
|
/// the physical address of something the kernel holds a physmap pointer to
|
||||||
|
/// (e.g. a page-table frame for CR3, a post-mortem breadcrumb's RAM location).
|
||||||
|
pub inline fn virtualToPhysical(virtual: u64) u64 {
|
||||||
|
return virtual - physmap_base;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// danos's own classification of a span of physical memory — deliberately not
|
||||||
|
/// UEFI's vocabulary. Each boot path (UEFI now, device tree later) translates its
|
||||||
|
/// native memory description into these kinds, so the kernel never learns what
|
||||||
|
/// booted it. [[architecture]] keeps the same discipline for CPU code.
|
||||||
|
pub const MemoryKind = enum(u32) {
|
||||||
|
/// Free RAM the kernel may allocate. Each boot path folds its own transient
|
||||||
|
/// memory into this once it's genuinely free (e.g. the UEFI loader classifies
|
||||||
|
/// boot-services memory as usable after ExitBootServices), so the kernel never
|
||||||
|
/// has to know about boot-protocol-specific "reclaimable" states.
|
||||||
|
usable,
|
||||||
|
/// Firmware, MMIO, the kernel image, our own boot buffers, the boot stack —
|
||||||
|
/// never hand out.
|
||||||
|
reserved,
|
||||||
|
/// ACPI tables: parse, then reclaim.
|
||||||
|
acpi_tables,
|
||||||
|
/// ACPI non-volatile storage: preserve across sleep, do not allocate.
|
||||||
|
acpi_nvs,
|
||||||
|
/// Not backed by RAM: memory-mapped device registers or a reserved
|
||||||
|
/// address-space window (e.g. PCIe configuration space). Kept distinct from
|
||||||
|
/// `reserved` so RAM accounting doesn't count device address space.
|
||||||
|
mmio,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One contiguous span of physical memory. Because danos defines this layout
|
||||||
|
/// itself (unlike the UEFI descriptor it's built from), `@sizeOf` is
|
||||||
|
/// authoritative — the kernel walks a plain `[]MemoryRegion`, with none of the
|
||||||
|
/// firmware's variable descriptor-stride to worry about.
|
||||||
|
pub const MemoryRegion = extern struct {
|
||||||
|
base: u64, // physical start address
|
||||||
|
pages: u64, // length in 4 KiB pages (the [[abi]] `page_size` unit)
|
||||||
|
kind: MemoryKind,
|
||||||
|
_pad: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The physical memory layout handed to the kernel: a pointer to an array of
|
||||||
|
/// `len` `MemoryRegion`s, in a buffer that outlives the loader.
|
||||||
|
pub const MemoryMap = extern struct {
|
||||||
|
regions: usize, // address of a `[len]MemoryRegion`
|
||||||
|
len: usize,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One PT_LOAD segment of the kernel image, so the kernel can re-map itself with
|
||||||
|
/// correct permissions (code R+X, rodata R, data R+W+NX). `flags` are raw ELF
|
||||||
|
/// segment flags: PF_X=1, PF_W=2, PF_R=4. `virtual` is the higher-half link address;
|
||||||
|
/// `physical` is where the loader actually placed the segment (they differ once the
|
||||||
|
/// kernel links high — the loader records the real load address here).
|
||||||
|
pub const KernelSegment = extern struct {
|
||||||
|
virtual: u64,
|
||||||
|
physical: u64,
|
||||||
|
pages: u64,
|
||||||
|
flags: u32,
|
||||||
|
_pad: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Handoff structure the bootloader fills in and passes to the kernel's
|
||||||
|
/// `_start` in RDI (the first argument under the SystemV AMD64 C ABI).
|
||||||
|
pub const BootInformation = extern struct {
|
||||||
|
framebuffer: Framebuffer,
|
||||||
|
memory_map: MemoryMap,
|
||||||
|
/// The kernel's own PT_LOAD segments (it has three: text, rodata, data).
|
||||||
|
kernel_segments: [8]KernelSegment,
|
||||||
|
kernel_segment_count: u32,
|
||||||
|
/// Physical address of the ACPI RSDP the firmware exposed, or 0 if none. The
|
||||||
|
/// kernel's device layer parses the ACPI tables from here to discover hardware.
|
||||||
|
/// A device-tree boot path leaves this 0 and (later) fills a `device_tree_blob`
|
||||||
|
/// field instead, so the kernel discovers devices without knowing what booted it.
|
||||||
|
acpi_rsdp: u64 = 0,
|
||||||
|
/// The raw `/system/services/init` ELF image, read off the boot volume by the loader
|
||||||
|
/// into memory that survives the handoff (classified reserved, so the kernel
|
||||||
|
/// identity-maps it and never allocates over it). 0/0 = no init found — the
|
||||||
|
/// kernel boots without user space. Grows into a full initial_ramdisk handoff later.
|
||||||
|
init_base: u64 = 0,
|
||||||
|
init_len: u64 = 0,
|
||||||
|
/// The initial_ramdisk image (a bundle of extra user binaries — the VFS server and
|
||||||
|
/// device drivers), read off the boot volume into memory that survives the
|
||||||
|
/// handoff, same as `init` above. 0/0 = no initial_ramdisk. See system/initial-ramdisk.zig.
|
||||||
|
initial_ramdisk_base: u64 = 0,
|
||||||
|
initial_ramdisk_len: u64 = 0,
|
||||||
|
};
|
||||||
@@ -0,0 +1,138 @@
|
|||||||
|
//! ACPI / PnP hardware-ID (`_HID`) names: the flat analog of pci-class.zig for
|
||||||
|
//! `acpi_device` nodes. Unlike PCI, ACPI has no class/subclass/prog-IF taxonomy — a
|
||||||
|
//! device's identity *is* its `_HID` string (`PNP0303` simply means "PS/2 keyboard"),
|
||||||
|
//! so this is a plain id <-> name registry rather than a hierarchical decoder.
|
||||||
|
//! The well-known PnP/ACPI IDs; vendor-specific ids (e.g. `QEMU0002`, `INTC1234`) have
|
||||||
|
//! no standard name and decode to nothing. Pure reference data, so it is shared by
|
||||||
|
//! kernel discovery (the device-tree dump) and any user-space driver or tool.
|
||||||
|
//!
|
||||||
|
//! Code that means a specific device names the `HardwareId` variant instead of its
|
||||||
|
//! `_HID` string — `HardwareId.ps2_keyboard.hid()` reads without a registry lookup,
|
||||||
|
//! where a bare `"PNP0303"` does not.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
/// The common standard PnP/ACPI hardware IDs, as named values. Prefix ranges hint at
|
||||||
|
/// the grouping (PNP03xx keyboards, PNP0Fxx pointing devices, PNP0Cxx ACPI
|
||||||
|
/// power/thermal, PNP0Axx buses), but there is no formal hierarchy — hence a flat
|
||||||
|
/// enum over a flat registry.
|
||||||
|
pub const HardwareId = enum {
|
||||||
|
programmable_interrupt_controller,
|
||||||
|
system_timer,
|
||||||
|
high_precision_event_timer,
|
||||||
|
dma_controller,
|
||||||
|
ps2_keyboard,
|
||||||
|
parallel_port,
|
||||||
|
ecp_parallel_port,
|
||||||
|
serial_port,
|
||||||
|
floppy_disk_controller,
|
||||||
|
system_speaker,
|
||||||
|
pci_bus,
|
||||||
|
generic_container,
|
||||||
|
/// The second id the ACPI spec assigns the same "Generic Container Device" name.
|
||||||
|
generic_container_extended,
|
||||||
|
pci_express_root_bridge,
|
||||||
|
real_time_clock,
|
||||||
|
system_board,
|
||||||
|
motherboard_reserved_resources,
|
||||||
|
math_coprocessor,
|
||||||
|
acpi_system_board,
|
||||||
|
embedded_controller,
|
||||||
|
control_method_battery,
|
||||||
|
fan,
|
||||||
|
power_button,
|
||||||
|
lid,
|
||||||
|
sleep_button,
|
||||||
|
pci_interrupt_link,
|
||||||
|
microsoft_ps2_mouse,
|
||||||
|
ps2_mouse,
|
||||||
|
ac_adapter,
|
||||||
|
processor_device,
|
||||||
|
processor_aggregator,
|
||||||
|
processor_container,
|
||||||
|
|
||||||
|
const Entry = struct { hid: []const u8, name: []const u8 };
|
||||||
|
|
||||||
|
/// The registry row for this id: its `_HID` string and human-readable name.
|
||||||
|
fn entry(self: HardwareId) Entry {
|
||||||
|
return switch (self) {
|
||||||
|
.programmable_interrupt_controller => .{ .hid = "PNP0000", .name = "Programmable Interrupt Controller (PIC)" },
|
||||||
|
.system_timer => .{ .hid = "PNP0100", .name = "System Timer (PIT)" },
|
||||||
|
.high_precision_event_timer => .{ .hid = "PNP0103", .name = "High Precision Event Timer (HPET)" },
|
||||||
|
.dma_controller => .{ .hid = "PNP0200", .name = "DMA Controller" },
|
||||||
|
.ps2_keyboard => .{ .hid = "PNP0303", .name = "PS/2 Keyboard" },
|
||||||
|
.parallel_port => .{ .hid = "PNP0400", .name = "Standard LPT Parallel Port" },
|
||||||
|
.ecp_parallel_port => .{ .hid = "PNP0401", .name = "ECP Parallel Port" },
|
||||||
|
.serial_port => .{ .hid = "PNP0501", .name = "16550A-compatible Serial Port" },
|
||||||
|
.floppy_disk_controller => .{ .hid = "PNP0700", .name = "PC Floppy Disk Controller" },
|
||||||
|
.system_speaker => .{ .hid = "PNP0800", .name = "System Speaker" },
|
||||||
|
.pci_bus => .{ .hid = "PNP0A03", .name = "PCI Bus" },
|
||||||
|
.generic_container => .{ .hid = "PNP0A05", .name = "Generic Container Device" },
|
||||||
|
.generic_container_extended => .{ .hid = "PNP0A06", .name = "Generic Container Device" },
|
||||||
|
.pci_express_root_bridge => .{ .hid = "PNP0A08", .name = "PCI Express Root Bridge" },
|
||||||
|
.real_time_clock => .{ .hid = "PNP0B00", .name = "Real-Time Clock (RTC)" },
|
||||||
|
.system_board => .{ .hid = "PNP0C01", .name = "System Board" },
|
||||||
|
.motherboard_reserved_resources => .{ .hid = "PNP0C02", .name = "Motherboard Reserved Resources" },
|
||||||
|
.math_coprocessor => .{ .hid = "PNP0C04", .name = "Math Coprocessor" },
|
||||||
|
.acpi_system_board => .{ .hid = "PNP0C08", .name = "ACPI System Board" },
|
||||||
|
.embedded_controller => .{ .hid = "PNP0C09", .name = "ACPI Embedded Controller" },
|
||||||
|
.control_method_battery => .{ .hid = "PNP0C0A", .name = "ACPI Control Method Battery" },
|
||||||
|
.fan => .{ .hid = "PNP0C0B", .name = "ACPI Fan" },
|
||||||
|
.power_button => .{ .hid = "PNP0C0C", .name = "ACPI Power Button" },
|
||||||
|
.lid => .{ .hid = "PNP0C0D", .name = "ACPI Lid" },
|
||||||
|
.sleep_button => .{ .hid = "PNP0C0E", .name = "ACPI Sleep Button" },
|
||||||
|
.pci_interrupt_link => .{ .hid = "PNP0C0F", .name = "PCI Interrupt Link Device" },
|
||||||
|
.microsoft_ps2_mouse => .{ .hid = "PNP0F03", .name = "Microsoft PS/2 Mouse" },
|
||||||
|
.ps2_mouse => .{ .hid = "PNP0F13", .name = "PS/2 Mouse" },
|
||||||
|
.ac_adapter => .{ .hid = "ACPI0003", .name = "AC Adapter" },
|
||||||
|
.processor_device => .{ .hid = "ACPI0007", .name = "Processor Device" },
|
||||||
|
.processor_aggregator => .{ .hid = "ACPI000C", .name = "Processor Aggregator" },
|
||||||
|
.processor_container => .{ .hid = "ACPI0010", .name = "Processor Container" },
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// This id's `_HID` string (e.g. `.ps2_keyboard` -> "PNP0303").
|
||||||
|
pub fn hid(self: HardwareId) []const u8 {
|
||||||
|
return self.entry().hid;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// This id's human-readable name (e.g. `.ps2_keyboard` -> "PS/2 Keyboard").
|
||||||
|
pub fn description(self: HardwareId) []const u8 {
|
||||||
|
return self.entry().name;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The named value for a `_HID` string, or null if it is not a known standard
|
||||||
|
/// id (vendor-specific ids are not in the registry).
|
||||||
|
pub fn fromHid(hid_string: []const u8) ?HardwareId {
|
||||||
|
for (std.enums.values(HardwareId)) |id| {
|
||||||
|
if (std.mem.eql(u8, id.hid(), hid_string)) return id;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The human-readable name for a `_HID` string, or "" if it is not a known standard
|
||||||
|
/// id (vendor-specific ids have no registry name — callers just print the raw HID).
|
||||||
|
pub fn description(hid: []const u8) []const u8 {
|
||||||
|
return (HardwareId.fromHid(hid) orelse return "").description();
|
||||||
|
}
|
||||||
|
|
||||||
|
test "decodes standard PnP/ACPI ids and leaves the rest alone" {
|
||||||
|
const eq = std.testing.expectEqualStrings;
|
||||||
|
try eq("PS/2 Keyboard", description("PNP0303"));
|
||||||
|
try eq("PS/2 Mouse", description("PNP0F13"));
|
||||||
|
try eq("PCI Express Root Bridge", description("PNP0A08"));
|
||||||
|
try eq("Real-Time Clock (RTC)", description("PNP0B00"));
|
||||||
|
try eq("", description("QEMU0002")); // vendor-specific: no standard name
|
||||||
|
try eq("", description("")); // no HID at all
|
||||||
|
}
|
||||||
|
|
||||||
|
test "named values round-trip through their _HID strings" {
|
||||||
|
const testing = std.testing;
|
||||||
|
try testing.expectEqualStrings("PNP0303", HardwareId.ps2_keyboard.hid());
|
||||||
|
try testing.expectEqual(@as(?HardwareId, .ps2_mouse), HardwareId.fromHid("PNP0F13"));
|
||||||
|
try testing.expectEqual(@as(?HardwareId, null), HardwareId.fromHid("QEMU0002"));
|
||||||
|
for (std.enums.values(HardwareId)) |id| {
|
||||||
|
try testing.expectEqual(@as(?HardwareId, id), HardwareId.fromHid(id.hid()));
|
||||||
|
}
|
||||||
|
}
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -3,26 +3,24 @@
|
|||||||
//!
|
//!
|
||||||
//! This module has two stages. `parser.zig` walks the entire byte stream and
|
//! This module has two stages. `parser.zig` walks the entire byte stream and
|
||||||
//! records every named object into a namespace tree (`namespace.zig`), capturing
|
//! records every named object into a namespace tree (`namespace.zig`), capturing
|
||||||
//! method bodies and field/region layout. `interp.zig` then *evaluates* control
|
//! method bodies and field/region layout. `interpreter.zig` then *evaluates* control
|
||||||
//! methods on demand — running operators, control flow, and OperationRegion field
|
//! methods on demand — running operators, control flow, and OperationRegion field
|
||||||
//! access — so callers can resolve device status (`_STA`), current resource
|
//! access — so callers can resolve device status (`_STA`), current resource
|
||||||
//! settings (`_CRS`), sleep states (`_Sx`), and the like against the live namespace.
|
//! settings (`_CRS`), sleep states (`_Sx`), and the like against the live namespace.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
const op = @import("opcodes.zig");
|
const opcode = @import("opcodes.zig");
|
||||||
const parser = @import("parser.zig");
|
const parser = @import("parser.zig");
|
||||||
const namespace = @import("namespace.zig");
|
|
||||||
const interp = @import("interp.zig");
|
|
||||||
|
|
||||||
pub const Namespace = namespace.Namespace;
|
pub const Namespace = @import("namespace.zig").Namespace;
|
||||||
pub const Node = namespace.Node;
|
pub const Node = @import("namespace.zig").Node;
|
||||||
pub const NodeKind = namespace.NodeKind;
|
pub const NodeKind = @import("namespace.zig").NodeKind;
|
||||||
|
|
||||||
/// The AML evaluator: interprets control methods (and reads Names/Fields) far
|
/// The AML evaluator: interprets control methods (and reads Names/Fields) far
|
||||||
/// enough for device discovery. See `interp.zig`.
|
/// enough for device discovery. See `interpreter.zig`.
|
||||||
pub const Interp = interp.Interp;
|
pub const Interpreter = @import("interpreter.zig").Interpreter;
|
||||||
pub const Object = interp.Object;
|
pub const Object = @import("interpreter.zig").Object;
|
||||||
pub const EvalHal = interp.Hal;
|
pub const EvaluateHal = @import("interpreter.zig").Hal;
|
||||||
|
|
||||||
/// The SLP_TYP values written to PM1a/PM1b control to enter a sleep state.
|
/// The SLP_TYP values written to PM1a/PM1b control to enter a sleep state.
|
||||||
pub const SleepType = struct {
|
pub const SleepType = struct {
|
||||||
@@ -41,22 +39,22 @@ pub const ParseResult = struct {
|
|||||||
/// Parse the given AML blocks (DSDT first, then SSDTs) into one namespace. Later
|
/// Parse the given AML blocks (DSDT first, then SSDTs) into one namespace. Later
|
||||||
/// blocks extend the namespace built by earlier ones, exactly as ACPI intends.
|
/// blocks extend the namespace built by earlier ones, exactly as ACPI intends.
|
||||||
pub fn parse(allocator: std.mem.Allocator, blocks: []const []const u8) !ParseResult {
|
pub fn parse(allocator: std.mem.Allocator, blocks: []const []const u8) !ParseResult {
|
||||||
var ns = try Namespace.init(allocator);
|
var namespace = try Namespace.init(allocator);
|
||||||
var consumed: usize = 0;
|
var consumed: usize = 0;
|
||||||
var total: usize = 0;
|
var total: usize = 0;
|
||||||
for (blocks) |block| {
|
for (blocks) |block| {
|
||||||
var p = parser.Parser.init(block, &ns);
|
var p = parser.Parser.init(block, &namespace);
|
||||||
consumed += p.parseAll();
|
consumed += p.parseAll();
|
||||||
total += block.len;
|
total += block.len;
|
||||||
}
|
}
|
||||||
return .{ .namespace = ns, .consumed = consumed, .total = total };
|
return .{ .namespace = namespace, .consumed = consumed, .total = total };
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Look up the `\_S{state}` sleep package in a parsed namespace and return its
|
/// Look up the `\_S{state}` sleep package in a parsed namespace and return its
|
||||||
/// first two integer elements (SLP_TYP for PM1a / PM1b), or null if absent.
|
/// first two integer elements (SLP_TYP for PM1a / PM1b), or null if absent.
|
||||||
pub fn sleepState(ns: *Namespace, state: u8) ?SleepType {
|
pub fn sleepState(namespace: *Namespace, state: u8) ?SleepType {
|
||||||
const seg = [4]u8{ '_', 'S', '0' + state, '_' };
|
const segment = [4]u8{ '_', 'S', '0' + state, '_' };
|
||||||
const node = ns.resolve(ns.root, false, 0, &.{seg}) orelse return null;
|
const node = namespace.resolve(namespace.root, false, 0, &.{segment}) orelse return null;
|
||||||
if (node.kind != .name) return null;
|
if (node.kind != .name) return null;
|
||||||
return parseSleepPackage(node.value);
|
return parseSleepPackage(node.value);
|
||||||
}
|
}
|
||||||
@@ -64,20 +62,20 @@ pub fn sleepState(ns: *Namespace, state: u8) ?SleepType {
|
|||||||
/// Decode a `Package(){ SLP_TYPa, SLP_TYPb, ... }` from the raw AML of a Name's
|
/// Decode a `Package(){ SLP_TYPa, SLP_TYPb, ... }` from the raw AML of a Name's
|
||||||
/// value. Returns the first two elements as bytes (missing elements default to 0).
|
/// value. Returns the first two elements as bytes (missing elements default to 0).
|
||||||
fn parseSleepPackage(value: []const u8) ?SleepType {
|
fn parseSleepPackage(value: []const u8) ?SleepType {
|
||||||
if (value.len == 0 or value[0] != op.package_op) return null;
|
if (value.len == 0 or value[0] != opcode.package_opcode) return null;
|
||||||
var p: usize = 1;
|
var p: usize = 1;
|
||||||
p += pkgLengthSize(value, p) orelse return null;
|
p += packageLengthSize(value, p) orelse return null;
|
||||||
if (p >= value.len) return null;
|
if (p >= value.len) return null;
|
||||||
const num_elements = value[p];
|
const number_elements = value[p];
|
||||||
p += 1;
|
p += 1;
|
||||||
|
|
||||||
const a: u8 = if (num_elements >= 1) @truncate(readInteger(value, &p) orelse 0) else 0;
|
const a: u8 = if (number_elements >= 1) @truncate(readInteger(value, &p) orelse 0) else 0;
|
||||||
const b: u8 = if (num_elements >= 2) @truncate(readInteger(value, &p) orelse 0) else 0;
|
const b: u8 = if (number_elements >= 2) @truncate(readInteger(value, &p) orelse 0) else 0;
|
||||||
return .{ .slp_typ_a = a, .slp_typ_b = b };
|
return .{ .slp_typ_a = a, .slp_typ_b = b };
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Bytes a PkgLength field occupies at `p` (we only need to step over it here).
|
/// Bytes a PkgLength field occupies at `p` (we only need to step over it here).
|
||||||
fn pkgLengthSize(bytes: []const u8, p: usize) ?usize {
|
fn packageLengthSize(bytes: []const u8, p: usize) ?usize {
|
||||||
if (p >= bytes.len) return null;
|
if (p >= bytes.len) return null;
|
||||||
const follow: usize = bytes[p] >> 6;
|
const follow: usize = bytes[p] >> 6;
|
||||||
if (p + 1 + follow > bytes.len) return null;
|
if (p + 1 + follow > bytes.len) return null;
|
||||||
@@ -87,16 +85,16 @@ fn pkgLengthSize(bytes: []const u8, p: usize) ?usize {
|
|||||||
/// Read one AML integer data object at `p`, advancing `p`.
|
/// Read one AML integer data object at `p`, advancing `p`.
|
||||||
fn readInteger(bytes: []const u8, p: *usize) ?u64 {
|
fn readInteger(bytes: []const u8, p: *usize) ?u64 {
|
||||||
if (p.* >= bytes.len) return null;
|
if (p.* >= bytes.len) return null;
|
||||||
const opcode = bytes[p.*];
|
const opcode_byte = bytes[p.*];
|
||||||
p.* += 1;
|
p.* += 1;
|
||||||
return switch (opcode) {
|
return switch (opcode_byte) {
|
||||||
op.zero_op => 0,
|
opcode.zero_opcode => 0,
|
||||||
op.one_op => 1,
|
opcode.one_opcode => 1,
|
||||||
op.ones_op => 0xFF,
|
opcode.ones_opcode => 0xFF,
|
||||||
op.byte_prefix => readLittle(bytes, p, 1),
|
opcode.byte_prefix => readLittle(bytes, p, 1),
|
||||||
op.word_prefix => readLittle(bytes, p, 2),
|
opcode.word_prefix => readLittle(bytes, p, 2),
|
||||||
op.dword_prefix => readLittle(bytes, p, 4),
|
opcode.dword_prefix => readLittle(bytes, p, 4),
|
||||||
op.qword_prefix => readLittle(bytes, p, 8),
|
opcode.qword_prefix => readLittle(bytes, p, 8),
|
||||||
else => null,
|
else => null,
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
@@ -125,20 +123,25 @@ test "parses a nested namespace and finds the sleep package" {
|
|||||||
const blob = [_]u8{
|
const blob = [_]u8{
|
||||||
// Name(_S5, Package(2){Byte 0x05, Byte 0x00})
|
// Name(_S5, Package(2){Byte 0x05, Byte 0x00})
|
||||||
0x08, 0x5F, 0x53, 0x35, 0x5F, 0x12, 0x06, 0x02, 0x0A, 0x05, 0x0A, 0x00,
|
0x08, 0x5F, 0x53, 0x35, 0x5F, 0x12, 0x06, 0x02, 0x0A, 0x05, 0x0A, 0x00,
|
||||||
// Scope(\_SB) pkglen=0x27
|
// Scope(\_SB) packagelen=0x27
|
||||||
0x10, 0x27, 0x5C, 0x5F, 0x53, 0x42, 0x5F,
|
0x10, 0x27, 0x5C, 0x5F, 0x53, 0x42, 0x5F,
|
||||||
// Device(PCI0) pkglen=0x1F
|
// Device(PCI0) packagelen=0x1F
|
||||||
0x5B, 0x82, 0x1F, 0x50, 0x43, 0x49, 0x30,
|
0x5B, 0x82, 0x1F, 0x50, 0x43,
|
||||||
|
0x49, 0x30,
|
||||||
// Name(_HID, 0x11)
|
// Name(_HID, 0x11)
|
||||||
0x08, 0x5F, 0x48, 0x49, 0x44, 0x0A, 0x11,
|
0x08, 0x5F, 0x48, 0x49, 0x44, 0x0A, 0x11,
|
||||||
// Method(MTHD, flags=1) empty, pkglen=0x06
|
// Method(MTHD, flags=1) empty, packagelen=0x06
|
||||||
0x14, 0x06, 0x4D, 0x54, 0x48, 0x44, 0x01,
|
0x14, 0x06, 0x4D,
|
||||||
// Method(CALL, flags=0) { MTHD(Zero) }, pkglen=0x0B
|
0x54, 0x48, 0x44, 0x01,
|
||||||
0x14, 0x0B, 0x43, 0x41, 0x4C, 0x4C, 0x00, 0x4D, 0x54, 0x48, 0x44, 0x00,
|
// Method(CALL, flags=0) { MTHD(Zero) }, packagelen=0x0B
|
||||||
|
0x14, 0x0B, 0x43, 0x41, 0x4C, 0x4C, 0x00, 0x4D,
|
||||||
|
0x54, 0x48, 0x44, 0x00,
|
||||||
// OperationRegion(DBG0, SystemIO, Word 0x0402, Byte 1)
|
// OperationRegion(DBG0, SystemIO, Word 0x0402, Byte 1)
|
||||||
0x5B, 0x80, 0x44, 0x42, 0x47, 0x30, 0x01, 0x0B, 0x02, 0x04, 0x0A, 0x01,
|
0x5B, 0x80, 0x44, 0x42, 0x47, 0x30, 0x01, 0x0B,
|
||||||
// Field(DBG0, flags=1) { DBGB, 8 }, pkglen=0x0B
|
0x02, 0x04, 0x0A, 0x01,
|
||||||
0x5B, 0x81, 0x0B, 0x44, 0x42, 0x47, 0x30, 0x01, 0x44, 0x42, 0x47, 0x42, 0x08,
|
// Field(DBG0, flags=1) { DBGB, 8 }, packagelen=0x0B
|
||||||
|
0x5B, 0x81, 0x0B, 0x44, 0x42, 0x47, 0x30, 0x01,
|
||||||
|
0x44, 0x42, 0x47, 0x42, 0x08,
|
||||||
};
|
};
|
||||||
|
|
||||||
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
||||||
@@ -149,31 +152,33 @@ test "parses a nested namespace and finds the sleep package" {
|
|||||||
try std.testing.expectEqual(blob.len, result.consumed);
|
try std.testing.expectEqual(blob.len, result.consumed);
|
||||||
try std.testing.expectEqual(blob.len, result.total);
|
try std.testing.expectEqual(blob.len, result.total);
|
||||||
|
|
||||||
const ns = &result.namespace;
|
const namespace = &result.namespace;
|
||||||
|
|
||||||
// Expected top-level nodes.
|
// Expected top-level nodes.
|
||||||
const sb = ns.resolve(ns.root, false, 0, &.{.{ '_', 'S', 'B', '_' }}) orelse return error.NoSB;
|
const sb = namespace.resolve(namespace.root, false, 0, &.{.{ '_', 'S', 'B', '_' }}) orelse return error.NoSB;
|
||||||
try std.testing.expectEqual(NodeKind.scope, sb.kind);
|
try std.testing.expectEqual(NodeKind.scope, sb.kind);
|
||||||
const pci0 = ns.resolve(sb, false, 0, &.{.{ 'P', 'C', 'I', '0' }}) orelse return error.NoPCI0;
|
const pci0 = namespace.resolve(sb, false, 0, &.{.{ 'P', 'C', 'I', '0' }}) orelse return error.NoPCI0;
|
||||||
try std.testing.expectEqual(NodeKind.device, pci0.kind);
|
try std.testing.expectEqual(NodeKind.device, pci0.kind);
|
||||||
_ = ns.resolve(pci0, false, 0, &.{.{ '_', 'H', 'I', 'D' }}) orelse return error.NoHID;
|
_ = namespace.resolve(pci0, false, 0, &.{.{ '_', 'H', 'I', 'D' }}) orelse return error.NoHID;
|
||||||
|
|
||||||
// The 1-arg method's arg count was parsed from its flags byte.
|
// The 1-arg method's arg count was parsed from its flags byte.
|
||||||
const mthd = ns.resolve(pci0, false, 0, &.{.{ 'M', 'T', 'H', 'D' }}) orelse return error.NoMTHD;
|
const mthd = namespace.resolve(pci0, false, 0, &.{.{ 'M', 'T', 'H', 'D' }}) orelse return error.NoMTHD;
|
||||||
try std.testing.expectEqual(NodeKind.method, mthd.kind);
|
try std.testing.expectEqual(NodeKind.method, mthd.kind);
|
||||||
try std.testing.expectEqual(@as(u8, 1), mthd.arg_count);
|
try std.testing.expectEqual(@as(u8, 1), mthd.arg_count);
|
||||||
|
|
||||||
// OperationRegion and the Field unit made it into the namespace.
|
// OperationRegion and the Field unit made it into the namespace.
|
||||||
_ = ns.resolve(ns.root, false, 0, &.{.{ 'D', 'B', 'G', '0' }}) orelse return error.NoRegion;
|
_ = namespace.resolve(namespace.root, false, 0, &.{.{ 'D', 'B', 'G', '0' }}) orelse return error.NoRegion;
|
||||||
_ = ns.resolve(ns.root, false, 0, &.{.{ 'D', 'B', 'G', 'B' }}) orelse return error.NoField;
|
_ = namespace.resolve(namespace.root, false, 0, &.{.{ 'D', 'B', 'G', 'B' }}) orelse return error.NoField;
|
||||||
|
|
||||||
// The sleep package decoded.
|
// The sleep package decoded.
|
||||||
const s5 = sleepState(ns, 5) orelse return error.NoS5;
|
const s5 = sleepState(namespace, 5) orelse return error.NoS5;
|
||||||
try std.testing.expectEqual(@as(u8, 5), s5.slp_typ_a);
|
try std.testing.expectEqual(@as(u8, 5), s5.slp_typ_a);
|
||||||
try std.testing.expectEqual(@as(u8, 0), s5.slp_typ_b);
|
try std.testing.expectEqual(@as(u8, 0), s5.slp_typ_b);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn noMap(_: u64, _: u64, _: bool) void {}
|
fn noMap(physical: u64, _: u64, _: bool) u64 {
|
||||||
|
return physical;
|
||||||
|
}
|
||||||
fn noRead(_: u8, _: u16) u32 {
|
fn noRead(_: u8, _: u16) u32 {
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
@@ -196,13 +201,13 @@ test "interpreter runs a method with args, arithmetic, and control flow" {
|
|||||||
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
||||||
defer arena.deinit();
|
defer arena.deinit();
|
||||||
var result = try parse(arena.allocator(), &.{&blob});
|
var result = try parse(arena.allocator(), &.{&blob});
|
||||||
const ns = &result.namespace;
|
const namespace = &result.namespace;
|
||||||
const tst = ns.resolve(ns.root, false, 0, &.{.{ 'T', 'S', 'T', '_' }}) orelse return error.NoMethod;
|
const tst = namespace.resolve(namespace.root, false, 0, &.{.{ 'T', 'S', 'T', '_' }}) orelse return error.NoMethod;
|
||||||
|
|
||||||
var ev = Interp.init(ns, .{ .mapMmio = noMap, .pioRead = noRead, .pioWrite = noWrite }, arena.allocator());
|
var interpreter = Interpreter.init(namespace, .{ .mapMmio = noMap, .pioRead = noRead, .pioWrite = noWrite }, arena.allocator());
|
||||||
|
|
||||||
const hi = try ev.evaluate(tst, &.{.{ .integer = 7 }}); // 7+5=12 > 10 -> 1
|
const hi = try interpreter.evaluate(tst, &.{.{ .integer = 7 }}); // 7+5=12 > 10 -> 1
|
||||||
try std.testing.expectEqual(@as(u64, 1), try hi.asInt());
|
try std.testing.expectEqual(@as(u64, 1), try hi.asInteger());
|
||||||
const lo = try ev.evaluate(tst, &.{.{ .integer = 2 }}); // 2+5=7 !> 10 -> 0
|
const lo = try interpreter.evaluate(tst, &.{.{ .integer = 2 }}); // 2+5=7 !> 10 -> 0
|
||||||
try std.testing.expectEqual(@as(u64, 0), try lo.asInt());
|
try std.testing.expectEqual(@as(u64, 0), try lo.asInteger());
|
||||||
}
|
}
|
||||||
@@ -0,0 +1,736 @@
|
|||||||
|
//! A tree-walking AML interpreter — the evaluation stage on top of the parser's
|
||||||
|
//! structural namespace. It executes control methods (their bodies captured by
|
||||||
|
//! the parser) far enough to serve device discovery: device status (`_STA`, is a
|
||||||
|
//! device present), current resource settings (`_CRS`), and the operators, control
|
||||||
|
//! flow, locals/args, and
|
||||||
|
//! OperationRegion field access those methods reach for.
|
||||||
|
//!
|
||||||
|
//! Scope: integers, buffers, strings, packages, and references; If/Else/While/
|
||||||
|
//! Return; the arithmetic/logic operators; method invocation; Name/Local/Arg
|
||||||
|
//! access; CreateField buffer patching (the common current-resource-settings
|
||||||
|
//! (`_CRS`) idiom); and field
|
||||||
|
//! reads/writes against SystemMemory and SystemIO regions. Opcodes outside this
|
||||||
|
//! set return `error.Unsupported`, which callers treat as "couldn't evaluate" and
|
||||||
|
//! fall back — never a hard failure.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const opcode = @import("opcodes.zig");
|
||||||
|
const Node = @import("namespace.zig").Node;
|
||||||
|
const Namespace = @import("namespace.zig").Namespace;
|
||||||
|
|
||||||
|
/// Injected hardware access for OperationRegion reads/writes (the architecture VMM + pio).
|
||||||
|
pub const Hal = struct {
|
||||||
|
mapMmio: *const fn (physical: u64, len: u64, writable: bool) u64,
|
||||||
|
pioRead: *const fn (width: u8, port: u16) u32,
|
||||||
|
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const Error = error{ Unsupported, Truncated, DivByZero } || std.mem.Allocator.Error;
|
||||||
|
|
||||||
|
/// A runtime AML value.
|
||||||
|
pub const Object = union(enum) {
|
||||||
|
uninitialized,
|
||||||
|
integer: u64,
|
||||||
|
buffer: []u8,
|
||||||
|
string: []u8,
|
||||||
|
package: []Object,
|
||||||
|
reference: *Node,
|
||||||
|
|
||||||
|
pub fn asInteger(self: Object) Error!u64 {
|
||||||
|
return switch (self) {
|
||||||
|
.integer => |v| v,
|
||||||
|
.buffer => |b| blk: {
|
||||||
|
var v: u64 = 0;
|
||||||
|
for (b, 0..) |byte, i| {
|
||||||
|
if (i >= 8) break;
|
||||||
|
v |= @as(u64, byte) << @intCast(i * 8);
|
||||||
|
}
|
||||||
|
break :blk v;
|
||||||
|
},
|
||||||
|
else => error.Unsupported,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
const maximum_segments = 16;
|
||||||
|
const NamePath = struct {
|
||||||
|
rooted: bool = false,
|
||||||
|
parents: u8 = 0,
|
||||||
|
segments: [maximum_segments][4]u8 = undefined,
|
||||||
|
count: usize = 0,
|
||||||
|
fn slice(self: *const NamePath) []const [4]u8 {
|
||||||
|
return self.segments[0..self.count];
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
const Cursor = struct {
|
||||||
|
b: []const u8,
|
||||||
|
i: usize = 0,
|
||||||
|
|
||||||
|
fn eof(self: *Cursor) bool {
|
||||||
|
return self.i >= self.b.len;
|
||||||
|
}
|
||||||
|
fn peek(self: *Cursor) ?u8 {
|
||||||
|
return if (self.eof()) null else self.b[self.i];
|
||||||
|
}
|
||||||
|
fn byte(self: *Cursor) Error!u8 {
|
||||||
|
if (self.eof()) return error.Truncated;
|
||||||
|
const v = self.b[self.i];
|
||||||
|
self.i += 1;
|
||||||
|
return v;
|
||||||
|
}
|
||||||
|
fn take(self: *Cursor, n: usize) Error![]const u8 {
|
||||||
|
if (self.i + n > self.b.len) return error.Truncated;
|
||||||
|
const s = self.b[self.i .. self.i + n];
|
||||||
|
self.i += n;
|
||||||
|
return s;
|
||||||
|
}
|
||||||
|
fn packageLength(self: *Cursor) Error!usize {
|
||||||
|
const lead = try self.byte();
|
||||||
|
const follow: usize = lead >> 6;
|
||||||
|
if (follow == 0) return lead & 0x3F;
|
||||||
|
var value: usize = lead & 0x0F;
|
||||||
|
var k: usize = 0;
|
||||||
|
while (k < follow) : (k += 1) value |= @as(usize, try self.byte()) << @intCast(4 + k * 8);
|
||||||
|
return value;
|
||||||
|
}
|
||||||
|
fn nameString(self: *Cursor) Error!NamePath {
|
||||||
|
var name_path = NamePath{};
|
||||||
|
if (self.peek() == opcode.root_char) {
|
||||||
|
name_path.rooted = true;
|
||||||
|
self.i += 1;
|
||||||
|
} else {
|
||||||
|
while (self.peek() == opcode.parent_prefix_char) : (self.i += 1) name_path.parents += 1;
|
||||||
|
}
|
||||||
|
const lead = self.peek() orelse return name_path;
|
||||||
|
switch (lead) {
|
||||||
|
0x00 => self.i += 1,
|
||||||
|
opcode.dual_name_prefix => {
|
||||||
|
self.i += 1;
|
||||||
|
try self.segment(&name_path);
|
||||||
|
try self.segment(&name_path);
|
||||||
|
},
|
||||||
|
opcode.multi_name_prefix => {
|
||||||
|
self.i += 1;
|
||||||
|
const count = try self.byte();
|
||||||
|
var k: usize = 0;
|
||||||
|
while (k < count) : (k += 1) try self.segment(&name_path);
|
||||||
|
},
|
||||||
|
else => try self.segment(&name_path),
|
||||||
|
}
|
||||||
|
return name_path;
|
||||||
|
}
|
||||||
|
fn segment(self: *Cursor, name_path: *NamePath) Error!void {
|
||||||
|
const s = try self.take(4);
|
||||||
|
if (name_path.count < maximum_segments) {
|
||||||
|
name_path.segments[name_path.count] = s[0..4].*;
|
||||||
|
name_path.count += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
const Frame = struct {
|
||||||
|
args: [7]Object = .{.uninitialized} ** 7,
|
||||||
|
locals: [8]Object = .{.uninitialized} ** 8,
|
||||||
|
scope: *Node,
|
||||||
|
ret: Object = .uninitialized,
|
||||||
|
returned: bool = false,
|
||||||
|
broke: bool = false,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A CreateField binding: a name that indexes into a buffer object.
|
||||||
|
const BufferField = struct { buffer: *Node, byte_off: usize, bit_width: u32 };
|
||||||
|
|
||||||
|
pub const Interpreter = struct {
|
||||||
|
namespace: *Namespace,
|
||||||
|
hal: Hal,
|
||||||
|
arena: std.mem.Allocator,
|
||||||
|
/// Runtime object overrides for Name nodes (Store targets, patched buffers).
|
||||||
|
dynamic_overrides: std.AutoHashMapUnmanaged(*Node, Object) = .{},
|
||||||
|
/// CreateField bindings active for the current evaluation.
|
||||||
|
fields: std.AutoHashMapUnmanaged(*Node, BufferField) = .{},
|
||||||
|
|
||||||
|
pub fn init(namespace: *Namespace, hal: Hal, arena: std.mem.Allocator) Interpreter {
|
||||||
|
return .{ .namespace = namespace, .hal = hal, .arena = arena };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Evaluate a namespace object: invoke a Method, read a Name's value, or read a
|
||||||
|
/// Field. Resets per-evaluation runtime state first.
|
||||||
|
pub fn evaluate(self: *Interpreter, node: *Node, args: []const Object) Error!Object {
|
||||||
|
self.dynamic_overrides.clearRetainingCapacity();
|
||||||
|
self.fields.clearRetainingCapacity();
|
||||||
|
return self.invoke(node, args);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn invoke(self: *Interpreter, node: *Node, args: []const Object) Error!Object {
|
||||||
|
switch (node.kind) {
|
||||||
|
.method => {
|
||||||
|
var frame = Frame{ .scope = node };
|
||||||
|
for (args, 0..) |a, i| {
|
||||||
|
if (i < frame.args.len) frame.args[i] = a;
|
||||||
|
}
|
||||||
|
var current = Cursor{ .b = node.value };
|
||||||
|
try self.executeList(¤t, &frame);
|
||||||
|
return frame.ret;
|
||||||
|
},
|
||||||
|
.name => {
|
||||||
|
if (self.dynamic_overrides.get(node)) |o| return o;
|
||||||
|
var current = Cursor{ .b = node.value };
|
||||||
|
var frame = Frame{ .scope = node.parent orelse self.namespace.root };
|
||||||
|
return self.term(¤t, &frame);
|
||||||
|
},
|
||||||
|
.field => return .{ .integer = try self.readField(node) },
|
||||||
|
else => return .{ .reference = node },
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Execute a TermList until it ends or the frame returns/breaks.
|
||||||
|
fn executeList(self: *Interpreter, current: *Cursor, frame: *Frame) Error!void {
|
||||||
|
while (!current.eof() and !frame.returned and !frame.broke) {
|
||||||
|
_ = try self.term(current, frame);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Evaluate/execute one term, returning its value (`.uninitialized` for pure
|
||||||
|
/// statements).
|
||||||
|
fn term(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const lead = current.peek() orelse return error.Truncated;
|
||||||
|
if (isNameStart(lead)) return self.nameReference(current, frame);
|
||||||
|
_ = try current.byte();
|
||||||
|
|
||||||
|
return switch (lead) {
|
||||||
|
opcode.zero_opcode => Object{ .integer = 0 },
|
||||||
|
opcode.one_opcode => Object{ .integer = 1 },
|
||||||
|
opcode.ones_opcode => Object{ .integer = ~@as(u64, 0) },
|
||||||
|
opcode.byte_prefix => Object{ .integer = try self.readConstant(current, 1) },
|
||||||
|
opcode.word_prefix => Object{ .integer = try self.readConstant(current, 2) },
|
||||||
|
opcode.dword_prefix => Object{ .integer = try self.readConstant(current, 4) },
|
||||||
|
opcode.qword_prefix => Object{ .integer = try self.readConstant(current, 8) },
|
||||||
|
opcode.string_prefix => try self.readString(current),
|
||||||
|
opcode.buffer_opcode => try self.buffer(current, frame),
|
||||||
|
opcode.package_opcode, opcode.var_package_opcode => try self.package(current, frame, lead == opcode.var_package_opcode),
|
||||||
|
|
||||||
|
opcode.local0_opcode...opcode.local7_opcode => frame.locals[lead - opcode.local0_opcode],
|
||||||
|
opcode.arg0_opcode...opcode.arg6_opcode => frame.args[lead - opcode.arg0_opcode],
|
||||||
|
|
||||||
|
opcode.return_opcode => blk: {
|
||||||
|
frame.ret = try self.term(current, frame);
|
||||||
|
frame.returned = true;
|
||||||
|
break :blk .uninitialized;
|
||||||
|
},
|
||||||
|
opcode.break_opcode => blk: {
|
||||||
|
frame.broke = true;
|
||||||
|
break :blk .uninitialized;
|
||||||
|
},
|
||||||
|
opcode.continue_opcode, opcode.noop_opcode => .uninitialized,
|
||||||
|
|
||||||
|
opcode.if_opcode => try self.ifElse(current, frame),
|
||||||
|
opcode.while_opcode => try self.whileLoop(current, frame),
|
||||||
|
opcode.store_opcode => try self.store(current, frame),
|
||||||
|
opcode.increment_opcode => try self.incDec(current, frame, 1),
|
||||||
|
opcode.decrement_opcode => try self.incDec(current, frame, -1),
|
||||||
|
|
||||||
|
opcode.add_opcode => try self.binary(current, frame, .add),
|
||||||
|
opcode.subtract_opcode => try self.binary(current, frame, .sub),
|
||||||
|
opcode.multiply_opcode => try self.binary(current, frame, .mul),
|
||||||
|
opcode.mod_opcode => try self.binary(current, frame, .mod),
|
||||||
|
opcode.and_opcode => try self.binary(current, frame, .band),
|
||||||
|
opcode.or_opcode => try self.binary(current, frame, .bor),
|
||||||
|
opcode.xor_opcode => try self.binary(current, frame, .bxor),
|
||||||
|
opcode.nand_opcode => try self.binary(current, frame, .nand),
|
||||||
|
opcode.nor_opcode => try self.binary(current, frame, .nor),
|
||||||
|
opcode.shift_left_opcode => try self.binary(current, frame, .shl),
|
||||||
|
opcode.shift_right_opcode => try self.binary(current, frame, .shr),
|
||||||
|
opcode.divide_opcode => try self.divide(current, frame),
|
||||||
|
|
||||||
|
opcode.land_opcode => try self.logic2(current, frame, .land),
|
||||||
|
opcode.lor_opcode => try self.logic2(current, frame, .lor),
|
||||||
|
opcode.lequal_opcode => try self.logic2(current, frame, .eq),
|
||||||
|
opcode.lgreater_opcode => try self.logic2(current, frame, .gt),
|
||||||
|
opcode.lless_opcode => try self.logic2(current, frame, .lt),
|
||||||
|
opcode.lnot_opcode => try self.lnot(current, frame),
|
||||||
|
|
||||||
|
opcode.not_opcode => blk: {
|
||||||
|
const v = try self.evaluateInteger(current, frame);
|
||||||
|
const r = ~v;
|
||||||
|
try self.storeTarget(current, frame, .{ .integer = r });
|
||||||
|
break :blk .{ .integer = r };
|
||||||
|
},
|
||||||
|
|
||||||
|
opcode.size_of_opcode => try self.sizeOf(current, frame),
|
||||||
|
opcode.index_opcode => try self.index(current, frame),
|
||||||
|
opcode.dereference_of_opcode => try self.dereferenceOf(current, frame),
|
||||||
|
opcode.to_integer_opcode => blk: {
|
||||||
|
const v = try self.evaluateInteger(current, frame);
|
||||||
|
try self.storeTarget(current, frame, .{ .integer = v });
|
||||||
|
break :blk .{ .integer = v };
|
||||||
|
},
|
||||||
|
opcode.to_buffer_opcode => try self.passThroughUnary(current, frame),
|
||||||
|
|
||||||
|
opcode.extended_opcode_prefix => try self.ext(current, frame),
|
||||||
|
|
||||||
|
// CreateXField: source, index, name (bit widths differ by op)
|
||||||
|
opcode.create_bit_field_opcode => try self.createField(current, frame, 1),
|
||||||
|
opcode.create_byte_field_opcode => try self.createField(current, frame, 8),
|
||||||
|
opcode.create_word_field_opcode => try self.createField(current, frame, 16),
|
||||||
|
opcode.create_dword_field_opcode => try self.createField(current, frame, 32),
|
||||||
|
opcode.create_qword_field_opcode => try self.createField(current, frame, 64),
|
||||||
|
|
||||||
|
else => error.Unsupported,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- name references ----------------------------------------------------
|
||||||
|
|
||||||
|
fn nameReference(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const name_path = try current.nameString();
|
||||||
|
const node = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice()) orelse
|
||||||
|
return .uninitialized; // unknown name -> treat as uninitialised
|
||||||
|
switch (node.kind) {
|
||||||
|
.method => {
|
||||||
|
var argbuf: [7]Object = undefined;
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < node.arg_count and i < argbuf.len) : (i += 1) argbuf[i] = try self.term(current, frame);
|
||||||
|
return self.invoke(node, argbuf[0..@min(node.arg_count, argbuf.len)]);
|
||||||
|
},
|
||||||
|
.field => return .{ .integer = try self.readField(node) },
|
||||||
|
.name => return self.invoke(node, &.{}),
|
||||||
|
else => return .{ .reference = node },
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- data objects -------------------------------------------------------
|
||||||
|
|
||||||
|
fn readConstant(self: *Interpreter, current: *Cursor, n: usize) Error!u64 {
|
||||||
|
_ = self;
|
||||||
|
const bytes = try current.take(n);
|
||||||
|
var v: u64 = 0;
|
||||||
|
for (bytes, 0..) |b, i| v |= @as(u64, b) << @intCast(i * 8);
|
||||||
|
return v;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn readString(self: *Interpreter, current: *Cursor) Error!Object {
|
||||||
|
const start = current.i;
|
||||||
|
while (current.peek()) |c| {
|
||||||
|
current.i += 1;
|
||||||
|
if (c == 0) break;
|
||||||
|
}
|
||||||
|
const raw = current.b[start .. current.i - 1];
|
||||||
|
const s = try self.arena.dupe(u8, raw);
|
||||||
|
return .{ .string = s };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn buffer(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const start = current.i;
|
||||||
|
const len = try current.packageLength();
|
||||||
|
const end = @min(start + len, current.b.len);
|
||||||
|
const size = try self.evaluateInteger(current, frame);
|
||||||
|
const data = current.b[@min(current.i, end)..end];
|
||||||
|
const bytes = try self.arena.alloc(u8, @intCast(size));
|
||||||
|
@memset(bytes, 0);
|
||||||
|
@memcpy(bytes[0..@min(bytes.len, data.len)], data[0..@min(bytes.len, data.len)]);
|
||||||
|
current.i = end;
|
||||||
|
return .{ .buffer = bytes };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn package(self: *Interpreter, current: *Cursor, frame: *Frame, variable: bool) Error!Object {
|
||||||
|
const start = current.i;
|
||||||
|
const len = try current.packageLength();
|
||||||
|
const end = @min(start + len, current.b.len);
|
||||||
|
const count: usize = if (variable) @intCast(try self.evaluateInteger(current, frame)) else try current.byte();
|
||||||
|
const elems = try self.arena.alloc(Object, count);
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < count and current.i < end) : (i += 1) elems[i] = try self.term(current, frame);
|
||||||
|
while (i < count) : (i += 1) elems[i] = .uninitialized;
|
||||||
|
current.i = end;
|
||||||
|
return .{ .package = elems };
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- operators ----------------------------------------------------------
|
||||||
|
|
||||||
|
const BinaryOperation = enum { add, sub, mul, mod, band, bor, bxor, nand, nor, shl, shr };
|
||||||
|
|
||||||
|
fn binary(self: *Interpreter, current: *Cursor, frame: *Frame, kind: BinaryOperation) Error!Object {
|
||||||
|
const a = try self.evaluateInteger(current, frame);
|
||||||
|
const b = try self.evaluateInteger(current, frame);
|
||||||
|
const r: u64 = switch (kind) {
|
||||||
|
.add => a +% b,
|
||||||
|
.sub => a -% b,
|
||||||
|
.mul => a *% b,
|
||||||
|
.mod => if (b == 0) return error.DivByZero else a % b,
|
||||||
|
.band => a & b,
|
||||||
|
.bor => a | b,
|
||||||
|
.bxor => a ^ b,
|
||||||
|
.nand => ~(a & b),
|
||||||
|
.nor => ~(a | b),
|
||||||
|
.shl => if (b >= 64) 0 else a << @intCast(b),
|
||||||
|
.shr => if (b >= 64) 0 else a >> @intCast(b),
|
||||||
|
};
|
||||||
|
try self.storeTarget(current, frame, .{ .integer = r });
|
||||||
|
return .{ .integer = r };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn divide(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const a = try self.evaluateInteger(current, frame);
|
||||||
|
const b = try self.evaluateInteger(current, frame);
|
||||||
|
if (b == 0) return error.DivByZero;
|
||||||
|
try self.storeTarget(current, frame, .{ .integer = a % b }); // remainder target
|
||||||
|
try self.storeTarget(current, frame, .{ .integer = a / b }); // quotient target
|
||||||
|
return .{ .integer = a / b };
|
||||||
|
}
|
||||||
|
|
||||||
|
const LogicOperation = enum { land, lor, eq, gt, lt };
|
||||||
|
|
||||||
|
fn logic2(self: *Interpreter, current: *Cursor, frame: *Frame, kind: LogicOperation) Error!Object {
|
||||||
|
const a = try self.evaluateInteger(current, frame);
|
||||||
|
const b = try self.evaluateInteger(current, frame);
|
||||||
|
const r = switch (kind) {
|
||||||
|
.land => a != 0 and b != 0,
|
||||||
|
.lor => a != 0 or b != 0,
|
||||||
|
.eq => a == b,
|
||||||
|
.gt => a > b,
|
||||||
|
.lt => a < b,
|
||||||
|
};
|
||||||
|
return .{ .integer = if (r) ~@as(u64, 0) else 0 };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn lnot(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
// 0x92 0x93/94/95 are the compound comparisons.
|
||||||
|
const b = current.peek() orelse return error.Truncated;
|
||||||
|
switch (b) {
|
||||||
|
opcode.lnot.not_equal => {
|
||||||
|
current.i += 1;
|
||||||
|
const x = try self.evaluateInteger(current, frame);
|
||||||
|
const y = try self.evaluateInteger(current, frame);
|
||||||
|
return .{ .integer = if (x != y) ~@as(u64, 0) else 0 };
|
||||||
|
},
|
||||||
|
opcode.lnot.less_equal => {
|
||||||
|
current.i += 1;
|
||||||
|
const x = try self.evaluateInteger(current, frame);
|
||||||
|
const y = try self.evaluateInteger(current, frame);
|
||||||
|
return .{ .integer = if (x <= y) ~@as(u64, 0) else 0 };
|
||||||
|
},
|
||||||
|
opcode.lnot.greater_equal => {
|
||||||
|
current.i += 1;
|
||||||
|
const x = try self.evaluateInteger(current, frame);
|
||||||
|
const y = try self.evaluateInteger(current, frame);
|
||||||
|
return .{ .integer = if (x >= y) ~@as(u64, 0) else 0 };
|
||||||
|
},
|
||||||
|
else => {
|
||||||
|
const x = try self.evaluateInteger(current, frame);
|
||||||
|
return .{ .integer = if (x == 0) ~@as(u64, 0) else 0 };
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn incDec(self: *Interpreter, current: *Cursor, frame: *Frame, delta: i64) Error!Object {
|
||||||
|
// Operand is a SuperName that is both read and written.
|
||||||
|
const save = current.i;
|
||||||
|
const current_value = try self.term(current, frame);
|
||||||
|
const v = try current_value.asInteger();
|
||||||
|
const r = if (delta > 0) v +% 1 else v -% 1;
|
||||||
|
var tcur = Cursor{ .b = current.b, .i = save };
|
||||||
|
try self.storeInto(&tcur, frame, .{ .integer = r });
|
||||||
|
return .{ .integer = r };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn sizeOf(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const o = try self.term(current, frame);
|
||||||
|
return .{ .integer = switch (o) {
|
||||||
|
.buffer => |b| b.len,
|
||||||
|
.string => |s| s.len,
|
||||||
|
.package => |p| p.len,
|
||||||
|
else => 0,
|
||||||
|
} };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn passThroughUnary(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const o = try self.term(current, frame);
|
||||||
|
try self.storeTarget(current, frame, o);
|
||||||
|
return o;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn index(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const source = try self.term(current, frame);
|
||||||
|
const element_index: usize = @intCast(try self.evaluateInteger(current, frame));
|
||||||
|
// Optional target (a reference); we don't materialise references, so store
|
||||||
|
// the indexed value if a target is present.
|
||||||
|
const value: Object = switch (source) {
|
||||||
|
.buffer => |b| .{ .integer = if (element_index < b.len) b[element_index] else 0 },
|
||||||
|
.package => |p| if (element_index < p.len) p[element_index] else .uninitialized,
|
||||||
|
.string => |s| .{ .integer = if (element_index < s.len) s[element_index] else 0 },
|
||||||
|
else => .uninitialized,
|
||||||
|
};
|
||||||
|
try self.storeTarget(current, frame, value);
|
||||||
|
return value;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn dereferenceOf(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const o = try self.term(current, frame);
|
||||||
|
return switch (o) {
|
||||||
|
.reference => |n| self.invoke(n, &.{}),
|
||||||
|
else => o,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- control flow -------------------------------------------------------
|
||||||
|
|
||||||
|
fn ifElse(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const start = current.i;
|
||||||
|
const end = @min(start + try current.packageLength(), current.b.len);
|
||||||
|
const cond = try self.evaluateInteger(current, frame);
|
||||||
|
if (cond != 0) {
|
||||||
|
var body = Cursor{ .b = current.b[0..end], .i = current.i };
|
||||||
|
try self.executeList(&body, frame);
|
||||||
|
current.i = end;
|
||||||
|
// Skip a trailing Else.
|
||||||
|
if (current.peek() == opcode.else_opcode) {
|
||||||
|
current.i += 1;
|
||||||
|
const es = current.i;
|
||||||
|
const ee = @min(es + try current.packageLength(), current.b.len);
|
||||||
|
current.i = ee;
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
current.i = end;
|
||||||
|
if (current.peek() == opcode.else_opcode) {
|
||||||
|
current.i += 1;
|
||||||
|
const es = current.i;
|
||||||
|
const ee = @min(es + try current.packageLength(), current.b.len);
|
||||||
|
var body = Cursor{ .b = current.b[0..ee], .i = current.i };
|
||||||
|
try self.executeList(&body, frame);
|
||||||
|
current.i = ee;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return .uninitialized;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn whileLoop(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const start = current.i;
|
||||||
|
const end = @min(start + try current.packageLength(), current.b.len);
|
||||||
|
const pred_at = current.i;
|
||||||
|
var guard: usize = 0;
|
||||||
|
while (guard < 100_000) : (guard += 1) {
|
||||||
|
var pc = Cursor{ .b = current.b[0..end], .i = pred_at };
|
||||||
|
const cond = try self.evaluateInteger(&pc, frame);
|
||||||
|
if (cond == 0) break;
|
||||||
|
var body = Cursor{ .b = current.b[0..end], .i = pc.i };
|
||||||
|
try self.executeList(&body, frame);
|
||||||
|
if (frame.returned) break;
|
||||||
|
if (frame.broke) {
|
||||||
|
frame.broke = false;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
current.i = end;
|
||||||
|
return .uninitialized;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- store --------------------------------------------------------------
|
||||||
|
|
||||||
|
fn store(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const value = try self.term(current, frame);
|
||||||
|
try self.storeInto(current, frame, value);
|
||||||
|
return value;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A Store *target* that may be NullName (no store).
|
||||||
|
fn storeTarget(self: *Interpreter, current: *Cursor, frame: *Frame, value: Object) Error!void {
|
||||||
|
if (current.peek() == 0x00) {
|
||||||
|
current.i += 1; // NullName
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
try self.storeInto(current, frame, value);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn storeInto(self: *Interpreter, current: *Cursor, frame: *Frame, value: Object) Error!void {
|
||||||
|
const lead = current.peek() orelse return error.Truncated;
|
||||||
|
if (isNameStart(lead)) {
|
||||||
|
const name_path = try current.nameString();
|
||||||
|
const node = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice()) orelse return;
|
||||||
|
if (self.fields.get(node)) |buffer_field| {
|
||||||
|
try self.writeBufferField(buffer_field, try value.asInteger());
|
||||||
|
} else if (node.kind == .field) {
|
||||||
|
try self.writeField(node, try value.asInteger());
|
||||||
|
} else {
|
||||||
|
try self.dynamic_overrides.put(self.arena, node, value);
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
_ = try current.byte();
|
||||||
|
switch (lead) {
|
||||||
|
0x00 => {}, // NullName
|
||||||
|
opcode.local0_opcode...opcode.local7_opcode => frame.locals[lead - opcode.local0_opcode] = value,
|
||||||
|
opcode.arg0_opcode...opcode.arg6_opcode => frame.args[lead - opcode.arg0_opcode] = value,
|
||||||
|
opcode.index_opcode => {
|
||||||
|
const source = try self.term(current, frame);
|
||||||
|
const element_index: usize = @intCast(try self.evaluateInteger(current, frame));
|
||||||
|
switch (source) {
|
||||||
|
.buffer => |b| if (element_index < b.len) {
|
||||||
|
b[element_index] = @truncate(try value.asInteger());
|
||||||
|
},
|
||||||
|
.package => |p| if (element_index < p.len) {
|
||||||
|
p[element_index] = value;
|
||||||
|
},
|
||||||
|
else => {},
|
||||||
|
}
|
||||||
|
},
|
||||||
|
else => return error.Unsupported,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- CreateField (buffer patching) --------------------------------------
|
||||||
|
|
||||||
|
fn createField(self: *Interpreter, current: *Cursor, frame: *Frame, bit_width: u32) Error!Object {
|
||||||
|
const source = try self.term(current, frame); // source buffer (as a reference or value)
|
||||||
|
const bit_index = try self.evaluateInteger(current, frame);
|
||||||
|
const name_path = try current.nameString();
|
||||||
|
const node = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice()) orelse return .uninitialized;
|
||||||
|
|
||||||
|
// Bind the new name to the source buffer's node so stores land in it.
|
||||||
|
const buffer_node: *Node = switch (source) {
|
||||||
|
.reference => |n| n,
|
||||||
|
else => return .uninitialized,
|
||||||
|
};
|
||||||
|
// Materialise the buffer into `dynamic_overrides` so patches persist and are returned.
|
||||||
|
if (self.dynamic_overrides.get(buffer_node) == null) {
|
||||||
|
const value = try self.invoke(buffer_node, &.{});
|
||||||
|
try self.dynamic_overrides.put(self.arena, buffer_node, value);
|
||||||
|
}
|
||||||
|
const byte_off: usize = @intCast(bit_index / 8);
|
||||||
|
try self.fields.put(self.arena, node, .{ .buffer = buffer_node, .byte_off = byte_off, .bit_width = bit_width });
|
||||||
|
return .uninitialized;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn writeBufferField(self: *Interpreter, buffer_field: BufferField, value: u64) Error!void {
|
||||||
|
const obj = self.dynamic_overrides.get(buffer_field.buffer) orelse return;
|
||||||
|
const bytes = switch (obj) {
|
||||||
|
.buffer => |b| b,
|
||||||
|
else => return,
|
||||||
|
};
|
||||||
|
const byte_count = (buffer_field.bit_width + 7) / 8;
|
||||||
|
var k: usize = 0;
|
||||||
|
while (k < byte_count and buffer_field.byte_off + k < bytes.len) : (k += 1) {
|
||||||
|
bytes[buffer_field.byte_off + k] = @truncate(value >> @intCast(k * 8));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- OperationRegion field access ---------------------------------------
|
||||||
|
|
||||||
|
fn readField(self: *Interpreter, field: *Node) Error!u64 {
|
||||||
|
const region = field.region orelse return error.Unsupported;
|
||||||
|
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
||||||
|
const base = try self.regionBase(region);
|
||||||
|
const start_byte = base + field.bit_offset / 8;
|
||||||
|
const shift: u7 = @intCast(field.bit_offset % 8);
|
||||||
|
const total = @as(usize, shift) + field.bit_width;
|
||||||
|
const byte_count = (total + 7) / 8;
|
||||||
|
var raw: u128 = 0;
|
||||||
|
var k: usize = 0;
|
||||||
|
while (k < byte_count) : (k += 1) {
|
||||||
|
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
||||||
|
}
|
||||||
|
const masked = (raw >> shift) & bitMask(field.bit_width);
|
||||||
|
return @truncate(masked);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn writeField(self: *Interpreter, field: *Node, value: u64) Error!void {
|
||||||
|
const region = field.region orelse return error.Unsupported;
|
||||||
|
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
||||||
|
const base = try self.regionBase(region);
|
||||||
|
const start_byte = base + field.bit_offset / 8;
|
||||||
|
const shift: u7 = @intCast(field.bit_offset % 8);
|
||||||
|
const total = @as(usize, shift) + field.bit_width;
|
||||||
|
const byte_count = (total + 7) / 8;
|
||||||
|
// Read-modify-write byte by byte.
|
||||||
|
var raw: u128 = 0;
|
||||||
|
var k: usize = 0;
|
||||||
|
while (k < byte_count) : (k += 1) {
|
||||||
|
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
||||||
|
}
|
||||||
|
const mask = bitMask(field.bit_width) << shift;
|
||||||
|
raw = (raw & ~mask) | ((@as(u128, value) << shift) & mask);
|
||||||
|
k = 0;
|
||||||
|
while (k < byte_count) : (k += 1) {
|
||||||
|
try self.writeRegionByte(region.region_space, start_byte + k, @truncate(raw >> @intCast(k * 8)));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn regionBase(self: *Interpreter, region: *Node) Error!u64 {
|
||||||
|
var current = Cursor{ .b = region.region_offset_aml };
|
||||||
|
var frame = Frame{ .scope = region.parent orelse self.namespace.root };
|
||||||
|
return (try self.term(¤t, &frame)).asInteger();
|
||||||
|
}
|
||||||
|
|
||||||
|
fn readRegionByte(self: *Interpreter, space: u8, address: u64) Error!u8 {
|
||||||
|
switch (space) {
|
||||||
|
0 => { // SystemMemory
|
||||||
|
const virtual = self.hal.mapMmio(address & ~@as(u64, 0xFFF), 0x1000, true);
|
||||||
|
const p: *align(1) const volatile u8 = @ptrFromInt(virtual + (address & 0xFFF));
|
||||||
|
return p.*;
|
||||||
|
},
|
||||||
|
1 => return @truncate(self.hal.pioRead(1, @intCast(address & 0xFFFF))), // SystemIO
|
||||||
|
else => return error.Unsupported,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn writeRegionByte(self: *Interpreter, space: u8, address: u64, value: u8) Error!void {
|
||||||
|
switch (space) {
|
||||||
|
0 => {
|
||||||
|
const virtual = self.hal.mapMmio(address & ~@as(u64, 0xFFF), 0x1000, true);
|
||||||
|
const p: *align(1) volatile u8 = @ptrFromInt(virtual + (address & 0xFFF));
|
||||||
|
p.* = value;
|
||||||
|
},
|
||||||
|
1 => self.hal.pioWrite(1, @intCast(address & 0xFFFF), value),
|
||||||
|
else => return error.Unsupported,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- extended opcodes ---------------------------------------------------
|
||||||
|
|
||||||
|
fn ext(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||||
|
const e = try current.byte();
|
||||||
|
switch (e) {
|
||||||
|
opcode.extended.debug => return .uninitialized,
|
||||||
|
opcode.extended.revision => return .{ .integer = 2 },
|
||||||
|
opcode.extended.timer => return .{ .integer = 0 },
|
||||||
|
// Mutex/Event ops are no-ops in this single-threaded evaluator.
|
||||||
|
opcode.extended.acquire => {
|
||||||
|
_ = try self.term(current, frame); // mutex SuperName
|
||||||
|
_ = try current.take(2); // timeout
|
||||||
|
return .{ .integer = 0 }; // acquired
|
||||||
|
},
|
||||||
|
opcode.extended.release, opcode.extended.reset, opcode.extended.signal => {
|
||||||
|
_ = try self.term(current, frame);
|
||||||
|
return .uninitialized;
|
||||||
|
},
|
||||||
|
opcode.extended.wait => {
|
||||||
|
_ = try self.term(current, frame);
|
||||||
|
_ = try self.term(current, frame);
|
||||||
|
return .{ .integer = 0 };
|
||||||
|
},
|
||||||
|
opcode.extended.sleep, opcode.extended.stall => {
|
||||||
|
_ = try self.term(current, frame);
|
||||||
|
return .uninitialized;
|
||||||
|
},
|
||||||
|
else => return error.Unsupported,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn evaluateInteger(self: *Interpreter, current: *Cursor, frame: *Frame) Error!u64 {
|
||||||
|
return (try self.term(current, frame)).asInteger();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
fn bitMask(width: u32) u128 {
|
||||||
|
if (width >= 128) return ~@as(u128, 0);
|
||||||
|
return (@as(u128, 1) << @intCast(width)) - 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn isNameStart(b: u8) bool {
|
||||||
|
return (b >= opcode.name_char_start and b <= opcode.name_char_end) or
|
||||||
|
b == opcode.name_char_underscore or
|
||||||
|
b == opcode.root_char or
|
||||||
|
b == opcode.parent_prefix_char or
|
||||||
|
b == opcode.dual_name_prefix or
|
||||||
|
b == opcode.multi_name_prefix;
|
||||||
|
}
|
||||||
@@ -18,7 +18,7 @@ pub const NodeKind = enum {
|
|||||||
mutex,
|
mutex,
|
||||||
event,
|
event,
|
||||||
processor,
|
processor,
|
||||||
power_res,
|
power_resource,
|
||||||
thermal_zone,
|
thermal_zone,
|
||||||
alias,
|
alias,
|
||||||
external,
|
external,
|
||||||
@@ -28,12 +28,12 @@ pub const NodeKind = enum {
|
|||||||
pub const Node = struct {
|
pub const Node = struct {
|
||||||
/// The 4-byte NameSeg identifying this node within its parent. The root uses
|
/// The 4-byte NameSeg identifying this node within its parent. The root uses
|
||||||
/// all-zero.
|
/// all-zero.
|
||||||
seg: [4]u8 = .{ 0, 0, 0, 0 },
|
segment: [4]u8 = .{ 0, 0, 0, 0 },
|
||||||
kind: NodeKind = .other,
|
kind: NodeKind = .other,
|
||||||
/// For Method / External: the declared argument count (0..7). Used to resolve
|
/// For Method / External: the declared argument count (0..7). Used to resolve
|
||||||
/// how many TermArgs a method invocation consumes.
|
/// how many TermArgs a method invocation consumes.
|
||||||
arg_count: u8 = 0,
|
arg_count: u8 = 0,
|
||||||
/// For Name: the AML bytes of its DataRefObject (so a value like a sleep
|
/// For Name: the AML bytes of its DataReferenceObject (so a value like a sleep
|
||||||
/// state's (`_Sx`) Package can be parsed on demand). For Method: the AML bytes of the body,
|
/// state's (`_Sx`) Package can be parsed on demand). For Method: the AML bytes of the body,
|
||||||
/// interpreted on demand by the evaluator. Empty otherwise.
|
/// interpreted on demand by the evaluator. Empty otherwise.
|
||||||
value: []const u8 = &.{},
|
value: []const u8 = &.{},
|
||||||
@@ -78,49 +78,49 @@ pub const Namespace = struct {
|
|||||||
return self.root.subtreeCount();
|
return self.root.subtreeCount();
|
||||||
}
|
}
|
||||||
|
|
||||||
fn findChild(parent: *Node, seg: [4]u8) ?*Node {
|
fn findChild(parent: *Node, segment: [4]u8) ?*Node {
|
||||||
var c = parent.first_child;
|
var c = parent.first_child;
|
||||||
while (c) |child| : (c = child.next_sibling) {
|
while (c) |child| : (c = child.next_sibling) {
|
||||||
if (std.mem.eql(u8, &child.seg, &seg)) return child;
|
if (std.mem.eql(u8, &child.segment, &segment)) return child;
|
||||||
}
|
}
|
||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The direct child of `node` named `seg`, or null. Unlike `resolve`, this does
|
/// The direct child of `node` named `segment`, or null. Unlike `resolve`, this does
|
||||||
/// not apply the search-rule walk-up — it looks only at immediate children (for
|
/// not apply the search-rule walk-up — it looks only at immediate children (for
|
||||||
/// reading a device's own hardware ID (`_HID`) / current resource settings (`_CRS`)).
|
/// reading a device's own hardware ID (`_HID`) / current resource settings (`_CRS`)).
|
||||||
pub fn childOf(node: *Node, seg: [4]u8) ?*Node {
|
pub fn childOf(node: *Node, segment: [4]u8) ?*Node {
|
||||||
return findChild(node, seg);
|
return findChild(node, segment);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn newChild(self: *Namespace, parent: *Node, seg: [4]u8, kind: NodeKind) !*Node {
|
fn newChild(self: *Namespace, parent: *Node, segment: [4]u8, kind: NodeKind) !*Node {
|
||||||
const n = try self.allocator.create(Node);
|
const n = try self.allocator.create(Node);
|
||||||
n.* = .{ .seg = seg, .kind = kind, .parent = parent };
|
n.* = .{ .segment = segment, .kind = kind, .parent = parent };
|
||||||
// Append at the tail so a dump reads in declaration order.
|
// Append at the tail so a dump reads in declaration order.
|
||||||
if (parent.first_child == null) {
|
if (parent.first_child == null) {
|
||||||
parent.first_child = n;
|
parent.first_child = n;
|
||||||
} else {
|
} else {
|
||||||
var cur = parent.first_child.?;
|
var current = parent.first_child.?;
|
||||||
while (cur.next_sibling) |sib| cur = sib;
|
while (current.next_sibling) |sib| current = sib;
|
||||||
cur.next_sibling = n;
|
current.next_sibling = n;
|
||||||
}
|
}
|
||||||
return n;
|
return n;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Create a Field unit node directly under `scope` (field units live in the
|
/// Create a Field unit node directly under `scope` (field units live in the
|
||||||
/// scope of the Field/IndexField/BankField, not under the region).
|
/// scope of the Field/IndexField/BankField, not under the region).
|
||||||
pub fn newFieldUnit(self: *Namespace, scope: *Node, seg: [4]u8) !*Node {
|
pub fn newFieldUnit(self: *Namespace, scope: *Node, segment: [4]u8) !*Node {
|
||||||
return self.findOrCreate(scope, seg, .field);
|
return self.findOrCreate(scope, segment, .field);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn findOrCreate(self: *Namespace, parent: *Node, seg: [4]u8, kind: NodeKind) !*Node {
|
fn findOrCreate(self: *Namespace, parent: *Node, segment: [4]u8, kind: NodeKind) !*Node {
|
||||||
if (findChild(parent, seg)) |existing| {
|
if (findChild(parent, segment)) |existing| {
|
||||||
// Reopening a scope (e.g. Scope(\_SB) after Device \_SB) keeps the more
|
// Reopening a scope (e.g. Scope(\_SB) after Device \_SB) keeps the more
|
||||||
// specific kind rather than downgrading to a plain scope.
|
// specific kind rather than downgrading to a plain scope.
|
||||||
if (existing.kind == .scope and kind != .scope) existing.kind = kind;
|
if (existing.kind == .scope and kind != .scope) existing.kind = kind;
|
||||||
return existing;
|
return existing;
|
||||||
}
|
}
|
||||||
return self.newChild(parent, seg, kind);
|
return self.newChild(parent, segment, kind);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The node a definition's NameString names, creating any intermediate scopes.
|
/// The node a definition's NameString names, creating any intermediate scopes.
|
||||||
@@ -131,16 +131,16 @@ pub const Namespace = struct {
|
|||||||
current: *Node,
|
current: *Node,
|
||||||
rooted: bool,
|
rooted: bool,
|
||||||
parents: u8,
|
parents: u8,
|
||||||
segs: []const [4]u8,
|
segments: []const [4]u8,
|
||||||
kind: NodeKind,
|
kind: NodeKind,
|
||||||
) !*Node {
|
) !*Node {
|
||||||
var base = startNode(self, current, rooted, parents);
|
var base = startNode(self, current, rooted, parents);
|
||||||
if (segs.len == 0) return base;
|
if (segments.len == 0) return base;
|
||||||
var i: usize = 0;
|
var i: usize = 0;
|
||||||
while (i + 1 < segs.len) : (i += 1) {
|
while (i + 1 < segments.len) : (i += 1) {
|
||||||
base = try self.findOrCreate(base, segs[i], .scope);
|
base = try self.findOrCreate(base, segments[i], .scope);
|
||||||
}
|
}
|
||||||
return self.findOrCreate(base, segs[segs.len - 1], kind);
|
return self.findOrCreate(base, segments[segments.len - 1], kind);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Resolve a NameString *reference* to an existing node, or null. A single
|
/// Resolve a NameString *reference* to an existing node, or null. A single
|
||||||
@@ -151,22 +151,22 @@ pub const Namespace = struct {
|
|||||||
current: *Node,
|
current: *Node,
|
||||||
rooted: bool,
|
rooted: bool,
|
||||||
parents: u8,
|
parents: u8,
|
||||||
segs: []const [4]u8,
|
segments: []const [4]u8,
|
||||||
) ?*Node {
|
) ?*Node {
|
||||||
if (segs.len == 0) return null;
|
if (segments.len == 0) return null;
|
||||||
|
|
||||||
if (!rooted and parents == 0 and segs.len == 1) {
|
if (!rooted and parents == 0 and segments.len == 1) {
|
||||||
// Search rule: this scope, then each ancestor up to the root.
|
// Search rule: this scope, then each ancestor up to the root.
|
||||||
var scope: ?*Node = current;
|
var scope: ?*Node = current;
|
||||||
while (scope) |s| : (scope = s.parent) {
|
while (scope) |s| : (scope = s.parent) {
|
||||||
if (findChild(s, segs[0])) |n| return n;
|
if (findChild(s, segments[0])) |n| return n;
|
||||||
}
|
}
|
||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
|
||||||
var base = startNode(self, current, rooted, parents);
|
var base = startNode(self, current, rooted, parents);
|
||||||
for (segs) |seg| {
|
for (segments) |segment| {
|
||||||
base = findChild(base, seg) orelse return null;
|
base = findChild(base, segment) orelse return null;
|
||||||
}
|
}
|
||||||
return base;
|
return base;
|
||||||
}
|
}
|
||||||
@@ -0,0 +1,137 @@
|
|||||||
|
//! AML opcode constants — the full ACPI Machine Language opcode table.
|
||||||
|
//!
|
||||||
|
//! Single-byte opcodes are plain values. Extended opcodes are a two-byte sequence
|
||||||
|
//! `ext_prefix` (0x5B) followed by a byte listed under `ext`. A few comparison
|
||||||
|
//! opcodes are `lnot_opcode` (0x92) followed by a second byte (see `lnot`).
|
||||||
|
|
||||||
|
// --- name / path characters -------------------------------------------------
|
||||||
|
pub const zero_opcode = 0x00;
|
||||||
|
pub const one_opcode = 0x01;
|
||||||
|
pub const alias_opcode = 0x06;
|
||||||
|
pub const name_opcode = 0x08;
|
||||||
|
pub const byte_prefix = 0x0A;
|
||||||
|
pub const word_prefix = 0x0B;
|
||||||
|
pub const dword_prefix = 0x0C;
|
||||||
|
pub const string_prefix = 0x0D;
|
||||||
|
pub const qword_prefix = 0x0E;
|
||||||
|
pub const scope_opcode = 0x10;
|
||||||
|
pub const buffer_opcode = 0x11;
|
||||||
|
pub const package_opcode = 0x12;
|
||||||
|
pub const var_package_opcode = 0x13;
|
||||||
|
pub const method_opcode = 0x14;
|
||||||
|
pub const external_opcode = 0x15;
|
||||||
|
|
||||||
|
pub const dual_name_prefix = 0x2E;
|
||||||
|
pub const multi_name_prefix = 0x2F;
|
||||||
|
pub const extended_opcode_prefix = 0x5B;
|
||||||
|
pub const root_char = 0x5C;
|
||||||
|
pub const parent_prefix_char = 0x5E;
|
||||||
|
pub const name_char_underscore = 0x5F;
|
||||||
|
|
||||||
|
pub const digit_char_start = 0x30;
|
||||||
|
pub const digit_char_end = 0x39;
|
||||||
|
pub const name_char_start = 0x41; // 'A'
|
||||||
|
pub const name_char_end = 0x5A; // 'Z'
|
||||||
|
|
||||||
|
// --- locals / args ----------------------------------------------------------
|
||||||
|
pub const local0_opcode = 0x60;
|
||||||
|
pub const local7_opcode = 0x67;
|
||||||
|
pub const arg0_opcode = 0x68;
|
||||||
|
pub const arg6_opcode = 0x6E;
|
||||||
|
|
||||||
|
// --- store / references / arithmetic ---------------------------------------
|
||||||
|
pub const store_opcode = 0x70;
|
||||||
|
pub const ref_of_opcode = 0x71;
|
||||||
|
pub const add_opcode = 0x72;
|
||||||
|
pub const concat_opcode = 0x73;
|
||||||
|
pub const subtract_opcode = 0x74;
|
||||||
|
pub const increment_opcode = 0x75;
|
||||||
|
pub const decrement_opcode = 0x76;
|
||||||
|
pub const multiply_opcode = 0x77;
|
||||||
|
pub const divide_opcode = 0x78;
|
||||||
|
pub const shift_left_opcode = 0x79;
|
||||||
|
pub const shift_right_opcode = 0x7A;
|
||||||
|
pub const and_opcode = 0x7B;
|
||||||
|
pub const nand_opcode = 0x7C;
|
||||||
|
pub const or_opcode = 0x7D;
|
||||||
|
pub const nor_opcode = 0x7E;
|
||||||
|
pub const xor_opcode = 0x7F;
|
||||||
|
pub const not_opcode = 0x80;
|
||||||
|
pub const find_set_left_bit_opcode = 0x81;
|
||||||
|
pub const find_set_right_bit_opcode = 0x82;
|
||||||
|
pub const dereference_of_opcode = 0x83;
|
||||||
|
pub const concat_resource_opcode = 0x84;
|
||||||
|
pub const mod_opcode = 0x85;
|
||||||
|
pub const notify_opcode = 0x86;
|
||||||
|
pub const size_of_opcode = 0x87;
|
||||||
|
pub const index_opcode = 0x88;
|
||||||
|
pub const match_opcode = 0x89;
|
||||||
|
pub const create_dword_field_opcode = 0x8A;
|
||||||
|
pub const create_word_field_opcode = 0x8B;
|
||||||
|
pub const create_byte_field_opcode = 0x8C;
|
||||||
|
pub const create_bit_field_opcode = 0x8D;
|
||||||
|
pub const object_type_opcode = 0x8E;
|
||||||
|
pub const create_qword_field_opcode = 0x8F;
|
||||||
|
|
||||||
|
pub const land_opcode = 0x90;
|
||||||
|
pub const lor_opcode = 0x91;
|
||||||
|
pub const lnot_opcode = 0x92; // may be followed by a second byte (see `lnot`)
|
||||||
|
pub const lequal_opcode = 0x93;
|
||||||
|
pub const lgreater_opcode = 0x94;
|
||||||
|
pub const lless_opcode = 0x95;
|
||||||
|
pub const to_buffer_opcode = 0x96;
|
||||||
|
pub const to_decimal_string_opcode = 0x97;
|
||||||
|
pub const to_hex_string_opcode = 0x98;
|
||||||
|
pub const to_integer_opcode = 0x99;
|
||||||
|
pub const to_string_opcode = 0x9C;
|
||||||
|
pub const copy_object_opcode = 0x9D;
|
||||||
|
pub const mid_opcode = 0x9E;
|
||||||
|
pub const continue_opcode = 0x9F;
|
||||||
|
pub const if_opcode = 0xA0;
|
||||||
|
pub const else_opcode = 0xA1;
|
||||||
|
pub const while_opcode = 0xA2;
|
||||||
|
pub const noop_opcode = 0xA3;
|
||||||
|
pub const return_opcode = 0xA4;
|
||||||
|
pub const break_opcode = 0xA5;
|
||||||
|
pub const break_point_opcode = 0xCC;
|
||||||
|
pub const ones_opcode = 0xFF;
|
||||||
|
|
||||||
|
/// Second bytes of the `lnot_opcode` (0x92) compound comparison opcodes.
|
||||||
|
pub const lnot = struct {
|
||||||
|
pub const not_equal = 0x93; // LNotEqualOp: 0x92 0x93
|
||||||
|
pub const less_equal = 0x94; // LLessEqualOp: 0x92 0x94
|
||||||
|
pub const greater_equal = 0x95; // LGreaterEqualOp: 0x92 0x95
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Second bytes of extended opcodes (prefixed by `extended_opcode_prefix`, 0x5B).
|
||||||
|
pub const extended = struct {
|
||||||
|
pub const mutex = 0x01;
|
||||||
|
pub const event = 0x02;
|
||||||
|
pub const conditional_reference_of = 0x12;
|
||||||
|
pub const create_field = 0x13;
|
||||||
|
pub const load_table = 0x1F;
|
||||||
|
pub const load = 0x20;
|
||||||
|
pub const stall = 0x21;
|
||||||
|
pub const sleep = 0x22;
|
||||||
|
pub const acquire = 0x23;
|
||||||
|
pub const signal = 0x24;
|
||||||
|
pub const wait = 0x25;
|
||||||
|
pub const reset = 0x26;
|
||||||
|
pub const release = 0x27;
|
||||||
|
pub const from_bcd = 0x28;
|
||||||
|
pub const to_bcd = 0x29;
|
||||||
|
pub const unload = 0x2A;
|
||||||
|
pub const revision = 0x30;
|
||||||
|
pub const debug = 0x31;
|
||||||
|
pub const fatal = 0x32;
|
||||||
|
pub const timer = 0x33;
|
||||||
|
pub const operation_region = 0x80;
|
||||||
|
pub const field = 0x81;
|
||||||
|
pub const device = 0x82;
|
||||||
|
pub const processor = 0x83;
|
||||||
|
pub const power_resource = 0x84;
|
||||||
|
pub const thermal_zone = 0x85;
|
||||||
|
pub const index_field = 0x86;
|
||||||
|
pub const bank_field = 0x87;
|
||||||
|
pub const data_region = 0x88;
|
||||||
|
};
|
||||||
@@ -0,0 +1,517 @@
|
|||||||
|
//! Recursive-descent AML parser. Walks the entire byte stream — including method
|
||||||
|
//! bodies — building the ACPI namespace as it goes. It does not *evaluate*
|
||||||
|
//! anything (no OperationRegion reads, no arithmetic); it parses structure so the
|
||||||
|
//! cursor stays aligned and every named object is recorded.
|
||||||
|
//!
|
||||||
|
//! The one genuine ambiguity in AML is method invocation: a bare NameString in an
|
||||||
|
//! operand position is a call whose argument count is only known from the method's
|
||||||
|
//! (earlier) declaration. Because we build the namespace in the same in-order pass,
|
||||||
|
//! `resolve` finds that declaration and tells us how many operands to consume.
|
||||||
|
//!
|
||||||
|
//! Safety net: every object delimited by a PkgLength (Scope/Device/Method/If/While/
|
||||||
|
//! Field/Buffer/Package/…) is parsed within its known extent, and the cursor is
|
||||||
|
//! snapped to that extent afterwards. So a mis-resolved invocation can only desync
|
||||||
|
//! *within* one such object; the enclosing walk realigns at the boundary.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const opcode = @import("opcodes.zig");
|
||||||
|
const Namespace = @import("namespace.zig").Namespace;
|
||||||
|
const Node = @import("namespace.zig").Node;
|
||||||
|
const NodeKind = @import("namespace.zig").NodeKind;
|
||||||
|
|
||||||
|
pub const Error = error{ Truncated, Malformed } || std.mem.Allocator.Error;
|
||||||
|
|
||||||
|
const maximum_segments = 64;
|
||||||
|
|
||||||
|
/// A parsed NameString: an optional root anchor or some parent hops, then a list
|
||||||
|
/// of 4-byte segments.
|
||||||
|
const NamePath = struct {
|
||||||
|
rooted: bool = false,
|
||||||
|
parents: u8 = 0,
|
||||||
|
segments: [maximum_segments][4]u8 = undefined,
|
||||||
|
count: usize = 0,
|
||||||
|
|
||||||
|
fn slice(self: *const NamePath) []const [4]u8 {
|
||||||
|
return self.segments[0..self.count];
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const Parser = struct {
|
||||||
|
aml: []const u8,
|
||||||
|
position: usize = 0,
|
||||||
|
namespace: *Namespace,
|
||||||
|
|
||||||
|
pub fn init(aml: []const u8, namespace: *Namespace) Parser {
|
||||||
|
return .{ .aml = aml, .namespace = namespace };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Parse the whole block as a TermList under the namespace root. Returns the
|
||||||
|
/// number of bytes consumed — equal to `aml.len` for a clean full traversal.
|
||||||
|
pub fn parseAll(self: *Parser) usize {
|
||||||
|
self.termList(self.aml.len, self.namespace.root);
|
||||||
|
return self.position;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- cursor primitives --------------------------------------------------
|
||||||
|
|
||||||
|
fn eof(self: *Parser) bool {
|
||||||
|
return self.position >= self.aml.len;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn peek(self: *Parser) ?u8 {
|
||||||
|
return if (self.eof()) null else self.aml[self.position];
|
||||||
|
}
|
||||||
|
|
||||||
|
fn readByte(self: *Parser) Error!u8 {
|
||||||
|
if (self.eof()) return error.Truncated;
|
||||||
|
const b = self.aml[self.position];
|
||||||
|
self.position += 1;
|
||||||
|
return b;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn skip(self: *Parser, n: usize) Error!void {
|
||||||
|
if (self.position + n > self.aml.len) return error.Truncated;
|
||||||
|
self.position += n;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn skipCString(self: *Parser) Error!void {
|
||||||
|
while (true) {
|
||||||
|
const b = try self.readByte();
|
||||||
|
if (b == 0) return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// AML PkgLength: the lead byte's top two bits give how many extra bytes
|
||||||
|
/// follow; the value counts from the start of the PkgLength field.
|
||||||
|
fn readPackageLength(self: *Parser) Error!usize {
|
||||||
|
const lead = try self.readByte();
|
||||||
|
const follow: usize = lead >> 6;
|
||||||
|
if (follow == 0) return lead & 0x3F;
|
||||||
|
var value: usize = lead & 0x0F;
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < follow) : (i += 1) {
|
||||||
|
const b = try self.readByte();
|
||||||
|
value |= @as(usize, b) << @intCast(4 + i * 8);
|
||||||
|
}
|
||||||
|
return value;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn readNameSegment(self: *Parser) Error![4]u8 {
|
||||||
|
if (self.position + 4 > self.aml.len) return error.Truncated;
|
||||||
|
const segment = self.aml[self.position..][0..4].*;
|
||||||
|
self.position += 4;
|
||||||
|
return segment;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn readNameString(self: *Parser) Error!NamePath {
|
||||||
|
var name_path = NamePath{};
|
||||||
|
// A NameString is either root-anchored or parent-relative, not both.
|
||||||
|
if (self.peek() == opcode.root_char) {
|
||||||
|
name_path.rooted = true;
|
||||||
|
self.position += 1;
|
||||||
|
} else {
|
||||||
|
while (self.peek() == opcode.parent_prefix_char) : (self.position += 1) name_path.parents += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
const lead = self.peek() orelse return name_path;
|
||||||
|
switch (lead) {
|
||||||
|
0x00 => self.position += 1, // NullName
|
||||||
|
opcode.dual_name_prefix => {
|
||||||
|
self.position += 1;
|
||||||
|
try self.appendSegment(&name_path);
|
||||||
|
try self.appendSegment(&name_path);
|
||||||
|
},
|
||||||
|
opcode.multi_name_prefix => {
|
||||||
|
self.position += 1;
|
||||||
|
const count = try self.readByte();
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < count) : (i += 1) try self.appendSegment(&name_path);
|
||||||
|
},
|
||||||
|
else => {
|
||||||
|
if (isNameStart(lead)) try self.appendSegment(&name_path);
|
||||||
|
},
|
||||||
|
}
|
||||||
|
return name_path;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn appendSegment(self: *Parser, name_path: *NamePath) Error!void {
|
||||||
|
const segment = try self.readNameSegment();
|
||||||
|
if (name_path.count < maximum_segments) {
|
||||||
|
name_path.segments[name_path.count] = segment;
|
||||||
|
name_path.count += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- term list / object -------------------------------------------------
|
||||||
|
|
||||||
|
/// Parse objects until `end`, then snap to `end`. Any parse error resyncs to
|
||||||
|
/// the boundary rather than propagating — containment for the rare desync.
|
||||||
|
fn termList(self: *Parser, end: usize, scope: *Node) void {
|
||||||
|
while (self.position < end) {
|
||||||
|
self.object(scope) catch break;
|
||||||
|
}
|
||||||
|
self.position = end;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Parse exactly one object/term at the cursor. Used for both TermObjs and
|
||||||
|
/// operands (TermArg / SuperName / Target all reduce to "one object" for the
|
||||||
|
/// purpose of advancing the cursor).
|
||||||
|
fn object(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const lead = self.peek() orelse return error.Truncated;
|
||||||
|
if (isNameStart(lead)) return self.nameInvocation(scope);
|
||||||
|
|
||||||
|
_ = try self.readByte();
|
||||||
|
switch (lead) {
|
||||||
|
// constants and no-operand statements
|
||||||
|
opcode.zero_opcode, opcode.one_opcode, opcode.ones_opcode => {},
|
||||||
|
opcode.noop_opcode, opcode.continue_opcode, opcode.break_opcode, opcode.break_point_opcode => {},
|
||||||
|
opcode.local0_opcode...opcode.local7_opcode => {},
|
||||||
|
opcode.arg0_opcode...opcode.arg6_opcode => {},
|
||||||
|
|
||||||
|
// literal data
|
||||||
|
opcode.byte_prefix => try self.skip(1),
|
||||||
|
opcode.word_prefix => try self.skip(2),
|
||||||
|
opcode.dword_prefix => try self.skip(4),
|
||||||
|
opcode.qword_prefix => try self.skip(8),
|
||||||
|
opcode.string_prefix => try self.skipCString(),
|
||||||
|
|
||||||
|
// data containers (contents skipped via their PkgLength)
|
||||||
|
opcode.buffer_opcode, opcode.package_opcode, opcode.var_package_opcode => try self.skipPackage(),
|
||||||
|
|
||||||
|
// namespace modifiers / named objects
|
||||||
|
opcode.name_opcode => try self.parseName(scope),
|
||||||
|
opcode.alias_opcode => try self.parseAlias(scope),
|
||||||
|
opcode.scope_opcode => try self.parseScopeLike(scope, .scope),
|
||||||
|
opcode.method_opcode => try self.parseMethod(scope),
|
||||||
|
opcode.external_opcode => try self.parseExternal(scope),
|
||||||
|
opcode.extended_opcode_prefix => try self.parseExtended(scope),
|
||||||
|
|
||||||
|
// control flow
|
||||||
|
opcode.if_opcode => try self.parseIf(scope),
|
||||||
|
opcode.else_opcode => try self.parseElse(scope),
|
||||||
|
opcode.while_opcode => try self.parseWhile(scope),
|
||||||
|
opcode.return_opcode => try self.object(scope),
|
||||||
|
opcode.notify_opcode => try self.args(scope, 2),
|
||||||
|
|
||||||
|
// stores / references / unary+target
|
||||||
|
opcode.store_opcode => try self.args(scope, 2),
|
||||||
|
opcode.ref_of_opcode, opcode.dereference_of_opcode, opcode.size_of_opcode, opcode.object_type_opcode => try self.args(scope, 1),
|
||||||
|
opcode.increment_opcode, opcode.decrement_opcode => try self.args(scope, 1),
|
||||||
|
opcode.not_opcode, opcode.find_set_left_bit_opcode, opcode.find_set_right_bit_opcode => try self.args(scope, 2),
|
||||||
|
opcode.to_buffer_opcode, opcode.to_decimal_string_opcode, opcode.to_hex_string_opcode, opcode.to_integer_opcode => try self.args(scope, 2),
|
||||||
|
opcode.copy_object_opcode => try self.args(scope, 2),
|
||||||
|
|
||||||
|
// binary + target
|
||||||
|
opcode.add_opcode, opcode.subtract_opcode, opcode.multiply_opcode, opcode.mod_opcode => try self.args(scope, 3),
|
||||||
|
opcode.and_opcode, opcode.nand_opcode, opcode.or_opcode, opcode.nor_opcode, opcode.xor_opcode => try self.args(scope, 3),
|
||||||
|
opcode.shift_left_opcode, opcode.shift_right_opcode, opcode.concat_opcode, opcode.concat_resource_opcode, opcode.index_opcode => try self.args(scope, 3),
|
||||||
|
opcode.divide_opcode => try self.args(scope, 4),
|
||||||
|
opcode.to_string_opcode => try self.args(scope, 3),
|
||||||
|
opcode.mid_opcode => try self.args(scope, 4),
|
||||||
|
|
||||||
|
// logical
|
||||||
|
opcode.land_opcode, opcode.lor_opcode => try self.args(scope, 2),
|
||||||
|
opcode.lequal_opcode, opcode.lgreater_opcode, opcode.lless_opcode => try self.args(scope, 2),
|
||||||
|
opcode.lnot_opcode => try self.parseLnot(scope),
|
||||||
|
|
||||||
|
opcode.match_opcode => try self.parseMatch(scope),
|
||||||
|
|
||||||
|
// CreateXField: <source> <index> NameString
|
||||||
|
opcode.create_dword_field_opcode,
|
||||||
|
opcode.create_word_field_opcode,
|
||||||
|
opcode.create_byte_field_opcode,
|
||||||
|
opcode.create_bit_field_opcode,
|
||||||
|
opcode.create_qword_field_opcode,
|
||||||
|
=> try self.parseCreateField(scope, 2),
|
||||||
|
|
||||||
|
else => return error.Malformed,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Parse `n` operands.
|
||||||
|
fn args(self: *Parser, scope: *Node, n: usize) Error!void {
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < n) : (i += 1) try self.object(scope);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A NameString in operand/statement position: a method invocation (consuming
|
||||||
|
/// the callee's declared argument count) or a plain name reference.
|
||||||
|
fn nameInvocation(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
if (self.namespace.resolve(scope, name_path.rooted, name_path.parents, name_path.slice())) |node| {
|
||||||
|
if ((node.kind == .method or node.kind == .external) and node.arg_count > 0) {
|
||||||
|
try self.args(scope, node.arg_count);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Skip a PkgLength-delimited body wholesale (Buffer / Package / VarPackage):
|
||||||
|
/// the contents are pure data, never namespace declarations.
|
||||||
|
fn skipPackage(self: *Parser) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const len = try self.readPackageLength();
|
||||||
|
const end = start + len;
|
||||||
|
if (end > self.aml.len) return error.Truncated;
|
||||||
|
self.position = end;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- namespace objects --------------------------------------------------
|
||||||
|
|
||||||
|
fn parseName(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
const value_start = self.position;
|
||||||
|
try self.object(scope); // the DataReferenceObject value
|
||||||
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .name);
|
||||||
|
node.value = self.aml[value_start..self.position];
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseAlias(self: *Parser, scope: *Node) Error!void {
|
||||||
|
_ = try self.readNameString(); // source
|
||||||
|
const name_path = try self.readNameString(); // the alias name
|
||||||
|
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .alias);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseMethod(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
const flags = try self.readByte();
|
||||||
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .method);
|
||||||
|
node.arg_count = flags & 0x7;
|
||||||
|
// Capture the body for on-demand evaluation and skip it — objects declared
|
||||||
|
// inside a method are created at *runtime*, not at load, so they must not
|
||||||
|
// become permanent namespace nodes.
|
||||||
|
node.value = self.aml[self.position..@min(end, self.aml.len)];
|
||||||
|
self.position = end;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseExternal(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
_ = try self.readByte(); // object type
|
||||||
|
const arg_count = try self.readByte();
|
||||||
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .external);
|
||||||
|
node.arg_count = arg_count;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Scope / Device / ThermalZone: PkgLength, NameString, then a nested TermList.
|
||||||
|
fn parseScopeLike(self: *Parser, scope: *Node, kind: NodeKind) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), kind);
|
||||||
|
self.termList(end, node);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseProcessor(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
try self.skip(6); // ProcID(byte) + PblkAddress(dword) + PblkLen(byte)
|
||||||
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .processor);
|
||||||
|
self.termList(end, node);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parsePowerResource(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
try self.skip(3); // SystemLevel(byte) + ResourceOrder(word)
|
||||||
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .power_resource);
|
||||||
|
self.termList(end, node);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// OperationRegion: NameString, RegionSpace(byte), Offset(TermArg), Len(TermArg).
|
||||||
|
/// The offset/length expressions are kept as AML for lazy evaluation.
|
||||||
|
fn parseRegion(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
const space = try self.readByte();
|
||||||
|
const off_start = self.position;
|
||||||
|
try self.object(scope);
|
||||||
|
const off_end = self.position;
|
||||||
|
try self.object(scope);
|
||||||
|
const len_end = self.position;
|
||||||
|
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .region);
|
||||||
|
node.region_space = space;
|
||||||
|
node.region_offset_aml = self.aml[off_start..off_end];
|
||||||
|
node.region_len_aml = self.aml[off_end..len_end];
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseDataRegion(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
try self.args(scope, 3); // signature, oem id, oem table id (TermArgs)
|
||||||
|
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .region);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseMutex(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
try self.skip(1); // sync flags
|
||||||
|
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .mutex);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseEvent(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .event);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// CreateXField: `count` TermArgs then the new field's NameString.
|
||||||
|
fn parseCreateField(self: *Parser, scope: *Node, count: usize) Error!void {
|
||||||
|
try self.args(scope, count);
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .name);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Field / IndexField / BankField: a region/bank reference, flags, then a
|
||||||
|
/// FieldList whose NamedFields become nodes in the current scope. For a plain
|
||||||
|
/// Field, the first NameString is the backing region — captured so field units
|
||||||
|
/// carry a region + bit position the evaluator can read/write.
|
||||||
|
fn parseField(self: *Parser, scope: *Node, name_strings: u8, bank: bool) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
var region: ?*Node = null;
|
||||||
|
var i: u8 = 0;
|
||||||
|
while (i < name_strings) : (i += 1) {
|
||||||
|
const name_path = try self.readNameString();
|
||||||
|
// Only a plain Field's single NameString denotes an OperationRegion.
|
||||||
|
if (name_strings == 1) region = self.namespace.resolve(scope, name_path.rooted, name_path.parents, name_path.slice());
|
||||||
|
}
|
||||||
|
if (bank) try self.object(scope); // bank value TermArg
|
||||||
|
const flags = try self.readByte();
|
||||||
|
self.fieldList(end, scope, region, flags & 0x0F);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn fieldList(self: *Parser, end: usize, scope: *Node, region: ?*Node, initial_access: u8) void {
|
||||||
|
var bit_offset: u32 = 0;
|
||||||
|
var access = initial_access;
|
||||||
|
while (self.position < end) {
|
||||||
|
const lead = self.peek() orelse break;
|
||||||
|
switch (lead) {
|
||||||
|
0x00 => { // ReservedField: advances the bit position
|
||||||
|
self.position += 1;
|
||||||
|
const width = self.readPackageLength() catch break;
|
||||||
|
bit_offset += @intCast(width);
|
||||||
|
},
|
||||||
|
0x01 => { // AccessField: AccessType (low nibble) + AccessAttrib
|
||||||
|
self.position += 1;
|
||||||
|
const at = self.readByte() catch break;
|
||||||
|
self.skip(1) catch break;
|
||||||
|
access = at & 0x0F;
|
||||||
|
},
|
||||||
|
0x02 => { // ConnectField: NameString | BufferData
|
||||||
|
self.position += 1;
|
||||||
|
self.object(scope) catch break;
|
||||||
|
},
|
||||||
|
0x03 => { // ExtendedAccessField: type + attrib + length
|
||||||
|
self.position += 1;
|
||||||
|
self.skip(3) catch break;
|
||||||
|
},
|
||||||
|
else => { // NamedField: NameSegment + PkgLength (bit width)
|
||||||
|
const segment = self.readNameSegment() catch break;
|
||||||
|
const width = self.readPackageLength() catch break;
|
||||||
|
const unit = self.namespace.newFieldUnit(scope, segment) catch break;
|
||||||
|
unit.region = region;
|
||||||
|
unit.bit_offset = bit_offset;
|
||||||
|
unit.bit_width = @intCast(width);
|
||||||
|
unit.access_type = access;
|
||||||
|
bit_offset += @intCast(width);
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
self.position = end;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- control flow -------------------------------------------------------
|
||||||
|
|
||||||
|
fn parseIf(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
try self.object(scope); // predicate
|
||||||
|
self.termList(end, scope);
|
||||||
|
if (self.peek() == opcode.else_opcode) {
|
||||||
|
self.position += 1;
|
||||||
|
try self.parseElse(scope);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseElse(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
self.termList(end, scope);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseWhile(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const start = self.position;
|
||||||
|
const end = start + try self.readPackageLength();
|
||||||
|
try self.object(scope); // predicate
|
||||||
|
self.termList(end, scope);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseLnot(self: *Parser, scope: *Node) Error!void {
|
||||||
|
// 0x92 followed by 0x93/94/95 is a compound comparison (two operands);
|
||||||
|
// otherwise it is a plain LNot of one operand.
|
||||||
|
const b = self.peek() orelse return error.Truncated;
|
||||||
|
switch (b) {
|
||||||
|
opcode.lnot.not_equal, opcode.lnot.less_equal, opcode.lnot.greater_equal => {
|
||||||
|
self.position += 1;
|
||||||
|
try self.args(scope, 2);
|
||||||
|
},
|
||||||
|
else => try self.object(scope),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parseMatch(self: *Parser, scope: *Node) Error!void {
|
||||||
|
try self.object(scope); // search package
|
||||||
|
try self.skip(1); // match opcode 1
|
||||||
|
try self.object(scope); // operand 1
|
||||||
|
try self.skip(1); // match opcode 2
|
||||||
|
try self.object(scope); // operand 2
|
||||||
|
try self.object(scope); // start index
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- extended opcodes (0x5B xx) -----------------------------------------
|
||||||
|
|
||||||
|
fn parseExtended(self: *Parser, scope: *Node) Error!void {
|
||||||
|
const e = try self.readByte();
|
||||||
|
switch (e) {
|
||||||
|
opcode.extended.mutex => try self.parseMutex(scope),
|
||||||
|
opcode.extended.event => try self.parseEvent(scope),
|
||||||
|
opcode.extended.operation_region => try self.parseRegion(scope),
|
||||||
|
opcode.extended.data_region => try self.parseDataRegion(scope),
|
||||||
|
opcode.extended.field => try self.parseField(scope, 1, false),
|
||||||
|
opcode.extended.index_field => try self.parseField(scope, 2, false),
|
||||||
|
opcode.extended.bank_field => try self.parseField(scope, 2, true),
|
||||||
|
opcode.extended.device => try self.parseScopeLike(scope, .device),
|
||||||
|
opcode.extended.thermal_zone => try self.parseScopeLike(scope, .thermal_zone),
|
||||||
|
opcode.extended.processor => try self.parseProcessor(scope),
|
||||||
|
opcode.extended.power_resource => try self.parsePowerResource(scope),
|
||||||
|
|
||||||
|
opcode.extended.conditional_reference_of => try self.args(scope, 2), // SuperName, Target
|
||||||
|
opcode.extended.create_field => try self.parseCreateField(scope, 3),
|
||||||
|
opcode.extended.load_table => try self.args(scope, 6),
|
||||||
|
opcode.extended.load => try self.args(scope, 2), // NameString, Target
|
||||||
|
opcode.extended.stall, opcode.extended.sleep => try self.args(scope, 1),
|
||||||
|
opcode.extended.acquire => {
|
||||||
|
try self.object(scope); // mutex SuperName
|
||||||
|
try self.skip(2); // timeout WordData
|
||||||
|
},
|
||||||
|
opcode.extended.signal, opcode.extended.reset, opcode.extended.release, opcode.extended.unload => try self.args(scope, 1),
|
||||||
|
opcode.extended.wait => try self.args(scope, 2),
|
||||||
|
opcode.extended.from_bcd, opcode.extended.to_bcd => try self.args(scope, 2),
|
||||||
|
opcode.extended.fatal => {
|
||||||
|
try self.skip(5); // Type(byte) + Code(dword)
|
||||||
|
try self.object(scope); // Arg TermArg
|
||||||
|
},
|
||||||
|
opcode.extended.revision, opcode.extended.debug, opcode.extended.timer => {},
|
||||||
|
|
||||||
|
else => return error.Malformed,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
fn isNameStart(b: u8) bool {
|
||||||
|
return (b >= opcode.name_char_start and b <= opcode.name_char_end) or
|
||||||
|
b == opcode.name_char_underscore or
|
||||||
|
b == opcode.root_char or
|
||||||
|
b == opcode.parent_prefix_char or
|
||||||
|
b == opcode.dual_name_prefix or
|
||||||
|
b == opcode.multi_name_prefix;
|
||||||
|
}
|
||||||
@@ -0,0 +1,85 @@
|
|||||||
|
//! The **device ABI**: the flat, `extern` device types that cross the system_call
|
||||||
|
//! boundary — what `device_enumerate` hands a user-space driver, what
|
||||||
|
//! `device_register` takes back. This is the devices sub-project's *public
|
||||||
|
//! interface*, exposed as its own `device-abi` module the same way the VFS server
|
||||||
|
//! exposes `vfs-protocol` — so both the kernel and user space depend on the contract
|
||||||
|
//! by name, and neither reaches into the other's files.
|
||||||
|
//!
|
||||||
|
//! It is also the **single source of truth** for `DeviceClass` and `ResourceKind`:
|
||||||
|
//! the kernel's rich, pointer-based device tree (system/devices/device-model.zig,
|
||||||
|
//! which user space must never import) re-exports these, so the enum that a driver
|
||||||
|
//! matches on and the enum the kernel classifies with are the *same* type — no
|
||||||
|
//! hand-kept "mirror in order" to drift. The core kernel↔user ABI is [[abi]]; the
|
||||||
|
//! loader↔kernel handoff is [[boot-handoff]].
|
||||||
|
|
||||||
|
/// A coarse classification of a device, independent of the describing firmware.
|
||||||
|
/// Kept small on purpose; refine as real drivers arrive. `enum(u32)` because the
|
||||||
|
/// `@intFromEnum` value crosses the system_call boundary in `DeviceDescriptor.class`.
|
||||||
|
pub const DeviceClass = enum(u32) {
|
||||||
|
/// The synthetic root every discovered device hangs beneath.
|
||||||
|
root,
|
||||||
|
processor,
|
||||||
|
interrupt_controller,
|
||||||
|
timer,
|
||||||
|
/// A PCI(e) host bridge — the root of a PCI segment (owns an ECAM window).
|
||||||
|
pci_host_bridge,
|
||||||
|
/// A single PCI function.
|
||||||
|
pci_device,
|
||||||
|
/// A device named in the ACPI namespace (from the DSDT/SSDT), carrying a
|
||||||
|
/// hardware ID (`_HID`) and, where static, current resource settings (`_CRS`).
|
||||||
|
acpi_device,
|
||||||
|
unknown,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The kind of hardware resource a device occupies. `enum(u32)` for the same
|
||||||
|
/// boundary-crossing reason as `DeviceClass` (see `ResourceDescriptor.kind`).
|
||||||
|
pub const ResourceKind = enum(u32) {
|
||||||
|
/// A memory-mapped I/O window: `start` is the physical base, `len` its size.
|
||||||
|
memory,
|
||||||
|
/// A legacy I/O-port range: `start` is the first port, `len` the count.
|
||||||
|
io_port,
|
||||||
|
/// An interrupt: `start` is the global system interrupt (GSI), `len` is 1.
|
||||||
|
irq,
|
||||||
|
/// A range of bus numbers owned by a bridge: `start`..`start+len`.
|
||||||
|
bus_range,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One device resource, as handed to a user-space driver (flat, extern).
|
||||||
|
pub const ResourceDescriptor = extern struct {
|
||||||
|
kind: u64, // a ResourceKind value
|
||||||
|
start: u64,
|
||||||
|
len: u64,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const maximum_device_resources = 8;
|
||||||
|
|
||||||
|
/// `DeviceDescriptor.parent` for a device with no parent — a root of the device tree.
|
||||||
|
pub const no_parent: u64 = ~@as(u64, 0);
|
||||||
|
|
||||||
|
/// `DeviceDescriptor.pci_class` for a device that is not a PCI function. (Zero would be
|
||||||
|
/// ambiguous: 0x000000 is a real class code, "unclassified device".)
|
||||||
|
pub const no_pci_class: u64 = ~@as(u64, 0);
|
||||||
|
|
||||||
|
/// A device, as snapshotted for user space by `device_enumerate`. A driver scans
|
||||||
|
/// these to find the hardware it owns, claims it, and maps its MMIO.
|
||||||
|
///
|
||||||
|
/// `parent` makes the table a tree rather than a list, which is what a **bus driver**
|
||||||
|
/// needs: it claims the bus, finds the devices below it, and publishes any it
|
||||||
|
/// discovers itself with `device_register`. A registered child's resources must lie
|
||||||
|
/// within its parent's (the kernel enforces this) — that containment is what makes
|
||||||
|
/// delegation safe, since a device descriptor is otherwise a licence to map physical
|
||||||
|
/// memory.
|
||||||
|
pub const DeviceDescriptor = extern struct {
|
||||||
|
id: u64,
|
||||||
|
parent: u64, // a device id, or `no_parent`
|
||||||
|
class: u64, // a DeviceClass value
|
||||||
|
// The PCI class/subclass/prog-IF triple packed as 0xCCSSPP when this device is a PCI
|
||||||
|
// function, or `no_pci_class` otherwise. This is how a manager tells *what* a
|
||||||
|
// `pci_device` is (an xHCI controller, an AHCI controller) — decode the triple into
|
||||||
|
// names with the pci-class module.
|
||||||
|
pci_class: u64,
|
||||||
|
hid_len: u64,
|
||||||
|
resource_count: u64,
|
||||||
|
hid: [8]u8,
|
||||||
|
resources: [maximum_device_resources]ResourceDescriptor,
|
||||||
|
};
|
||||||
@@ -3,7 +3,7 @@
|
|||||||
//! Discovery backends (ACPI today, device-tree later) translate their native
|
//! Discovery backends (ACPI today, device-tree later) translate their native
|
||||||
//! hardware description into this one shape, so the rest of the kernel walks a
|
//! hardware description into this one shape, so the rest of the kernel walks a
|
||||||
//! plain `Device` tree without knowing which firmware described the machine —
|
//! plain `Device` tree without knowing which firmware described the machine —
|
||||||
//! the same discipline `root.zig`'s `MemoryKind` applies to memory and `arch`
|
//! the same discipline `ps2-library.zig`'s `MemoryKind` applies to memory and `architecture`
|
||||||
//! applies to the CPU.
|
//! applies to the CPU.
|
||||||
//!
|
//!
|
||||||
//! This is deliberately minimal: enough to *describe* what was discovered (a
|
//! This is deliberately minimal: enough to *describe* what was discovered (a
|
||||||
@@ -12,28 +12,27 @@
|
|||||||
//! on top of this — nothing here presumes them.
|
//! on top of this — nothing here presumes them.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
|
const device_abi = @import("device-abi");
|
||||||
|
const pci_class = @import("pci-class");
|
||||||
|
const acpi_ids = @import("acpi-ids");
|
||||||
|
|
||||||
/// The hardware primitives a discovery backend needs but can't express portably.
|
/// The hardware primitives a discovery backend needs but can't express portably.
|
||||||
/// The kernel injects an implementation (the arch VMM + port I/O), so the device
|
/// The kernel injects an implementation (the architecture VMM + port I/O), so the device
|
||||||
/// layer touches hardware without importing `arch` — the same discipline that lets
|
/// layer touches hardware without importing `architecture` — the same discipline that lets
|
||||||
/// it stay firmware-agnostic. `pioRead`/`pioWrite` take a width in bytes (1/2/4).
|
/// it stay firmware-agnostic. `pioRead`/`pioWrite` take a width in bytes (1/2/4).
|
||||||
pub const Hal = struct {
|
pub const Hal = struct {
|
||||||
mapMmio: *const fn (virt: u64, phys: u64, writable: bool) void,
|
/// Map a physical MMIO range and return the virtual address to reach it at.
|
||||||
|
/// The device layer dereferences the returned address and never learns how
|
||||||
|
/// the kernel places it (identity, physmap, a window — the kernel's choice).
|
||||||
|
mapMmio: *const fn (physical: u64, len: u64, writable: bool) u64,
|
||||||
pioRead: *const fn (width: u8, port: u16) u32,
|
pioRead: *const fn (width: u8, port: u16) u32,
|
||||||
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
||||||
};
|
};
|
||||||
|
|
||||||
/// The kind of hardware resource a device occupies.
|
/// The kind of hardware resource a device occupies. Canonically defined by the
|
||||||
pub const ResourceKind = enum {
|
/// device ABI (system/devices/device-abi.zig) and re-exported here, so the kernel's
|
||||||
/// A memory-mapped I/O window: `start` is the physical base, `len` its size.
|
/// internal tree and the descriptors it hands user space share one enum.
|
||||||
memory,
|
pub const ResourceKind = device_abi.ResourceKind;
|
||||||
/// A legacy I/O-port range: `start` is the first port, `len` the count.
|
|
||||||
io_port,
|
|
||||||
/// An interrupt: `start` is the global system interrupt (GSI), `len` is 1.
|
|
||||||
irq,
|
|
||||||
/// A range of bus numbers owned by a bridge: `start`..`start+len`.
|
|
||||||
bus_range,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// One hardware resource claimed by a device.
|
/// One hardware resource claimed by a device.
|
||||||
pub const Resource = struct {
|
pub const Resource = struct {
|
||||||
@@ -43,22 +42,10 @@ pub const Resource = struct {
|
|||||||
};
|
};
|
||||||
|
|
||||||
/// A coarse classification of a device, independent of the describing firmware.
|
/// A coarse classification of a device, independent of the describing firmware.
|
||||||
/// Kept small on purpose; refine as real drivers arrive.
|
/// Canonically defined by the device ABI (system/devices/device-abi.zig) and
|
||||||
pub const DeviceClass = enum {
|
/// re-exported here — the enum a driver matches on and the one the kernel classifies
|
||||||
/// The synthetic root every discovered device hangs beneath.
|
/// with are the same type. Kept small on purpose; refine as real drivers arrive.
|
||||||
root,
|
pub const DeviceClass = device_abi.DeviceClass;
|
||||||
processor,
|
|
||||||
interrupt_controller,
|
|
||||||
timer,
|
|
||||||
/// A PCI(e) host bridge — the root of a PCI segment (owns an ECAM window).
|
|
||||||
pci_host_bridge,
|
|
||||||
/// A single PCI function.
|
|
||||||
pci_device,
|
|
||||||
/// A device named in the ACPI namespace (from the DSDT/SSDT), carrying a
|
|
||||||
/// hardware ID (`_HID`) and, where static, current resource settings (`_CRS`).
|
|
||||||
acpi_device,
|
|
||||||
unknown,
|
|
||||||
};
|
|
||||||
|
|
||||||
/// Firmware-independent identity. Each backend fills only the fields it knows;
|
/// Firmware-independent identity. Each backend fills only the fields it knows;
|
||||||
/// the rest stay null. The generic layer never branches on *how* an id was
|
/// the rest stay null. The generic layer never branches on *how* an id was
|
||||||
@@ -71,29 +58,29 @@ pub const Ids = struct {
|
|||||||
pci_device: ?u16 = null,
|
pci_device: ?u16 = null,
|
||||||
/// PCI class/subclass/prog-if packed as 0xCCSSPP.
|
/// PCI class/subclass/prog-if packed as 0xCCSSPP.
|
||||||
pci_class: ?u24 = null,
|
pci_class: ?u24 = null,
|
||||||
/// PCI bus/device/function packed as (bus << 8) | (dev << 3) | func — the key
|
/// PCI bus/device/function packed as (bus << 8) | (device << 3) | function — the key
|
||||||
/// the ACPI address (`_ADR`) merge uses to match a namespace device to this node.
|
/// the ACPI address (`_ADR`) merge uses to match a namespace device to this node.
|
||||||
pci_bdf: ?u16 = null,
|
pci_bdf: ?u16 = null,
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Upper bound on resources tracked per device (6 PCI BARs + a couple of IRQs is
|
/// Upper bound on resources tracked per device (6 PCI BARs + a couple of IRQs is
|
||||||
/// the busy case). Stored inline so a device is a single allocation.
|
/// the busy case). Stored inline so a device is a single allocation.
|
||||||
pub const max_resources = 8;
|
pub const maximum_resources = 8;
|
||||||
|
|
||||||
/// One node in the device tree. Nodes are individually heap-allocated and linked
|
/// One node in the device tree. Nodes are individually heap-allocated and linked
|
||||||
/// intrusively (first-child / next-sibling), the classic device-tree layout —
|
/// intrusively (first-child / next-sibling), the classic device-tree layout —
|
||||||
/// no per-node dynamic arrays to manage.
|
/// no per-node dynamic arrays to manage.
|
||||||
pub const Device = struct {
|
pub const Device = struct {
|
||||||
name_buf: [24]u8 = undefined,
|
name_buffer: [24]u8 = undefined,
|
||||||
name_len: u8 = 0,
|
name_len: u8 = 0,
|
||||||
class: DeviceClass = .unknown,
|
class: DeviceClass = .unknown,
|
||||||
ids: Ids = .{},
|
ids: Ids = .{},
|
||||||
/// Human-readable hardware id (e.g. "PNP0A03"), when known. Backed inline like
|
/// Human-readable hardware id (e.g. "PNP0A03"), when known. Backed inline like
|
||||||
/// `name`; empty when unset. The generic layer stores/prints it without knowing
|
/// `name`; empty when unset. The generic layer stores/prints it without knowing
|
||||||
/// how a backend encoded it.
|
/// how a backend encoded it.
|
||||||
hid_buf: [8]u8 = undefined,
|
hid_buffer: [8]u8 = undefined,
|
||||||
hid_len: u8 = 0,
|
hid_len: u8 = 0,
|
||||||
resources: [max_resources]Resource = undefined,
|
resources: [maximum_resources]Resource = undefined,
|
||||||
resource_count: u8 = 0,
|
resource_count: u8 = 0,
|
||||||
|
|
||||||
parent: ?*Device = null,
|
parent: ?*Device = null,
|
||||||
@@ -103,30 +90,30 @@ pub const Device = struct {
|
|||||||
/// The device's short name (e.g. "cpu0", "pci0:00:1f.0"). Backed by an inline
|
/// The device's short name (e.g. "cpu0", "pci0:00:1f.0"). Backed by an inline
|
||||||
/// buffer, so it stays valid for the life of the node with no extra allocation.
|
/// buffer, so it stays valid for the life of the node with no extra allocation.
|
||||||
pub fn name(self: *const Device) []const u8 {
|
pub fn name(self: *const Device) []const u8 {
|
||||||
return self.name_buf[0..self.name_len];
|
return self.name_buffer[0..self.name_len];
|
||||||
}
|
}
|
||||||
|
|
||||||
fn setName(self: *Device, s: []const u8) void {
|
fn setName(self: *Device, s: []const u8) void {
|
||||||
const n: u8 = @intCast(@min(s.len, self.name_buf.len));
|
const n: u8 = @intCast(@min(s.len, self.name_buffer.len));
|
||||||
@memcpy(self.name_buf[0..n], s[0..n]);
|
@memcpy(self.name_buffer[0..n], s[0..n]);
|
||||||
self.name_len = n;
|
self.name_len = n;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The device's hardware id string, or empty if none is set.
|
/// The device's hardware id string, or empty if none is set.
|
||||||
pub fn hid(self: *const Device) []const u8 {
|
pub fn hid(self: *const Device) []const u8 {
|
||||||
return self.hid_buf[0..self.hid_len];
|
return self.hid_buffer[0..self.hid_len];
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn setHid(self: *Device, s: []const u8) void {
|
pub fn setHid(self: *Device, s: []const u8) void {
|
||||||
const n: u8 = @intCast(@min(s.len, self.hid_buf.len));
|
const n: u8 = @intCast(@min(s.len, self.hid_buffer.len));
|
||||||
@memcpy(self.hid_buf[0..n], s[0..n]);
|
@memcpy(self.hid_buffer[0..n], s[0..n]);
|
||||||
self.hid_len = n;
|
self.hid_len = n;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Record a resource. Silently drops beyond `max_resources` — discovery logs
|
/// Record a resource. Silently drops beyond `maximum_resources` — discovery logs
|
||||||
/// the truncation rather than failing the whole tree.
|
/// the truncation rather than failing the whole tree.
|
||||||
pub fn addResource(self: *Device, kind: ResourceKind, start: u64, len: u64) bool {
|
pub fn addResource(self: *Device, kind: ResourceKind, start: u64, len: u64) bool {
|
||||||
if (self.resource_count >= max_resources) return false;
|
if (self.resource_count >= maximum_resources) return false;
|
||||||
self.resources[self.resource_count] = .{ .kind = kind, .start = start, .len = len };
|
self.resources[self.resource_count] = .{ .kind = kind, .start = start, .len = len };
|
||||||
self.resource_count += 1;
|
self.resource_count += 1;
|
||||||
return true;
|
return true;
|
||||||
@@ -167,17 +154,17 @@ pub const DeviceTree = struct {
|
|||||||
self: *DeviceTree,
|
self: *DeviceTree,
|
||||||
parent: *Device,
|
parent: *Device,
|
||||||
class: DeviceClass,
|
class: DeviceClass,
|
||||||
dev_name: []const u8,
|
device_name: []const u8,
|
||||||
) !*Device {
|
) !*Device {
|
||||||
const d = try self.allocator.create(Device);
|
const d = try self.allocator.create(Device);
|
||||||
d.* = .{ .class = class, .parent = parent };
|
d.* = .{ .class = class, .parent = parent };
|
||||||
d.setName(dev_name);
|
d.setName(device_name);
|
||||||
if (parent.first_child == null) {
|
if (parent.first_child == null) {
|
||||||
parent.first_child = d;
|
parent.first_child = d;
|
||||||
} else {
|
} else {
|
||||||
var cur = parent.first_child.?;
|
var current = parent.first_child.?;
|
||||||
while (cur.next_sibling) |sib| cur = sib;
|
while (current.next_sibling) |sib| current = sib;
|
||||||
cur.next_sibling = d;
|
current.next_sibling = d;
|
||||||
}
|
}
|
||||||
return d;
|
return d;
|
||||||
}
|
}
|
||||||
@@ -199,18 +186,37 @@ fn firstOfClassIn(node: *Device, class: DeviceClass) ?*Device {
|
|||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn dumpNode(dev: *const Device, depth: usize, emit: *const fn ([]const u8) void) void {
|
fn dumpNode(device: *const Device, depth: usize, emit: *const fn ([]const u8) void) void {
|
||||||
const indent = @min(depth * 2, 40);
|
const indent = @min(depth * 2, 40);
|
||||||
|
|
||||||
var buf: [200]u8 = undefined;
|
var buffer: [200]u8 = undefined;
|
||||||
@memset(buf[0..indent], ' ');
|
@memset(buffer[0..indent], ' ');
|
||||||
const body = if (dev.hid_len != 0)
|
const body = if (device.hid_len != 0) blk: {
|
||||||
std.fmt.bufPrint(buf[indent..], "{s} [{s}] hid={s}\n", .{ dev.name(), @tagName(dev.class), dev.hid() }) catch return
|
// Decode the _HID to a human name when it's a known standard PnP/ACPI id.
|
||||||
|
const desc = acpi_ids.description(device.hid());
|
||||||
|
break :blk if (desc.len != 0)
|
||||||
|
std.fmt.bufPrint(buffer[indent..], "{s} [{s}] hid={s} ({s})\n", .{ device.name(), @tagName(device.class), device.hid(), desc }) catch return
|
||||||
else
|
else
|
||||||
std.fmt.bufPrint(buf[indent..], "{s} [{s}]\n", .{ dev.name(), @tagName(dev.class) }) catch return;
|
std.fmt.bufPrint(buffer[indent..], "{s} [{s}] hid={s}\n", .{ device.name(), @tagName(device.class), device.hid() }) catch return;
|
||||||
emit(buf[0 .. indent + body.len]);
|
} else std.fmt.bufPrint(buffer[indent..], "{s} [{s}]\n", .{ device.name(), @tagName(device.class) }) catch return;
|
||||||
|
emit(buffer[0 .. indent + body.len]);
|
||||||
|
|
||||||
for (dev.resources[0..dev.resource_count]) |r| {
|
// For a PCI function, decode its class code — the (class / subclass / prog-IF)
|
||||||
|
// triple that says what it actually is, which the coarse `DeviceClass` can't.
|
||||||
|
if (device.ids.pci_class) |packed_code| {
|
||||||
|
const cc = pci_class.ClassCode.unpack(packed_code);
|
||||||
|
var cbuf: [200]u8 = undefined;
|
||||||
|
const pad = @min(indent + 2, 42);
|
||||||
|
@memset(cbuf[0..pad], ' ');
|
||||||
|
const pif = pci_class.progIfName(cc.base, cc.subclass, cc.prog_if);
|
||||||
|
const cline = if (pif.len != 0)
|
||||||
|
std.fmt.bufPrint(cbuf[pad..], "class 0x{x:0>2} ({s}) subclass 0x{x:0>2} ({s}) progif 0x{x:0>2} ({s})\n", .{ cc.base, pci_class.className(cc.base), cc.subclass, pci_class.subclassName(cc.base, cc.subclass), cc.prog_if, pif }) catch return
|
||||||
|
else
|
||||||
|
std.fmt.bufPrint(cbuf[pad..], "class 0x{x:0>2} ({s}) subclass 0x{x:0>2} ({s}) progif 0x{x:0>2}\n", .{ cc.base, pci_class.className(cc.base), cc.subclass, pci_class.subclassName(cc.base, cc.subclass), cc.prog_if }) catch return;
|
||||||
|
emit(cbuf[0 .. pad + cline.len]);
|
||||||
|
}
|
||||||
|
|
||||||
|
for (device.resources[0..device.resource_count]) |r| {
|
||||||
var rbuf: [200]u8 = undefined;
|
var rbuf: [200]u8 = undefined;
|
||||||
const pad = @min(indent + 2, 42);
|
const pad = @min(indent + 2, 42);
|
||||||
@memset(rbuf[0..pad], ' ');
|
@memset(rbuf[0..pad], ' ');
|
||||||
@@ -222,6 +228,6 @@ fn dumpNode(dev: *const Device, depth: usize, emit: *const fn ([]const u8) void)
|
|||||||
emit(rbuf[0 .. pad + rline.len]);
|
emit(rbuf[0 .. pad + rline.len]);
|
||||||
}
|
}
|
||||||
|
|
||||||
var child = dev.first_child;
|
var child = device.first_child;
|
||||||
while (child) |c| : (child = c.next_sibling) dumpNode(c, depth + 1, emit);
|
while (child) |c| : (child = c.next_sibling) dumpNode(c, depth + 1, emit);
|
||||||
}
|
}
|
||||||
@@ -7,10 +7,10 @@
|
|||||||
//! already routes to a backend rather than hard-coding ACPI — wiring the FDT
|
//! already routes to a backend rather than hard-coding ACPI — wiring the FDT
|
||||||
//! parser in later is a local change here, not an architectural one.
|
//! parser in later is a local change here, not an architectural one.
|
||||||
|
|
||||||
const device = @import("device.zig");
|
const device_model = @import("device-model.zig");
|
||||||
|
|
||||||
/// Populate `dt` from a device-tree blob. Not implemented yet.
|
/// Populate `device_tree` from a device-tree blob. Not implemented yet.
|
||||||
pub fn discover(dt: *device.DeviceTree) !void {
|
pub fn discover(device_tree: *device_model.DeviceTree) !void {
|
||||||
_ = dt;
|
_ = device_tree;
|
||||||
return error.Unsupported;
|
return error.Unsupported;
|
||||||
}
|
}
|
||||||
@@ -0,0 +1,261 @@
|
|||||||
|
//! PCI class-code decoding: turn the (class, subclass, prog-IF) triple a PCI function
|
||||||
|
//! reports in its configuration header into human-readable names. Every PCI function
|
||||||
|
//! carries a 24-bit class code — base class (config byte 0x0B), subclass (0x0A), and
|
||||||
|
//! programming interface (0x09) — that says *what it is* far more precisely than
|
||||||
|
//! danos's coarse `DeviceClass`: an ISA bridge, a SATA/AHCI controller, and an xHCI USB
|
||||||
|
//! controller are all just `pci_device` by class, and only this triple tells them
|
||||||
|
//! apart. Pure reference data (from the PCI spec; see https://wiki.osdev.org/PCI) — no
|
||||||
|
//! hardware access — so it is shared by kernel discovery (the device-tree dump) and any
|
||||||
|
//! user-space tool (a future lspci, driver matching).
|
||||||
|
|
||||||
|
/// The three bytes of a PCI class code, unpacked from the `0xCCSSPP` value discovery
|
||||||
|
/// records in `Device.ids.pci_class` (CC = base class, SS = subclass, PP = prog-IF).
|
||||||
|
pub const ClassCode = struct {
|
||||||
|
base: u8, // class code (config offset 0x0B)
|
||||||
|
subclass: u8, // subclass (0x0A)
|
||||||
|
prog_if: u8, // programming interface (0x09)
|
||||||
|
|
||||||
|
pub fn unpack(packed_code: u24) ClassCode {
|
||||||
|
return .{
|
||||||
|
.base = @intCast((packed_code >> 16) & 0xFF),
|
||||||
|
.subclass = @intCast((packed_code >> 8) & 0xFF),
|
||||||
|
.prog_if = @intCast(packed_code & 0xFF),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Name of the base class (byte 0x0B), e.g. `0x06` -> "Bridge".
|
||||||
|
pub fn className(base: u8) []const u8 {
|
||||||
|
return switch (base) {
|
||||||
|
0x00 => "Unclassified",
|
||||||
|
0x01 => "Mass Storage Controller",
|
||||||
|
0x02 => "Network Controller",
|
||||||
|
0x03 => "Display Controller",
|
||||||
|
0x04 => "Multimedia Controller",
|
||||||
|
0x05 => "Memory Controller",
|
||||||
|
0x06 => "Bridge",
|
||||||
|
0x07 => "Simple Communication Controller",
|
||||||
|
0x08 => "Base System Peripheral",
|
||||||
|
0x09 => "Input Device Controller",
|
||||||
|
0x0A => "Docking Station",
|
||||||
|
0x0B => "Processor",
|
||||||
|
0x0C => "Serial Bus Controller",
|
||||||
|
0x0D => "Wireless Controller",
|
||||||
|
0x0E => "Intelligent Controller",
|
||||||
|
0x0F => "Satellite Communication Controller",
|
||||||
|
0x10 => "Encryption Controller",
|
||||||
|
0x11 => "Signal Processing Controller",
|
||||||
|
0x12 => "Processing Accelerator",
|
||||||
|
0x13 => "Non-Essential Instrumentation",
|
||||||
|
0x40 => "Co-Processor",
|
||||||
|
0xFF => "Unassigned Class (Vendor specific)",
|
||||||
|
else => "Unknown",
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Name of the subclass within its base class, e.g. `(0x06, 0x01)` -> "ISA Bridge".
|
||||||
|
/// Subclass `0x80` is "Other" by PCI convention; anything unlisted is "Unknown".
|
||||||
|
pub fn subclassName(base: u8, subclass: u8) []const u8 {
|
||||||
|
return switch (base) {
|
||||||
|
0x01 => switch (subclass) {
|
||||||
|
0x00 => "SCSI Bus Controller",
|
||||||
|
0x01 => "IDE Controller",
|
||||||
|
0x02 => "Floppy Disk Controller",
|
||||||
|
0x03 => "IPI Bus Controller",
|
||||||
|
0x04 => "RAID Controller",
|
||||||
|
0x05 => "ATA Controller",
|
||||||
|
0x06 => "Serial ATA Controller",
|
||||||
|
0x07 => "Serial Attached SCSI Controller",
|
||||||
|
0x08 => "Non-Volatile Memory Controller",
|
||||||
|
else => defaultSubclass(subclass),
|
||||||
|
},
|
||||||
|
0x02 => switch (subclass) {
|
||||||
|
0x00 => "Ethernet Controller",
|
||||||
|
0x01 => "Token Ring Controller",
|
||||||
|
0x02 => "FDDI Controller",
|
||||||
|
0x03 => "ATM Controller",
|
||||||
|
0x04 => "ISDN Controller",
|
||||||
|
0x06 => "PICMG 2.14 Multi Computing Controller",
|
||||||
|
0x07 => "Infiniband Controller",
|
||||||
|
0x08 => "Fabric Controller",
|
||||||
|
else => defaultSubclass(subclass),
|
||||||
|
},
|
||||||
|
0x03 => switch (subclass) {
|
||||||
|
0x00 => "VGA Compatible Controller",
|
||||||
|
0x01 => "XGA Controller",
|
||||||
|
0x02 => "3D Controller (Not VGA-Compatible)",
|
||||||
|
else => defaultSubclass(subclass),
|
||||||
|
},
|
||||||
|
0x04 => switch (subclass) {
|
||||||
|
0x00 => "Multimedia Video Controller",
|
||||||
|
0x01 => "Multimedia Audio Controller",
|
||||||
|
0x02 => "Computer Telephony Device",
|
||||||
|
0x03 => "Audio Device",
|
||||||
|
else => defaultSubclass(subclass),
|
||||||
|
},
|
||||||
|
0x05 => switch (subclass) {
|
||||||
|
0x00 => "RAM Controller",
|
||||||
|
0x01 => "Flash Controller",
|
||||||
|
else => defaultSubclass(subclass),
|
||||||
|
},
|
||||||
|
0x06 => switch (subclass) {
|
||||||
|
0x00 => "Host Bridge",
|
||||||
|
0x01 => "ISA Bridge",
|
||||||
|
0x02 => "EISA Bridge",
|
||||||
|
0x03 => "MCA Bridge",
|
||||||
|
0x04 => "PCI-to-PCI Bridge",
|
||||||
|
0x05 => "PCMCIA Bridge",
|
||||||
|
0x06 => "NuBus Bridge",
|
||||||
|
0x07 => "CardBus Bridge",
|
||||||
|
0x08 => "RACEway Bridge",
|
||||||
|
0x09 => "PCI-to-PCI Bridge (Semi-Transparent)",
|
||||||
|
0x0A => "InfiniBand-to-PCI Host Bridge",
|
||||||
|
else => defaultSubclass(subclass),
|
||||||
|
},
|
||||||
|
0x07 => switch (subclass) {
|
||||||
|
0x00 => "Serial Controller",
|
||||||
|
0x01 => "Parallel Controller",
|
||||||
|
0x02 => "Multiport Serial Controller",
|
||||||
|
0x03 => "Modem",
|
||||||
|
0x04 => "IEEE 488.1/2 (GPIB) Controller",
|
||||||
|
0x05 => "Smart Card Controller",
|
||||||
|
else => defaultSubclass(subclass),
|
||||||
|
},
|
||||||
|
0x08 => switch (subclass) {
|
||||||
|
0x00 => "PIC",
|
||||||
|
0x01 => "DMA Controller",
|
||||||
|
0x02 => "Timer",
|
||||||
|
0x03 => "RTC Controller",
|
||||||
|
0x04 => "PCI Hot-Plug Controller",
|
||||||
|
0x05 => "SD Host Controller",
|
||||||
|
0x06 => "IOMMU",
|
||||||
|
else => defaultSubclass(subclass),
|
||||||
|
},
|
||||||
|
0x09 => switch (subclass) {
|
||||||
|
0x00 => "Keyboard Controller",
|
||||||
|
0x01 => "Digitizer Pen",
|
||||||
|
0x02 => "Mouse Controller",
|
||||||
|
0x03 => "Scanner Controller",
|
||||||
|
0x04 => "Gameport Controller",
|
||||||
|
else => defaultSubclass(subclass),
|
||||||
|
},
|
||||||
|
0x0C => switch (subclass) {
|
||||||
|
0x00 => "FireWire (IEEE 1394) Controller",
|
||||||
|
0x01 => "ACCESS Bus Controller",
|
||||||
|
0x02 => "SSA",
|
||||||
|
0x03 => "USB Controller",
|
||||||
|
0x04 => "Fibre Channel",
|
||||||
|
0x05 => "SMBus Controller",
|
||||||
|
0x06 => "InfiniBand Controller",
|
||||||
|
0x07 => "IPMI Interface",
|
||||||
|
0x08 => "SERCOS Interface (IEC 61491)",
|
||||||
|
0x09 => "CANbus Controller",
|
||||||
|
else => defaultSubclass(subclass),
|
||||||
|
},
|
||||||
|
0x0D => switch (subclass) {
|
||||||
|
0x00 => "iRDA Compatible Controller",
|
||||||
|
0x01 => "Consumer IR Controller",
|
||||||
|
0x10 => "RF Controller",
|
||||||
|
0x11 => "Bluetooth Controller",
|
||||||
|
0x12 => "Broadband Controller",
|
||||||
|
0x20 => "Ethernet Controller (802.1a)",
|
||||||
|
0x21 => "Ethernet Controller (802.1b)",
|
||||||
|
else => defaultSubclass(subclass),
|
||||||
|
},
|
||||||
|
else => defaultSubclass(subclass),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
fn defaultSubclass(subclass: u8) []const u8 {
|
||||||
|
return if (subclass == 0x80) "Other" else "Unknown";
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Name of the programming interface, for the subclasses that define standard ones
|
||||||
|
/// (IDE modes, SATA/AHCI, NVMe, PCI-bridge decode, UART generation, USB host type).
|
||||||
|
/// Returns "" when the prog-IF carries no standard meaning for this class/subclass —
|
||||||
|
/// callers just print the hex byte in that case.
|
||||||
|
pub fn progIfName(base: u8, subclass: u8, prog_if: u8) []const u8 {
|
||||||
|
return switch (base) {
|
||||||
|
0x01 => switch (subclass) {
|
||||||
|
0x06 => switch (prog_if) { // Serial ATA
|
||||||
|
0x00 => "Vendor Specific Interface",
|
||||||
|
0x01 => "AHCI 1.0",
|
||||||
|
0x02 => "Serial Storage Bus",
|
||||||
|
else => "",
|
||||||
|
},
|
||||||
|
0x08 => switch (prog_if) { // Non-Volatile Memory
|
||||||
|
0x01 => "NVMHCI",
|
||||||
|
0x02 => "NVM Express",
|
||||||
|
else => "",
|
||||||
|
},
|
||||||
|
else => "",
|
||||||
|
},
|
||||||
|
0x03 => switch (subclass) {
|
||||||
|
0x00 => switch (prog_if) { // VGA Compatible
|
||||||
|
0x00 => "VGA Controller",
|
||||||
|
0x01 => "8514-Compatible Controller",
|
||||||
|
else => "",
|
||||||
|
},
|
||||||
|
else => "",
|
||||||
|
},
|
||||||
|
0x06 => switch (subclass) {
|
||||||
|
0x04 => switch (prog_if) { // PCI-to-PCI Bridge
|
||||||
|
0x00 => "Normal Decode",
|
||||||
|
0x01 => "Subtractive Decode",
|
||||||
|
else => "",
|
||||||
|
},
|
||||||
|
else => "",
|
||||||
|
},
|
||||||
|
0x07 => switch (subclass) {
|
||||||
|
0x00 => switch (prog_if) { // Serial Controller
|
||||||
|
0x00 => "8250-Compatible (Generic XT)",
|
||||||
|
0x01 => "16450-Compatible",
|
||||||
|
0x02 => "16550-Compatible",
|
||||||
|
0x03 => "16650-Compatible",
|
||||||
|
0x04 => "16750-Compatible",
|
||||||
|
0x05 => "16850-Compatible",
|
||||||
|
0x06 => "16950-Compatible",
|
||||||
|
else => "",
|
||||||
|
},
|
||||||
|
else => "",
|
||||||
|
},
|
||||||
|
0x0C => switch (subclass) {
|
||||||
|
0x03 => switch (prog_if) { // USB Controller
|
||||||
|
0x00 => "UHCI Controller",
|
||||||
|
0x10 => "OHCI Controller",
|
||||||
|
0x20 => "EHCI (USB2) Controller",
|
||||||
|
0x30 => "XHCI (USB3) Controller",
|
||||||
|
0x80 => "Unspecified",
|
||||||
|
0xFE => "USB Device (not a host controller)",
|
||||||
|
else => "",
|
||||||
|
},
|
||||||
|
else => "",
|
||||||
|
},
|
||||||
|
else => "",
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
test "decodes the common class codes" {
|
||||||
|
const std = @import("std");
|
||||||
|
const eq = std.testing.expectEqualStrings;
|
||||||
|
|
||||||
|
const isa = ClassCode.unpack(0x06_01_00);
|
||||||
|
try std.testing.expectEqual(@as(u8, 0x06), isa.base);
|
||||||
|
try std.testing.expectEqual(@as(u8, 0x01), isa.subclass);
|
||||||
|
try eq("Bridge", className(isa.base));
|
||||||
|
try eq("ISA Bridge", subclassName(isa.base, isa.subclass));
|
||||||
|
|
||||||
|
const ahci = ClassCode.unpack(0x01_06_01);
|
||||||
|
try eq("Mass Storage Controller", className(ahci.base));
|
||||||
|
try eq("Serial ATA Controller", subclassName(ahci.base, ahci.subclass));
|
||||||
|
try eq("AHCI 1.0", progIfName(ahci.base, ahci.subclass, ahci.prog_if));
|
||||||
|
|
||||||
|
const xhci = ClassCode.unpack(0x0C_03_30);
|
||||||
|
try eq("USB Controller", subclassName(xhci.base, xhci.subclass));
|
||||||
|
try eq("XHCI (USB3) Controller", progIfName(xhci.base, xhci.subclass, xhci.prog_if));
|
||||||
|
|
||||||
|
// Unknowns and the "Other" convention.
|
||||||
|
try eq("Other", subclassName(0x02, 0x80));
|
||||||
|
try eq("Unknown", subclassName(0x06, 0x7E));
|
||||||
|
try eq("", progIfName(0x06, 0x00, 0x00)); // host bridge: prog-IF has no standard name
|
||||||
|
}
|
||||||
@@ -0,0 +1,95 @@
|
|||||||
|
//! The firmware-agnostic discovery facade.
|
||||||
|
//!
|
||||||
|
//! The kernel calls `platform.discover()` and gets back a generic `DeviceTree`
|
||||||
|
//! without ever naming ACPI or device-tree — the same way it imports `architecture`
|
||||||
|
//! without naming x86_64. Which backend runs is decided *at runtime* from what
|
||||||
|
//! the bootloader handed us (an ACPI RSDP today, a device-tree blob later),
|
||||||
|
//! because a single image — a future ARM kernel especially — may boot under
|
||||||
|
//! either firmware. That's a deliberate divergence from `architecture`, which is a
|
||||||
|
//! compile-time choice.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const boot_handoff = @import("boot-handoff");
|
||||||
|
const device_model = @import("device-model.zig");
|
||||||
|
const acpi = @import("acpi.zig");
|
||||||
|
const power = @import("power.zig");
|
||||||
|
const devicetree = @import("device-tree.zig");
|
||||||
|
|
||||||
|
pub const DeviceTree = device_model.DeviceTree;
|
||||||
|
pub const Device = device_model.Device;
|
||||||
|
pub const DeviceClass = device_model.DeviceClass;
|
||||||
|
pub const Resource = device_model.Resource;
|
||||||
|
pub const ResourceKind = device_model.ResourceKind;
|
||||||
|
pub const Hal = device_model.Hal;
|
||||||
|
pub const PowerInformation = acpi.PowerInformation;
|
||||||
|
pub const AmlStats = acpi.AmlStats;
|
||||||
|
pub const PlatformInformation = acpi.PlatformInformation;
|
||||||
|
pub const RegisterAccess = acpi.RegisterAccess;
|
||||||
|
pub const IsoEntry = acpi.IsoEntry;
|
||||||
|
pub const Cpu = acpi.Cpu;
|
||||||
|
|
||||||
|
/// The register map + sleep types discovery extracted, for logging/diagnostics.
|
||||||
|
pub fn powerInformation() PowerInformation {
|
||||||
|
return acpi.power_information;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The scalar firmware facts the architecture layer needs to avoid legacy assumptions
|
||||||
|
/// (8259 presence, LAPIC base, PM timer, SPCR UART, IRQ overrides).
|
||||||
|
pub fn platformInformation() PlatformInformation {
|
||||||
|
return acpi.platform_information;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// AML parse integrity/diagnostics (namespace node count, bytes consumed).
|
||||||
|
pub fn amlStats() AmlStats {
|
||||||
|
return acpi.aml_stats;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The usable logical processors discovered during enumeration — one entry per
|
||||||
|
/// core danos may schedule on, each carrying the Local APIC ID an SMP wake targets.
|
||||||
|
/// `len` is the hardware's degree of parallelism: how many tasks *could* run at the
|
||||||
|
/// same instant once the application processors are started. Today only the
|
||||||
|
/// bootstrap processor is actually running, so starting the rest is the pending SMP
|
||||||
|
/// step (see docs/smp.md). Borrowed from static storage populated by `discover`.
|
||||||
|
pub fn cpus() []const Cpu {
|
||||||
|
return acpi.cpu_information.cpus[0..acpi.cpu_information.count];
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Non-zero only if enumeration found more processors than the static pool holds
|
||||||
|
/// (the surplus were dropped from `cpus()`); surfaced so the cap is never silent.
|
||||||
|
pub fn cpusDropped() usize {
|
||||||
|
return acpi.cpu_information.dropped;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Enumerate hardware into a fresh device tree. `hal` supplies the hardware
|
||||||
|
/// primitives the backend needs (MMIO mapping for PCIe configuration space, port I/O for
|
||||||
|
/// ACPI registers); pass the architecture implementation. Errors leave nothing to clean up
|
||||||
|
/// beyond the tree's own allocations.
|
||||||
|
pub fn discover(
|
||||||
|
boot_information: *const boot_handoff.BootInformation,
|
||||||
|
allocator: std.mem.Allocator,
|
||||||
|
hal: Hal,
|
||||||
|
) !DeviceTree {
|
||||||
|
var device_tree = try DeviceTree.init(allocator);
|
||||||
|
|
||||||
|
if (boot_information.acpi_rsdp != 0) {
|
||||||
|
try acpi.discover(boot_information.acpi_rsdp, &device_tree, hal);
|
||||||
|
} else {
|
||||||
|
// No ACPI RSDP. A device-tree boot would parse its blob here; today that
|
||||||
|
// path is a stub, so this reports the machine described itself no way we
|
||||||
|
// understand yet.
|
||||||
|
try devicetree.discover(&device_tree);
|
||||||
|
}
|
||||||
|
|
||||||
|
return device_tree;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Restart the machine. Never returns on success; returns only if no reset method
|
||||||
|
/// worked (extremely unlikely). Backend-agnostic entry the kernel calls.
|
||||||
|
pub fn reboot(hal: Hal) void {
|
||||||
|
power.reboot(hal);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Power the machine off (ACPI S5). Never returns on success.
|
||||||
|
pub fn shutdown(hal: Hal) void {
|
||||||
|
power.shutdown(hal);
|
||||||
|
}
|
||||||
@@ -9,8 +9,8 @@
|
|||||||
//! re-initialisation, a milestone of its own.
|
//! re-initialisation, a milestone of its own.
|
||||||
|
|
||||||
const acpi = @import("acpi.zig");
|
const acpi = @import("acpi.zig");
|
||||||
const device = @import("device.zig");
|
const device_model = @import("device-model.zig");
|
||||||
const Hal = device.Hal;
|
const Hal = device_model.Hal;
|
||||||
|
|
||||||
const slp_en: u32 = 1 << 13; // SLP_EN: writing 1 triggers the sleep transition
|
const slp_en: u32 = 1 << 13; // SLP_EN: writing 1 triggers the sleep transition
|
||||||
const sci_en: u32 = 1 << 0; // SCI_EN in PM1 control: set once ACPI mode is active
|
const sci_en: u32 = 1 << 0; // SCI_EN in PM1 control: set once ACPI mode is active
|
||||||
@@ -19,27 +19,27 @@ const sci_en: u32 = 1 << 0; // SCI_EN in PM1 control: set once ACPI mode is acti
|
|||||||
/// register is live. A no-op when the firmware exposes no SMI command port (ACPI
|
/// register is live. A no-op when the firmware exposes no SMI command port (ACPI
|
||||||
/// already enabled, as under QEMU/OVMF) — we still verify SCI_EN first.
|
/// already enabled, as under QEMU/OVMF) — we still verify SCI_EN first.
|
||||||
pub fn enable(hal: Hal) void {
|
pub fn enable(hal: Hal) void {
|
||||||
const pi = acpi.power_info;
|
const pi = acpi.power_information;
|
||||||
if (!pi.pm1a_cnt.present()) return;
|
if (!pi.pm1a_cnt.present()) return;
|
||||||
if (readReg(hal, pi.pm1a_cnt) & sci_en != 0) return; // already in ACPI mode
|
if (readRegister(hal, pi.pm1a_cnt) & sci_en != 0) return; // already in ACPI mode
|
||||||
if (pi.smi_cmd == 0 or pi.acpi_enable == 0) return; // no way to switch; assume fine
|
if (pi.smi_cmd == 0 or pi.acpi_enable == 0) return; // no way to switch; assume fine
|
||||||
|
|
||||||
hal.pioWrite(1, pi.smi_cmd, pi.acpi_enable);
|
hal.pioWrite(1, pi.smi_cmd, pi.acpi_enable);
|
||||||
var spins: usize = 0;
|
var spins: usize = 0;
|
||||||
while (readReg(hal, pi.pm1a_cnt) & sci_en == 0 and spins < 1_000_000) : (spins += 1) {}
|
while (readRegister(hal, pi.pm1a_cnt) & sci_en == 0 and spins < 1_000_000) : (spins += 1) {}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Restart the machine. Tries the ACPI reset register first, then the two legacy
|
/// Restart the machine. Tries the ACPI reset register first, then the two legacy
|
||||||
/// fallbacks. Returns only if every method failed (very unlikely).
|
/// fallbacks. Returns only if every method failed (very unlikely).
|
||||||
pub fn reboot(hal: Hal) void {
|
pub fn reboot(hal: Hal) void {
|
||||||
const pi = acpi.power_info;
|
const pi = acpi.power_information;
|
||||||
|
|
||||||
// 1. The FADT reset register, when the firmware advertises support.
|
// 1. The FADT reset register, when the firmware advertises support.
|
||||||
if (pi.reset_supported and pi.reset.present()) {
|
if (pi.reset_supported and pi.reset.present()) {
|
||||||
writeReg(hal, pi.reset, pi.reset_value);
|
writeRegister(hal, pi.reset, pi.reset_value);
|
||||||
delay();
|
delay();
|
||||||
}
|
}
|
||||||
// 2. The PCI reset-control register at port 0xCF9 (RST_CPU | SYS_RST).
|
// 2. The PCI reset-control register at port 0xCF9 (RST_CPU | SYSTEM_RST).
|
||||||
hal.pioWrite(1, 0xCF9, 0x0E);
|
hal.pioWrite(1, 0xCF9, 0x0E);
|
||||||
hal.pioWrite(1, 0xCF9, 0x06);
|
hal.pioWrite(1, 0xCF9, 0x06);
|
||||||
delay();
|
delay();
|
||||||
@@ -52,14 +52,14 @@ pub fn reboot(hal: Hal) void {
|
|||||||
/// it wasn't found in the AML, there is nothing safe to do and this returns.
|
/// it wasn't found in the AML, there is nothing safe to do and this returns.
|
||||||
pub fn shutdown(hal: Hal) void {
|
pub fn shutdown(hal: Hal) void {
|
||||||
enable(hal);
|
enable(hal);
|
||||||
const pi = acpi.power_info;
|
const pi = acpi.power_information;
|
||||||
const s5 = pi.s5 orelse return;
|
const s5 = pi.s5 orelse return;
|
||||||
|
|
||||||
if (pi.pm1a_cnt.present()) {
|
if (pi.pm1a_cnt.present()) {
|
||||||
writeReg(hal, pi.pm1a_cnt, sleepValue(s5.slp_typ_a));
|
writeRegister(hal, pi.pm1a_cnt, sleepValue(s5.slp_typ_a));
|
||||||
}
|
}
|
||||||
if (pi.pm1b_cnt.present()) {
|
if (pi.pm1b_cnt.present()) {
|
||||||
writeReg(hal, pi.pm1b_cnt, sleepValue(s5.slp_typ_b));
|
writeRegister(hal, pi.pm1b_cnt, sleepValue(s5.slp_typ_b));
|
||||||
}
|
}
|
||||||
delay();
|
delay();
|
||||||
}
|
}
|
||||||
@@ -76,27 +76,25 @@ fn sleepValue(slp_typ: u8) u32 {
|
|||||||
return (@as(u32, slp_typ & 0x7) << 10) | slp_en;
|
return (@as(u32, slp_typ & 0x7) << 10) | slp_en;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn readReg(hal: Hal, reg: acpi.RegAccess) u32 {
|
fn readRegister(hal: Hal, register: acpi.RegisterAccess) u32 {
|
||||||
if (reg.mmio) {
|
if (register.mmio) {
|
||||||
hal.mapMmio(reg.address, reg.address, true);
|
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true));
|
||||||
const p: *align(1) volatile u32 = @ptrFromInt(reg.address);
|
|
||||||
return p.*;
|
return p.*;
|
||||||
}
|
}
|
||||||
return hal.pioRead(reg.width, @intCast(reg.address));
|
return hal.pioRead(register.width, @intCast(register.address));
|
||||||
}
|
}
|
||||||
|
|
||||||
fn writeReg(hal: Hal, reg: acpi.RegAccess, value: u32) void {
|
fn writeRegister(hal: Hal, register: acpi.RegisterAccess, value: u32) void {
|
||||||
if (reg.mmio) {
|
if (register.mmio) {
|
||||||
hal.mapMmio(reg.address, reg.address, true);
|
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true));
|
||||||
const p: *align(1) volatile u32 = @ptrFromInt(reg.address);
|
|
||||||
p.* = value;
|
p.* = value;
|
||||||
} else {
|
} else {
|
||||||
hal.pioWrite(reg.width, @intCast(reg.address), value);
|
hal.pioWrite(register.width, @intCast(register.address), value);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A short busy-wait so a reset/power-off takes effect before we fall through to
|
/// A short busy-wait so a reset/power-off takes effect before we fall through to
|
||||||
/// the next method. The empty asm is an arch-neutral barrier that keeps the loop
|
/// the next method. The empty asm is an architecture-neutral barrier that keeps the loop
|
||||||
/// from being optimised away.
|
/// from being optimised away.
|
||||||
fn delay() void {
|
fn delay() void {
|
||||||
var i: usize = 0;
|
var i: usize = 0;
|
||||||
@@ -0,0 +1,857 @@
|
|||||||
|
//! USB device-framework wire ABI: the set-up packets, standard requests, and standard
|
||||||
|
//! descriptors every USB device speaks over its default control pipe, as defined by chapter 9
|
||||||
|
//! of the USB 2.0 specification (see https://wiki.osdev.org/Universal_Serial_Bus). Pure data
|
||||||
|
//! definitions — no hardware access — shared by the host-controller bus drivers (which build
|
||||||
|
//! the requests) and anything that parses what devices return (device naming, driver
|
||||||
|
//! matching, configuration). The structs mirror the wire byte-for-byte: multi-byte fields are
|
||||||
|
//! little-endian and align(1), so a descriptor can be bit-cast straight out of a transfer
|
||||||
|
//! buffer at any offset, and bitmap bytes are packed structs so no caller ever needs a magic
|
||||||
|
//! mask. Class, subclass, and protocol code tables live in usb-ids.zig.
|
||||||
|
|
||||||
|
const DeviceState = enum(u8) {
|
||||||
|
// Immediately after the USB device is attached to the USB system, it is in this state.
|
||||||
|
// The USB specifications do not define the state of a USB device that is detached from
|
||||||
|
// a USB system.
|
||||||
|
attached,
|
||||||
|
// A device is in this state after it has both been attached to the bus, and the VBUS line is
|
||||||
|
// applied to the device (the host controller drives the VBUS at +5V, however this is only
|
||||||
|
// particularly important for hardware developers). In this state, the device must not respond
|
||||||
|
// to any bus transactions. The USB specification recognizes three potential scenarios with
|
||||||
|
// respect to how a device draws power:
|
||||||
|
// - Self-Powered Devices draw power from an external power source (e.g, a USB printer plugs
|
||||||
|
// into the wall as well as a USB port). Although the device may be considered
|
||||||
|
// technically "powered" even before attachment to the USB, it is still only considered
|
||||||
|
// powered after the VBUS line is applied to the device.
|
||||||
|
// - Bus-Powered Devices draw power solely from the USB up to 100mA.
|
||||||
|
// - Self- or Bus-Powered Devices may draw power from either the bus or an external power
|
||||||
|
// source, depending on the configuration. These devices may change power source at any
|
||||||
|
// time. If a device is currently self-powered and requires more than 100mA of power, but
|
||||||
|
// switches to being bus-powered, then the device must return to the Address state.
|
||||||
|
powered,
|
||||||
|
// A device in the powered state enters the default state after receiving a bus reset. In this
|
||||||
|
// state, the device is addressable at the default, reserved address of 0. At this point, the
|
||||||
|
// device is operating at the correct speed. The host is expected to allow 10 milliseconds
|
||||||
|
// before expecting the device to respond to data transfers after reset.
|
||||||
|
default,
|
||||||
|
// A device enters this state after the host assigns it an address via the default control pipe,
|
||||||
|
// which is always accessible whether the device's address has been set or not.
|
||||||
|
address,
|
||||||
|
// A device is in this state after the host examines its possible configurations and selects
|
||||||
|
// one. All endpoint's data toggle bits are initialized to zero when a device enters this state.
|
||||||
|
configured,
|
||||||
|
// When no traffic is observed on the bus for a period of 1 millisecond, a USB device enters
|
||||||
|
// this state, characterized by its low power consumption. The device's address and
|
||||||
|
// configuration settings are maintained while suspended. A device exits the suspended state as
|
||||||
|
// soon as it begins seeing bus activity again. The host is expected to allow 10 milliseconds
|
||||||
|
// before expecting the device to respond to data transfers after resume.
|
||||||
|
suspended,
|
||||||
|
};
|
||||||
|
|
||||||
|
const RequestCode = enum(u8) {
|
||||||
|
get_status = 0,
|
||||||
|
clear_feature = 1,
|
||||||
|
set_feature = 3,
|
||||||
|
set_address = 5,
|
||||||
|
get_descriptor = 6,
|
||||||
|
set_descriptor = 7,
|
||||||
|
get_configuration = 8,
|
||||||
|
set_configuration = 9,
|
||||||
|
get_interface = 10,
|
||||||
|
set_interface = 11,
|
||||||
|
sync_frame = 12,
|
||||||
|
};
|
||||||
|
|
||||||
|
// Direction of an endpoint, from the host's point of view
|
||||||
|
const EndpointDirection = enum(u1) {
|
||||||
|
out = 0,
|
||||||
|
in = 1,
|
||||||
|
};
|
||||||
|
|
||||||
|
// Identifier newtypes: distinct wire-sized types for values that identify something on the
|
||||||
|
// device rather than count something. Each is a non-exhaustive enum whose values originate
|
||||||
|
// in the descriptors below and flow, still typed, into the standard request constructors —
|
||||||
|
// so an interface number can never be passed where a configuration value is expected.
|
||||||
|
|
||||||
|
// The bus address of a device, assigned by the host with SET_ADDRESS. Addresses are 7 bits
|
||||||
|
// wide.
|
||||||
|
const DeviceAddress = enum(u7) {
|
||||||
|
// The default address every device answers at after a reset, until SET_ADDRESS
|
||||||
|
// completes
|
||||||
|
default = 0,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
|
||||||
|
// Identifies a configuration; from ConfigurationDescriptor.configuration_value.
|
||||||
|
const ConfigurationValue = enum(u8) {
|
||||||
|
// Not configured: returned by GET_CONFIGURATION while the device is in the address
|
||||||
|
// state, and passed to SET_CONFIGURATION to return a configured device to the address
|
||||||
|
// state
|
||||||
|
none = 0,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
|
||||||
|
// Identifies an interface within a configuration; from
|
||||||
|
// InterfaceDescriptor.interface_number.
|
||||||
|
const InterfaceNumber = enum(u8) { _ };
|
||||||
|
|
||||||
|
// Selects between the alternate settings of one interface; from
|
||||||
|
// InterfaceDescriptor.alternate_setting.
|
||||||
|
const AlternateSetting = enum(u8) {
|
||||||
|
// The default setting of an interface
|
||||||
|
default = 0,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
|
||||||
|
// The number of an endpoint within a device, 4 bits wide. The direction bit carried
|
||||||
|
// alongside it tells the two endpoints sharing a number apart.
|
||||||
|
const EndpointNumber = enum(u4) {
|
||||||
|
// Endpoint zero: the default control pipe every device provides
|
||||||
|
default_control = 0,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
|
||||||
|
// Index of a STRING descriptor, stored in descriptors that reference a string and passed to
|
||||||
|
// GET_DESCRIPTOR to read it.
|
||||||
|
const StringIndex = enum(u8) {
|
||||||
|
// The device has no string descriptor for this field
|
||||||
|
none = 0,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
|
||||||
|
// Characteristics of a device request (the bmRequestType field of a set-up packet). Fields are
|
||||||
|
// declared least-significant first: recipient occupies bits 4...0, kind bits 6...5, and
|
||||||
|
// direction bit 7.
|
||||||
|
const RequestType = packed struct(u8) {
|
||||||
|
// The recipient of the request (values 4...31 are reserved)
|
||||||
|
recipient: Recipient,
|
||||||
|
// The type of the request
|
||||||
|
kind: Kind,
|
||||||
|
// Data transfer direction. The value of this bit is ignored when length is zero.
|
||||||
|
direction: Direction,
|
||||||
|
|
||||||
|
const Recipient = enum(u5) {
|
||||||
|
device = 0,
|
||||||
|
interface = 1,
|
||||||
|
endpoint = 2,
|
||||||
|
other = 3,
|
||||||
|
};
|
||||||
|
|
||||||
|
const Kind = enum(u2) {
|
||||||
|
standard = 0,
|
||||||
|
class = 1,
|
||||||
|
vendor = 2,
|
||||||
|
reserved = 3,
|
||||||
|
};
|
||||||
|
|
||||||
|
const Direction = enum(u1) {
|
||||||
|
host_to_device = 0,
|
||||||
|
device_to_host = 1,
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
const Request = extern struct {
|
||||||
|
// Characteristics of the request
|
||||||
|
request_type: RequestType,
|
||||||
|
// Specific request
|
||||||
|
request_code: RequestCode,
|
||||||
|
// Word-sized field that may (or may not) serve as a parameter to the request, depending
|
||||||
|
// on the specific request. For GET_DESCRIPTOR and SET_DESCRIPTOR, bit-cast a
|
||||||
|
// DescriptorValue into this field.
|
||||||
|
value: u16 align(1),
|
||||||
|
// Word-sized field that may (or may not) serve as a parameter to the request, depending
|
||||||
|
// on the specific request. Typically this field holds an index or an offset value. When
|
||||||
|
// request_type specifies an endpoint or an interface as the recipient, bit-cast an
|
||||||
|
// EndpointIndex or an InterfaceIndex into this field.
|
||||||
|
index: u16 align(1),
|
||||||
|
// Number of bytes to transfer if there is a DATA stage.
|
||||||
|
// - If this field is non-zero, and request_type indicates a transfer from
|
||||||
|
// device-to-host, then the device must never return more than length bytes of data.
|
||||||
|
// However, a device may return less.
|
||||||
|
// - If this field is non-zero, and request_type indicates a transfer from
|
||||||
|
// host-to-device, then the host must send exactly length bytes of data. If the host
|
||||||
|
// sends more than length bytes, the behavior of the device is undefined.
|
||||||
|
length: u16 align(1),
|
||||||
|
|
||||||
|
// The format of the index field when request_type specifies an endpoint as the
|
||||||
|
// recipient. The host should always set the direction bit to zero (but the device
|
||||||
|
// should accept either value) when the endpoint is part of a control pipe.
|
||||||
|
const EndpointIndex = packed struct(u16) {
|
||||||
|
// Endpoint number
|
||||||
|
number: EndpointNumber,
|
||||||
|
// Reserved (reset to zero)
|
||||||
|
reserved: u3 = 0,
|
||||||
|
// Selects the OUT or the IN endpoint with the specified endpoint number
|
||||||
|
direction: EndpointDirection,
|
||||||
|
// Reserved (reset to zero)
|
||||||
|
reserved_high: u8 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
// The format of the index field when request_type specifies an interface as the
|
||||||
|
// recipient.
|
||||||
|
const InterfaceIndex = packed struct(u16) {
|
||||||
|
// Interface number
|
||||||
|
number: u8,
|
||||||
|
// Reserved (reset to zero)
|
||||||
|
reserved: u8 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
// The format of the value field of GET_DESCRIPTOR and SET_DESCRIPTOR requests: the
|
||||||
|
// descriptor type in the high byte, and the descriptor index in the low byte. The index
|
||||||
|
// is used to select a specific descriptor (only for CONFIGURATION and STRING
|
||||||
|
// descriptors) when several descriptors of that type are implemented by a device.
|
||||||
|
const DescriptorValue = packed struct(u16) {
|
||||||
|
// Descriptor index
|
||||||
|
index: u8 = 0,
|
||||||
|
// Descriptor type
|
||||||
|
kind: DescriptorType,
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
// Feature selectors, used as the value field of CLEAR_FEATURE and SET_FEATURE requests. The
|
||||||
|
// comment on each value notes the recipient the selector applies to.
|
||||||
|
const FeatureSelector = enum(u16) {
|
||||||
|
// Halts an endpoint (recipient: endpoint)
|
||||||
|
endpoint_halt = 0,
|
||||||
|
// Enables or disables the device's remote wakeup capability (recipient: device)
|
||||||
|
device_remote_wakeup = 1,
|
||||||
|
// Puts a hi-speed device into a test mode, selected by a TestMode value in the high
|
||||||
|
// byte of the index field (recipient: device)
|
||||||
|
test_mode = 2,
|
||||||
|
};
|
||||||
|
|
||||||
|
// Test mode selectors, passed in the high byte of the index field of a SET_FEATURE request
|
||||||
|
// with the test_mode feature selector. Values 06h...3Fh are reserved for standard test
|
||||||
|
// selectors and C0h...FFh for vendor-specific test modes; all other unlisted values are
|
||||||
|
// reserved.
|
||||||
|
const TestMode = enum(u8) {
|
||||||
|
test_j = 0x01,
|
||||||
|
test_k = 0x02,
|
||||||
|
test_se0_nak = 0x03,
|
||||||
|
test_packet = 0x04,
|
||||||
|
test_force_enable = 0x05,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
|
||||||
|
// The two bytes returned by a GET_STATUS request directed at a device. Fields are declared
|
||||||
|
// least-significant first.
|
||||||
|
const DeviceStatus = packed struct(u16) {
|
||||||
|
// Whether the device is currently self-powered (as opposed to bus-powered). This bit
|
||||||
|
// cannot be changed with the SET_FEATURE or CLEAR_FEATURE requests.
|
||||||
|
self_powered: bool,
|
||||||
|
// Whether the device is currently enabled to request remote wakeup. Changed with the
|
||||||
|
// SET_FEATURE and CLEAR_FEATURE requests using the device_remote_wakeup feature
|
||||||
|
// selector.
|
||||||
|
remote_wakeup: bool,
|
||||||
|
// Reserved (reset to zero)
|
||||||
|
reserved: u14,
|
||||||
|
};
|
||||||
|
|
||||||
|
// The two bytes returned by a GET_STATUS request directed at an endpoint. (A GET_STATUS
|
||||||
|
// request directed at an interface returns two bytes that are entirely reserved.)
|
||||||
|
const EndpointStatus = packed struct(u16) {
|
||||||
|
// Whether the endpoint is currently halted. Set with the SET_FEATURE request using the
|
||||||
|
// endpoint_halt feature selector, and cleared with CLEAR_FEATURE.
|
||||||
|
halted: bool,
|
||||||
|
// Reserved (reset to zero)
|
||||||
|
reserved: u15,
|
||||||
|
};
|
||||||
|
|
||||||
|
// A target for the standard requests that may be directed at the device, an interface, or
|
||||||
|
// an endpoint.
|
||||||
|
const Target = union(enum) {
|
||||||
|
device,
|
||||||
|
interface: InterfaceNumber,
|
||||||
|
endpoint: Request.EndpointIndex,
|
||||||
|
|
||||||
|
fn recipient(target: Target) RequestType.Recipient {
|
||||||
|
return switch (target) {
|
||||||
|
.device => .device,
|
||||||
|
.interface => .interface,
|
||||||
|
.endpoint => .endpoint,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
fn index(target: Target) u16 {
|
||||||
|
return switch (target) {
|
||||||
|
.device => 0,
|
||||||
|
.interface => |number| @intFromEnum(number),
|
||||||
|
.endpoint => |endpoint| @bitCast(endpoint),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
// Constructors for the standard device requests, one per RequestCode. Each returns a
|
||||||
|
// ready-to-send set-up packet with the request_type, value, index, and length fields the
|
||||||
|
// specification prescribes for that request.
|
||||||
|
|
||||||
|
// Reads the status of the given target: bit-cast the two bytes the device returns into a
|
||||||
|
// DeviceStatus or an EndpointStatus. (The two bytes returned for an interface are entirely
|
||||||
|
// reserved.)
|
||||||
|
fn getStatus(target: Target) Request {
|
||||||
|
return .{
|
||||||
|
.request_type = .{
|
||||||
|
.recipient = target.recipient(),
|
||||||
|
.kind = .standard,
|
||||||
|
.direction = .device_to_host,
|
||||||
|
},
|
||||||
|
.request_code = .get_status,
|
||||||
|
.value = 0,
|
||||||
|
.index = target.index(),
|
||||||
|
.length = 2,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// Clears or disables the given feature. A device cannot be taken out of a test mode with
|
||||||
|
// this request; test_mode is only cleared by cycling power.
|
||||||
|
fn clearFeature(feature: FeatureSelector, target: Target) Request {
|
||||||
|
return .{
|
||||||
|
.request_type = .{
|
||||||
|
.recipient = target.recipient(),
|
||||||
|
.kind = .standard,
|
||||||
|
.direction = .host_to_device,
|
||||||
|
},
|
||||||
|
.request_code = .clear_feature,
|
||||||
|
.value = @intFromEnum(feature),
|
||||||
|
.index = target.index(),
|
||||||
|
.length = 0,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// Sets or enables the given feature. For the test_mode feature selector, use setTestMode
|
||||||
|
// instead: the test selector rides in the high byte of the index field.
|
||||||
|
fn setFeature(feature: FeatureSelector, target: Target) Request {
|
||||||
|
return .{
|
||||||
|
.request_type = .{
|
||||||
|
.recipient = target.recipient(),
|
||||||
|
.kind = .standard,
|
||||||
|
.direction = .host_to_device,
|
||||||
|
},
|
||||||
|
.request_code = .set_feature,
|
||||||
|
.value = @intFromEnum(feature),
|
||||||
|
.index = target.index(),
|
||||||
|
.length = 0,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// Puts a hi-speed device into the given test mode: a SET_FEATURE request with the test_mode
|
||||||
|
// feature selector and the test selector in the high byte of the index field.
|
||||||
|
fn setTestMode(mode: TestMode) Request {
|
||||||
|
return .{
|
||||||
|
.request_type = .{
|
||||||
|
.recipient = .device,
|
||||||
|
.kind = .standard,
|
||||||
|
.direction = .host_to_device,
|
||||||
|
},
|
||||||
|
.request_code = .set_feature,
|
||||||
|
.value = @intFromEnum(FeatureSelector.test_mode),
|
||||||
|
.index = @as(u16, @intFromEnum(mode)) << 8,
|
||||||
|
.length = 0,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// Assigns the device its bus address, moving it from the default state to the address
|
||||||
|
// state. The device does not answer at the new address until the status stage of this
|
||||||
|
// request completes.
|
||||||
|
fn setAddress(address: DeviceAddress) Request {
|
||||||
|
return .{
|
||||||
|
.request_type = .{
|
||||||
|
.recipient = .device,
|
||||||
|
.kind = .standard,
|
||||||
|
.direction = .host_to_device,
|
||||||
|
},
|
||||||
|
.request_code = .set_address,
|
||||||
|
.value = @intFromEnum(address),
|
||||||
|
.index = 0,
|
||||||
|
.length = 0,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// Reads a descriptor from the device.
|
||||||
|
// - descriptor_index selects among descriptors of the same type, and is only used for
|
||||||
|
// configuration and string descriptors.
|
||||||
|
// - language_id selects the language of a string descriptor, and is zero otherwise.
|
||||||
|
// - length is the number of bytes to read; a device never returns more than length bytes,
|
||||||
|
// but may return less if the descriptor is shorter.
|
||||||
|
fn getDescriptor(kind: DescriptorType, descriptor_index: u8, language_id: u16, length: u16) Request {
|
||||||
|
return .{
|
||||||
|
.request_type = .{
|
||||||
|
.recipient = .device,
|
||||||
|
.kind = .standard,
|
||||||
|
.direction = .device_to_host,
|
||||||
|
},
|
||||||
|
.request_code = .get_descriptor,
|
||||||
|
.value = @bitCast(Request.DescriptorValue{ .index = descriptor_index, .kind = kind }),
|
||||||
|
.index = language_id,
|
||||||
|
.length = length,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// Updates an existing descriptor or adds a new one (optional; many devices do not support
|
||||||
|
// this request). The parameters mirror getDescriptor; the descriptor itself is sent in the
|
||||||
|
// DATA stage.
|
||||||
|
fn setDescriptor(kind: DescriptorType, descriptor_index: u8, language_id: u16, length: u16) Request {
|
||||||
|
return .{
|
||||||
|
.request_type = .{
|
||||||
|
.recipient = .device,
|
||||||
|
.kind = .standard,
|
||||||
|
.direction = .host_to_device,
|
||||||
|
},
|
||||||
|
.request_code = .set_descriptor,
|
||||||
|
.value = @bitCast(Request.DescriptorValue{ .index = descriptor_index, .kind = kind }),
|
||||||
|
.index = language_id,
|
||||||
|
.length = length,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// Reads the currently active configuration: @enumFromInt the byte the device returns into a
|
||||||
|
// ConfigurationValue, which is none while the device is not configured.
|
||||||
|
fn getConfiguration() Request {
|
||||||
|
return .{
|
||||||
|
.request_type = .{
|
||||||
|
.recipient = .device,
|
||||||
|
.kind = .standard,
|
||||||
|
.direction = .device_to_host,
|
||||||
|
},
|
||||||
|
.request_code = .get_configuration,
|
||||||
|
.value = 0,
|
||||||
|
.index = 0,
|
||||||
|
.length = 1,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// Selects the configuration with the given configuration_value (from
|
||||||
|
// ConfigurationDescriptor.configuration_value), moving the device from the address state to
|
||||||
|
// the configured state. Selecting none returns the device to the address state.
|
||||||
|
fn setConfiguration(configuration_value: ConfigurationValue) Request {
|
||||||
|
return .{
|
||||||
|
.request_type = .{
|
||||||
|
.recipient = .device,
|
||||||
|
.kind = .standard,
|
||||||
|
.direction = .host_to_device,
|
||||||
|
},
|
||||||
|
.request_code = .set_configuration,
|
||||||
|
.value = @intFromEnum(configuration_value),
|
||||||
|
.index = 0,
|
||||||
|
.length = 0,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// Reads the alternate setting currently selected for the given interface: @enumFromInt the
|
||||||
|
// byte the device returns into an AlternateSetting.
|
||||||
|
fn getInterface(interface: InterfaceNumber) Request {
|
||||||
|
return .{
|
||||||
|
.request_type = .{
|
||||||
|
.recipient = .interface,
|
||||||
|
.kind = .standard,
|
||||||
|
.direction = .device_to_host,
|
||||||
|
},
|
||||||
|
.request_code = .get_interface,
|
||||||
|
.value = 0,
|
||||||
|
.index = @intFromEnum(interface),
|
||||||
|
.length = 1,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// Selects an alternate setting (from InterfaceDescriptor.alternate_setting) for the given
|
||||||
|
// interface.
|
||||||
|
fn setInterface(interface: InterfaceNumber, alternate_setting: AlternateSetting) Request {
|
||||||
|
return .{
|
||||||
|
.request_type = .{
|
||||||
|
.recipient = .interface,
|
||||||
|
.kind = .standard,
|
||||||
|
.direction = .host_to_device,
|
||||||
|
},
|
||||||
|
.request_code = .set_interface,
|
||||||
|
.value = @intFromEnum(alternate_setting),
|
||||||
|
.index = @intFromEnum(interface),
|
||||||
|
.length = 0,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// Reads the two-byte number of the frame in which the given isochronous endpoint's
|
||||||
|
// repeating pattern of transfers begins.
|
||||||
|
fn syncFrame(endpoint: Request.EndpointIndex) Request {
|
||||||
|
return .{
|
||||||
|
.request_type = .{
|
||||||
|
.recipient = .endpoint,
|
||||||
|
.kind = .standard,
|
||||||
|
.direction = .device_to_host,
|
||||||
|
},
|
||||||
|
.request_code = .sync_frame,
|
||||||
|
.value = 0,
|
||||||
|
.index = @bitCast(endpoint),
|
||||||
|
.length = 2,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
const DescriptorType = enum(u8) {
|
||||||
|
device = 1,
|
||||||
|
configuration = 2,
|
||||||
|
string = 3,
|
||||||
|
interface = 4,
|
||||||
|
endpoint = 5,
|
||||||
|
device_qualifier = 6,
|
||||||
|
other_speed_configuration = 7,
|
||||||
|
interface_power = 8,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
|
||||||
|
const DeviceDescriptor = extern struct {
|
||||||
|
// Size of this descriptor in bytes
|
||||||
|
length: u8,
|
||||||
|
// DEVICE Descriptor Type
|
||||||
|
descriptor_type: DescriptorType,
|
||||||
|
// USB Specification Release Number in Binary-Coded Decimal (i.e, 2.10 is expressed as 210h).
|
||||||
|
// Identifies the release of the USB Specification with with the device and its
|
||||||
|
// descriptors are compliant.
|
||||||
|
bcd_usb: u16 align(1),
|
||||||
|
// Class code (assigned by the USB-IF)
|
||||||
|
// - This field is reset to zero if each interface within a configuration specifies its own
|
||||||
|
// class information and the various interfaces operate independently.
|
||||||
|
// - A value of FFh in this field indicates the device class is vendor-specific.
|
||||||
|
device_class: u8,
|
||||||
|
// Subclass Code (assigned by the USB-IF)
|
||||||
|
// - The subclass code of a device is qualified by the class code of that device.
|
||||||
|
// - If device_class is reset to zero, then this field must also be reset to zero.
|
||||||
|
// - When device_class is not set to FFh, then all values for this field are reserved for
|
||||||
|
// assignment by the USB-IF.
|
||||||
|
device_subclass: u8,
|
||||||
|
// Protocol code (assigned by the USB-IF)
|
||||||
|
// - The protocol code of a device is qualified by both the class and subclass codes of
|
||||||
|
// that device.
|
||||||
|
// - A value of 00h in this field means that the device may specify class-specific
|
||||||
|
// protocols on an interface basis, though this is not a requirement.
|
||||||
|
// - If this field is set to FFh, then the device uses a vendor-specific protocol.
|
||||||
|
device_protocol: u8,
|
||||||
|
// Maximum packet size for endpoint zero (8, 16, 32, or 64 are the only valid options)
|
||||||
|
max_packet_size_0: u8,
|
||||||
|
// Vendor ID (assigned by the USB-IF)
|
||||||
|
vendor_id: u16 align(1),
|
||||||
|
// Product ID (assigned by the USB-IF)
|
||||||
|
product_id: u16 align(1),
|
||||||
|
// Device release number in binary-coded decimal
|
||||||
|
bcd_device: u16 align(1),
|
||||||
|
// Index of STRING descriptor describing manufacturer
|
||||||
|
manufacturer_index: StringIndex,
|
||||||
|
// Index of STRING descriptor describing product
|
||||||
|
product_index: StringIndex,
|
||||||
|
// Index of STRING descriptor describing the device's serial number
|
||||||
|
serial_number_index: StringIndex,
|
||||||
|
// Number of possible configurations
|
||||||
|
configuration_count: u8,
|
||||||
|
};
|
||||||
|
|
||||||
|
const DeviceQualifierDescriptor = extern struct {
|
||||||
|
// Size of this descriptor in bytes
|
||||||
|
length: u8,
|
||||||
|
// DEVICE_QUALIFIER Descriptor Type
|
||||||
|
descriptor_type: DescriptorType,
|
||||||
|
// USB Specification Release Number in Binary-Coded Decimal (i.e, 2.00 is expressed as 200h).
|
||||||
|
// Identifies the release of the USB Specification with with the device and its
|
||||||
|
// descriptors are compliant. This field must be at least 0200h.
|
||||||
|
bcd_usb: u16 align(1),
|
||||||
|
// Class code (assigned by the USB-IF)
|
||||||
|
device_class: u8,
|
||||||
|
// Subclass Code (assigned by the USB-IF)
|
||||||
|
device_subclass: u8,
|
||||||
|
// Protocol code (assigned by the USB-IF)
|
||||||
|
device_protocol: u8,
|
||||||
|
// Maximum packet size for endpoint zero (8, 16, 32, or 64 are the only valid options)
|
||||||
|
max_packet_size_0: u8,
|
||||||
|
// Number of possible configurations
|
||||||
|
configuration_count: u8,
|
||||||
|
// Reserved for future uses, must be zero.
|
||||||
|
reserved: u8,
|
||||||
|
};
|
||||||
|
|
||||||
|
const ConfigurationDescriptor = extern struct {
|
||||||
|
// Size of this descriptor in bytes
|
||||||
|
length: u8,
|
||||||
|
// CONFIGURATION Descriptor Type
|
||||||
|
descriptor_type: DescriptorType,
|
||||||
|
// The total combined length in bytes of all the descriptors returned with the request for
|
||||||
|
// this CONFIGURATION descriptor (including CONFIGURATION, INTERFACE, ENDPOINT, class- and
|
||||||
|
// vendor-specific descriptors).
|
||||||
|
total_length: u16 align(1),
|
||||||
|
// Number of interfaces supported by this configuration
|
||||||
|
interface_count: u8,
|
||||||
|
// Value which when used as an argument in the SET_CONFIGURATION request, causes the device
|
||||||
|
// to assume the configuration described by this descriptor.
|
||||||
|
configuration_value: ConfigurationValue,
|
||||||
|
// Index of STRING descriptor describing this configuration.
|
||||||
|
configuration_index: StringIndex,
|
||||||
|
// Configuration Characteristics
|
||||||
|
attributes: Attributes,
|
||||||
|
// Maximum power consumption of this device from the bus when fully operational and using
|
||||||
|
// this configuration. Expressed in units of 2mA (i.e., a value of 50 in this field
|
||||||
|
// indicates 100mA).
|
||||||
|
// - A device reports with the attributes field whether the configuration is bus- or
|
||||||
|
// self-powered, but the device status (retrieved with a GET_STATUS request) reports
|
||||||
|
// whether the device is currently self-powered.
|
||||||
|
// - If a device is disconnected from an external power source, it may not draw more
|
||||||
|
// power from the bus than specified in this field.
|
||||||
|
max_power: u8,
|
||||||
|
|
||||||
|
// Configuration characteristics. Fields are declared least-significant first.
|
||||||
|
const Attributes = packed struct(u8) {
|
||||||
|
// Reserved, reset to zero (D4...0)
|
||||||
|
reserved: u5,
|
||||||
|
// Whether Remote Wakeup is supported by this configuration (D5)
|
||||||
|
remote_wakeup: bool,
|
||||||
|
// Self-Powered (D6)
|
||||||
|
// - false: Device runs on power supplied by the bus
|
||||||
|
// - true: Device provides a local power source; if max_power is non-zero, the
|
||||||
|
// device also may use bus power.
|
||||||
|
self_powered: bool,
|
||||||
|
// Reserved, must be set to one for historical reasons (D7)
|
||||||
|
reserved_one: u1,
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
// This descriptor describes the configuration of a high-speed device if it were operating at
|
||||||
|
// its alternative speed. The structure of the OTHER_SPEED_CONFIGURATION is identical to that
|
||||||
|
// of the CONFIGURATION descriptor; the only difference is that the descriptor_type field
|
||||||
|
// reflects that the descriptor is an OTHER_SPEED_CONFIGURATION descriptor.
|
||||||
|
const OtherSpeedConfigurationDescriptor = ConfigurationDescriptor;
|
||||||
|
|
||||||
|
const InterfaceDescriptor = extern struct {
|
||||||
|
// Size of this descriptor in bytes
|
||||||
|
length: u8,
|
||||||
|
// INTERFACE Descriptor Type
|
||||||
|
descriptor_type: DescriptorType,
|
||||||
|
// Number of this interface. Zero-based value which identifies the index of this interface
|
||||||
|
// in the array of interfaces supported within a configuration.
|
||||||
|
interface_number: InterfaceNumber,
|
||||||
|
// Value used to select the alternate settings described by this INTERFACE descriptor for
|
||||||
|
// the interface with the interface_number in the previous field. This value is zero if
|
||||||
|
// this descriptor describes the default settings for a particular interface.
|
||||||
|
alternate_setting: AlternateSetting,
|
||||||
|
// Number of endpoints used by this interface, not including endpoint zero.
|
||||||
|
endpoint_count: u8,
|
||||||
|
// Class code (assigned by the USB-IF)
|
||||||
|
// - A value of zero here is reserved for future standardization.
|
||||||
|
// - If this value is FFh, the interface class is vendor-specific.
|
||||||
|
// - All other values are reserved for assignment by the USB-IF.
|
||||||
|
interface_class: u8,
|
||||||
|
// Subclass code (assigned by the USB-IF)
|
||||||
|
// - The subclass code in this field is qualified by the value of the interface_class
|
||||||
|
// field.
|
||||||
|
// - If interface_class is reset to zero, then this field must also be reset to zero.
|
||||||
|
// - If interface_class is not set to the value of FFh, then all values of this field are
|
||||||
|
// reserved for assignment by the USB-IF.
|
||||||
|
interface_subclass: u8,
|
||||||
|
// Protocol code (assigned by the USB-IF)
|
||||||
|
// - The protocol code in this field is qualified by the values of the interface_class
|
||||||
|
// and interface_subclass fields.
|
||||||
|
// - If an interface supports class-specific requests, then this field identifies the
|
||||||
|
// protocols that the device uses as defined by the specifications of the device class.
|
||||||
|
// - If this field is reset to zero, then the device does not use a class-specific
|
||||||
|
// protocol on this interface.
|
||||||
|
// - If this field is set to FFh, then the device uses a vendor-specific protocol on
|
||||||
|
// this interface.
|
||||||
|
interface_protocol: u8,
|
||||||
|
// Index of STRING descriptor describing this interface
|
||||||
|
interface_index: StringIndex,
|
||||||
|
};
|
||||||
|
|
||||||
|
const EndpointDescriptor = extern struct {
|
||||||
|
// Size of this descriptor in bytes
|
||||||
|
length: u8,
|
||||||
|
// ENDPOINT Descriptor Type
|
||||||
|
descriptor_type: DescriptorType,
|
||||||
|
// The address of the endpoint on the USB device described by this descriptor
|
||||||
|
endpoint_address: Address,
|
||||||
|
// The endpoint's attributes
|
||||||
|
attributes: Attributes,
|
||||||
|
// Maximum packet size that this endpoint is capable of sending or receiving. For
|
||||||
|
// isochronous endpoints, this value is used to reserve bus time; the pipe, however, may
|
||||||
|
// not always use all of the reserved bus time.
|
||||||
|
max_packet_size: MaxPacketSize align(1),
|
||||||
|
// Interval for polling a device during a data transfer, expressed in units of microframes
|
||||||
|
// for high-speed devices, and frames for low- and full-speed devices. The exact meaning of
|
||||||
|
// the value in this field depends on the endpoint type and the operating speed of the
|
||||||
|
// device:
|
||||||
|
// - Full- and High-speed isochronous endpoints, and high-speed interrupt endpoints:
|
||||||
|
// This field must be in the range from 1 to 16, and is used to calculate the period
|
||||||
|
// as 2^(interval - 1). That is, a value of 4 calculates to 2^(4 - 1) = 2^3 = 8.
|
||||||
|
// - Full- and Low-speed interrupt endpoints: This field must be in the range from
|
||||||
|
// 1 to 255.
|
||||||
|
// - High-speed bulk and control OUT endpoints: This field must be in the range from
|
||||||
|
// 0 to 255, and specifies the maximum NAK rate of the endpoint. A value of zero
|
||||||
|
// indicates that the endpoint never NAKs; other values indicate at most 1 NAK each
|
||||||
|
// interval number of microframes.
|
||||||
|
interval: u8,
|
||||||
|
|
||||||
|
// The address of an endpoint. Fields are declared least-significant first.
|
||||||
|
const Address = packed struct(u8) {
|
||||||
|
// Endpoint Number (D3...0)
|
||||||
|
number: EndpointNumber,
|
||||||
|
// Reserved, reset to zero (D6...4)
|
||||||
|
reserved: u3,
|
||||||
|
// Direction, ignored for control endpoints (D7)
|
||||||
|
direction: EndpointDirection,
|
||||||
|
};
|
||||||
|
|
||||||
|
// An endpoint's attributes. Fields are declared least-significant first.
|
||||||
|
const Attributes = packed struct(u8) {
|
||||||
|
// Transfer Type (D1...0)
|
||||||
|
transfer_type: TransferType,
|
||||||
|
// Synchronization Type; isochronous endpoints only, reserved and reset to zero for
|
||||||
|
// other endpoint types (D3...2)
|
||||||
|
synchronization: Synchronization,
|
||||||
|
// Usage Type; isochronous endpoints only, reserved and reset to zero for other
|
||||||
|
// endpoints (D5...4)
|
||||||
|
usage: Usage,
|
||||||
|
// Reserved, reset to zero (D7...6)
|
||||||
|
reserved: u2,
|
||||||
|
};
|
||||||
|
|
||||||
|
const TransferType = enum(u2) {
|
||||||
|
control = 0,
|
||||||
|
isochronous = 1,
|
||||||
|
bulk = 2,
|
||||||
|
interrupt = 3,
|
||||||
|
};
|
||||||
|
|
||||||
|
const Synchronization = enum(u2) {
|
||||||
|
none = 0,
|
||||||
|
asynchronous = 1,
|
||||||
|
adaptive = 2,
|
||||||
|
synchronous = 3,
|
||||||
|
};
|
||||||
|
|
||||||
|
const Usage = enum(u2) {
|
||||||
|
data = 0,
|
||||||
|
feedback = 1,
|
||||||
|
implicit_feedback_data = 2,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
|
||||||
|
// The maximum packet size of an endpoint. Fields are declared least-significant first.
|
||||||
|
const MaxPacketSize = packed struct(u16) {
|
||||||
|
// Maximum packet size in bytes (bits 10...0)
|
||||||
|
size: u11,
|
||||||
|
// Number of additional transaction opportunities per microframe, for high-speed
|
||||||
|
// isochronous and interrupt endpoints; reserved and reset to zero for other
|
||||||
|
// endpoints (bits 12...11)
|
||||||
|
additional_transactions: AdditionalTransactions,
|
||||||
|
// Reserved, must be reset to zero (bits 15...13)
|
||||||
|
reserved: u3,
|
||||||
|
};
|
||||||
|
|
||||||
|
const AdditionalTransactions = enum(u2) {
|
||||||
|
// None (1 transaction per microframe)
|
||||||
|
none = 0,
|
||||||
|
// 1 additional (2 transactions per microframe)
|
||||||
|
one = 1,
|
||||||
|
// 2 additional (3 transactions per microframe)
|
||||||
|
two = 2,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
// A STRING descriptor at index zero returns the list of LANGID codes supported by the
|
||||||
|
// device; all other indices return a Unicode string. Both forms start with this two-byte
|
||||||
|
// header, followed by the variable-length payload:
|
||||||
|
// - index 0: an array of two-byte LANGID codes (wLangID[0] through wLangID[x])
|
||||||
|
// - other indices: a Unicode string of N bytes
|
||||||
|
const StringDescriptor = extern struct {
|
||||||
|
// Size of this descriptor in bytes
|
||||||
|
length: u8,
|
||||||
|
// STRING Descriptor Type
|
||||||
|
descriptor_type: DescriptorType,
|
||||||
|
};
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
test "wire sizes and offsets match the specification" {
|
||||||
|
const expectEqual = std.testing.expectEqual;
|
||||||
|
|
||||||
|
try expectEqual(8, @sizeOf(Request));
|
||||||
|
try expectEqual(18, @sizeOf(DeviceDescriptor));
|
||||||
|
try expectEqual(10, @sizeOf(DeviceQualifierDescriptor));
|
||||||
|
try expectEqual(9, @sizeOf(ConfigurationDescriptor));
|
||||||
|
try expectEqual(9, @sizeOf(InterfaceDescriptor));
|
||||||
|
try expectEqual(7, @sizeOf(EndpointDescriptor));
|
||||||
|
try expectEqual(2, @sizeOf(StringDescriptor));
|
||||||
|
|
||||||
|
try expectEqual(2, @offsetOf(DeviceDescriptor, "bcd_usb"));
|
||||||
|
try expectEqual(8, @offsetOf(DeviceDescriptor, "vendor_id"));
|
||||||
|
try expectEqual(17, @offsetOf(DeviceDescriptor, "configuration_count"));
|
||||||
|
try expectEqual(2, @offsetOf(ConfigurationDescriptor, "total_length"));
|
||||||
|
try expectEqual(4, @offsetOf(EndpointDescriptor, "max_packet_size"));
|
||||||
|
}
|
||||||
|
|
||||||
|
test "bitmap packings match the specification" {
|
||||||
|
const expectEqual = std.testing.expectEqual;
|
||||||
|
const expect = std.testing.expect;
|
||||||
|
|
||||||
|
// bmRequestType for GET_DESCRIPTOR: device-to-host | standard | device = 80h
|
||||||
|
const request_type = RequestType{
|
||||||
|
.recipient = .device,
|
||||||
|
.kind = .standard,
|
||||||
|
.direction = .device_to_host,
|
||||||
|
};
|
||||||
|
try expectEqual(0x80, @as(u8, @bitCast(request_type)));
|
||||||
|
|
||||||
|
// wValue for GET_DESCRIPTOR(CONFIGURATION, index 0) = 0200h
|
||||||
|
const descriptor_value = Request.DescriptorValue{ .kind = .configuration };
|
||||||
|
try expectEqual(0x0200, @as(u16, @bitCast(descriptor_value)));
|
||||||
|
|
||||||
|
// wIndex for the IN endpoint 1 = 0081h
|
||||||
|
const endpoint_index = Request.EndpointIndex{ .number = @enumFromInt(1), .direction = .in };
|
||||||
|
try expectEqual(0x0081, @as(u16, @bitCast(endpoint_index)));
|
||||||
|
|
||||||
|
// Endpoint address 81h = IN endpoint 1
|
||||||
|
const address: EndpointDescriptor.Address = @bitCast(@as(u8, 0x81));
|
||||||
|
try expectEqual(1, @intFromEnum(address.number));
|
||||||
|
try expectEqual(.in, address.direction);
|
||||||
|
|
||||||
|
// Endpoint attributes 03h = interrupt transfer
|
||||||
|
const attributes: EndpointDescriptor.Attributes = @bitCast(@as(u8, 0x03));
|
||||||
|
try expectEqual(.interrupt, attributes.transfer_type);
|
||||||
|
|
||||||
|
// wMaxPacketSize 0008h = 8 bytes, no additional transactions
|
||||||
|
const max_packet_size: EndpointDescriptor.MaxPacketSize = @bitCast(@as(u16, 0x0008));
|
||||||
|
try expectEqual(8, max_packet_size.size);
|
||||||
|
try expectEqual(.none, max_packet_size.additional_transactions);
|
||||||
|
|
||||||
|
// Configuration attributes C0h = self-powered, with the historical D7 bit set
|
||||||
|
const configuration_attributes: ConfigurationDescriptor.Attributes = @bitCast(@as(u8, 0xC0));
|
||||||
|
try expect(configuration_attributes.self_powered);
|
||||||
|
try expect(!configuration_attributes.remote_wakeup);
|
||||||
|
try expectEqual(1, configuration_attributes.reserved_one);
|
||||||
|
|
||||||
|
// GET_STATUS words: device 0001h = self-powered; endpoint 0001h = halted
|
||||||
|
const device_status: DeviceStatus = @bitCast(@as(u16, 0x0001));
|
||||||
|
try expect(device_status.self_powered and !device_status.remote_wakeup);
|
||||||
|
const endpoint_status: EndpointStatus = @bitCast(@as(u16, 0x0001));
|
||||||
|
try expect(endpoint_status.halted);
|
||||||
|
|
||||||
|
// DescriptorType is non-exhaustive: class-specific values (HID = 21h) pass through
|
||||||
|
const hid_type: DescriptorType = @enumFromInt(0x21);
|
||||||
|
try expectEqual(0x21, @intFromEnum(hid_type));
|
||||||
|
try expect(hid_type != .device);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn expectRequestBytes(request: Request, expected: [8]u8) !void {
|
||||||
|
try std.testing.expectEqualSlices(u8, &expected, std.mem.asBytes(&request));
|
||||||
|
}
|
||||||
|
|
||||||
|
test "standard request constructors encode the specification's set-up packets" {
|
||||||
|
try expectRequestBytes(getStatus(.device), .{ 0x80, 0, 0, 0, 0, 0, 2, 0 });
|
||||||
|
try expectRequestBytes(getStatus(.{ .interface = @enumFromInt(3) }), .{ 0x81, 0, 0, 0, 3, 0, 2, 0 });
|
||||||
|
try expectRequestBytes(getStatus(.{ .endpoint = .{ .number = @enumFromInt(2), .direction = .in } }), .{ 0x82, 0, 0, 0, 0x82, 0, 2, 0 });
|
||||||
|
try expectRequestBytes(clearFeature(.endpoint_halt, .{ .endpoint = .{ .number = @enumFromInt(1), .direction = .out } }), .{ 0x02, 1, 0, 0, 0x01, 0, 0, 0 });
|
||||||
|
try expectRequestBytes(setFeature(.device_remote_wakeup, .device), .{ 0x00, 3, 1, 0, 0, 0, 0, 0 });
|
||||||
|
try expectRequestBytes(setTestMode(.test_packet), .{ 0x00, 3, 2, 0, 0, 0x04, 0, 0 });
|
||||||
|
try expectRequestBytes(setAddress(@enumFromInt(5)), .{ 0x00, 5, 5, 0, 0, 0, 0, 0 });
|
||||||
|
try expectRequestBytes(getDescriptor(.device, 0, 0, 18), .{ 0x80, 6, 0, 1, 0, 0, 18, 0 });
|
||||||
|
try expectRequestBytes(getDescriptor(.string, 2, 0x0409, 255), .{ 0x80, 6, 2, 3, 0x09, 0x04, 255, 0 });
|
||||||
|
try expectRequestBytes(setDescriptor(.string, 2, 0x0409, 16), .{ 0x00, 7, 2, 3, 0x09, 0x04, 16, 0 });
|
||||||
|
try expectRequestBytes(getConfiguration(), .{ 0x80, 8, 0, 0, 0, 0, 1, 0 });
|
||||||
|
try expectRequestBytes(setConfiguration(@enumFromInt(1)), .{ 0x00, 9, 1, 0, 0, 0, 0, 0 });
|
||||||
|
try expectRequestBytes(getInterface(@enumFromInt(2)), .{ 0x81, 10, 0, 0, 2, 0, 1, 0 });
|
||||||
|
try expectRequestBytes(setInterface(@enumFromInt(2), @enumFromInt(1)), .{ 0x01, 11, 1, 0, 2, 0, 0, 0 });
|
||||||
|
try expectRequestBytes(syncFrame(.{ .number = @enumFromInt(3), .direction = .in }), .{ 0x82, 12, 0, 0, 0x83, 0, 2, 0 });
|
||||||
|
}
|
||||||
@@ -0,0 +1,264 @@
|
|||||||
|
//! USB class-code decoding: turn the (class, subclass, protocol) triple a USB device or
|
||||||
|
//! interface reports in its descriptors into typed values. The device descriptor carries one
|
||||||
|
//! triple for the whole device, and each interface descriptor carries its own; a class code
|
||||||
|
//! of zero at the device level defers entirely to the interfaces. Subclass and protocol
|
||||||
|
//! codes are qualified by the class code — the same value means different things under
|
||||||
|
//! different classes — so there is no single SubClass or Protocol enum: each class with
|
||||||
|
//! spec-defined codes gets its own namespace below. Pure reference data (from the USB-IF
|
||||||
|
//! defined class codes; see https://www.usb.org/defined-class-codes) — no hardware access —
|
||||||
|
//! so it is shared by kernel discovery and any user-space tool (device naming, driver
|
||||||
|
//! matching).
|
||||||
|
|
||||||
|
// Base class codes (assigned by the USB-IF). The comment on each value notes where the code
|
||||||
|
// may legally appear: in the device descriptor, in interface descriptors, or both.
|
||||||
|
const Class = enum(u8) {
|
||||||
|
// Use class information in the interface descriptors (device descriptor only). Each
|
||||||
|
// interface within a configuration specifies its own class information and the various
|
||||||
|
// interfaces operate independently.
|
||||||
|
per_interface = 0x00,
|
||||||
|
// Audio: speakers, microphones, sound cards (interface)
|
||||||
|
audio = 0x01,
|
||||||
|
// Communications and CDC control: modems, network adapters (both)
|
||||||
|
communications = 0x02,
|
||||||
|
// Human Interface Device: keyboards, mice, game controllers (interface)
|
||||||
|
hid = 0x03,
|
||||||
|
// Physical: force-feedback devices (interface)
|
||||||
|
physical = 0x05,
|
||||||
|
// Image: still-imaging cameras, scanners (interface)
|
||||||
|
image = 0x06,
|
||||||
|
// Printer (interface)
|
||||||
|
printer = 0x07,
|
||||||
|
// Mass storage: flash drives, external disks, card readers (interface)
|
||||||
|
mass_storage = 0x08,
|
||||||
|
// Hub (device descriptor only)
|
||||||
|
hub = 0x09,
|
||||||
|
// CDC-Data: the data interfaces paired with a communications control interface
|
||||||
|
// (interface)
|
||||||
|
cdc_data = 0x0A,
|
||||||
|
// Smart card readers (interface)
|
||||||
|
smart_card = 0x0B,
|
||||||
|
// Content security (interface)
|
||||||
|
content_security = 0x0D,
|
||||||
|
// Video: webcams (interface)
|
||||||
|
video = 0x0E,
|
||||||
|
// Personal healthcare devices (interface)
|
||||||
|
personal_healthcare = 0x0F,
|
||||||
|
// Audio/Video devices (interface)
|
||||||
|
audio_video = 0x10,
|
||||||
|
// Billboard: describes alternate modes a USB Type-C device supports (device descriptor
|
||||||
|
// only)
|
||||||
|
billboard = 0x11,
|
||||||
|
// USB Type-C bridge (interface)
|
||||||
|
type_c_bridge = 0x12,
|
||||||
|
// USB Bulk Display Protocol devices (interface)
|
||||||
|
bulk_display = 0x13,
|
||||||
|
// MCTP over USB protocol endpoint devices (interface)
|
||||||
|
mctp = 0x14,
|
||||||
|
// I3C devices (interface)
|
||||||
|
i3c = 0x3C,
|
||||||
|
// Diagnostic devices (both)
|
||||||
|
diagnostic = 0xDC,
|
||||||
|
// Wireless controllers: Bluetooth adapters (interface)
|
||||||
|
wireless_controller = 0xE0,
|
||||||
|
// Miscellaneous (both)
|
||||||
|
miscellaneous = 0xEF,
|
||||||
|
// Application-specific: firmware upgrade, IrDA bridges, test and measurement
|
||||||
|
// (interface)
|
||||||
|
application_specific = 0xFE,
|
||||||
|
// Vendor-specific (both)
|
||||||
|
vendor_specific = 0xFF,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
|
||||||
|
// Subclass and protocol codes qualified by Class.hub. Hubs have no subclass codes; the
|
||||||
|
// protocol distinguishes the hub's transaction-translator arrangement.
|
||||||
|
const hub = struct {
|
||||||
|
const Protocol = enum(u8) {
|
||||||
|
// Full-speed hub
|
||||||
|
full_speed = 0x00,
|
||||||
|
// Hi-speed hub with a single transaction translator
|
||||||
|
hi_speed_single_tt = 0x01,
|
||||||
|
// Hi-speed hub with multiple transaction translators
|
||||||
|
hi_speed_multi_tt = 0x02,
|
||||||
|
// SuperSpeed hub (USB 3)
|
||||||
|
super_speed = 0x03,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
// Subclass and protocol codes qualified by Class.hid.
|
||||||
|
const hid = struct {
|
||||||
|
const SubClass = enum(u8) {
|
||||||
|
// No subclass
|
||||||
|
none = 0x00,
|
||||||
|
// Boot interface: the device also supports the simplified boot protocol, usable by
|
||||||
|
// firmware before a full HID report-descriptor parser is available
|
||||||
|
boot = 0x01,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
|
||||||
|
// Only meaningful when the subclass is boot
|
||||||
|
const Protocol = enum(u8) {
|
||||||
|
none = 0x00,
|
||||||
|
keyboard = 0x01,
|
||||||
|
mouse = 0x02,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
// Subclass and protocol codes qualified by Class.mass_storage. The subclass identifies the
|
||||||
|
// command set the device understands; the protocol identifies the transport used to carry
|
||||||
|
// commands, data, and status over the bus.
|
||||||
|
const mass_storage = struct {
|
||||||
|
const SubClass = enum(u8) {
|
||||||
|
// SCSI command set not reported; de facto, treat as scsi
|
||||||
|
not_reported = 0x00,
|
||||||
|
// Reduced Block Commands: typically flash devices
|
||||||
|
rbc = 0x01,
|
||||||
|
// MMC-5 (ATAPI): CD and DVD drives
|
||||||
|
atapi = 0x02,
|
||||||
|
// QIC-157 tape drives (obsolete)
|
||||||
|
qic_157 = 0x03,
|
||||||
|
// UFI: floppy disk drives
|
||||||
|
ufi = 0x04,
|
||||||
|
// SFF-8070i (obsolete)
|
||||||
|
sff_8070i = 0x05,
|
||||||
|
// Transparent SCSI command set: the common case for flash drives and disks
|
||||||
|
scsi = 0x06,
|
||||||
|
// LSD FS: negotiated access to large storage devices
|
||||||
|
lsd_fs = 0x07,
|
||||||
|
// IEEE 1667
|
||||||
|
ieee_1667 = 0x08,
|
||||||
|
// Vendor-specific
|
||||||
|
vendor_specific = 0xFF,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
|
||||||
|
const Protocol = enum(u8) {
|
||||||
|
// Control/Bulk/Interrupt with command completion interrupt
|
||||||
|
cbi_completion_interrupt = 0x00,
|
||||||
|
// Control/Bulk/Interrupt without command completion interrupt
|
||||||
|
cbi = 0x01,
|
||||||
|
// Bulk-only transport: the common case for flash drives and disks
|
||||||
|
bulk_only = 0x50,
|
||||||
|
// USB attached SCSI
|
||||||
|
uas = 0x62,
|
||||||
|
// Vendor-specific
|
||||||
|
vendor_specific = 0xFF,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
// Subclass and protocol codes qualified by Class.communications (CDC). The protocol codes
|
||||||
|
// are model-specific; the useful invariant is the subclass, which selects the control model
|
||||||
|
// the interface implements.
|
||||||
|
const communications = struct {
|
||||||
|
const SubClass = enum(u8) {
|
||||||
|
// Direct line control model
|
||||||
|
direct_line = 0x01,
|
||||||
|
// Abstract control model: USB modems and serial adapters
|
||||||
|
abstract_control = 0x02,
|
||||||
|
// Telephone control model
|
||||||
|
telephone = 0x03,
|
||||||
|
// Multi-channel control model
|
||||||
|
multi_channel = 0x04,
|
||||||
|
// CAPI control model
|
||||||
|
capi = 0x05,
|
||||||
|
// Ethernet networking control model
|
||||||
|
ethernet = 0x06,
|
||||||
|
// ATM networking control model
|
||||||
|
atm = 0x07,
|
||||||
|
// Wireless handset control model
|
||||||
|
wireless_handset = 0x08,
|
||||||
|
// Device management
|
||||||
|
device_management = 0x09,
|
||||||
|
// Mobile direct line model
|
||||||
|
mobile_direct_line = 0x0A,
|
||||||
|
// OBEX
|
||||||
|
obex = 0x0B,
|
||||||
|
// Ethernet emulation model
|
||||||
|
ethernet_emulation = 0x0C,
|
||||||
|
// Network control model
|
||||||
|
network_control = 0x0D,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
// Subclass and protocol codes qualified by Class.wireless_controller.
|
||||||
|
const wireless_controller = struct {
|
||||||
|
const SubClass = enum(u8) {
|
||||||
|
// Radio frequency controllers
|
||||||
|
radio_frequency = 0x01,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
|
||||||
|
// Only meaningful when the subclass is radio_frequency
|
||||||
|
const Protocol = enum(u8) {
|
||||||
|
// Bluetooth programming interface
|
||||||
|
bluetooth = 0x01,
|
||||||
|
// Ultra-wideband radio control
|
||||||
|
ultra_wideband = 0x02,
|
||||||
|
// Remote NDIS
|
||||||
|
remote_ndis = 0x03,
|
||||||
|
// Bluetooth AMP controller
|
||||||
|
bluetooth_amp = 0x04,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
// Subclass and protocol codes qualified by Class.miscellaneous.
|
||||||
|
const miscellaneous = struct {
|
||||||
|
const SubClass = enum(u8) {
|
||||||
|
// Common class
|
||||||
|
common = 0x02,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
|
||||||
|
// Only meaningful when the subclass is common
|
||||||
|
const Protocol = enum(u8) {
|
||||||
|
// Interface association descriptor: at the device level, announces that the
|
||||||
|
// configuration groups interfaces into functions with IADs
|
||||||
|
interface_association = 0x01,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
// Subclass and protocol codes qualified by Class.application_specific.
|
||||||
|
const application_specific = struct {
|
||||||
|
const SubClass = enum(u8) {
|
||||||
|
// Device firmware upgrade
|
||||||
|
firmware_upgrade = 0x01,
|
||||||
|
// IrDA bridge
|
||||||
|
irda_bridge = 0x02,
|
||||||
|
// Test and measurement
|
||||||
|
test_and_measurement = 0x03,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
test "class codes match the USB-IF assignments" {
|
||||||
|
const std = @import("std");
|
||||||
|
const expectEqual = std.testing.expectEqual;
|
||||||
|
|
||||||
|
try expectEqual(0x03, @intFromEnum(Class.hid));
|
||||||
|
try expectEqual(0x09, @intFromEnum(Class.hub));
|
||||||
|
try expectEqual(0xFF, @intFromEnum(Class.vendor_specific));
|
||||||
|
|
||||||
|
// A typical flash drive: mass storage, transparent SCSI, bulk-only transport.
|
||||||
|
try expectEqual(0x06, @intFromEnum(mass_storage.SubClass.scsi));
|
||||||
|
try expectEqual(0x50, @intFromEnum(mass_storage.Protocol.bulk_only));
|
||||||
|
|
||||||
|
// A boot keyboard: HID, boot subclass, keyboard protocol.
|
||||||
|
try expectEqual(0x01, @intFromEnum(hid.SubClass.boot));
|
||||||
|
try expectEqual(0x01, @intFromEnum(hid.Protocol.keyboard));
|
||||||
|
|
||||||
|
// Class codes are non-exhaustive: unlisted values pass through undamaged.
|
||||||
|
const unknown: Class = @enumFromInt(0x42);
|
||||||
|
try expectEqual(0x42, @intFromEnum(unknown));
|
||||||
|
|
||||||
|
_ = hub.Protocol.hi_speed_multi_tt;
|
||||||
|
_ = communications.SubClass.abstract_control;
|
||||||
|
_ = wireless_controller.Protocol.bluetooth;
|
||||||
|
_ = miscellaneous.Protocol.interface_association;
|
||||||
|
_ = application_specific.SubClass.firmware_upgrade;
|
||||||
|
}
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user