Compare commits
117
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0628944b15 | ||
|
|
e6d0bb7ef0 | ||
|
|
5ca804d827 | ||
|
|
a299363b59 | ||
|
|
d8dd62c639 | ||
|
|
d106b6e8dc | ||
|
|
af2c766f42 | ||
|
|
d26262bf56 | ||
|
|
10b89c06ff | ||
|
|
a2a05d0b3d | ||
|
|
75d62660b0 | ||
|
|
a53c2b0193 | ||
|
|
bf481c080c | ||
|
|
3a78dcab3f | ||
|
|
470f93a83d | ||
|
|
7798706b41 | ||
|
|
ad40de03c2 | ||
|
|
d8778b4b70 | ||
|
|
79d859a111 | ||
|
|
37fb09f75e | ||
|
|
34ebeb968d | ||
|
|
3cc1d38dd0 | ||
|
|
36e804b848 | ||
|
|
be83a42d42 | ||
|
|
650a1b1595 | ||
|
|
d8c55c6f2f | ||
|
|
2ebfb0c3b0 | ||
|
|
888eaa74e1 | ||
|
|
ed76cbbc79 | ||
|
|
140229b88d | ||
|
|
cb2379fd06 | ||
|
|
1665b239b0 | ||
|
|
70ed0337f8 | ||
|
|
116b8f6c41 | ||
|
|
77901bbba6 | ||
|
|
78582d24d2 | ||
|
|
1cdffe21b1 | ||
|
|
4df90bc212 | ||
|
|
713e77354b | ||
|
|
abb7b1b634 | ||
|
|
f5f0e15769 | ||
|
|
8652b4a724 | ||
|
|
e8233127c7 | ||
|
|
88e92254e9 | ||
|
|
5725d35e5b | ||
|
|
8c95525793 | ||
|
|
5bba5d3363 | ||
|
|
80b72db676 | ||
|
|
aa0c97353a | ||
|
|
dd93204b44 | ||
|
|
c7b17aaa0e | ||
|
|
d7a154a596 | ||
|
|
1bf91115dd | ||
|
|
65244e3103 | ||
|
|
75ccfff171 | ||
|
|
2a583d55a8 | ||
|
|
d218d93f79 | ||
|
|
a5fe63c1dd | ||
|
|
6b3ae0c997 | ||
|
|
59104dd988 | ||
|
|
e499f500c3 | ||
|
|
2ab0d129a2 | ||
|
|
d702d2e9ae | ||
|
|
3ec2d1828a | ||
|
|
21657943a8 | ||
|
|
e612d948d2 | ||
|
|
4ef21fa083 | ||
|
|
125a3b4993 | ||
|
|
e7c7e7b94c | ||
|
|
a581712b09 | ||
|
|
8eb4210251 | ||
|
|
56110b0019 | ||
|
|
afbf10f7fc | ||
|
|
b61b7775b9 | ||
|
|
be81394be3 | ||
|
|
47610e8ee2 | ||
|
|
193fd71a50 | ||
|
|
1c2b3ae64d | ||
|
|
d19a0ae38d | ||
|
|
3d1de37d0e | ||
|
|
ceacc6b514 | ||
|
|
8754d4e46a | ||
|
|
15b70856c9 | ||
|
|
83881641ca | ||
|
|
b0f894f50c | ||
|
|
750a73f050 | ||
|
|
eabb81684a | ||
|
|
0a8c81b17a | ||
|
|
9316f9f1c3 | ||
|
|
7384a730be | ||
|
|
b9d9e1e523 | ||
|
|
37fb3cb0cf | ||
|
|
5d57e7b01c | ||
|
|
8a7235c725 | ||
|
|
f4590fcc19 | ||
|
|
bb7597ea0b | ||
|
|
f57a73e8a1 | ||
|
|
724de7bbd0 | ||
|
|
75bc429aa1 | ||
|
|
91aff2bc4a | ||
|
|
546dd44a2a | ||
|
|
7501bd1703 | ||
|
|
1b47de5058 | ||
|
|
26ac97df31 | ||
|
|
37f72a4df8 | ||
|
|
f02259cae0 | ||
|
|
5940864958 | ||
|
|
91f2cfa17b | ||
|
|
debe815a5c | ||
|
|
dba3939a0f | ||
|
|
43afe6bf2e | ||
|
|
ed7f542006 | ||
|
|
36c29d2d6d | ||
|
|
941ab091db | ||
|
|
cf7c6df41c | ||
|
|
26d2f5259c | ||
|
|
cb63d2e31b |
@@ -0,0 +1,16 @@
|
||||
# EditorConfig: https://editorconfig.org/
|
||||
# Follows the Zig style guide: https://ziglang.org/documentation/0.16.0/#Style-Guide
|
||||
|
||||
root = true
|
||||
|
||||
[*]
|
||||
charset = utf-8
|
||||
end_of_line = lf
|
||||
indent_style = space
|
||||
indent_size = 4
|
||||
trim_trailing_whitespace = true
|
||||
insert_final_newline = true
|
||||
|
||||
[*.zig]
|
||||
# "Line length: aim for 100; use common sense."
|
||||
max_line_length = 100
|
||||
@@ -0,0 +1 @@
|
||||
*.zig text eol=lf
|
||||
@@ -4,3 +4,6 @@ zig-out/
|
||||
|
||||
# JetBrains IDE
|
||||
.idea/
|
||||
|
||||
.claude/
|
||||
.github/
|
||||
@@ -2,12 +2,31 @@
|
||||
Codename: Shodan
|
||||
Version: 1
|
||||
|
||||
A small operating system, written from scratch in Zig — a bootloader (`src/boot/`)
|
||||
and a microkernel (`src/kernel/`), sharing a neutral handoff contract (`src/root.zig`).
|
||||
It boots x86-64 via UEFI, and so far has a framebuffer console, a physical frame
|
||||
allocator, its own paging with W^X permissions, interrupt/exception handling, a
|
||||
LAPIC timer, a kernel heap, a fixed-priority preemptive scheduler, and in-kernel IPC
|
||||
channels. See [`docs/`](docs/README.md) for how each piece works.
|
||||
A small resilient operating system, written from scratch in Zig.
|
||||
|
||||
## Zen of DanOS:
|
||||
|
||||
- Resilient Micro-Kernel Architecture.
|
||||
- Every process run in an isolated user space not kernel space.
|
||||
- Processes cannot take down the entire OS with it when they die or is killed
|
||||
- Stable public runtime library, private OS ABI.
|
||||
- Keeps a stable runtime for user space processes between OS versions (great for backwards compatibility)
|
||||
- Allows the underlying OS to be changed without effecting applications
|
||||
- Provides a boundary to enable compatibility between OS's e.g. POSIX, MUSL etc
|
||||
- Drivers are just isolated processes in user space.
|
||||
- Thin binaries that can be restarted like applications.
|
||||
- Useful during driver development.
|
||||
- Drivers can claim MMIO / ports
|
||||
- Driver resources (e.g. IRQ/Port/MMIO) claims are automatically cleaned up if the driver dies or is killed
|
||||
- Drivers can also hook into the process lifecyle to clean up or reset hardware
|
||||
- No legacy to deal with
|
||||
- Zig code uses a clean coding style (Zen of Zig)
|
||||
- Favor reading code over writing code.
|
||||
- No magic numbers.
|
||||
- No shortend names unless its for ABI compatibility or acronyms
|
||||
- Inter-Process Communication (IPC)
|
||||
- Publish and subscribe to Asynchronous Messages
|
||||
- Talk to services and processes synchronously
|
||||
|
||||
## Prerequisites
|
||||
|
||||
@@ -30,8 +49,10 @@ channels. See [`docs/`](docs/README.md) for how each piece works.
|
||||
zig build
|
||||
```
|
||||
|
||||
Produces the UEFI bootloader (`zig-out/bin/BOOTX64.efi`) and the kernel ELF
|
||||
(`zig-out/bin/kernel`).
|
||||
Produces a FHS-shaped `zig-out/` that *is* the danos filesystem and the boot volume:
|
||||
the UEFI bootloader at `zig-out/EFI/BOOT/BOOTX64.efi`, the kernel at
|
||||
`zig-out/system/kernel`, init at `zig-out/system/services/init`, drivers under
|
||||
`zig-out/system/drivers/`, and the initial-ramdisk at `zig-out/boot/`.
|
||||
|
||||
## Run
|
||||
|
||||
@@ -58,7 +79,7 @@ straight into CI.
|
||||
|
||||
## Documentation
|
||||
|
||||
Design notes explaining the *why* behind the code live in
|
||||
Design notes explaining *why* behind the code live in
|
||||
[`docs/`](docs/README.md) — start with [`docs/README.md`](docs/README.md).
|
||||
|
||||
## Logo
|
||||
|
||||
+274
-53
@@ -1,15 +1,24 @@
|
||||
const std = @import("std");
|
||||
const uefi = std.os.uefi;
|
||||
const elf = std.elf;
|
||||
const danos = @import("danos");
|
||||
const BootInfo = danos.BootInfo;
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const BootInformation = boot_handoff.BootInformation;
|
||||
const GraphicsOutput = uefi.protocol.GraphicsOutput;
|
||||
const EdidActive = uefi.protocol.edid.Active;
|
||||
const MemoryMapSlice = uefi.tables.MemoryMapSlice;
|
||||
|
||||
/// Name of the kernel ELF on the boot volume (installed to the ESP root by
|
||||
/// build.zig). UEFI wants a UTF-16, null-terminated path.
|
||||
const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("kernel");
|
||||
// The boot volume is the FHS-shaped zig-out (see build.zig / docs/README.md), so the
|
||||
// loader reads each artifact from its addressed FHS path. UEFI paths use backslashes;
|
||||
// the FAT driver walks the components itself, so no per-directory dance is needed.
|
||||
|
||||
/// The kernel image: /system/kernel.
|
||||
const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("system\\kernel");
|
||||
|
||||
/// The init program: /system/services/init.
|
||||
const init_file_name = std.unicode.utf8ToUtf16LeStringLiteral("system\\services\\init");
|
||||
|
||||
/// The initial-ramdisk (the VFS server + drivers), in /boot.
|
||||
const initial_ramdisk_file_name = std.unicode.utf8ToUtf16LeStringLiteral("boot\\initial-ramdisk.img");
|
||||
|
||||
/// Physical page size, and the sentinel UEFI uses to seek to end-of-file.
|
||||
const page_size = 4096;
|
||||
@@ -33,8 +42,16 @@ fn boot() !noreturn {
|
||||
|
||||
// Everything the kernel needs must be gathered *before* we exit boot
|
||||
// services, since afterwards none of these calls are usable.
|
||||
var boot_info: BootInfo = .{
|
||||
.framebuffer = try queryFramebuffer(bs),
|
||||
var boot_information: BootInformation = .{
|
||||
// A missing GOP (a headless machine) is not fatal — hand the kernel a
|
||||
// "no framebuffer" descriptor (base 0) and let it log to serial instead.
|
||||
.framebuffer = queryFramebuffer(bs) catch boot_handoff.Framebuffer{
|
||||
.base = 0,
|
||||
.width = 0,
|
||||
.height = 0,
|
||||
.pitch = 0,
|
||||
.format = .bgrx,
|
||||
},
|
||||
.memory_map = undefined, // filled by exitBootServices, just below
|
||||
.kernel_segments = undefined, // filled by loadKernel
|
||||
.kernel_segment_count = 0,
|
||||
@@ -44,16 +61,38 @@ fn boot() !noreturn {
|
||||
.acpi_rsdp = if (acpiRootSystemDescriptorPointer()) |p| @intFromPtr(p) else 0,
|
||||
};
|
||||
|
||||
const entry = try loadKernel(bs, &boot_info);
|
||||
const entry = try loadKernel(bs, &boot_information);
|
||||
|
||||
// Best effort: a volume without /system/services/init still boots (kernel-only).
|
||||
loadInit(bs, &boot_information) catch |err| {
|
||||
log("danos: no /system/services/init (");
|
||||
logBytes(@errorName(err));
|
||||
log(") - booting without user space\r\n");
|
||||
};
|
||||
|
||||
// Best effort: the initial_ramdisk (VFS server + drivers) is optional too.
|
||||
loadInitialRamdisk(bs, &boot_information) catch |err| {
|
||||
log("danos: no initial_ramdisk (");
|
||||
logBytes(@errorName(err));
|
||||
log(")\r\n");
|
||||
};
|
||||
|
||||
// Build the page tables the kernel starts life on: identity + a physmap of
|
||||
// low RAM, plus the higher-half kernel image once it links high. Allocated
|
||||
// now, while boot services (and the memory map) are still stable — nothing
|
||||
// is allocatable after ExitBootServices, and any allocation between fetching
|
||||
// the map and exiting would invalidate the map key.
|
||||
const cr3 = try buildBootstrapTables(bs, &boot_information);
|
||||
|
||||
log("danos: kernel loaded, exiting boot services\r\n");
|
||||
boot_info.memory_map = try exitBootServices(bs);
|
||||
boot_information.memory_map = try exitBootServices(bs);
|
||||
|
||||
// Hand control to the kernel. `danos.kernel_abi` is SysV, so the pointer is
|
||||
// passed in RDI as the kernel expects — not RCX, which this UEFI binary's
|
||||
// default `.c` convention (Microsoft x64) would use.
|
||||
const kernel: *const fn (*const BootInfo) callconv(danos.kernel_abi) noreturn = @ptrFromInt(entry);
|
||||
kernel(&boot_info);
|
||||
// Switch onto our tables and jump to the kernel in one uninterruptible step.
|
||||
// We load RDI explicitly (SystemV first arg) rather than trusting this UEFI
|
||||
// binary's Microsoft-x64 default, and jump straight to the (possibly
|
||||
// higher-half) entry — the bootstrap tables map both the low loader code
|
||||
// executing this and the kernel's link address.
|
||||
handoff(cr3, entry, &boot_information);
|
||||
}
|
||||
|
||||
/// A display resolution in pixels.
|
||||
@@ -61,7 +100,7 @@ const Resolution = struct { width: u32, height: u32 };
|
||||
|
||||
/// Switch the GPU to the monitor's native resolution (when we can determine it)
|
||||
/// and read the resulting graphics mode into our own framebuffer description.
|
||||
fn queryFramebuffer(bs: *uefi.tables.BootServices) !danos.Framebuffer {
|
||||
fn queryFramebuffer(bs: *uefi.tables.BootServices) !boot_handoff.Framebuffer {
|
||||
// Enumerate the handles carrying the Graphics Output Protocol. We go through
|
||||
// handles (rather than locateProtocol) so we can also ask them for their EDID,
|
||||
// which is what tells us the panel's native resolution.
|
||||
@@ -93,7 +132,7 @@ fn queryFramebuffer(bs: *uefi.tables.BootServices) !danos.Framebuffer {
|
||||
|
||||
/// Map a GOP pixel format to ours. bit_mask / blt_only have no linear 32bpp
|
||||
/// layout we can paint into, so they're rejected.
|
||||
fn pixelFormat(fmt: GraphicsOutput.PixelFormat) !danos.PixelFormat {
|
||||
fn pixelFormat(fmt: GraphicsOutput.PixelFormat) !boot_handoff.PixelFormat {
|
||||
return switch (fmt) {
|
||||
.red_green_blue_reserved_8_bit_per_color => .rgbx,
|
||||
.blue_green_red_reserved_8_bit_per_color => .bgrx,
|
||||
@@ -158,7 +197,7 @@ fn edidNative(edid: []const u8) ?Resolution {
|
||||
|
||||
/// Open the kernel on the volume we booted from, read it into a pool buffer,
|
||||
/// load its segments, and return the physical entry-point address.
|
||||
fn loadKernel(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !usize {
|
||||
fn loadKernel(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !usize {
|
||||
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
|
||||
return error.NoLoadedImage;
|
||||
const device = loaded.device_handle orelse return error.NoBootDevice;
|
||||
@@ -187,13 +226,190 @@ fn loadKernel(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !usize {
|
||||
read_total += n;
|
||||
}
|
||||
|
||||
return loadElf(bs, image, boot_info);
|
||||
return loadElf(bs, image, boot_information);
|
||||
}
|
||||
|
||||
// --- bootstrap page tables -------------------------------------------------
|
||||
// The kernel is (or will be) linked in the higher half but loaded low; the
|
||||
// firmware's identity map doesn't cover the higher half, so the loader builds
|
||||
// the first set of real page tables and switches CR3 before jumping in. They
|
||||
// carry: an identity map of low RAM (so the loader's own code/stack executing
|
||||
// the switch stays valid, and the low-linked kernel keeps working during the
|
||||
// staged move), a physmap at boot_handoff.physmap_base (the kernel's permanent way to
|
||||
// reach physical memory), and 4 KiB mappings of any higher-half kernel segment.
|
||||
// The kernel later builds its own precise tables (paging.init) and abandons
|
||||
// these; they leak as reserved LoaderData (~a handful of frames).
|
||||
|
||||
const pte_present: u64 = 1 << 0;
|
||||
const pte_write: u64 = 1 << 1;
|
||||
const pte_ps: u64 = 1 << 7; // page-size: a 2 MiB leaf at the PD level
|
||||
const pte_address: u64 = 0x000F_FFFF_FFFF_F000;
|
||||
const gib: u64 = 1 << 30;
|
||||
|
||||
/// A bump allocator over a pre-reserved block of zeroed frames, for page tables.
|
||||
const TablePool = struct {
|
||||
base: usize,
|
||||
next: usize,
|
||||
cap: usize,
|
||||
|
||||
fn alloc(self: *TablePool) !u64 {
|
||||
if (self.next >= self.cap) return error.OutOfBootstrapFrames;
|
||||
const frame = self.base + self.next * page_size;
|
||||
self.next += 1;
|
||||
@memset(@as(*[512]u64, @ptrFromInt(frame)), 0);
|
||||
return frame;
|
||||
}
|
||||
|
||||
fn table(physical: u64) *[512]u64 {
|
||||
return @ptrFromInt(physical);
|
||||
}
|
||||
|
||||
/// Return the next-level table an entry points at, creating it if absent.
|
||||
fn descend(self: *TablePool, entry: *u64) !u64 {
|
||||
if (entry.* & pte_present != 0) return entry.* & pte_address;
|
||||
const frame = try self.alloc();
|
||||
entry.* = frame | pte_present | pte_write;
|
||||
return frame;
|
||||
}
|
||||
|
||||
fn map2M(self: *TablePool, pml4: u64, virtual: u64, physical: u64) !void {
|
||||
const pml4e = &table(pml4)[(virtual >> 39) & 0x1FF];
|
||||
const pdpt = try self.descend(pml4e);
|
||||
const pdpte = &table(pdpt)[(virtual >> 30) & 0x1FF];
|
||||
const pd = try self.descend(pdpte);
|
||||
table(pd)[(virtual >> 21) & 0x1FF] = (physical & ~@as(u64, 0x1F_FFFF)) | pte_present | pte_write | pte_ps;
|
||||
}
|
||||
|
||||
fn map4K(self: *TablePool, pml4: u64, virtual: u64, physical: u64) !void {
|
||||
const pml4e = &table(pml4)[(virtual >> 39) & 0x1FF];
|
||||
const pdpt = try self.descend(pml4e);
|
||||
const pdpte = &table(pdpt)[(virtual >> 30) & 0x1FF];
|
||||
const pd = try self.descend(pdpte);
|
||||
const pde = &table(pd)[(virtual >> 21) & 0x1FF];
|
||||
const pt = try self.descend(pde);
|
||||
table(pt)[(virtual >> 12) & 0x1FF] = (physical & pte_address) | pte_present | pte_write;
|
||||
}
|
||||
};
|
||||
|
||||
/// Build the bootstrap tables and return the physical PML4 address (for CR3).
|
||||
/// No NX bits are set anywhere, so EFER.NXE (still off here) is irrelevant.
|
||||
fn buildBootstrapTables(bs: *uefi.tables.BootServices, boot_information: *const BootInformation) !u64 {
|
||||
// 64 frames (256 KiB) — comfortably covers a PML4, two PDPTs, eight PDs for
|
||||
// the 4 GiB identity+physmap ranges, plus the kernel image's PTs.
|
||||
const pool_pages = 64;
|
||||
const block = try bs.allocatePages(.any, .loader_data, pool_pages);
|
||||
var pool = TablePool{ .base = @intFromPtr(block.ptr), .next = 0, .cap = pool_pages };
|
||||
|
||||
const pml4 = try pool.alloc();
|
||||
|
||||
// Identity + physmap for low RAM. 4 GiB covers all of QEMU's RAM and MMIO
|
||||
// (LAPIC/IOAPIC/HPET/ECAM/framebuffer under q35); a machine with RAM or a
|
||||
// framebuffer above 4 GiB would extend this — see the fb window below.
|
||||
var address: u64 = 0;
|
||||
while (address < 4 * gib) : (address += 2 << 20) {
|
||||
try pool.map2M(pml4, address, address); // identity
|
||||
try pool.map2M(pml4, boot_handoff.physicalToVirtual(address), address); // physmap
|
||||
}
|
||||
|
||||
// A framebuffer above the 4 GiB window needs its own identity + physmap
|
||||
// pages (the kernel touches fb.base before it builds its own tables).
|
||||
const fb = boot_information.framebuffer;
|
||||
if (fb.present() and fb.base + @as(u64, fb.pitch) * fb.height > 4 * gib) {
|
||||
var p: u64 = fb.base & ~@as(u64, 0x1F_FFFF);
|
||||
const fb_end = fb.base + @as(u64, fb.pitch) * fb.height;
|
||||
while (p < fb_end) : (p += 2 << 20) {
|
||||
try pool.map2M(pml4, p, p);
|
||||
try pool.map2M(pml4, boot_handoff.physicalToVirtual(p), p);
|
||||
}
|
||||
}
|
||||
|
||||
// Higher-half kernel segments (virtual != physical). While the kernel still links
|
||||
// low its segments sit in the identity range and need no separate mapping
|
||||
// (and 4 KiB-mapping them would collide with the 2 MiB identity leaves), so
|
||||
// only map segments that actually live in the higher half.
|
||||
for (boot_information.kernel_segments[0..boot_information.kernel_segment_count]) |seg| {
|
||||
if (seg.virtual < boot_handoff.kernel_virt_base) continue;
|
||||
var off: u64 = 0;
|
||||
while (off < seg.pages * page_size) : (off += page_size) {
|
||||
try pool.map4K(pml4, seg.virtual + off, seg.physical + off);
|
||||
}
|
||||
}
|
||||
|
||||
return pml4;
|
||||
}
|
||||
|
||||
/// Switch onto `cr3` and jump to the kernel `entry` with `boot_information` in RDI,
|
||||
/// interrupts off, in one block so nothing runs between the CR3 load and the
|
||||
/// jump. The identity mapping keeps this low loader code valid across the CR3
|
||||
/// load; the jump target is mapped (identity while low, higher-half once high).
|
||||
fn handoff(cr3: u64, entry: usize, boot_information: *const BootInformation) noreturn {
|
||||
asm volatile (
|
||||
\\cli
|
||||
\\movq %[cr3], %%cr3
|
||||
\\movq %[bi], %%rdi
|
||||
\\callq *%[entry]
|
||||
:
|
||||
: [cr3] "r" (cr3),
|
||||
[bi] "r" (boot_information),
|
||||
[entry] "r" (entry),
|
||||
: .{ .memory = true });
|
||||
unreachable;
|
||||
}
|
||||
|
||||
/// Read a whole file off the boot volume into a pool buffer that outlives the
|
||||
/// loader. The buffer is deliberately NOT freed: it's LoaderData, which the
|
||||
/// memory-map conversion classifies as reserved, so the kernel identity-maps it
|
||||
/// and reads from there. Returns the buffer (pointer + length).
|
||||
fn loadFile(bs: *uefi.tables.BootServices, name: [*:0]const u16) ![]u8 {
|
||||
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
|
||||
return error.NoLoadedImage;
|
||||
const device = loaded.device_handle orelse return error.NoBootDevice;
|
||||
const fs = (try bs.handleProtocol(uefi.protocol.SimpleFileSystem, device)) orelse
|
||||
return error.NoFileSystem;
|
||||
|
||||
const root = try fs.openVolume();
|
||||
defer _ = root.close() catch {};
|
||||
|
||||
const file = try root.open(name, .read, .{});
|
||||
defer _ = file.close() catch {};
|
||||
|
||||
try file.setPosition(seek_end);
|
||||
const size: usize = @intCast(try file.getPosition());
|
||||
try file.setPosition(0);
|
||||
if (size == 0) return error.EmptyFile;
|
||||
|
||||
const image = try bs.allocatePool(.loader_data, size); // survives the handoff
|
||||
|
||||
var read_total: usize = 0;
|
||||
while (read_total < size) {
|
||||
const n = try file.read(image[read_total..]);
|
||||
if (n == 0) return error.UnexpectedEof;
|
||||
read_total += n;
|
||||
}
|
||||
return image[0..size];
|
||||
}
|
||||
|
||||
/// Ferry the init program (/system/services/init) to the kernel. The kernel does the ELF
|
||||
/// loading itself (into ring-3 mappings) — the loader just carries the bytes.
|
||||
fn loadInit(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||
const image = try loadFile(bs, init_file_name);
|
||||
boot_information.init_base = @intFromPtr(image.ptr);
|
||||
boot_information.init_len = image.len;
|
||||
log("danos: /system/services/init loaded\r\n");
|
||||
}
|
||||
|
||||
/// Ferry the initial_ramdisk (the VFS server + drivers) to the kernel, same as init.
|
||||
fn loadInitialRamdisk(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||
const image = try loadFile(bs, initial_ramdisk_file_name);
|
||||
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
||||
boot_information.initial_ramdisk_len = image.len;
|
||||
log("danos: initial_ramdisk loaded\r\n");
|
||||
}
|
||||
|
||||
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
||||
/// record each segment's layout so the kernel can re-map itself with the right
|
||||
/// permissions.
|
||||
fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !usize {
|
||||
fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_information: *BootInformation) !usize {
|
||||
if (image.len < @sizeOf(elf.Elf64_Ehdr)) return error.NotElf;
|
||||
const ehdr: *const elf.Elf64_Ehdr = @ptrCast(@alignCast(image.ptr));
|
||||
|
||||
@@ -210,7 +426,9 @@ fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !us
|
||||
|
||||
// Reserve the exact physical pages this segment is linked at. This
|
||||
// requires the segment's p_paddr to be free in the firmware memory map;
|
||||
// if it collides, adjust `image_base` in build.zig.
|
||||
// if it collides, adjust `image_base` in build.zig. (Once the kernel
|
||||
// links high — M2 step 4 — p_paddr becomes a separate low load address
|
||||
// via the linker's AT(), and this stays a valid physical allocation.)
|
||||
const mem_sz: usize = @intCast(phdr.p_memsz);
|
||||
const pages = (mem_sz + page_size - 1) / page_size;
|
||||
const dest: [*]align(page_size) uefi.Page = @ptrFromInt(phdr.p_paddr);
|
||||
@@ -223,15 +441,18 @@ fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !us
|
||||
@memcpy(bytes[0..file_sz], image[off..][0..file_sz]);
|
||||
@memset(bytes[file_sz..mem_sz], 0);
|
||||
|
||||
// Record it (identity-loaded: virtual == physical) for the kernel's VMM.
|
||||
const n = boot_info.kernel_segment_count;
|
||||
if (n < boot_info.kernel_segments.len) {
|
||||
boot_info.kernel_segments[n] = .{
|
||||
.virt = phdr.p_vaddr,
|
||||
// Record the virtual link address and the physical load address so the
|
||||
// kernel can map itself with the right permissions post-switch. They're
|
||||
// equal while the kernel links low; they diverge once it links high.
|
||||
const n = boot_information.kernel_segment_count;
|
||||
if (n < boot_information.kernel_segments.len) {
|
||||
boot_information.kernel_segments[n] = .{
|
||||
.virtual = phdr.p_vaddr,
|
||||
.physical = phdr.p_paddr,
|
||||
.pages = pages,
|
||||
.flags = phdr.p_flags,
|
||||
};
|
||||
boot_info.kernel_segment_count = n + 1;
|
||||
boot_information.kernel_segment_count = n + 1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -242,27 +463,27 @@ fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !us
|
||||
/// neutral form. Allocating the buffers can itself change the map (invalidating
|
||||
/// the key), so retry until it takes. Both buffers are LoaderData, which survives
|
||||
/// the exit, so the returned map stays valid for the kernel.
|
||||
fn exitBootServices(bs: *uefi.tables.BootServices) !danos.MemoryMap {
|
||||
fn exitBootServices(bs: *uefi.tables.BootServices) !boot_handoff.MemoryMap {
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 8) : (attempts += 1) {
|
||||
const info = try bs.getMemoryMapInfo();
|
||||
// Spare descriptors to absorb the growth from the allocations below.
|
||||
const cap = info.len + 8;
|
||||
const map_buf = try bs.allocatePool(.loader_data, cap * info.descriptor_size);
|
||||
const regions_buf = try bs.allocatePool(.loader_data, cap * @sizeOf(danos.MemoryRegion));
|
||||
const map = bs.getMemoryMap(map_buf) catch {
|
||||
_ = bs.freePool(map_buf.ptr) catch {};
|
||||
_ = bs.freePool(regions_buf.ptr) catch {};
|
||||
const map_buffer = try bs.allocatePool(.loader_data, cap * info.descriptor_size);
|
||||
const regions_buffer = try bs.allocatePool(.loader_data, cap * @sizeOf(boot_handoff.MemoryRegion));
|
||||
const map = bs.getMemoryMap(map_buffer) catch {
|
||||
_ = bs.freePool(map_buffer.ptr) catch {};
|
||||
_ = bs.freePool(regions_buffer.ptr) catch {};
|
||||
continue;
|
||||
};
|
||||
bs.exitBootServices(uefi.handle, map.info.key) catch {
|
||||
_ = bs.freePool(map_buf.ptr) catch {};
|
||||
_ = bs.freePool(regions_buf.ptr) catch {};
|
||||
_ = bs.freePool(map_buffer.ptr) catch {};
|
||||
_ = bs.freePool(regions_buffer.ptr) catch {};
|
||||
continue;
|
||||
};
|
||||
// Boot services are gone; do not touch `bs` again. Converting the map is
|
||||
// pure computation on memory we already hold, so it's safe here.
|
||||
return convertMemoryMap(map, regions_buf);
|
||||
return convertMemoryMap(map, regions_buffer);
|
||||
}
|
||||
return error.ExitBootServicesFailed;
|
||||
}
|
||||
@@ -271,8 +492,8 @@ fn exitBootServices(bs: *uefi.tables.BootServices) !danos.MemoryMap {
|
||||
/// into `out` (sized for at least `map.info.len` regions). Adjacent regions of
|
||||
/// the same kind are coalesced. This is the loader's job precisely so the kernel
|
||||
/// never sees UEFI's vocabulary — the same seam the framebuffer already uses.
|
||||
fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
||||
const regions: [*]danos.MemoryRegion = @ptrCast(@alignCast(out.ptr));
|
||||
fn convertMemoryMap(map: MemoryMapSlice, out: []u8) boot_handoff.MemoryMap {
|
||||
const regions: [*]boot_handoff.MemoryRegion = @ptrCast(@alignCast(out.ptr));
|
||||
// We're about to call boot-services memory `usable`, but our own stack lives
|
||||
// in it and the kernel starts out running on it. Keep the region holding the
|
||||
// current stack pointer reserved so it's never handed out.
|
||||
@@ -289,16 +510,16 @@ fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
||||
if (d.number_of_pages == 0) continue;
|
||||
var kind = classify(d);
|
||||
// The descriptor we're executing on stays reserved (see rsp above).
|
||||
const region_end = d.physical_start + d.number_of_pages * danos.page_size;
|
||||
const region_end = d.physical_start + d.number_of_pages * page_size;
|
||||
if (kind == .usable and rsp >= d.physical_start and rsp < region_end) kind = .reserved;
|
||||
|
||||
// Coalesce with the previous region if it's the same kind and contiguous.
|
||||
if (count > 0) {
|
||||
const prev = ®ions[count - 1];
|
||||
if (prev.kind == kind and
|
||||
prev.base + prev.pages * danos.page_size == d.physical_start)
|
||||
const previous = ®ions[count - 1];
|
||||
if (previous.kind == kind and
|
||||
previous.base + previous.pages * page_size == d.physical_start)
|
||||
{
|
||||
prev.pages += d.number_of_pages;
|
||||
previous.pages += d.number_of_pages;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
@@ -314,7 +535,7 @@ fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
||||
|
||||
/// Map a UEFI descriptor to danos's neutral kind. A region that isn't
|
||||
/// writeback-cacheable (`wb`) isn't backed by real RAM — it's device registers or
|
||||
/// a reserved address-space window (e.g. PCIe config space) — so it's `mmio`
|
||||
/// a reserved address-space window (e.g. PCIe configuration space) — so it's `mmio`
|
||||
/// regardless of type. UEFI overloads `reserved_memory_type` for both reserved RAM
|
||||
/// and such holes, and the cache attribute is what actually tells them apart.
|
||||
///
|
||||
@@ -323,7 +544,7 @@ fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
||||
/// ever the firmware's (the one live piece, our stack, is reserved by the caller).
|
||||
/// Anything unrecognised is `reserved` — the safe default; our own LoaderData (the
|
||||
/// kernel image and these buffers) lands there and stays reserved.
|
||||
fn classify(d: *const uefi.tables.MemoryDescriptor) danos.MemoryKind {
|
||||
fn classify(d: *const uefi.tables.MemoryDescriptor) boot_handoff.MemoryKind {
|
||||
if (!d.attribute.wb) return .mmio;
|
||||
return switch (d.type) {
|
||||
.conventional_memory, .boot_services_code, .boot_services_data => .usable,
|
||||
@@ -335,33 +556,33 @@ fn classify(d: *const uefi.tables.MemoryDescriptor) danos.MemoryKind {
|
||||
}
|
||||
|
||||
/// Write a compile-time string to the console (best effort).
|
||||
fn log(comptime msg: []const u8) void {
|
||||
fn log(comptime message: []const u8) void {
|
||||
const out = uefi.system_table.con_out orelse return;
|
||||
_ = out.outputString(std.unicode.utf8ToUtf16LeStringLiteral(msg)) catch {};
|
||||
_ = out.outputString(std.unicode.utf8ToUtf16LeStringLiteral(message)) catch {};
|
||||
}
|
||||
|
||||
/// Write a runtime ASCII byte string (e.g. an @errorName) by widening to UTF-16.
|
||||
fn logBytes(bytes: []const u8) void {
|
||||
const out = uefi.system_table.con_out orelse return;
|
||||
var buf: [128]u16 = undefined;
|
||||
var buffer: [128]u16 = undefined;
|
||||
var i: usize = 0;
|
||||
for (bytes) |b| {
|
||||
if (i + 1 >= buf.len) break;
|
||||
buf[i] = b;
|
||||
if (i + 1 >= buffer.len) break;
|
||||
buffer[i] = b;
|
||||
i += 1;
|
||||
}
|
||||
buf[i] = 0;
|
||||
_ = out.outputString(buf[0..i :0].ptr) catch {};
|
||||
buffer[i] = 0;
|
||||
_ = out.outputString(buffer[0..i :0].ptr) catch {};
|
||||
}
|
||||
|
||||
fn acpiRootSystemDescriptorPointer() ?*const anyopaque {
|
||||
const table_entries = uefi.system_table.number_of_table_entries;
|
||||
const config_tables = uefi.system_table.configuration_table;
|
||||
const configuration_tables = uefi.system_table.configuration_table;
|
||||
const acpi2 = uefi.tables.ConfigurationTable.acpi_20_table_guid;
|
||||
const acpi1 = uefi.tables.ConfigurationTable.acpi_10_table_guid;
|
||||
|
||||
for (0..table_entries) |i| {
|
||||
const entry = config_tables[i];
|
||||
const entry = configuration_tables[i];
|
||||
if (entry.vendor_guid.eql(acpi2) or entry.vendor_guid.eql(acpi1)) {
|
||||
return entry.vendor_table;
|
||||
}
|
||||
@@ -48,50 +48,229 @@ fn timestamp(b: *std.Build) []const u8 {
|
||||
});
|
||||
}
|
||||
|
||||
/// Build one user-space binary the same way for every program (init, and later
|
||||
/// the VFS server + drivers): freestanding, ReleaseSmall, `.large` code model
|
||||
/// (the image base is above 4 GiB — smaller models emit 32-bit relocations that
|
||||
/// can't reach), linked against the `runtime` runtime library with the shared user
|
||||
/// link script. Pinned to LLVM + LLD so the script's PHDRS (segment permissions)
|
||||
/// are authoritative — the kernel's W^X user-ELF loader requires exact perms.
|
||||
fn addUserBinary(
|
||||
b: *std.Build,
|
||||
target: std.Build.ResolvedTarget,
|
||||
runtime_module: *std.Build.Module,
|
||||
posix_module: *std.Build.Module,
|
||||
mmio_module: *std.Build.Module,
|
||||
xkeyboard_config_module: *std.Build.Module,
|
||||
acpi_ids_module: *std.Build.Module,
|
||||
name: []const u8,
|
||||
root: []const u8,
|
||||
) *std.Build.Step.Compile {
|
||||
const exe = b.addExecutable(.{
|
||||
.name = name,
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(root),
|
||||
.target = target,
|
||||
.optimize = .ReleaseSmall,
|
||||
.code_model = .large,
|
||||
.single_threaded = true,
|
||||
.sanitize_c = .off,
|
||||
.stack_check = false,
|
||||
.stack_protector = false,
|
||||
.imports = &.{
|
||||
.{ .name = "runtime", .module = runtime_module },
|
||||
// POSIX/C compatibility layer, available to any program that wants it
|
||||
// (danos-native code uses `runtime` directly). See library/posix/.
|
||||
.{ .name = "posix", .module = posix_module },
|
||||
// Typed volatile MMIO + memory barriers, for drivers. See library/mmio/.
|
||||
.{ .name = "mmio", .module = mmio_module },
|
||||
// Keyboard layouts (keycode + modifiers -> keysym/character), available
|
||||
// to any program that wants it. See library/xkeyboard-config/.
|
||||
.{ .name = "xkeyboard-config", .module = xkeyboard_config_module },
|
||||
// ACPI/PnP hardware-ID registry, so drivers name devices
|
||||
// (HardwareId.ps2_keyboard) instead of magic "_HID" strings.
|
||||
.{ .name = "acpi-ids", .module = acpi_ids_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
exe.setLinkerScript(b.path("library/runtime/user.ld"));
|
||||
exe.entry = .{ .symbol_name = "_start" };
|
||||
exe.image_base = 0x7000_0000_0000;
|
||||
exe.use_llvm = true;
|
||||
exe.use_lld = true;
|
||||
return exe;
|
||||
}
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
ensureZigVersion();
|
||||
|
||||
const target = b.standardTargetOptions(.{});
|
||||
const optimize = b.standardOptimizeOption(.{});
|
||||
|
||||
// Shared handoff definitions (BootInfo, Framebuffer, ...). No target is set,
|
||||
// so the module inherits the target of whichever binary imports it — the
|
||||
// freestanding kernel or the UEFI bootloader.
|
||||
const mod = b.addModule("danos", .{
|
||||
.root_source_file = b.path("src/root.zig"),
|
||||
// The three shared contracts, each with its own audience so every import
|
||||
// declares which one it speaks (no target is set, so each inherits the target of
|
||||
// whichever binary imports it). See docs/coding-standards.md.
|
||||
// boot-handoff : loader <-> kernel (BootInformation, framebuffer, VM layout)
|
||||
// abi : kernel <-> runtime, core (SystemCall, mmap prot flags, page_size)
|
||||
// device-abi : kernel <-> user, devices (DeviceDescriptor, DeviceClass, ...)
|
||||
const boot_handoff_module = b.addModule("boot-handoff", .{
|
||||
.root_source_file = b.path("system/boot-handoff.zig"),
|
||||
});
|
||||
const abi_module = b.addModule("abi", .{
|
||||
.root_source_file = b.path("system/abi.zig"),
|
||||
});
|
||||
// The devices sub-project's public interface (the flat wire types), exposed as
|
||||
// its own module like vfs-protocol — importable by user space, unlike the
|
||||
// kernel-internal device model it also feeds (system/devices/device-model.zig).
|
||||
const device_abi_module = b.addModule("device-abi", .{
|
||||
.root_source_file = b.path("system/devices/device-abi.zig"),
|
||||
});
|
||||
// PCI class-code decoding (class/subclass/prog-IF -> names). Pure reference data,
|
||||
// shared by kernel discovery (the device-tree dump) and any user-space PCI tool.
|
||||
const pci_class_module = b.addModule("pci-class", .{
|
||||
.root_source_file = b.path("system/devices/pci-class.zig"),
|
||||
});
|
||||
// ACPI/PnP hardware-ID (_HID) names — the flat analog of pci-class for acpi_device
|
||||
// nodes. Also shared reference data.
|
||||
// The AML interpreter, a build module so the ring-3 acpi service can run the
|
||||
// same parser the kernel does (docs/m19-m20-plan.md decision 1). Pure Zig,
|
||||
// no kernel imports — one source, two builds.
|
||||
const aml_module = b.addModule("aml", .{
|
||||
.root_source_file = b.path("system/devices/aml/aml.zig"),
|
||||
});
|
||||
|
||||
const acpi_ids_module = b.addModule("acpi-ids", .{
|
||||
.root_source_file = b.path("system/devices/acpi-ids.zig"),
|
||||
});
|
||||
|
||||
// Kernel tunables (maximum_cpus, stack sizes, tick rate). A dependency-free module of
|
||||
// compile-time constants, imported wherever a knob is read; keeps the trade-offs
|
||||
// in one place instead of scattered across the tree. See system/parameters.zig.
|
||||
const parameters_module = b.addModule("parameters", .{
|
||||
.root_source_file = b.path("system/parameters.zig"),
|
||||
});
|
||||
|
||||
// Architecture-specific kernel code (CPU ops, entry, later GDT/IDT/paging).
|
||||
// The generic kernel imports this as "arch" and never names x86_64, so a new
|
||||
// The generic kernel imports this as "architecture" and never names x86_64, so a new
|
||||
// architecture is a matter of pointing this module at a different directory.
|
||||
const arch_mod = b.addModule("arch", .{
|
||||
.root_source_file = b.path("src/kernel/arch/x86_64/cpu.zig"),
|
||||
const architecture_module = b.addModule("architecture", .{
|
||||
.root_source_file = b.path("system/kernel/architecture/x86_64/cpu.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "danos", .module = mod }, // paging uses the shared BootInfo/memory-map types
|
||||
.{ .name = "boot-handoff", .module = boot_handoff_module }, // paging uses BootInformation/memory-map + physicalToVirtual
|
||||
.{ .name = "abi", .module = abi_module }, // paging works in page_size units
|
||||
.{ .name = "parameters", .module = parameters_module }, // maximum_cpus, ist_stack_size, timer_hz
|
||||
},
|
||||
});
|
||||
// CPU-exception stubs — real assembly, since they need cross-symbol
|
||||
// jumps/calls that Zig inline asm can't express (see the file's header).
|
||||
arch_mod.addAssemblyFile(b.path("src/kernel/arch/x86_64/isr.s"));
|
||||
architecture_module.addAssemblyFile(b.path("system/kernel/architecture/x86_64/isr.s"));
|
||||
// The AP bring-up trampoline: 16-/32-/64-bit mode-switch code that can't be
|
||||
// inline asm (it runs relocated to a low page, not at its link address).
|
||||
architecture_module.addAssemblyFile(b.path("system/kernel/architecture/x86_64/trampoline.s"));
|
||||
|
||||
// Firmware-agnostic device discovery. The generic kernel imports this as
|
||||
// "platform" and asks it to enumerate hardware into a backend-neutral device
|
||||
// tree, never naming ACPI (or, later, device-tree) — the same discipline the
|
||||
// arch module applies to CPU code. The backend is selected at runtime from
|
||||
// the boot handoff (see src/device/platform.zig).
|
||||
const platform_mod = b.addModule("platform", .{
|
||||
.root_source_file = b.path("src/device/platform.zig"),
|
||||
// architecture module applies to CPU code. The backend is selected at runtime from
|
||||
// the boot handoff (see system/devices/platform.zig).
|
||||
const platform_module = b.addModule("platform", .{
|
||||
.root_source_file = b.path("system/devices/platform.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "danos", .module = mod }, // BootInfo (carries the ACPI RSDP)
|
||||
.{ .name = "boot-handoff", .module = boot_handoff_module }, // BootInformation (carries the ACPI RSDP), physicalToVirtual
|
||||
.{ .name = "abi", .module = abi_module }, // acpi.zig works in page_size units
|
||||
.{ .name = "device-abi", .module = device_abi_module }, // device-model's DeviceClass/ResourceKind live here
|
||||
.{ .name = "pci-class", .module = pci_class_module }, // decode PCI class codes in the device dump
|
||||
.{ .name = "acpi-ids", .module = acpi_ids_module }, // decode ACPI _HID names in the device dump
|
||||
.{ .name = "parameters", .module = parameters_module }, // maximum_cpus (the discovery pool)
|
||||
},
|
||||
});
|
||||
|
||||
// Compile-time config the kernel reads as `@import("build_options")`. The
|
||||
// The VFS wire protocol: the vfs sub-project's public interface, exposed as its
|
||||
// own module. Both the vfs server and the runtime's file layer (unistd/stdio)
|
||||
// depend on this contract by name — neither reaches into the other's files. This
|
||||
// is the first "protocol module" (see docs/driver-model.md); usb/block will
|
||||
// expose theirs the same way.
|
||||
const vfs_protocol_module = b.addModule("vfs-protocol", .{
|
||||
.root_source_file = b.path("system/services/vfs/protocol.zig"),
|
||||
});
|
||||
|
||||
// The input wire protocol: the input service's public interface, exposed as its own
|
||||
// module the same way vfs-protocol is. Shared by the input service, the runtime's
|
||||
// `input` helper (subscribe/publish), and every source and subscriber.
|
||||
const input_protocol_module = b.addModule("input-protocol", .{
|
||||
.root_source_file = b.path("system/services/input/protocol.zig"),
|
||||
});
|
||||
|
||||
// The danos-native user-space runtime: system_call wrappers, the C-convention
|
||||
// heap, IPC helpers, the process start shim, device access. This is the stable
|
||||
// application ABI; POSIX compatibility is a separate library on top (see below).
|
||||
// Compiled into every user binary (see addUserBinary), so it inherits each exe's
|
||||
// `.large` code model — do NOT set a target/code_model here. It imports `abi`
|
||||
// for the shared SystemCall numbers / mmap flags, `device-abi` for the device
|
||||
// types its `device` helper wraps, and re-exports `vfs-protocol` for the VFS
|
||||
// server. It never touches `boot-handoff` — user space has no business with the
|
||||
// loader↔kernel handoff.
|
||||
const runtime_module = b.addModule("runtime", .{
|
||||
.root_source_file = b.path("library/runtime/runtime.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
.{ .name = "device-abi", .module = device_abi_module },
|
||||
.{ .name = "vfs-protocol", .module = vfs_protocol_module },
|
||||
.{ .name = "input-protocol", .module = input_protocol_module },
|
||||
},
|
||||
});
|
||||
|
||||
// The device-manager protocol: hello + (M18.2) tree reports, exposed as its
|
||||
// own module like the other protocol modules. Imported through the runtime.
|
||||
const device_manager_protocol_module = b.addModule("device-manager-protocol", .{
|
||||
.root_source_file = b.path("system/services/device-manager/device-manager-protocol.zig"),
|
||||
});
|
||||
runtime_module.addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
|
||||
// Typed volatile MMIO register access + memory-ordering barriers, for drivers on
|
||||
// top of an mmio_map grant. Depends only on `builtin` (arch-conditional barriers);
|
||||
// no target set, so it inherits each driver's. See library/mmio/mmio.zig.
|
||||
const mmio_module = b.addModule("mmio", .{
|
||||
.root_source_file = b.path("library/mmio/mmio.zig"),
|
||||
});
|
||||
|
||||
// Keyboard layouts compiled from the X11 xkeyboard-config database into native Zig
|
||||
// (keycode + modifiers -> keysym/character). The `layouts` tables are generated by
|
||||
// tools/make-xkeyboard-config.py; `xkeyboard-config` is the hand-written API over them.
|
||||
// No target set, so each inherits its importer's. See library/xkeyboard-config/.
|
||||
const xkb_layouts_module = b.addModule("layouts", .{
|
||||
.root_source_file = b.path("library/xkeyboard-config/generated/layouts.zig"),
|
||||
});
|
||||
const xkeyboard_config_module = b.addModule("xkeyboard-config", .{
|
||||
.root_source_file = b.path("library/xkeyboard-config/xkeyboard-config.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "layouts", .module = xkb_layouts_module },
|
||||
},
|
||||
});
|
||||
|
||||
// The POSIX / C compatibility layer, a separate library layered strictly over the
|
||||
// runtime (it calls the runtime's IPC/heap, never system calls directly). This is
|
||||
// the one place POSIX/C spellings are allowed verbatim — see docs/coding-standards.md
|
||||
// and library/posix/posix.zig.
|
||||
const posix_module = b.addModule("posix", .{
|
||||
.root_source_file = b.path("library/posix/posix.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "runtime", .module = runtime_module },
|
||||
.{ .name = "vfs-protocol", .module = vfs_protocol_module },
|
||||
},
|
||||
});
|
||||
|
||||
// The initial_ramdisk container format, shared by the kernel (unpacks it) and the
|
||||
// build-time packer tools/make-initial-ramdisk.py (produces it). No dependencies.
|
||||
const initial_ramdisk_module = b.addModule("initial-ramdisk", .{
|
||||
.root_source_file = b.path("system/initial-ramdisk.zig"),
|
||||
});
|
||||
|
||||
// Compile-time configuration the kernel reads as `@import("build_options")`. The
|
||||
// QEMU test harness sets -Dtest-case=<name> to run one self-test at boot.
|
||||
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see src/kernel/tests.zig)");
|
||||
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see system/kernel/tests.zig)");
|
||||
const build_options = b.addOptions();
|
||||
build_options.addOption(?[]const u8, "test_case", test_case);
|
||||
const build_options_mod = build_options.createModule();
|
||||
const build_options_module = build_options.createModule();
|
||||
|
||||
// --- Kernel: freestanding x86_64 ELF, jumped to by the bootloader ---
|
||||
// SSE2 is part of the x86_64 baseline and UEFI leaves it enabled at handoff,
|
||||
@@ -106,62 +285,194 @@ pub fn build(b: *std.Build) void {
|
||||
const exe = b.addExecutable(.{
|
||||
.name = "kernel",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/kernel/main.zig"),
|
||||
.root_source_file = b.path("system/kernel/kernel.zig"),
|
||||
.target = kernel_target,
|
||||
.optimize = optimize,
|
||||
.code_model = .small, // kernel is linked in the low 2 GiB (see image_base)
|
||||
.red_zone = false, // interrupts would corrupt the SysV red zone
|
||||
.single_threaded = true, // no scheduler yet; avoids pulling in TLS/atomics
|
||||
.code_model = .kernel, // kernel runs in the top 2 GiB (higher half)
|
||||
.red_zone = false, // interrupts would corrupt the SystemV red zone
|
||||
.single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores
|
||||
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
||||
.stack_check = false, // stack-probe calls have no runtime to land in
|
||||
.stack_protector = false,
|
||||
.imports = &.{
|
||||
.{ .name = "danos", .module = mod },
|
||||
.{ .name = "arch", .module = arch_mod },
|
||||
.{ .name = "platform", .module = platform_mod },
|
||||
.{ .name = "build_options", .module = build_options_mod },
|
||||
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
.{ .name = "device-abi", .module = device_abi_module },
|
||||
.{ .name = "architecture", .module = architecture_module },
|
||||
.{ .name = "platform", .module = platform_module },
|
||||
.{ .name = "parameters", .module = parameters_module },
|
||||
.{ .name = "build_options", .module = build_options_module },
|
||||
.{ .name = "initial-ramdisk", .module = initial_ramdisk_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
exe.setLinkerScript(b.path("src/kernel/arch/x86_64/linker.ld"));
|
||||
exe.setLinkerScript(b.path("system/kernel/architecture/x86_64/linker.ld"));
|
||||
exe.entry = .{ .symbol_name = "_start" };
|
||||
// Physical address the bootloader loads the kernel to (identity-mapped under
|
||||
// UEFI). Overrides Zig's default image base so the linker script's layout is
|
||||
// honoured; adjust here if it collides with firmware-reserved memory.
|
||||
exe.image_base = 0x100000; // 1 MiB
|
||||
// The self-hosted linker ignores parts of the linker script (PHDRS,
|
||||
// /DISCARD/, AT(), section order); the higher-half layout depends on the
|
||||
// script being authoritative, so pin the kernel to LLVM + LLD.
|
||||
exe.use_llvm = true;
|
||||
exe.use_lld = true;
|
||||
// Higher-half virtual base (matches KERNEL_VIRT_BASE in linker.ld); the
|
||||
// linker's AT() clauses give each segment a low physical load address
|
||||
// (.text at 1 MiB), which the loader allocates and copies into.
|
||||
exe.image_base = 0xFFFFFFFF80100000;
|
||||
|
||||
b.installArtifact(exe);
|
||||
// Everything installs into a FHS-shaped zig-out: it IS the danos filesystem *and*
|
||||
// the boot volume. Each binary lands at its addressed, leaf-collapsed path — the
|
||||
// kernel at zig-out/system/kernel (from system/kernel/kernel.zig), init at
|
||||
// zig-out/system/services/init, and so on (see docs/README.md). The bootloader
|
||||
// then loads these FHS paths off the volume.
|
||||
const kernel_install = b.addInstallArtifact(exe, .{ .dest_dir = .{ .override = .{ .custom = "system" } } });
|
||||
b.getInstallStep().dependOn(&kernel_install.step);
|
||||
|
||||
// Boot methods live in src/boot/, one per way of getting the kernel running.
|
||||
// --- init: the first user-space program (a system service) ---
|
||||
// Built by the shared user-binary recipe (see addUserBinary): freestanding,
|
||||
// linked into the kernel's user region against the `runtime` runtime library, and
|
||||
// started in ring 3 by the kernel's user-ELF loader.
|
||||
const init_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "init", "system/services/init/init.zig");
|
||||
const init_install = b.addInstallArtifact(init_exe, .{ .dest_dir = .{ .override = .{ .custom = "system/services" } } });
|
||||
b.getInstallStep().dependOn(&init_install.step);
|
||||
|
||||
// --- initial_ramdisk: a bundle of extra user binaries (VFS server + drivers) ---
|
||||
// Each is built by the same user-binary recipe, then packed into one image by
|
||||
// the host-side make-initial-ramdisk tool. The bootloader ferries the image to the kernel,
|
||||
// which unpacks it and spawns each program (system/initial-ramdisk.zig).
|
||||
const vfs_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs", "system/services/vfs/vfs.zig");
|
||||
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs-test", "system/services/vfs/vfs-test.zig");
|
||||
const hpet_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "hpet", "system/drivers/hpet/hpet.zig");
|
||||
const bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "bus", "system/drivers/bus/bus.zig");
|
||||
const ps2_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-bus", "system/drivers/ps2-bus/ps2-bus.zig");
|
||||
const ps2_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig");
|
||||
const ps2_mouse_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig");
|
||||
const usb_xhci_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-xhci-bus", "system/drivers/usb-xhci-bus/usb-xhci-bus.zig");
|
||||
const pci_bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "pci-bus", "system/drivers/pci-bus/pci-bus.zig");
|
||||
// A test fixture, not a real driver: hellos to the device manager, then faults —
|
||||
// what the driver-restart scenario drives the crash-loop cap with.
|
||||
const crash_test_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "crash-test", "system/services/crash-test/crash-test.zig");
|
||||
const device_list_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-list", "system/services/device-list/device-list.zig");
|
||||
// The discovery service: one swappable process per firmware
|
||||
// (docs/m19-m20-plan.md decision 7), bundled under the neutral ramdisk name
|
||||
// "discovery" so the device manager never learns which firmware it is on.
|
||||
// x86 boots describe hardware with ACPI; the Raspberry Pis hand over a
|
||||
// flattened device tree — the aarch64 target flips the default when it
|
||||
// lands (docs/arm.md). Both are placeholders until M20.1 (acpi) and the
|
||||
// ARM bring-up (fdt).
|
||||
const Discovery = enum { acpi, fdt };
|
||||
const discovery = b.option(Discovery, "discovery", "Which discovery service fills the ramdisk's 'discovery' slot (default: acpi)") orelse Discovery.acpi;
|
||||
const discovery_source: []const u8 = switch (discovery) {
|
||||
.acpi => "system/services/acpi/acpi.zig",
|
||||
.fdt => "system/services/fdt/fdt.zig",
|
||||
};
|
||||
const discovery_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "discovery", discovery_source);
|
||||
if (discovery == .acpi) discovery_exe.root_module.addImport("aml", aml_module);
|
||||
const device_manager_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-manager", "system/services/device-manager/device-manager.zig");
|
||||
// The input service and its exercisers: the fan-out server, a hardware-free synthetic
|
||||
// source, and a subscriber that doubles as the `input` test's oracle. See docs/input.md.
|
||||
const input_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input", "system/services/input/input.zig");
|
||||
const input_source_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-source", "system/services/input-source/input-source.zig");
|
||||
const input_test_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-test", "system/services/input-test/input-test.zig");
|
||||
const args_echo_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "args-echo", "system/services/args-echo/args-echo.zig");
|
||||
const process_test_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "process-test", "system/services/process-test/process-test.zig");
|
||||
|
||||
// Pack the user binaries into the initial_ramdisk image with the host-side Python tool
|
||||
// (the container format is trivial, and Python sidesteps std API churn). Args:
|
||||
// make-initial-ramdisk.py <out> [<name> <file>]... — one name/file pair per binary.
|
||||
const mk_run = b.addSystemCommand(&.{"python3"});
|
||||
mk_run.addFileArg(b.path("tools/make-initial-ramdisk.py"));
|
||||
const initial_ramdisk_img = mk_run.addOutputFileArg("initial-ramdisk.img");
|
||||
mk_run.addArg("vfs");
|
||||
mk_run.addFileArg(vfs_exe.getEmittedBin());
|
||||
mk_run.addArg("vfs-test");
|
||||
mk_run.addFileArg(vfstest_exe.getEmittedBin());
|
||||
mk_run.addArg("hpet");
|
||||
mk_run.addFileArg(hpet_exe.getEmittedBin());
|
||||
mk_run.addArg("bus");
|
||||
mk_run.addFileArg(bus_exe.getEmittedBin());
|
||||
mk_run.addArg("ps2-bus");
|
||||
mk_run.addFileArg(ps2_bus_exe.getEmittedBin());
|
||||
mk_run.addArg("ps2-keyboard");
|
||||
mk_run.addFileArg(ps2_keyboard_exe.getEmittedBin());
|
||||
mk_run.addArg("ps2-mouse");
|
||||
mk_run.addFileArg(ps2_mouse_exe.getEmittedBin());
|
||||
mk_run.addArg("usb-xhci-bus");
|
||||
mk_run.addFileArg(usb_xhci_bus_exe.getEmittedBin());
|
||||
mk_run.addArg("pci-bus");
|
||||
mk_run.addFileArg(pci_bus_exe.getEmittedBin());
|
||||
mk_run.addArg("crash-test");
|
||||
mk_run.addFileArg(crash_test_exe.getEmittedBin());
|
||||
mk_run.addArg("device-list");
|
||||
mk_run.addFileArg(device_list_exe.getEmittedBin());
|
||||
mk_run.addArg("discovery");
|
||||
mk_run.addFileArg(discovery_exe.getEmittedBin());
|
||||
mk_run.addArg("device-manager");
|
||||
mk_run.addFileArg(device_manager_exe.getEmittedBin());
|
||||
mk_run.addArg("input");
|
||||
mk_run.addFileArg(input_exe.getEmittedBin());
|
||||
mk_run.addArg("input-source");
|
||||
mk_run.addFileArg(input_source_exe.getEmittedBin());
|
||||
mk_run.addArg("input-test");
|
||||
mk_run.addFileArg(input_test_exe.getEmittedBin());
|
||||
mk_run.addArg("args-echo");
|
||||
mk_run.addFileArg(args_echo_exe.getEmittedBin());
|
||||
mk_run.addArg("process-test");
|
||||
mk_run.addFileArg(process_test_exe.getEmittedBin());
|
||||
|
||||
// Also install the packed binaries to their FHS homes, so zig-out is a true image
|
||||
// of the filesystem — even though at boot they arrive inside the initial-ramdisk.
|
||||
for ([_]struct { *std.Build.Step.Compile, []const u8 }{
|
||||
.{ vfs_exe, "system/services" },
|
||||
.{ device_manager_exe, "system/services" },
|
||||
.{ input_exe, "system/services" },
|
||||
.{ hpet_exe, "system/drivers" },
|
||||
.{ bus_exe, "system/drivers" },
|
||||
.{ ps2_bus_exe, "system/drivers" },
|
||||
.{ ps2_keyboard_exe, "system/drivers" },
|
||||
.{ ps2_mouse_exe, "system/drivers" },
|
||||
.{ usb_xhci_bus_exe, "system/drivers" },
|
||||
}) |entry| {
|
||||
const step = b.addInstallArtifact(entry[0], .{ .dest_dir = .{ .override = .{ .custom = entry[1] } } });
|
||||
b.getInstallStep().dependOn(&step.step);
|
||||
}
|
||||
|
||||
// The initial-ramdisk itself installs to /boot (with the loaders).
|
||||
const initial_ramdisk_install = b.addInstallFile(initial_ramdisk_img, "boot/initial-ramdisk.img");
|
||||
b.getInstallStep().dependOn(&initial_ramdisk_install.step);
|
||||
|
||||
// Boot methods live in boot/, one per way of getting the kernel running.
|
||||
// Each is its own binary/entry (a loader is built for its own target); today
|
||||
// that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis.
|
||||
const efiexe = b.addExecutable(.{
|
||||
.name = "BOOTX64",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/boot/efi.zig"),
|
||||
.root_source_file = b.path("boot/efi.zig"),
|
||||
.target = b.resolveTargetQuery(.{
|
||||
.cpu_arch = .x86_64,
|
||||
.os_tag = .uefi,
|
||||
}),
|
||||
.optimize = optimize,
|
||||
.imports = &.{
|
||||
.{ .name = "danos", .module = mod },
|
||||
// The bootloader speaks only the handoff contract — never the user ABI.
|
||||
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
|
||||
b.installArtifact(efiexe);
|
||||
// UEFI firmware requires the removable-media loader at exactly \EFI\BOOT\BOOTX64.efi,
|
||||
// so that path is fixed by the firmware (it is /boot's EFI stub, conceptually).
|
||||
const efi_install = b.addInstallArtifact(efiexe, .{ .dest_dir = .{ .override = .{ .custom = "EFI/BOOT" } } });
|
||||
b.getInstallStep().dependOn(&efi_install.step);
|
||||
|
||||
// --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF ---
|
||||
// Firmware lives in different places per OS/distro, so probe the known
|
||||
// layouts (Arch, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
|
||||
// layouts (Architecture, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
|
||||
// that exists. Override with -Dovmf-code / -Dovmf-vars if yours is elsewhere.
|
||||
const ovmf_code = b.option(
|
||||
[]const u8,
|
||||
"ovmf-code",
|
||||
"Path to the OVMF_CODE firmware image",
|
||||
) orelse firstExisting(b.graph.io, &.{
|
||||
"/usr/share/edk2/x64/OVMF_CODE.4m.fd", // Arch
|
||||
"/usr/share/edk2/x64/OVMF_CODE.4m.fd", // Architecture
|
||||
"/usr/share/OVMF/OVMF_CODE_4M.fd", // Debian/Ubuntu
|
||||
"/usr/share/OVMF/OVMF_CODE.fd", // older Debian/Ubuntu
|
||||
"/usr/share/edk2-ovmf/x64/OVMF_CODE.fd", // Fedora
|
||||
@@ -173,7 +484,7 @@ pub fn build(b: *std.Build) void {
|
||||
"ovmf-vars",
|
||||
"Path to the OVMF_VARS firmware image (a writable copy is made)",
|
||||
) orelse firstExisting(b.graph.io, &.{
|
||||
"/usr/share/edk2/x64/OVMF_VARS.4m.fd", // Arch
|
||||
"/usr/share/edk2/x64/OVMF_VARS.4m.fd", // Architecture
|
||||
"/usr/share/OVMF/OVMF_VARS_4M.fd", // Debian/Ubuntu
|
||||
"/usr/share/OVMF/OVMF_VARS.fd", // older Debian/Ubuntu
|
||||
"/usr/share/edk2-ovmf/x64/OVMF_VARS.fd", // Fedora
|
||||
@@ -181,15 +492,8 @@ pub fn build(b: *std.Build) void {
|
||||
"/usr/local/share/qemu/edk2-i386-vars.fd", // macOS Homebrew (Intel)
|
||||
});
|
||||
|
||||
// Assemble an EFI System Partition layout: esp/EFI/BOOT/BOOTX64.efi
|
||||
const efi_install = b.addInstallArtifact(efiexe, .{
|
||||
.dest_dir = .{ .override = .{ .custom = "esp/EFI/BOOT" } },
|
||||
});
|
||||
// The bootloader loads the kernel by name from the volume root, so drop the
|
||||
// kernel ELF at esp/kernel.
|
||||
const kernel_install = b.addInstallArtifact(exe, .{
|
||||
.dest_dir = .{ .override = .{ .custom = "esp" } },
|
||||
});
|
||||
// The FHS zig-out (installed above) *is* the boot volume — no separate ESP to
|
||||
// assemble. QEMU presents it to the guest as a FAT drive below.
|
||||
|
||||
// The firmware needs to write NVRAM, so give it a writable copy of the vars.
|
||||
const vars_copy = b.addSystemCommand(&.{ "cp", "-f", ovmf_vars });
|
||||
@@ -197,6 +501,19 @@ pub fn build(b: *std.Build) void {
|
||||
|
||||
const run_efi = b.addSystemCommand(&.{
|
||||
"qemu-system-x86_64",
|
||||
"-device",
|
||||
"qemu-xhci,id=xhci",
|
||||
"-device",
|
||||
"usb-mouse,bus=xhci.0",
|
||||
"-device",
|
||||
"usb-kbd,bus=xhci.0",
|
||||
// "-usb",
|
||||
// "-device",
|
||||
// "usb-ehci,id=ehci",
|
||||
// "-device",
|
||||
// "usb-tablet,bus=usb-bus.0",
|
||||
// "-device",
|
||||
// "usb-mouse,bus=ehci.0",
|
||||
"-machine",
|
||||
"q35",
|
||||
"-m",
|
||||
@@ -206,10 +523,10 @@ pub fn build(b: *std.Build) void {
|
||||
});
|
||||
run_efi.addArg("-drive");
|
||||
run_efi.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
||||
// Present the ESP directory to the guest as a FAT drive.
|
||||
// Present the FHS zig-out to the guest as a FAT drive — it is the boot volume.
|
||||
run_efi.addArgs(&.{
|
||||
"-drive",
|
||||
b.fmt("format=raw,file=fat:rw:{s}/esp", .{b.install_path}),
|
||||
b.fmt("format=raw,file=fat:rw:{s}", .{b.install_path}),
|
||||
"-net",
|
||||
"none",
|
||||
// Emulated display advertising 1280x720 as its native (EDID preferred)
|
||||
@@ -220,14 +537,19 @@ pub fn build(b: *std.Build) void {
|
||||
"-device",
|
||||
"VGA,edid=on,xres=1280,yres=720",
|
||||
});
|
||||
// Always capture the guest's serial0 (the kernel's machine-readable log) to a
|
||||
// timestamped file under zig-out, so each run leaves its own log behind.
|
||||
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ b.install_path, timestamp(b) });
|
||||
// Capture the guest's serial0 (danos's machine-readable log) to the qemu-test
|
||||
// scratch area — a dev/host artifact, kept out of the FHS boot volume we mount.
|
||||
// (/var/log/system is reserved for the kernel's own logging system later.) One
|
||||
// timestamped file per run.
|
||||
const log_dir = b.fmt("{s}/qemu-test", .{b.install_path});
|
||||
const make_log_dir = b.addSystemCommand(&.{ "mkdir", "-p", log_dir });
|
||||
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
||||
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
|
||||
run_efi.step.dependOn(&efi_install.step);
|
||||
run_efi.step.dependOn(&kernel_install.step);
|
||||
// The whole FHS zig-out must be installed (and the scratch dir created) before we mount it.
|
||||
run_efi.step.dependOn(b.getInstallStep());
|
||||
run_efi.step.dependOn(&make_log_dir.step);
|
||||
|
||||
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/run-x86-64-serial0-<timestamp>.log");
|
||||
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/qemu-test/run-x86-64-serial0-<timestamp>.log");
|
||||
run_efi_step.dependOn(&run_efi.step);
|
||||
|
||||
// const run_cmd = b.addRunArtifact(exe);
|
||||
@@ -240,18 +562,50 @@ pub fn build(b: *std.Build) void {
|
||||
// }
|
||||
|
||||
// Tests run on the host. The kernel and bootloader target freestanding/UEFI
|
||||
// and can't be executed natively, so only the shared module is unit-tested
|
||||
// here (compiled for the host rather than inheriting a freestanding target).
|
||||
const mod_tests = b.addTest(.{
|
||||
// and can't be executed natively, so only the shared contracts are unit-tested
|
||||
// here (compiled for the host rather than inheriting a freestanding target) —
|
||||
// which also compile-checks that the three-way split stays self-consistent.
|
||||
const test_step = b.step("test", "Run tests");
|
||||
for ([_][]const u8{
|
||||
"system/boot-handoff.zig",
|
||||
"system/abi.zig",
|
||||
"system/devices/device-abi.zig",
|
||||
"system/devices/pci-class.zig", // class/subclass/prog-IF name decoding
|
||||
"system/devices/acpi-ids.zig", // _HID name decoding
|
||||
"system/devices/usb-abi.zig", // wire sizes + bit packings + set-up packet encodings
|
||||
"system/devices/usb-ids.zig", // class/subclass/protocol code assignments
|
||||
"library/mmio/mmio.zig", // barriers assemble + registers round-trip
|
||||
"system/drivers/ps2-bus/scancode.zig", // set-2 decode + keyboard state machine
|
||||
"system/drivers/ps2-bus/mouse-packet.zig", // 3-byte mouse packet assembly
|
||||
}) |root| {
|
||||
const mod_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(root),
|
||||
.target = target,
|
||||
.optimize = optimize,
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(mod_tests).step);
|
||||
}
|
||||
|
||||
// The xkeyboard-config keymap tests need its generated `layouts` import wired, so they
|
||||
// don't fit the plain loop above. Its keycode->character assertions are the end-to-end
|
||||
// proof that the xkb-data -> generator -> Zig-lookup pipeline is correct.
|
||||
const xkb_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/root.zig"),
|
||||
.root_source_file = b.path("library/xkeyboard-config/xkeyboard-config.zig"),
|
||||
.target = target,
|
||||
.optimize = optimize,
|
||||
.imports = &.{
|
||||
.{ .name = "layouts", .module = xkb_layouts_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(xkb_tests).step);
|
||||
|
||||
const run_mod_tests = b.addRunArtifact(mod_tests);
|
||||
|
||||
const test_step = b.step("test", "Run tests");
|
||||
test_step.dependOn(&run_mod_tests.step);
|
||||
// Convenience: `zig build gen-xkeyboard-config` regenerates the layout tables from the
|
||||
// vendored data (offline). `fetch` (the network step) stays a manual script run.
|
||||
const gen_xkb = b.addSystemCommand(&.{ "python3", "tools/make-xkeyboard-config.py", "generate" });
|
||||
const gen_xkb_step = b.step("gen-xkeyboard-config", "Regenerate library/xkeyboard-config/generated from the vendored data");
|
||||
gen_xkb_step.dependOn(&gen_xkb.step);
|
||||
}
|
||||
|
||||
+129
-13
@@ -35,9 +35,41 @@ rather than restate it. Roughly in the order things happen at runtime:
|
||||
multitasking: kernel threads, the context switch, O(1) priority selection, and
|
||||
blocking (sleep, wait queues) — the leap to a running system.
|
||||
11. **[ipc.md](ipc.md) — inter-process communication.** Bounded blocking
|
||||
message-passing channels — the backbone the microkernel's isolated servers will
|
||||
talk over.
|
||||
12. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
||||
message-passing channels, then synchronous call/reply between *processes* over
|
||||
endpoints — the backbone the microkernel's isolated servers talk over.
|
||||
12. **[syscall.md](syscall.md) — system calls.** How ring 3 asks the kernel for
|
||||
something: the `syscall`/`sysret` fast path, the trap frame, and why the table is
|
||||
deliberately tiny.
|
||||
13. **[drivers.md](drivers.md) — writing a driver.** The payoff: a driver is an
|
||||
ordinary ring-3 process that claims a device, maps its registers, and **sleeps
|
||||
until its hardware interrupts it**. The claim is the capability; `irq_ack` is the
|
||||
unmask.
|
||||
14. **[driver-model.md](driver-model.md) — buses, classes and host controllers.** How
|
||||
real driver stacks factor into three shapes and how families share code. The
|
||||
three primitives it proposed are long since built (M13 capability passing,
|
||||
M14 DMA + barriers, M15 MSI), and the driver *contract* on top of them —
|
||||
hello, supervision, restart — is built too (device-manager.md, M18).
|
||||
15. **[process-management.md](process-management.md) — process management.** The
|
||||
microkernel's `ps`/`kill`/SIGCHLD: enumerate as a table snapshot, the
|
||||
supervision link as the kill authority, and child-exit notifications over the
|
||||
same endpoints IRQs arrive on.
|
||||
16. **[process-lifecycle.md](process-lifecycle.md) — the process lifecycle.** Built
|
||||
(M17): signals over IPC as the one lifecycle vocabulary every process speaks — the
|
||||
POSIX.1-1990 words with message delivery instead of stack hijack, the stable
|
||||
`runtime.process` interface, exit reasons, published exit events any stateful
|
||||
service can subscribe to (the VFS releasing dead clients' handles), and the two
|
||||
iron rules (cleanup is the kernel's job; kill is not a signal).
|
||||
17. **[device-manager.md](device-manager.md) — the device manager.** Built (M18,
|
||||
through the app surface): the
|
||||
tree, the matcher, and the supervisor. Tree structure lives in the manager,
|
||||
authority stays in the kernel; bus drivers report what they see; drivers are
|
||||
restarted through the lifecycle vocabulary — the plan that turns
|
||||
[resilience.md](resilience.md)'s restart goal into increments.
|
||||
18. **[input.md](input.md) — the input module.** Broadcasting input events (keyboard,
|
||||
mouse, joystick): why a synchronous rendezvous can't fan out to many listeners, the
|
||||
asynchronous `ipc_send` primitive built to fix it, and the per-device subscribe/publish
|
||||
service layered on top.
|
||||
19. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
||||
how `while (true) hlt` parks the CPU safely once there's nothing left to do.
|
||||
|
||||
Start with the north star:
|
||||
@@ -68,11 +100,19 @@ Cutting across all of these:
|
||||
- **[smp.md](smp.md) — multiple cores.** A design/research note on how microkernels
|
||||
(L4, seL4) handle SMP — big kernel lock vs per-CPU vs multikernel — and how the
|
||||
right choice depends on whether danos is chasing real-time or resilience.
|
||||
- **[coding-standards.md](coding-standards.md) — coding standards.** The naming rule the
|
||||
tree follows: non-acronyms are spelled out in full (`message`, not `msg`), files are
|
||||
`kebab-case`, code follows Zig's case conventions, and the handful of exceptions
|
||||
(POSIX/C ABI names, `init`/`len`/`ptr`, acronyms).
|
||||
- **[sysv.md](sysv.md) — the calling convention.** What "the kernel is SysV" means,
|
||||
and why the loader→kernel boundary has to pin it (the RDI-vs-RCX handoff).
|
||||
- **[testing.md](testing.md) — testing.** How the kernel is tested by booting it in
|
||||
QEMU and asserting on its serial output — reproducibly, and structured so the
|
||||
same tests run across architectures.
|
||||
- **[logging.md](logging.md) — logging.** The multi-sink diagnostic log (serial,
|
||||
0xE9 debugcon, file later) kept separate from the framebuffer display, plus the
|
||||
robustness path: optional framebuffer, POST-code checkpoints, and a persistent
|
||||
panic breadcrumb so the kernel survives — and can be diagnosed — with no output.
|
||||
|
||||
## How the pieces relate
|
||||
|
||||
@@ -90,19 +130,95 @@ passing messages over **[IPC](ipc.md)** channels — runs, its CPU-specific bits
|
||||
behind the [arch](arch.md) boundary, and when idle, or on a panic, it **halts**
|
||||
([halting.md](halting.md)).
|
||||
|
||||
Above that line the microkernel proper begins: **discovery** ([discovery.md](discovery.md),
|
||||
[acpi.md](acpi.md)) learns what hardware exists, ring-3 processes ask the kernel for
|
||||
things through the small **[syscall](syscall.md)** table, isolated servers reach each
|
||||
other over IPC **endpoints** ([ipc.md](ipc.md)), and a **[driver](drivers.md)** claims
|
||||
a device, maps its registers, and sleeps until the hardware interrupts it — which is
|
||||
the whole reason for the arrangement ([vision.md](vision.md)).
|
||||
|
||||
## Repository layout
|
||||
|
||||
danos is a **monorepo of sub-projects**. Each service or driver is a directory that is
|
||||
its own Zig module — it can hold as many files as it needs, and other sub-projects
|
||||
reach it *by module name*, never by a path into its files. The source tree deliberately
|
||||
**mirrors the runtime FHS** ([danos-file-system-hierarchy-FSH.md](danos-file-system-hierarchy-FSH.md)):
|
||||
what you see under `system/` in the source is what a running danos represents under
|
||||
`/system`.
|
||||
|
||||
**A sub-project is addressed by its directory; its entry point repeats the directory's
|
||||
name.** `system/services/init/` contains `init.zig` (its root), and produces a binary
|
||||
addressed as **`system/services/init`** — the repeated leaf resolves away:
|
||||
|
||||
| Source (root file) | Addressed as (module / binary / FHS path) |
|
||||
|----------------------------------------|--------------------------------------------|
|
||||
| `system/services/init/init.zig` | `system/services/init` → `/system/services/init` |
|
||||
| `system/drivers/hpet/hpet.zig` | `system/drivers/hpet` → `/system/drivers/hpet` |
|
||||
| `library/runtime/runtime.zig` | `library/runtime` (the `runtime` module) |
|
||||
|
||||
In **source**, a sub-project is a directory so it can hold many files — the entry is
|
||||
`init/init.zig`, beside it `vfs/vfs-test.zig`, `vfs/protocol.zig`, and so on. When
|
||||
**addressed or installed**, that collapses to the single canonical path: the `init`
|
||||
binary installs to `/system/services/init` (a file at that path), not
|
||||
`/system/services/init/init`. The repeated leaf exists only in source; the directory is
|
||||
the identity, the entry file is its implementation. (Same idea as a Go package being its
|
||||
directory, or a macOS `.app` bundle addressed by the bundle, not the executable within.)
|
||||
A sub-project's extra files are reached through the module, never as separate paths.
|
||||
|
||||
```
|
||||
system/ → /system danos's own internals (the self-representation)
|
||||
boot-handoff.zig the loader↔kernel contract (the `boot-handoff` module)
|
||||
abi.zig the private kernel↔runtime syscall ABI (the `abi` module)
|
||||
parameters.zig initial-ramdisk.zig shared contracts
|
||||
kernel/ IPC, memory, scheduling, the private syscall dispatch
|
||||
architecture/x86_64/ the `architecture` module (never named by generic code)
|
||||
devices/ the device model /system/devices reflects (+ aml/)
|
||||
device-abi.zig the device wire types (the `device-abi` module)
|
||||
drivers/ hpet/ bus/ one sub-project per driver → /system/drivers
|
||||
services/ init/ vfs/ device-manager/ system servers → /system/services (vfs/ holds
|
||||
vfs.zig, vfs-test.zig, protocol.zig)
|
||||
library/ → /lib libraries, one sub-directory each
|
||||
runtime/ the danos-native runtime — the stable application ABI
|
||||
posix/ POSIX/C compatibility, layered over runtime
|
||||
boot/ → /boot the loaders
|
||||
tools/ test/ host-side build + QEMU test harness
|
||||
```
|
||||
|
||||
A sub-project exposes its **public interface as a module**: `system/services/vfs/` owns
|
||||
the VFS wire protocol (`protocol.zig`, the `vfs-protocol` module), which the POSIX
|
||||
layer imports by name. `usb`/`block` drivers will expose their protocols the same way.
|
||||
|
||||
`library/posix/` is special: it is the **one place** POSIX/C spellings are allowed
|
||||
verbatim (`stat`, `O_CREAT`, `fopen`, `errno`). Everywhere else follows the danos
|
||||
naming rule with no exception — see [coding-standards.md](coding-standards.md). The
|
||||
POSIX layer calls the runtime, never the kernel's system calls directly, so it never
|
||||
appears in the private-ABI path.
|
||||
|
||||
## Source map
|
||||
|
||||
| Area | Code |
|
||||
|------|------|
|
||||
| Boot methods (one per way of booting the kernel) | `src/boot/` — `efi.zig` (UEFI) → `BOOTX64.efi` |
|
||||
| Kernel entry, panic, bring-up | `src/kernel/main.zig` |
|
||||
| Shared loader↔kernel contract (`BootInfo`, `Framebuffer`, `MemoryMap`, ABI) | `src/root.zig` |
|
||||
| Physical frame allocator | `src/kernel/pmm.zig` |
|
||||
| Kernel heap (`std.mem.Allocator`) | `src/kernel/heap.zig` |
|
||||
| Scheduler (fixed-priority preemptive; blocking, wait queues) | `src/kernel/sched.zig` |
|
||||
| IPC channels (message passing) | `src/kernel/ipc.zig` |
|
||||
| Framebuffer text console (mirrors to serial) | `src/kernel/console.zig` |
|
||||
| In-kernel test cases | `src/kernel/tests.zig` |
|
||||
| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/timer, serial, linker script) | `src/kernel/arch/x86_64/` |
|
||||
| Boot methods (one per way of booting the kernel) | `boot/` — `efi.zig` (UEFI) → `BOOTX64.efi` |
|
||||
| Kernel entry, panic, bring-up | `system/kernel/kernel.zig` |
|
||||
| Loader↔kernel handoff (`BootInfo`, `Framebuffer`, `MemoryMap`, VM layout) | `system/boot-handoff.zig` |
|
||||
| Private kernel↔runtime syscall ABI (`SystemCall`, mmap prot flags, `page_size`) — the runtime speaks it, not apps | `system/abi.zig` |
|
||||
| Device wire types (`DeviceDescriptor`, `DeviceClass`, …) | `system/devices/device-abi.zig` |
|
||||
| Physical frame allocator | `system/kernel/pmm.zig` |
|
||||
| Kernel heap (`std.mem.Allocator`) | `system/kernel/heap.zig` |
|
||||
| Scheduler (fixed-priority preemptive; blocking, wait queues) | `system/kernel/scheduler.zig` |
|
||||
| Big kernel lock + interrupt-safe critical sections | `system/kernel/sync.zig` |
|
||||
| IPC channels between kernel threads (message passing) | `system/kernel/ipc.zig` |
|
||||
| IPC endpoints: cross-address-space call/reply, handles, notifications | `system/kernel/ipc-synchronous.zig` |
|
||||
| User processes: ELF loading, address spaces, the syscall table | `system/kernel/process.zig` |
|
||||
| Device tree + claim capability + `device_register` containment | `system/kernel/devices-broker.zig` |
|
||||
| IRQ-as-IPC: routing a device interrupt to a driver's endpoint | `system/kernel/irq.zig` |
|
||||
| Hardware discovery (ACPI/device tree) behind one neutral device model | `system/devices/` |
|
||||
| Framebuffer text console (mirrors to serial) | `system/kernel/console.zig` |
|
||||
| In-kernel test cases | `system/kernel/tests.zig` |
|
||||
| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/IO-APIC/timer, serial, linker script) | `system/kernel/architecture/x86_64/` |
|
||||
| danos-native runtime (`runtime`): syscall wrappers, heap, IPC, device access — the stable application ABI | `library/runtime/` |
|
||||
| POSIX/C compatibility (`posix`): unistd, stdio — the one place POSIX names are allowed | `library/posix/` |
|
||||
| System services (init, the VFS server + `protocol`, the device-manager) | `system/services/` |
|
||||
| Device drivers, one sub-project each (`hpet` leaf driver, `bus` bus driver) | `system/drivers/` |
|
||||
| Build + `run-x86-64` (QEMU/OVMF) | `build.zig` |
|
||||
| QEMU integration test harness | `test/qemu_test.py` |
|
||||
|
||||
+6
-6
@@ -16,13 +16,13 @@ the RSDT's address is a field *inside* the RSDP. The platform follows that point
|
||||
UEFI configuration table
|
||||
│ the loader reads the RSDP's physical address
|
||||
▼
|
||||
BootInfo.acpi_rsdp (u64, in the shared `danos` module) src/root.zig
|
||||
BootInfo.acpi_rsdp (u64, in the loader↔kernel handoff) system/boot-handoff.zig
|
||||
│ the kernel forwards the whole BootInfo
|
||||
▼
|
||||
platform.discover(boot_info, …) src/device/platform.zig
|
||||
platform.discover(boot_info, …) system/devices/platform.zig
|
||||
│ reads boot_info.acpi_rsdp, hands it to the ACPI backend
|
||||
▼
|
||||
acpi.discover(rsdp_phys, …) src/device/acpi.zig
|
||||
acpi.discover(rsdp_phys, …) system/devices/acpi.zig
|
||||
│ dereferences the RSDP, reads the pointer it contains
|
||||
▼
|
||||
RSDP ──(a field in the struct)──► RSDT / XSDT ──► SDTs (MADT, MCFG, FADT, HPET, DSDT…)
|
||||
@@ -35,7 +35,7 @@ successor the **XSDT** (ACPI 2.0+) — which in turn lists every other SDT.
|
||||
## Step 1 — the loader finds the RSDP
|
||||
|
||||
Only the firmware knows where ACPI lives, so the RSDP must be grabbed while UEFI is
|
||||
still up. `acpiRootSystemDescriptorPointer()` in `src/boot/efi.zig` walks the UEFI
|
||||
still up. `acpiRootSystemDescriptorPointer()` in `boot/efi.zig` walks the UEFI
|
||||
**configuration table** for the ACPI GUID and returns the vendor pointer — the same
|
||||
"grab it before `ExitBootServices`" pattern as the [framebuffer](framebuffer.md) and
|
||||
the [memory map](memory-map.md).
|
||||
@@ -44,11 +44,11 @@ the [memory map](memory-map.md).
|
||||
|
||||
The loader can't just call the device module: the bootloader binary and the kernel
|
||||
binary are compiled separately, and **the loader isn't linked against the `platform`
|
||||
module at all** (it imports only the shared `danos` module). So instead of a call, it
|
||||
module at all** (it imports only the `boot-handoff` contract). So instead of a call, it
|
||||
deposits a value in the handoff struct:
|
||||
|
||||
```zig
|
||||
// src/boot/efi.zig — while boot services are still up
|
||||
// boot/efi.zig — while boot services are still up
|
||||
.acpi_rsdp = if (acpiRootSystemDescriptorPointer()) |p| @intFromPtr(p) else 0,
|
||||
```
|
||||
|
||||
|
||||
+13
-13
@@ -13,7 +13,7 @@ runtime dispatch. `build.zig` exposes one architecture's code as a module called
|
||||
|
||||
```zig
|
||||
const arch_mod = b.addModule("arch", .{
|
||||
.root_source_file = b.path("src/kernel/arch/x86_64/cpu.zig"),
|
||||
.root_source_file = b.path("system/kernel/architecture/x86_64/cpu.zig"),
|
||||
});
|
||||
```
|
||||
|
||||
@@ -26,7 +26,7 @@ arch.halt(); // never says "x86_64"
|
||||
```
|
||||
|
||||
Adding a second architecture is then a build-time choice: create
|
||||
`src/kernel/arch/aarch64/`, and point the `arch` module at it when the target CPU is
|
||||
`system/kernel/arch/aarch64/`, and point the `arch` module at it when the target CPU is
|
||||
AArch64. `main.zig` and `console.zig` don't change. **That compiler-checked module
|
||||
boundary _is_ the architecture interface** — when a new arch is missing a function
|
||||
the generic kernel calls, the build fails and names exactly what's missing.
|
||||
@@ -36,7 +36,7 @@ the generic kernel calls, the build fails and names exactly what's missing.
|
||||
The split follows a simple test: does it name a CPU instruction, a hardware
|
||||
register, or a memory-management structure? If so, it's arch-specific.
|
||||
|
||||
| Arch-specific — `src/kernel/arch/x86_64/` | Generic — kernel core |
|
||||
| Arch-specific — `system/kernel/architecture/x86_64/` | Generic — kernel core |
|
||||
|---|---|
|
||||
| `cpu.zig`: `halt()` (`hlt`), later GDT/IDT/paging | `console.zig` — pure pixel math, works anywhere |
|
||||
| `linker.ld` — link layout, load address | `main.zig` — `kmain` orchestration, panic handler |
|
||||
@@ -51,9 +51,9 @@ should end up on the generic side; the arch module stays small.
|
||||
There are really two independent questions, and it's worth not conflating them:
|
||||
|
||||
- **CPU architecture** (x86_64 vs AArch64): instructions, MMU, interrupts →
|
||||
`src/kernel/arch/<cpu>/`.
|
||||
`system/kernel/arch/<cpu>/`.
|
||||
- **Boot protocol** (UEFI vs Raspberry Pi firmware + device tree): handled
|
||||
*separately*, because loaders are their own binaries. `src/boot/efi.zig` builds
|
||||
*separately*, because loaders are their own binaries. `boot/efi.zig` builds
|
||||
`BOOTX64.efi`, a distinct executable from the kernel ELF. On a Pi there is no
|
||||
separate loader at all — the firmware jumps straight into the kernel with a
|
||||
device-tree pointer, so that entry work would live in the AArch64 arch code.
|
||||
@@ -61,29 +61,29 @@ There are really two independent questions, and it's worth not conflating them:
|
||||
|
||||
## Current x86_64 contents
|
||||
|
||||
- **`src/kernel/arch/x86_64/cpu.zig`** — the `arch` module root. Exposes `halt()` (see
|
||||
- **`system/kernel/architecture/x86_64/cpu.zig`** — the `arch` module root. Exposes `halt()` (see
|
||||
[halting.md](halting.md)), `init()` (bring up the descriptor tables),
|
||||
`enablePaging()`, `setFaultHandler`, `readCr2`/`readCr3`, and the `CpuState`
|
||||
trap frame.
|
||||
- **`src/kernel/arch/x86_64/gdt.zig`** / **`idt.zig`** / **`tss.zig`** — the GDT, IDT and
|
||||
- **`system/kernel/architecture/x86_64/gdt.zig`** / **`idt.zig`** / **`tss.zig`** — the GDT, IDT and
|
||||
TSS plus CPU-exception handling (see [interrupts.md](interrupts.md)).
|
||||
- **`src/kernel/arch/x86_64/paging.zig`** — the kernel's page tables (see
|
||||
- **`system/kernel/architecture/x86_64/paging.zig`** — the kernel's page tables (see
|
||||
[paging.md](paging.md)).
|
||||
- **`src/kernel/arch/x86_64/apic.zig`** — the Local APIC and its timer, the source of
|
||||
- **`system/kernel/architecture/x86_64/apic.zig`** — the Local APIC and its timer, the source of
|
||||
device interrupts (see [device-interrupts.md](device-interrupts.md)).
|
||||
- **`src/kernel/arch/x86_64/serial.zig`** / **`io.zig`** — the COM1 UART (the kernel's
|
||||
- **`system/kernel/architecture/x86_64/serial.zig`** / **`io.zig`** — the COM1 UART (the kernel's
|
||||
machine-readable log channel, see [testing.md](testing.md)) and the shared
|
||||
port-I/O + MSR primitives.
|
||||
- **`src/kernel/arch/x86_64/isr.s`** — the exception stubs, the `lgdt`/`lidt`/`ltr` load
|
||||
- **`system/kernel/architecture/x86_64/isr.s`** — the exception stubs, the `lgdt`/`lidt`/`ltr` load
|
||||
helpers, and the context switch (`switch_context` / `task_trampoline`, see
|
||||
[scheduling.md](scheduling.md)) — real assembly, since Zig inline asm can't
|
||||
express them.
|
||||
- **`src/kernel/arch/x86_64/linker.ld`** — the kernel link layout (fixed low load
|
||||
- **`system/kernel/architecture/x86_64/linker.ld`** — the kernel link layout (fixed low load
|
||||
address, one PT_LOAD per permission set).
|
||||
|
||||
The kernel entry point `_start` currently still lives in the generic `main.zig` as
|
||||
a thin trampoline into `kmain`. It's arch-adjacent (its calling convention is
|
||||
x86_64 [SysV](sysv.md), via the shared `danos.kernel_abi`), but it's three lines
|
||||
x86_64 [SysV](sysv.md), via the shared `system.kernel_abi`), but it's three lines
|
||||
and mostly generic, so it stays put for now. When AArch64 arrives — where entry means setting
|
||||
up a stack and reading a device-tree pointer from a register — the entry work will
|
||||
be substantial and per-arch, and *that* is when we extract an entry interface into
|
||||
|
||||
+4
-4
@@ -18,7 +18,7 @@ matters for understanding why. This page maps the landscape so the
|
||||
new ISA.
|
||||
|
||||
They are as different from each other as either is from x86-64: separate registers,
|
||||
page-table formats, and calling conventions. Each needs its own `src/kernel/arch/<name>/`.
|
||||
page-table formats, and calling conventions. Each needs its own `system/kernel/arch/<name>/`.
|
||||
|
||||
## The Raspberry Pi models
|
||||
|
||||
@@ -57,16 +57,16 @@ the DTB/ACPI tells you what devices exist.
|
||||
|
||||
## What danos needs, layer by layer
|
||||
|
||||
- **One CPU arch module: `src/kernel/arch/aarch64/`** — covering the Zero 2 W and Pi 3-5,
|
||||
- **One CPU arch module: `system/kernel/arch/aarch64/`** — covering the Zero 2 W and Pi 3-5,
|
||||
providing the same `arch` interface as x86_64: `halt`, context switch,
|
||||
interrupt/exception vectors, page tables, a UART, a timer. No `src/kernel/arch/arm/` is
|
||||
interrupt/exception vectors, page tables, a UART, a timer. No `system/kernel/arch/arm/` is
|
||||
planned (see the decision above), so there's a single ARM backend to write.
|
||||
- **A device-tree boot path.** Since stock Pis boot via DTB, danos needs an entry
|
||||
that parses the DTB's `/memory` and `/reserved-memory` into the neutral
|
||||
[`MemoryMap`](memory-map.md) — the same neutral handoff `efi.zig` produces, just
|
||||
from a different source. This is where keeping boot-protocol knowledge on the
|
||||
loader side (as we did for the UEFI memory-map classification) pays off.
|
||||
- **The UEFI loader mostly carries over.** `src/boot/efi.zig` is largely
|
||||
- **The UEFI loader mostly carries over.** `boot/efi.zig` is largely
|
||||
boot-*protocol* code (`std.os.uefi` protocol calls), not x86 code. Its only truly
|
||||
x86-specific bits are the ELF machine check (`.X86_64`) and the SysV calling
|
||||
convention for the kernel jump. So an `aarch64`-UEFI target (QEMU `virt` + AAVMF)
|
||||
|
||||
@@ -0,0 +1,168 @@
|
||||
# Coding standards
|
||||
|
||||
Conventions for danos source. The overriding one, from which most of the rest follows:
|
||||
|
||||
> **Names are spelled out in full. An identifier is not abbreviated unless the
|
||||
> abbreviation is an acronym.**
|
||||
|
||||
`interruptDispatch`, not `intDisp`. `message_len`, not `message_len` (`msg` expands, `len`
|
||||
is a Zig idiom — see the exceptions). `devices_broker`, not `devices_broker`. `scheduler`, not
|
||||
`sched`. The cost of a longer name is paid once, at the keyboard; the cost of a
|
||||
cryptic one is paid every time the code is read, by everyone who reads it. In a
|
||||
microkernel whose whole argument is that a human can hold each piece in their head,
|
||||
that trade is not close.
|
||||
|
||||
## The rule, precisely
|
||||
|
||||
**Acronyms and initialisms stay.** They *are* the full name — expanding them would make
|
||||
the code worse, not better. `IPC`, `MMIO`, `DMA`, `IRQ`, `TSS`, `GDT`, `IDT`, `APIC`,
|
||||
`GSI`, `HPET`, `ACPI`, `PCI`, `EOI`, `BAR`, `ECAM`, `MSI`, `CPU`, `ELF`, `ABI`, `UEFI`,
|
||||
`MMU`, `TLB`, `ISR`, `ISA`, `GAS`, `HAL`, `PMM`, `VMM`, `VFS`, `HID`, `HCD`, `SMP`,
|
||||
`AML`, `MADT`, `MCFG`, `FADT`, `RSDP`, `XSDT`, `RSDT`, `GOP`, `EDID`, `TSC`, `PIT`,
|
||||
`RTC`, `LAPIC`, `SIPI`. In code they carry whatever case the surrounding convention
|
||||
demands: `Hal` the type, `hal` the variable, `mapMmio` the function.
|
||||
|
||||
**Everything else is spelled out.** If it's a word with letters removed, restore them:
|
||||
|
||||
| Abbreviation | Full |
|
||||
|---|---|
|
||||
| `proto` | `protocol` |
|
||||
| `msg` | `message` |
|
||||
| `desc` | `descriptor` |
|
||||
| `res` | `resource` |
|
||||
| `recv` | `receive` |
|
||||
| `buf` | `buffer` |
|
||||
| `cur` | `current` |
|
||||
| `src` / `dst` | `source` / `destination` |
|
||||
| `idx` | `index` |
|
||||
| `addr` | `address` |
|
||||
| `reg` | `register` |
|
||||
| `prev` | `previous` |
|
||||
| `cfg` / `config` | `configuration` |
|
||||
| `arch` | `architecture` |
|
||||
| `sched` | `scheduler` |
|
||||
| `dev` | `device` |
|
||||
| `sys` / `syscall` | `system` / `system_call` |
|
||||
| `info` | `information` |
|
||||
| `dt` | `device_tree` |
|
||||
| `ep` | `endpoint` |
|
||||
| `rt` | `runtime` |
|
||||
| `func` | `function` |
|
||||
| `phys` / `virt` | `physical` / `virtual` |
|
||||
| `wq` | `wait_queue` |
|
||||
|
||||
This list is illustrative, not exhaustive. The rule is the rule; when you meet a new
|
||||
abbreviation, expand it.
|
||||
|
||||
## Exceptions
|
||||
|
||||
Three, and only three.
|
||||
|
||||
1. **Foreign ABI names are spelled exactly as the ABI spells them — but only inside
|
||||
the layer that *is* that ABI.** A function that *is* the C or POSIX interface keeps
|
||||
its name: `fopen`, `fwrite`, `fread`, `malloc`, `calloc`, `realloc`, `free`,
|
||||
`memcpy`, `mmap`, `munmap`, `open`, `read`, `write`, `close`, `lseek`, `stat`,
|
||||
`errno`, `O_CREAT`. We don't get to rename `fwrite` to `fileWrite` — it wouldn't be
|
||||
`fwrite` any more.
|
||||
|
||||
**This exception is scoped to one place: `library/posix/`.** A file under
|
||||
`library/posix/` *is* the foreign ABI, so it keeps the ABI's spellings — that is the
|
||||
whole rule for that directory. **Everywhere else, Zig/danos naming applies with no
|
||||
POSIX exception**, so there is nothing to get wrong: if you're not in
|
||||
`library/posix/`, expand it. A concept POSIX also has gets a danos name outside that
|
||||
layer — the VFS wire protocol carries a `FileStatus`, not a `Stat`, and a `create`
|
||||
flag, not `O_CREAT`; `library/posix/` is what maps `stat`→`status` and
|
||||
`O_CREAT`→`create` at the boundary. (The `syscall` *wrappers* elsewhere are not an
|
||||
exception to this — they wrap the private danos ABI, so they use danos names.)
|
||||
|
||||
2. **Zig idioms are spelled the way Zig spells them.** Three names are the language's,
|
||||
not ours, and are left alone:
|
||||
- **`init` / `deinit`** — the constructor convention (`std.ArrayList.init`), not a
|
||||
shortening of "initialize".
|
||||
- **`len` / `ptr`** — the slice field names (`slice.len`, `slice.ptr`). Our own
|
||||
structs use bare `len`/`ptr` fields to mirror them, so a reader carries one
|
||||
mental model. (Compounds still expand: a field is `message_len`, not
|
||||
`message_length` — `len` is kept, `msg` is not.)
|
||||
- The builtins (`@min`, `@max`, `@memcpy`) and `allocator.alloc` / `.create` are
|
||||
Zig's spelling.
|
||||
|
||||
The rule governs the names *we* coin.
|
||||
|
||||
3. **Single-letter variables in a trivial local scope.** `for (items) |item, i|` may
|
||||
keep `i`; a coordinate may be `x`, `y`. The moment the scope is big enough that the
|
||||
letter's meaning isn't obvious on sight, give it a real name. When in doubt, name it.
|
||||
|
||||
That's all — no Unix-abbreviation exception. The source directories are full words
|
||||
(`system`, `library`, not `src`/`lib`), and there is no daemon `d` suffix: a driver
|
||||
lives in `system/drivers/` and a service in `system/services/`, so the *location*
|
||||
already says what it is. Encoding the role in the name too (`busd`, `vfsd`) is
|
||||
redundant — the program is just `bus`, `vfs`. Don't put in a name what its directory
|
||||
already tells you.
|
||||
|
||||
## A note on collisions
|
||||
|
||||
Two identifiers can legitimately expand to the same word. When they do, keep both
|
||||
meaningful by renaming one to its *specific* identity rather than the generic
|
||||
expansion. Two cases resolved this way:
|
||||
|
||||
- The `config` module (compile-time tunables — `maximum_cpus`, `timer_hz`) would
|
||||
collide with `cfg` (a `PlatformConfiguration` value) at `configuration`. The module
|
||||
became **`parameters`**, which is what it holds.
|
||||
- The kernel `device.zig` module would collide with `dev` (a device value) at
|
||||
`device`. The module alias became **`device_model`**, which is what it is — the
|
||||
device data model (`Device`, `DeviceTree`, `ResourceKind`).
|
||||
- The `Namespace` module alias (`ns`/`nsp` across the AML files) collides with a
|
||||
`Namespace` **instance**. Resolved by dropping the module alias entirely — the two
|
||||
types it provided are imported directly (`const Node = @import("namespace.zig").Node;`)
|
||||
— which frees `namespace` for the instance.
|
||||
|
||||
A related case is one abbreviation with two meanings. In the AML code, `op` means
|
||||
**opcode** (`opcodes.zig`, the `*_opcode` constants) but `Op` in `BinaryOperation` /
|
||||
`LogicOperation` means **operation** — distinguished by case. The per-opcode parser
|
||||
handlers, formerly `opName`/`opField`, are `parseName`/`parseField`: they *parse* the
|
||||
opcode's structure, which says what they do without overloading "op".
|
||||
|
||||
## Case and file names
|
||||
|
||||
Within those spelling rules, follow Zig's own conventions:
|
||||
|
||||
- **Types** — `PascalCase`: `DeviceDescriptor`, `Endpoint`, `WaitQueue`.
|
||||
- **Functions** — `camelCase`: `mapUserDeviceInto`, `notifyFromIsr`.
|
||||
- **Variables, fields, constants** — `snake_case`: `message_length`, `devices_broker`,
|
||||
`notify_badge_bit`.
|
||||
|
||||
**File names are `kebab-case`.** A file named for a multi-word thing hyphenates it:
|
||||
`device-tree.zig`, `ipc-synchronous.zig`, `vfs-protocol.zig`, `devices-broker.zig`. A
|
||||
single word or acronym needs no hyphen: `scheduler.zig`, `paging.zig`, `apic.zig`,
|
||||
`idt.zig`. (The module *alias* a file is imported under still follows the code
|
||||
conventions above — `snake_case` — because it's an identifier, not a filename.)
|
||||
|
||||
**A sub-project's entry point repeats its directory's name** — `init/init.zig`,
|
||||
`runtime/runtime.zig`, `hpet/hpet.zig` — and the sub-project is addressed by the
|
||||
*directory* (`system/services/init`, `library/runtime`), with the repeated leaf
|
||||
resolving away. See the repository-layout section of [README.md](README.md).
|
||||
|
||||
## Why acronyms are the line
|
||||
|
||||
Because an acronym has no letters to restore. `MMIO` doesn't become "memory mapped
|
||||
input output" in code — that expansion is what the acronym *is for*. But `msg` is just
|
||||
`message` with three letters stolen, and stealing them buys nothing a reader wants. The
|
||||
test for "is this an abbreviation I must expand" is simply: *is there a longer word this
|
||||
is a clipped form of?* If yes, write the word. If it's an initialism standing in for a
|
||||
phrase, leave it.
|
||||
|
||||
## Zen of Zig
|
||||
|
||||
* Communicate intent precisely.
|
||||
* Edge cases matter.
|
||||
* Favor reading code over writing code.
|
||||
* Only one obvious way to do things.
|
||||
* Runtime crashes are better than bugs.
|
||||
* Compile errors are better than runtime crashes.
|
||||
* Incremental improvements.
|
||||
* Avoid local maximums.
|
||||
* Reduce the amount one must remember.
|
||||
* Focus on code rather than style.
|
||||
* Resource allocation may fail; resource deallocation must succeed.
|
||||
* Memory is a resource.
|
||||
* Together we serve the users.
|
||||
@@ -0,0 +1,121 @@
|
||||
# DanOS Filesystem Hierarchy Standard (DFHS)
|
||||
|
||||
Most modern Unix and Unix-like operating systems follow the FHS. DanOS has its own FHS structure which extends the unix FHS. This is provided by virtual file system driver (VFS).
|
||||
|
||||
## Directory structure
|
||||
|
||||
| Path | Description |
|
||||
|------------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| / | Primary hierarchy root and root directory of the entire file system hierarchy. |
|
||||
| /bin | Essential command binaries that need to be available in single-user mode, including to bring up the system or repair it, for all users (e.g., cat, ls, cp). |
|
||||
| /boot | Boot loader files (e.g., EFI, initial-ramdisk.img ). |
|
||||
| /dev | POSIX Device files (e.g., /dev/null, /dev/disk0, /dev/tty, /dev/random). |
|
||||
| /etc | Host-specific system-wide configuration files. |
|
||||
| /home | Users' home directories, containing saved files, personal settings, etc. |
|
||||
| /lib | Libraries essential for the binaries in /bin and /sbin. eg realtime, system, ipc etc. |
|
||||
| /sbin | Essential system binaries (e.g init) |
|
||||
| /srv | Site-specific data served by this system, such as data and scripts for web servers, data offered by FTP servers, and repositories for version control systems |
|
||||
| /system | DanOS operating system files (similar idea to C:\Windows). A true representation of danos — its layout mirrors the source tree, so `/system` is what danos *is*. |
|
||||
| /system/devices | danos virtual device tree e.g. similar to /sys on linux but with danos device tree conventions (the structures in the devices module) |
|
||||
| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/hpet) |
|
||||
| /system/services | system-service binaries — the VFS server, init, and other user-mode servers (e.g. /system/services/vfs, /system/services/init) |
|
||||
| /system/kernel | the kernel image |
|
||||
| /tmp | Directory for temporary files (see also /var/tmp). Often not preserved between system reboots and may be severely size-restricted. |
|
||||
| /usr | Secondary hierarchy for read-only user data; contains the majority of (multi-)user utilities and applications. Should be shareable and read-only. |
|
||||
| /var | Variable files: files whose content is expected to continually change during normal operation of the system, such as logs, spool files, and temporary e-mail files. |
|
||||
|
||||
## File types
|
||||
|
||||
POSIX specifies the long format of the ls command to represent the Unix file type as the first letter for an entry.
|
||||
|
||||
| type | symbol | Description |
|
||||
|-------------------|--------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| regular | - | An ordinary file holding an uninterpreted byte stream. Reads and writes are positional, and the file grows on demand (e.g., a binary in /bin, a config file in /etc). |
|
||||
| directory | d | A container mapping names to other files. It may only be modified through directory operations, never written to directly. |
|
||||
| symbolic link | l | A file whose contents are a path that is resolved in its place. The target need not exist, and may cross mount points. |
|
||||
| FIFO special | p | A named pipe: an in-order byte stream between processes, where writers block until a reader opens the other end. |
|
||||
| block special | b | A device node addressed in fixed-size blocks with the kernel free to buffer and reorder access (e.g., /dev/disk0). |
|
||||
| character special | c | A device node addressed as an unbuffered byte stream, delivered to the driver in order (e.g., /dev/tty, /dev/null). |
|
||||
| socket | s | A named endpoint for bidirectional message-passing between processes, bound to a path rather than an address. |
|
||||
|
||||
## /dev
|
||||
|
||||
`/dev` holds the names through which processes reach devices. It is deliberately not
|
||||
the device tree: the tree — every node discovered by ACPI or PCI enumeration, with its
|
||||
resources and its parent — lives under [/system/devices](#directory-structure) and is
|
||||
addressed by device id. `/dev` is the much smaller set of devices that have a driver
|
||||
willing to serve them, addressed by name.
|
||||
|
||||
A device node is not a file the VFS can read. The bytes live in a driver process
|
||||
([drivers.md](drivers.md)), so opening a `/dev` name has to resolve to that driver's
|
||||
IPC endpoint, and subsequent reads and writes are calls against it. This is what
|
||||
`system/services/vfs/vfs.zig` reserves for M10 and what the `Stat.kind` field is for; **none of it is
|
||||
implemented today.** The current VFS is a flat, in-memory ramfs of eight nodes, with no
|
||||
directories at all and `kind` hardcoded to zero. The three sections below describe the
|
||||
intended shape, and are honest about which parts the kernel can already support.
|
||||
|
||||
### Character devices
|
||||
|
||||
A character device is a byte stream with no addressable position: bytes are delivered
|
||||
to the driver in the order written, and a read consumes what is there. Terminals,
|
||||
serial lines, keyboards and mice are all of this shape. These are the natural first
|
||||
device nodes in danos, because a character driver needs nothing the kernel doesn't
|
||||
already provide — it claims its device, maps its registers with `mmio_map`, and blocks
|
||||
on `replyWait` for either an interrupt or a client request. `system/drivers/hpet/hpet.zig` is already
|
||||
that program, minus the client half.
|
||||
|
||||
The obstacle was never the file type; it is which hardware a ring-3 driver can reach.
|
||||
Direct `in`/`out` from user space is still a #GP (no TSS I/O bitmap, IOPL never raised),
|
||||
but a driver no longer needs it: **`io_read`/`io_write`** grant port access the same way
|
||||
`mmio_map` grants memory — gated by `device_claim` and the device's discovered `io_port`
|
||||
resource. So the 16550 UART at `0x3F8` and the PS/2 controller at `0x60`/`0x64` (and thus
|
||||
`/dev/ttyS0` and a keyboard node) are now writable as ordinary ring-3 drivers; the
|
||||
low-rate legacy hardware that needs port I/O is fine with a syscall per access. A
|
||||
memory-mapped device such as the framebuffer, needing no port I/O at all, remains the
|
||||
easiest first entry.
|
||||
|
||||
### Block devices
|
||||
|
||||
A block device is addressed in fixed-size blocks and, unlike a character device, the
|
||||
layer above is free to buffer, reorder, coalesce and retry requests against it. Disks
|
||||
and other persistent storage are the whole population of this class.
|
||||
|
||||
A block driver is now **writable, but not yet memory-safe.** Every storage controller
|
||||
worth naming is a bus master: it is programmed by handing it the physical address of a
|
||||
descriptor ring and left to read and write memory on its own. That ring is exactly what
|
||||
**`dma_alloc`** now provides — physically contiguous, pinned, uncacheable, with its
|
||||
physical address disclosed — and **`/lib/mmio`**'s barriers order the descriptor writes
|
||||
against the doorbell, and **`msi_bind`** delivers completions. So an AHCI or NVMe driver
|
||||
can be written today (the M14/M15 work in [driver-model.md](driver-model.md); the earlier
|
||||
"cannot host a block driver at all" is no longer true).
|
||||
|
||||
What is *not* yet true is that it is safe. A device programmed with an arbitrary physical
|
||||
address writes to arbitrary physical memory, and page tables do not sit between a device
|
||||
and RAM — an IOMMU does. The IOMMU is now *detected* (M16), but no translation domains
|
||||
are programmed, so granting a DMA-capable device to a driver process is still equivalent
|
||||
to granting ring 0. Until per-device domains confine a driver's DMA to the buffers it
|
||||
`dma_alloc`'d, a block driver works but forfeits the isolation that motivates user-space
|
||||
drivers — enforcement is the next step, and lands with that first driver. A ramdisk over
|
||||
the initial ramdisk remains the one block-shaped thing that needs no driver process at all.
|
||||
|
||||
### Pseudo-devices
|
||||
|
||||
A pseudo-device has the interface of a device and no hardware behind it: `/dev/null`
|
||||
discarding writes and reading as end-of-file, `/dev/zero` reading as an endless run of
|
||||
zero bytes, `/dev/full` failing writes with `ENOSPC`, `/dev/random` and `/dev/urandom`
|
||||
yielding unpredictable bytes.
|
||||
|
||||
These are the only `/dev` entries danos can implement immediately, and they are the
|
||||
sensible place to start, because they are exactly the entries that need no driver
|
||||
process, no `device_claim`, no MMIO grant and no interrupt. The VFS server answers them
|
||||
out of its own address space — `null` and `zero` are a few lines each in
|
||||
`system/services/vfs/vfs.zig`'s `read` and `write` handlers. Doing so forces the two pieces of
|
||||
structure that every later device node depends on and that the flat ramfs currently
|
||||
lacks: a directory, so that `/dev/null` is a path rather than a name; and a populated
|
||||
`Stat.kind`, so that a caller can tell a character device from a regular file.
|
||||
|
||||
`/dev/random` is the one that is not free. It needs an entropy source, and the honest
|
||||
options on this kernel are `RDRAND`/`RDSEED` where CPUID advertises them, and the HPET
|
||||
counter's low bits as a poor fallback. Neither is a seeded CSPRNG, and a `/dev/random`
|
||||
that is merely unpredictable-looking is worse than none — nothing should be keyed from
|
||||
it until it is a real one.
|
||||
+34
-12
@@ -18,7 +18,7 @@ Interrupt delivery on modern x86 goes through the **APIC**, not the legacy 8259
|
||||
PIC. There are two halves; we only need one so far:
|
||||
|
||||
- The **Local APIC** (per-CPU, memory-mapped at physical `0xFEE00000`) handles the
|
||||
CPU's own timer and receives interrupts routed to it. `src/kernel/arch/x86_64/apic.zig`.
|
||||
CPU's own timer and receives interrupts routed to it. `system/kernel/architecture/x86_64/apic.zig`.
|
||||
- The **IO-APIC** routes *external* device lines (keyboard, etc.) to LAPIC vectors.
|
||||
Not needed for the timer — it'll arrive with the keyboard.
|
||||
|
||||
@@ -89,7 +89,6 @@ if (state.vector < 32) {
|
||||
on_fault(state); // exception: report and halt (never returns)
|
||||
} else if (handlers[state.vector]) |handler| {
|
||||
handler(); // device: run the registered handler
|
||||
apic.eoi(); // ...acknowledge the LAPIC
|
||||
}
|
||||
// else: spurious/unhandled — deliberately no EOI
|
||||
```
|
||||
@@ -100,10 +99,23 @@ Two things make device interrupts *return* where exceptions don't:
|
||||
flows back to `isr_common`, which restores every register it saved and executes
|
||||
`iretq` — resuming the interrupted instruction exactly. (This is why the stub
|
||||
saves *all* the general registers.)
|
||||
2. **End-of-interrupt.** After handling, we write the LAPIC's EOI register. Miss
|
||||
2. **End-of-interrupt.** Somewhere in there we write the LAPIC's EOI register. Miss
|
||||
this and the LAPIC thinks we're still busy and never delivers the next
|
||||
interrupt. It's the single most common "my timer fired once and stopped" bug.
|
||||
|
||||
**Each handler issues its own EOI**, rather than the dispatcher doing it around the
|
||||
call. That looks like a needless devolution while the timer is the only device, and
|
||||
`apic.timerTick` indeed does nothing but `eoi()` before bumping its counter (early,
|
||||
because the tick hook is the scheduler, which may switch tasks and not return
|
||||
promptly — the LAPIC mustn't wait on it).
|
||||
|
||||
It stops looking needless with the second device. A *routed* interrupt — one arriving
|
||||
through the I/O APIC from a real device line — must be **masked before it is
|
||||
acknowledged**, because a level-triggered line is still asserted at EOI time and would
|
||||
redeliver instantly, forever. Only the handler knows which discipline its source
|
||||
needs, so only the handler can sequence it. See [drivers.md](drivers.md), where the
|
||||
device is quieted by a driver in ring 3, long after the ISR has returned.
|
||||
|
||||
A device handler is a plain `fn () void` — a timer or keyboard handler doesn't need
|
||||
the interrupted registers. (Note: the stubs don't save the SSE/vector registers, so
|
||||
a handler must not use them; ours don't.)
|
||||
@@ -131,14 +143,24 @@ If the APIC weren't enabled, or `sti` were missing, or EOI were forgotten, the
|
||||
count would stay put and the test would fail. That it advances — while the CPU was
|
||||
spinning in unrelated code — is the whole mechanism working end to end.
|
||||
|
||||
## Since (done elsewhere)
|
||||
|
||||
- **Preemption**: the timer handler is where the scheduler decides to switch — the
|
||||
reason a *returning* interrupt matters. See [scheduling.md](scheduling.md).
|
||||
- **`sleep()` / timeouts** built on the calibrated clock.
|
||||
- **The I/O APIC, routed**: external device lines now reach a vector, and the
|
||||
interrupt is delivered onward to a *user-space* driver as an IPC message. See
|
||||
[drivers.md](drivers.md).
|
||||
- **Uncacheable MMIO**: device grants are mapped `PCD|PWT` (strong-uncacheable) for
|
||||
user drivers — see [paging.md](paging.md).
|
||||
|
||||
## What's next (not done here)
|
||||
|
||||
- **The keyboard**: bring up the IO-APIC, route its IRQ to a vector, and read
|
||||
scancodes from the PS/2 controller — the first *input* device.
|
||||
- **`sleep()` / timeouts** built on the calibrated clock (the monotonic
|
||||
`uptimeMs()` is in place).
|
||||
- **Uncacheable MMIO**: the LAPIC page is currently mapped writeback-cacheable like
|
||||
the rest of the identity map. QEMU tolerates it, but real hardware wants MMIO
|
||||
marked uncacheable (via the page's cache bits or an MTRR).
|
||||
- **Preemption**: once there are tasks, the timer handler is where the scheduler
|
||||
decides to switch — the reason a *returning* interrupt matters.
|
||||
- **The keyboard**: the PS/2 controller is port-mapped (`0x60`/`0x64`), and port I/O is
|
||||
now available to ring 3 via the claim-gated `io_read`/`io_write` syscalls
|
||||
([drivers.md](drivers.md)) — so the first *input* device is unblocked; it just needs
|
||||
writing (claim the controller, `irq_bind` GSI 1, read scancodes from `0x60`).
|
||||
- **MSI-X**: `msi_bind` gives one per-device edge-triggered vector (M15); MSI-X's
|
||||
multi-vector table (many queues per device, e.g. NVMe) is the remaining extension.
|
||||
- **The LAPIC's own page** is still mapped writeback-cacheable like the rest of the
|
||||
identity map. QEMU tolerates it; real hardware wants it uncacheable.
|
||||
|
||||
@@ -0,0 +1,156 @@
|
||||
# The device manager
|
||||
|
||||
**Status: the protocol and supervision are built** (M18.1, 2026-07-13): `hello`
|
||||
with its deadline, supervised spawn, restart with backoff, and the crash-loop
|
||||
cap are in — usb-xhci-bus is the first conforming driver, and the
|
||||
`driver-restart` scenario proves fault → backoff → re-claim → cap end to end.
|
||||
Tree reports are built too (M18.2, 2026-07-13): the xHCI driver scans its
|
||||
root-hub ports and reports each connected device (`child_added`); the manager
|
||||
mirrors them and prunes a dead reporter's children, and the `usb-report`
|
||||
scenario proves report → prune → respawn → re-report. The application surface is built (M18.3, 2026-07-13):
|
||||
`enumerate` and `subscribe` over IPC, with `device-list` as the first client —
|
||||
the manager is now the one answer to "what devices exist" for applications.
|
||||
The primitives underneath are real ([process-management.md](process-management.md):
|
||||
spawn/supervise/kill/exit-notification; [driver-model.md](driver-model.md): the device
|
||||
table as a capability system; [drivers.md](drivers.md): claim/map/IRQ), and the first
|
||||
per-device driver spawn works (the device manager matches the xHCI controller by PCI
|
||||
class and spawns `usb-xhci-bus` with the device id as argv[1]). This document designs
|
||||
the rest: the device manager as **the tree, the matcher, and the supervisor** — the
|
||||
policy process that turns [resilience.md](resilience.md)'s restart goal into practice
|
||||
for drivers.
|
||||
|
||||
How processes stop, reload, and report their deaths is deliberately **not** in this
|
||||
document: that is the universal lifecycle every danos process speaks —
|
||||
[process-lifecycle.md](process-lifecycle.md), signals over IPC and the stable
|
||||
`runtime.process` interface. The device manager is that design's first serious
|
||||
customer, not its owner. Its own protocol contains nothing lifecycle-shaped; a
|
||||
driver is stopped, health-checked, and buried exactly like any other process.
|
||||
|
||||
## The tree: structure in the manager, authority in the kernel
|
||||
|
||||
The device tree is two things fused: *information* (what exists, how it nests) and
|
||||
*authority* (a descriptor is a licence to map physical memory). They separate:
|
||||
|
||||
- The **kernel keeps the capability system** — device, I/O-port, and interrupt
|
||||
claims, resource containment on `device_register`, the
|
||||
`mmio_map`/`irq_bind`/`msi_bind` gates — and **cleans all of it up when a process
|
||||
dies** (settled; it is increment 1 of
|
||||
[process-lifecycle.md](process-lifecycle.md)). The three invariants in
|
||||
[driver-model.md](driver-model.md) stay exactly where they are. A device manager
|
||||
that could mint MMIO mappings by its own say-so would be a second kernel, and a
|
||||
buggy one would un-earn everything the microkernel bought.
|
||||
- The **device manager owns the tree as data** — identity, topology, naming, driver
|
||||
matching, hotplug events, and being the one process everything else asks about
|
||||
devices. Firmware discovery seeds it (today via the kernel's snapshot); **bus
|
||||
drivers grow it** by reporting what they see; applications query and watch it.
|
||||
`device_enumerate` fades to a manager-internal (then deleted) seam.
|
||||
|
||||
Long-term, discovery itself leaves the kernel — but not *into* the manager. PCI
|
||||
enumeration is a **pci-bus driver**: the manager spawns it against the host bridge
|
||||
(already a device with the ECAM window as a resource), it scans, it reports functions
|
||||
like any bus reports children. ACPI becomes an **acpi service** that interprets the
|
||||
tables and reports the namespace. The manager only orchestrates and merges. Moving
|
||||
AML interpretation out of ring 0 is its own project on its own track; nothing here
|
||||
depends on when it lands.
|
||||
|
||||
## The protocol
|
||||
|
||||
A `device-manager-protocol` module (the vfs-protocol pattern): extern-struct
|
||||
messages, a version in the handshake, reserved fields everywhere. The manager is a
|
||||
well-known endpoint (`ipc.register(.device_manager)`); the badge tells it who is
|
||||
talking; the same endpoint receives its children's exit notifications — one loop,
|
||||
one world.
|
||||
|
||||
| Direction | Message | Purpose |
|
||||
|---|---|---|
|
||||
| driver → manager | `hello { version, role, device_id }` | confirms the argv assignment, starts the deadline clock |
|
||||
| bus → manager | `child_added { parent, identity, resources }` | one node the bus discovered |
|
||||
| bus → manager | `child_removed { id }` | unplug, or the bus lost it |
|
||||
| app → manager | `enumerate` | snapshot of the tree (read-only) |
|
||||
| app → manager | `subscribe` | receive published add/remove events |
|
||||
|
||||
`hello` is the one deadline the manager enforces itself: spawned and silent past the
|
||||
deadline means wrong binary, wrong protocol version, or wedged before main — apply
|
||||
the stop sequence and the restart policy. Everything else lifecycle-shaped
|
||||
(terminate, the common `ping` liveness call, exit reasons) arrives through
|
||||
[process-lifecycle.md](process-lifecycle.md)'s vocabulary, not this protocol.
|
||||
|
||||
Assignment stays argv (`usb-xhci-bus <device id>`) for now — simple, and it works.
|
||||
The step after `hello` exists is delegation: the manager claims (or is granted) the
|
||||
devices and passes the claim to the driver over IPC (the M13 capability-transfer
|
||||
mechanism), replacing first-come-first-served `device_claim` with policy. Identity in
|
||||
`child_added` is per-bus: PCI children carry the class triple (`pci_class`, as the
|
||||
xHCI match already uses); USB children carry the (class, subclass, protocol) triple
|
||||
from usb-ids.zig — each bus's native language, decoded by the shared ids modules.
|
||||
|
||||
## Supervision and restart
|
||||
|
||||
Every driver is spawned with the manager's exit endpoint (`spawnSupervised` — built).
|
||||
On a death notification:
|
||||
|
||||
1. **Read the reason** ([process-lifecycle.md](process-lifecycle.md) increment 2).
|
||||
Clean exit → it meant to; don't restart. Fault or missed `hello` deadline →
|
||||
restart with **backoff**, and a crash-loop cap (three fast deaths → mark failed,
|
||||
stop respawning, log loudly; a later `reload` to the manager can retry).
|
||||
2. **Prune the subtree** the dead bus driver reported. Its children describe
|
||||
protocol state (xHCI slot ids, transfer rings) that died with the process;
|
||||
keeping the nodes would be keeping a lie. Watchers receive `child_removed` — the
|
||||
input service losing, then regaining, a keyboard is the *honest* description of
|
||||
what happened. The restarted instance rediscovers and re-reports.
|
||||
3. **The claim is already free** because the kernel released it at death — the
|
||||
restarted instance claims the same controller and comes up.
|
||||
|
||||
Who supervises the supervisor: **init** (PID 1), which already supervises the
|
||||
services it starts. If the manager dies, drivers keep running (they hold their
|
||||
claims; the kernel doesn't care who their supervisor was — though their exit
|
||||
notifications now dangle harmlessly). The restarted manager re-learns the world:
|
||||
kernel snapshot, then a re-`hello` round — drivers answer a broadcast or are stopped
|
||||
and respawned. Full state handoff is deliberately not attempted.
|
||||
|
||||
## Thin drivers, class protocols
|
||||
|
||||
The [driver-model.md](driver-model.md) three-shape split, restated as processes:
|
||||
|
||||
- A **bus driver** (usb-xhci-bus) owns its controller — claim, MMIO, IRQ/MSI, DMA
|
||||
rings — and offers a *transfer* protocol ("submit a control transfer to device N",
|
||||
built from the usb-abi request constructors) plus tree reports to the manager.
|
||||
- A **class driver** (usb-hid, usb-storage) owns nothing: it is matched to a reported
|
||||
child by its identity triple, speaks the bus's transfer protocol downward and its
|
||||
service's protocol upward — HID reports to the input service, blocks to the block
|
||||
service. It works unchanged over any controller.
|
||||
- **Services** (input, display, block) aggregate class drivers and face applications.
|
||||
|
||||
Each arrow is a protocol module. The manager routes none of the data plane — it
|
||||
introduces the parties (matching), supervises them (lifecycle), and gets out of the
|
||||
way.
|
||||
|
||||
## Increments
|
||||
|
||||
Increments 1–4 are the lifecycle prerequisites and live in
|
||||
[process-lifecycle.md](process-lifecycle.md) (claim cleanup on death, exit reasons,
|
||||
published exit events, signals + `runtime.process`). On top of those:
|
||||
|
||||
5. **device-manager-protocol**: `hello`, supervised spawn with restart policy;
|
||||
usb-xhci-bus becomes the first conforming driver.
|
||||
6. **Tree reports**: `child_added`/`child_removed`; the manager mirrors; xHCI reports
|
||||
the mouse and keyboard QEMU already hangs off it.
|
||||
7. **App surface**: `enumerate`/`subscribe` over IPC; `device_enumerate` retreats
|
||||
to a manager-internal seam.
|
||||
8. **Discovery migration** — DONE (M19–M20, 2026-07-13): pci-bus driver (M19)
|
||||
then the acpi service (M20) moved enumeration to ring 3; the kernel seeds
|
||||
only the host bridge and the acpi-tables node. See
|
||||
[m19-m20-plan.md](m19-m20-plan.md).
|
||||
|
||||
## Settled questions (2026-07-12)
|
||||
|
||||
- **Stateful buses**: pruning the subtree on bus-driver death is right for USB. A
|
||||
future storage bus with in-flight writes wants drain-before-terminate — which is
|
||||
exactly the `deadline_ms` parameter `stop()` already has; a per-driver deadline
|
||||
is one value in the manager's policy table when such a bus arrives. No design
|
||||
change.
|
||||
- **Manager death**: drivers survive the manager; the restarted manager re-learns
|
||||
the world (above). Checkpointing driver state with the manager is deferred until
|
||||
something demonstrates the need.
|
||||
- **Matching stays code until the third bus.** `driverFor`/`pciDriverFor` are
|
||||
honest at two bus types; the third triggers the manifest (a driver declares what
|
||||
it binds: a PCI class triple, a USB class triple, an ACPI `_HID`).
|
||||
@@ -167,3 +167,27 @@ free; discovery on x86 is partly about *finding* what ARM just tells you.
|
||||
- [ipc.md](ipc.md) — the channels that interrupts-as-messages and the device manager
|
||||
will ride on.
|
||||
- [vision.md](vision.md) — why drivers belong in isolated user space at all.
|
||||
|
||||
## Update (M19.3, 2026-07-13): PCI enumeration left the kernel
|
||||
|
||||
The kernel now seeds only the `pci_host_bridge` node (ECAM window, MMIO
|
||||
apertures derived from the memory map's holes, bus range, and the 16-bit I/O
|
||||
window). The per-function walk moved to the ring-3 `pci-bus` driver
|
||||
([device-manager.md](device-manager.md)): it claims the bridge, repeats the
|
||||
ECAM scan through its mmio grant, and `device_register`s what it finds, which
|
||||
the device manager mirrors and matches. The ACPI namespace walk follows in M20;
|
||||
the static tables (MADT, HPET, MCFG, FADT + `\\_S5`) stay kernel-side.
|
||||
|
||||
## Update (M20.3, 2026-07-13): ACPI enumeration left the kernel too
|
||||
|
||||
The kernel no longer folds the AML namespace's Device objects into the device
|
||||
tree. It still parses the *static* tables (MADT for SMP, HPET for the tick, MCFG
|
||||
for the host bridge, FADT) and still builds the AML namespace — but only to read
|
||||
the `\\_S5` sleep type for poweroff. Device discovery is the ring-3 **acpi
|
||||
service** ([device-manager.md](device-manager.md)): it claims the `acpi-tables`
|
||||
node the kernel publishes (the AML blobs, a broad io_port grant, the SCI),
|
||||
re-parses the same blobs with the shared AML module, evaluates `_STA`/`_CRS`,
|
||||
and registers + reports each `_HID` device — the device manager matches drivers
|
||||
(ps2-bus) from those reports. With M19's pci-bus driver, discovery now runs
|
||||
entirely in user space; the kernel seeds only the host bridge and the
|
||||
acpi-tables node.
|
||||
|
||||
@@ -0,0 +1,367 @@
|
||||
# The driver model: buses, classes, and host controllers
|
||||
|
||||
[drivers.md](drivers.md) shows how to write *a* driver — claim a device, map its
|
||||
registers, sleep on its interrupt. That's enough for a leaf device like the HPET. It is
|
||||
not enough for a disk, a keyboard, or a network card, because those hang off a
|
||||
*controller*, on a *bus*, speaking a *protocol*, and no single process should have to
|
||||
know all three.
|
||||
|
||||
Real driver stacks factor into three shapes. This document is about what each one is,
|
||||
what the kernel must give it, how they share code — and precisely which primitive each
|
||||
is still blocked on.
|
||||
|
||||
## Three shapes
|
||||
|
||||
| Shape | Owns | Reaches hardware by | Talks to |
|
||||
|---|---|---|---|
|
||||
| **Host controller driver** (HCD) | a controller — an xHCI PCI function, an AHCI port block | `mmio_map` + `irq_bind` + DMA | the devices behind it, in its bus's language |
|
||||
| **Bus driver** | a bus — a PCI bridge, a USB hub | `device_register`, to publish what it finds | class drivers, over IPC |
|
||||
| **Class / protocol driver** | *nothing* | *nothing* | its bus driver, over IPC |
|
||||
|
||||
The last row is the surprising one and the whole point. A USB keyboard driver touches
|
||||
no registers, takes no interrupts, and maps no memory. It sends HID protocol messages
|
||||
to whatever published the device, and it works identically whether the controller
|
||||
below is xHCI, EHCI, or a Raspberry Pi's DWC2. That is what buys you drivers that
|
||||
outlive the hardware they were written for.
|
||||
|
||||
In practice **HCD and bus driver are usually the same process**. An xHCI driver is a
|
||||
host controller driver (it owns the PCI function, its BARs, its interrupt, its DMA
|
||||
rings) *and* a bus driver (it enumerates USB devices and publishes them). Splitting
|
||||
them is a fiction; what matters is that both *roles* have kernel support, because a
|
||||
plain bus driver with no controller — a USB hub — is also a real thing.
|
||||
|
||||
## The device table is the spine
|
||||
|
||||
danos already has the right central structure. `system/kernel/devices-broker.zig` holds a table of
|
||||
`DeviceDesc`, each with a parent, a class, and a set of resources. Firmware discovery
|
||||
seeds it ([discovery.md](discovery.md)); `device_register` grows it.
|
||||
|
||||
Three invariants make it a capability system rather than a directory:
|
||||
|
||||
1. **A claim is exclusive.** `device_claim(id)` succeeds once. Everything downstream —
|
||||
`mmio_map`, `irq_bind`, `device_register` — checks `devices_broker.ownerOf(id) == me`.
|
||||
2. **A descriptor is a licence to map physical memory.** Whoever claims a device may
|
||||
map its `.memory` resources and bind its `.irq` resources. This is why
|
||||
`device_register` cannot be a free-for-all.
|
||||
3. **Therefore: containment.** Every resource of a registered child must lie inside a
|
||||
resource of the same kind on its parent (`devices_broker.contains`). A bus driver can only
|
||||
ever *subdivide* what it already holds. Without this, `device_register` would be a
|
||||
syscall named "map any physical page you like."
|
||||
|
||||
Containment is transitive by construction: a grandchild is contained in its child,
|
||||
which is contained in the bus. Nothing can be laundered through a chain.
|
||||
|
||||
Note that firmware topology does **not** obey containment, and isn't asked to — a PCI
|
||||
function's BAR is not inside its host bridge's `bus_range`, because a bus-number range
|
||||
is not an address window. Discovery is trusted; user space is not.
|
||||
|
||||
### What a bus driver looks like
|
||||
|
||||
`system/drivers/bus/bus.zig` is the smallest honest one. Its "bus" is the HPET's register block and
|
||||
its "devices" are the block's comparators:
|
||||
|
||||
```zig
|
||||
_ = dev.claim(bus.id); // 1. own the bus
|
||||
const base = dev.mmioMap(bus.id, 0).?; // 2. enumerate it — from the hardware
|
||||
const n = ((cap.* >> 8) & 0x1F) + 1; // GENERAL_CAP says how many children
|
||||
|
||||
for (0..n) |i| { // 3. publish each child
|
||||
var child = std.mem.zeroes(dev.DeviceDesc);
|
||||
child.class = @intFromEnum(dev.DeviceClass.timer);
|
||||
child.resource_count = 1;
|
||||
child.resources[0] = .{ .kind = memory,
|
||||
.start = bus_mmio.start + 0x100 + 0x20 * i,
|
||||
.len = 0x20 };
|
||||
_ = dev.register(bus.id, &child).?; // kernel checks containment
|
||||
}
|
||||
```
|
||||
|
||||
Each child is left **unclaimed**, which is the handoff: a comparator driver can now
|
||||
`device_claim` one and `mmio_map` it, and will see only its own 0x20-byte window. A child
|
||||
whose window escapes the bus is refused — `bus` asserts that, and the `bus` test
|
||||
asserts the kernel's table upholds it.
|
||||
|
||||
A USB device has *no* resources at all: `resource_count = 0`, because it's addressed
|
||||
through its controller, not by MMIO. That case is allowed and is the common one.
|
||||
|
||||
## Families: sharing code between drivers
|
||||
|
||||
A "family" is two modules, not one:
|
||||
|
||||
- **A logic module** — the parts of the bus that every driver on it re-derives. Config
|
||||
space walking and BAR decode for PCI. Descriptor parsing, control transfers, and hub
|
||||
protocol for USB.
|
||||
- **A protocol module** — the IPC message types that let a class driver talk to
|
||||
*whatever* published its device. This is the part that makes class drivers portable.
|
||||
|
||||
danos already has one of each: `library/runtime/device.zig` is a logic module,
|
||||
[`system/services/vfs/protocol.zig`](system/services/vfs/protocol.zig) is a protocol module shared by `system/services/vfs/vfs.zig`
|
||||
and its clients. The pattern generalises directly:
|
||||
|
||||
```
|
||||
library/
|
||||
runtime/ module "runtime" — syscalls, heap, ipc, device, stdio
|
||||
mmio/ module "mmio" — volatile register access + barriers [M14]
|
||||
bus/
|
||||
pci/ module "pci" — ECAM, BAR decode, capability walk
|
||||
usb/ module "usb" — descriptors, control transfers, hubs
|
||||
proto/
|
||||
vfs/ module "vfs-protocol" (today: system/services/vfs/protocol.zig)
|
||||
block/ module "block-protocol"
|
||||
hid/ module "hid-protocol"
|
||||
|
||||
system/drivers/ one sub-project each → /system/drivers (no `d` suffix)
|
||||
xhci/ HCD + bus driver imports runtime, pci, usb, mmio
|
||||
usb-hid/ class driver imports runtime, usb, hid-protocol
|
||||
block/ class driver imports runtime, block-protocol
|
||||
```
|
||||
|
||||
The only build change needed: [`addUserBinary`](build.zig) currently takes exactly one
|
||||
module (`rt_mod`) and injects it. It should take a slice of modules. That's a
|
||||
five-line change, and it's the *entire* mechanism — Zig modules already give you
|
||||
everything else.
|
||||
|
||||
The discipline that makes this work: **a class driver must not import a bus's logic
|
||||
module.** `usbhid` imports `proto.hid` and `usb` (for descriptor types), never `pci`.
|
||||
If a class driver needs `mmio`, it has become an HCD and should be one.
|
||||
|
||||
## What exists today
|
||||
|
||||
- **M10** — `device_enumerate`, `device_claim`, `mmio_map`. Strong-uncacheable device
|
||||
grants, `device_grant` teardown.
|
||||
- **M11** — `irq_bind` / `irq_ack`. IRQ delivered as an IPC notification; mask before
|
||||
EOI; `irq_ack` is the unmask.
|
||||
- **M12** — `parent` in `DeviceDesc`, `device_register` with resource containment.
|
||||
- **M13** — capability passing. `ipc_call` / `ipc_reply_wait` grew a `send_cap` argument
|
||||
and a `received_cap` return (r8): an endpoint travels with a message, installed into
|
||||
the receiver's handle table (shared, refcount-bumped — a copy, not a move). A full
|
||||
table fails `-ENOSPC` and does not half-deliver. This is the "open" primitive — a bus
|
||||
driver mints a per-device endpoint and hands it to a class driver. The runtime exposes
|
||||
`callCap` and `replyWait(..., send_cap)`; no class driver consumes it yet.
|
||||
- **M14** — DMA memory + the memory-ordering layer. `/lib/mmio` gives drivers typed
|
||||
volatile access and `mb`/`rmb`/`wmb` (per-arch); `dma_alloc`/`dma_free` grant
|
||||
physically-contiguous, pinned, uncacheable, reclaim-on-teardown buffers with the
|
||||
physical address exposed (`pmm.allocContiguous`, a DMA arena, `mapUserDmaInto`).
|
||||
`dma_below_4g` caps the address for legacy engines; `dma_write_combining` is accepted
|
||||
but falls back to coherent until PAT is programmed. hpet is refactored onto `/lib/mmio`;
|
||||
no DMA driver consumes `dma_alloc` yet.
|
||||
- **M15** — interrupts for PCI devices, the MSI half. Discovery now gives every PCI
|
||||
function its 4 KiB ECAM config space as resource 0 (unblocking the capability walk
|
||||
with no new syscall), and `msi_bind(device_id, endpoint) -> address, data` allocates a
|
||||
per-device edge-triggered vector, delivered as an IPC notification with no mask and no
|
||||
ack cycle. Legacy INTx (`_PRT` parsing + shared lines) is deliberately skipped — MSI
|
||||
is the real answer. QEMU's HPET has no MSI, so delivery is proven with a self-IPI; the
|
||||
first PCI driver is the first real consumer.
|
||||
- **Port I/O** — `io_read`/`io_write(device_id, resource_index, offset, width[, value])`:
|
||||
a claimed device's `io_port` resource lets a driver read/write its ports, gated exactly
|
||||
like `mmio_map` gates memory (direct ring-3 `in`/`out` stays a #GP). This is what makes
|
||||
a PS/2 or 16550 driver possible; the low-rate legacy hardware that needs it is fine with
|
||||
a syscall per access. `io_port` resources were recorded by discovery and ignored — now
|
||||
they're used.
|
||||
- **M16 (detection)** — the IOMMU is now *found*: discovery parses the ACPI DMAR table,
|
||||
maps the first VT-d unit, and reads its version + capabilities (`iommu_present` in the
|
||||
platform info). This is detection only — **no translation domains are programmed, so
|
||||
DMA is still unprotected** (the caveat below). Enforcement lands with the first DMA
|
||||
driver, which is what there is to protect and test against. Proven in the `iommu` test,
|
||||
booted with an emulated `intel-iommu`.
|
||||
- **`system_spawn`** — a user-space supervisor starts a driver:
|
||||
`system_spawn(name, arguments)` loads a binary bundled in the initial-ramdisk as a
|
||||
fresh ring-3 process; `name` becomes the child's argv[0] and the optional
|
||||
NUL-separated `arguments` blob its argv[1..], delivered on a SysV entry stack
|
||||
([sysv.md](sysv.md)). This is what
|
||||
turned the device manager from "log the match" into "run the driver": the kernel now
|
||||
spawns only `init`, `init` spawns the services, and the **device-manager** discovers
|
||||
the hardware and spawns each driver ([drivers.md](drivers.md)). Ungated for now — a
|
||||
spawn capability is future work.
|
||||
|
||||
So: **bus drivers work now, and they're started by the device manager, not the kernel.**
|
||||
HCDs and class drivers do not work yet. Here is exactly why, and exactly what would fix it.
|
||||
|
||||
---
|
||||
|
||||
# Proposed ABI
|
||||
|
||||
## M13 — capability passing, for class drivers ✅ done
|
||||
|
||||
*Implemented as described below (see "What exists today"). The signatures landed
|
||||
verbatim: `send_cap` in r9, `received_cap` returned in r8, `-ENOSPC` on a full receiver
|
||||
table with no delivery. The rest of this section is the original design note.*
|
||||
|
||||
**The blocker.** A class driver has to reach *its* device. Today the only way to find
|
||||
an endpoint is the name registry: `ipc_register(service_id, h)` / `ipc_lookup(id)`,
|
||||
where `ServiceId` is a global integer namespace with `max_services = 8`. You cannot
|
||||
mint one endpoint per USB device that way, and there is no way for a bus driver to
|
||||
*hand* a class driver an endpoint. M7 deferred this deliberately.
|
||||
|
||||
**The fix.** Let a message carry one handle. Sender names a handle in its own table;
|
||||
the kernel installs the endpoint into the receiver's table (bumping `refcount`) and
|
||||
tells the receiver the index it landed at.
|
||||
|
||||
```
|
||||
ipc_call(h, msg, message_len, reply, reply_cap, send_cap) -> reply_len
|
||||
ipc_reply_wait(h, reply, reply_len, recv, recv_cap, send_cap)
|
||||
-> recv_len (rax), badge (rdx), received_cap (r8)
|
||||
```
|
||||
|
||||
`send_cap` is a handle or `no_cap` (`~0`). `received_cap` is the index the transferred
|
||||
endpoint was installed at in the receiver's table, or `no_cap`.
|
||||
|
||||
- Both calls grow from 5 args to 6, which fits: `syscall5` uses `rdi/rsi/rdx/r10/r8`,
|
||||
leaving `r9`. `ipc_reply_wait` already returns two values via `setSyscallResult2`;
|
||||
this needs a third (`setSyscallResult3`).
|
||||
- If the receiver's handle table is full, the call fails `-ENOSPC` and **the message is
|
||||
not delivered** — a half-delivered capability is worse than a failed send.
|
||||
- `closeHandles` already drops references on exit, so the lifetime story is unchanged.
|
||||
|
||||
That single primitive gives you the standard `open` pattern:
|
||||
|
||||
```zig
|
||||
// class driver // bus driver
|
||||
const h = ipc.lookup(.usb).?; const r = ipc.replyWait(ep, ...);
|
||||
const dev_ep = ipc.callCap(h, // ... mint a per-device endpoint,
|
||||
.{ .op = .open, .id = dev_id }); // reply with it as send_cap
|
||||
// now dev_ep is a private channel to that one device
|
||||
```
|
||||
|
||||
## M14 — DMA memory and the memory-ordering contract, for HCDs ✅ done
|
||||
|
||||
*Implemented: `/lib/mmio` (typed volatile access + `mb`/`rmb`/`wmb`, per-arch) and
|
||||
`dma_alloc`/`dma_free` (contiguous, pinned, uncacheable, reclaim-on-teardown, physical
|
||||
address exposed). `dma_write_combining` still falls back to coherent — real WC needs
|
||||
PAT, a small follow-up. The rest of this section is the original design note.*
|
||||
|
||||
**The blocker.** An HCD is a DMA-engine programmer. It needs a descriptor ring the
|
||||
device can read, which means memory that is (a) physically contiguous, (b) at a
|
||||
physical address the driver knows, (c) of the right cacheability, and (d) pinned.
|
||||
[`sysMmap`](system/kernel/process.zig) gives you *none* of the four: it calls `pmm.alloc()`
|
||||
once per page, maps writeback-cached, and never reveals a physical address.
|
||||
|
||||
**The fix.**
|
||||
|
||||
```
|
||||
dma_alloc(len, flags) -> vaddr (rax), paddr (rdx)
|
||||
dma_free(vaddr, len) -> 0
|
||||
|
||||
flags: dma_coherent (1) uncacheable; the default and the only one that's portable
|
||||
dma_wc (2) write-combining — needs PAT programmed; for framebuffers
|
||||
dma_below_4g (4) for devices with 32-bit DMA addressing
|
||||
```
|
||||
|
||||
Guarantees: page-aligned, physically contiguous, zeroed, pinned for the life of the
|
||||
mapping, and the physical address is stable. It needs one thing the kernel lacks —
|
||||
`pmm.allocContiguous(n, max_phys)`; today `pmm.alloc()` hands out one frame at a time
|
||||
with no adjacency guarantee.
|
||||
|
||||
**The memory-ordering contract.** danos has, at the time of writing, **zero memory
|
||||
barriers anywhere in the tree.** That is currently correct-by-accident and won't
|
||||
survive the first DMA driver, or the first ARM boot.
|
||||
|
||||
`volatile` is not a barrier. In Zig it means: don't elide this access, and don't
|
||||
reorder it against *other volatile* accesses. It says nothing about your *ordinary*
|
||||
stores — the descriptor you just filled in normal WB memory — which LLVM may freely
|
||||
sink past a volatile MMIO write. The canonical bug:
|
||||
|
||||
```zig
|
||||
ring[i] = descriptor; // ordinary store to WB RAM
|
||||
doorbell.* = i; // volatile store to UC MMIO
|
||||
// nothing stops the compiler reordering these; the device reads a stale descriptor
|
||||
```
|
||||
|
||||
So the rules, which belong in `library/mmio.zig` and behind `arch`:
|
||||
|
||||
| Situation | Required |
|
||||
|---|---|
|
||||
| MMIO register read/write | `mmio.read` / `mmio.write` (volatile) |
|
||||
| Fill DMA descriptor, then ring doorbell | `wmb()` between them |
|
||||
| Woken by IRQ, then read what the device wrote | `rmb()` before the read |
|
||||
| MMIO write that must complete before the next read | `mb()` |
|
||||
|
||||
And the per-arch lowering — the reason this must be an `arch` primitive and not a
|
||||
sprinkling of `asm volatile`:
|
||||
|
||||
| | x86_64 | aarch64 |
|
||||
|---|---|---|
|
||||
| `mb()` | `mfence` | `dsb sy` |
|
||||
| `rmb()` | `lfence` | `dsb ld` |
|
||||
| `wmb()` | `sfence` | `dsb st` |
|
||||
| DMA cache coherency | coherent; nothing to do | **not guaranteed**; needs non-cacheable buffers or cache maintenance |
|
||||
|
||||
x86 is forgiving here — TSO plus strong-uncacheable MMIO means you usually get away
|
||||
with a compiler barrier alone. ARM is not, and [vision.md](vision.md) makes ARM the win
|
||||
condition. Build the abstraction while there is one caller to fix.
|
||||
|
||||
(Zig note: `@fence` was **removed in 0.16**. Use `@atomicRmw(..., .seq_cst)` for a full
|
||||
barrier, or per-arch inline asm — which is what `library/mmio.zig` should hide.)
|
||||
|
||||
## M15 — interrupts for PCI devices ✅ done (MSI)
|
||||
|
||||
*Implemented the MSI half: ECAM config space per PCI function (resource 0) and
|
||||
`msi_bind` (per-device edge-triggered vector, delivered as a notification). Legacy INTx
|
||||
`_PRT` parsing is skipped on purpose. `msi_bind` returns (address, data) as two values
|
||||
rather than an out-struct. The rest of this section is the original design note.*
|
||||
|
||||
**The blocker, and it's a hard one.** No PCI device can take an interrupt today.
|
||||
[`addBars`](system/devices/acpi.zig) records `.memory` and `.io_port` BARs and never an
|
||||
`.irq`; there is no `_PRT` parsing anywhere in the tree. `hpet` only works because the
|
||||
HPET advertises its own routing options in its own registers — a privilege no ordinary
|
||||
device has.
|
||||
|
||||
**The fix, in two halves.**
|
||||
|
||||
*Legacy INTx*: parse `_PRT` from the DSDT to map (device, INTA–D) → GSI, and record it
|
||||
as an `.irq` resource. Then `irq_bind` works unchanged. But INTx lines are **shared**,
|
||||
and `irq.bound[gsi]` holds one endpoint. Sharing needs a list, and every driver on the
|
||||
line must be polled on each interrupt — the reason everyone left INTx behind.
|
||||
|
||||
*MSI/MSI-X*, which is the real answer: per-device vectors, edge-triggered, unshared, no
|
||||
mask/ack cycle, no 24-GSI ceiling. The kernel allocates a vector and hands the driver
|
||||
the (address, data) pair to program into its own MSI capability:
|
||||
|
||||
```
|
||||
msi_bind(dev_id, endpoint, out) -> 0 // out: extern struct { addr: u64, data: u32 }
|
||||
```
|
||||
|
||||
The driver writes those into config space itself — which means it needs config space,
|
||||
which means **discovery should give each `pci_device` a `.memory` resource for its
|
||||
4 KiB ECAM slot**. That's a small change to `parseMcfg` and it unblocks the whole
|
||||
capability walk (MSI, MSI-X, PCIe extended caps) without any new syscall.
|
||||
|
||||
Note QEMU's HPET reports `Tn_FSB_INT_DEL_CAP = 0` — no MSI — so `hpet` can never
|
||||
exercise this path. The first MSI driver will be the first PCI driver.
|
||||
|
||||
## M16 — the IOMMU, and the honest caveat ◑ detection done, enforcement pending
|
||||
|
||||
*The IOMMU is now detected (DMAR parsed, VT-d unit mapped and read — see the `iommu`
|
||||
test), but **enforcement is not built**: no translation domains are programmed, so the
|
||||
caveat below still holds in full. Detection can't be taken further usefully until there
|
||||
is a DMA driver to protect and QEMU's `intel-iommu` to test the protection against —
|
||||
building the per-device domains alongside that first driver is both the natural order
|
||||
and the only way to verify them. The rest of this section is the original caveat.*
|
||||
|
||||
Everything above is capability-gated at the *CPU*. None of it is gated at the *device*.
|
||||
A driver that can program a bus-mastering engine can make that device write to any
|
||||
physical address, because page tables sit between the CPU and RAM, not between a device
|
||||
and RAM. Until VT-d/DMAR (or SMMU on ARM) is programmed from the DMAR table, **`device_claim`
|
||||
on any DMA-capable device is equivalent to granting ring 0.**
|
||||
|
||||
This does not make the model useless — it's the same position Linux is in with the
|
||||
IOMMU off, and every other guarantee (crash isolation, restart, no shared address
|
||||
space) still holds. But "user-space drivers are memory-safe" is not true yet, and the
|
||||
gap should be named rather than implied.
|
||||
|
||||
## Ordering
|
||||
|
||||
`M13` (capability passing) is independent of `M14`/`M15` and is the cheapest. It
|
||||
unlocks class drivers, which are the shape with no hardware requirements at all — you
|
||||
could write a real one against `bus`'s comparators tomorrow.
|
||||
|
||||
`M14` and `M15` together unlock the first HCD. `M14`'s barrier layer is worth landing
|
||||
on its own regardless: it's small, obviously correct, and stops every future driver
|
||||
from hand-rolling `*volatile` and getting ARM wrong.
|
||||
|
||||
## See also
|
||||
|
||||
- [drivers.md](drivers.md) — how to write one, concretely.
|
||||
- [discovery.md](discovery.md) / [acpi.md](acpi.md) — where the device table comes from.
|
||||
- [ipc.md](ipc.md) — endpoints, badges, and the notification path an IRQ arrives on.
|
||||
- [resilience.md](resilience.md) — restart, the reason any of this is worth the trouble.
|
||||
+384
@@ -0,0 +1,384 @@
|
||||
# Writing a driver
|
||||
|
||||
In a monolithic kernel a driver is a function call away from everything: it runs in
|
||||
ring 0, dereferences any physical address, and its interrupt handler *is* the ISR. In
|
||||
danos a driver is **an ordinary ring-3 process**. It has its own address space, it
|
||||
can crash without taking the kernel with it, and — the point of this document — it
|
||||
can be restarted ([resilience](resilience.md)).
|
||||
|
||||
That leaves three questions the kernel has to answer, because a process can't answer
|
||||
them for itself:
|
||||
|
||||
1. **What hardware exists?** → `device_enumerate`, over the device table discovery built
|
||||
([discovery](discovery.md), [acpi](acpi.md)).
|
||||
2. **How do I touch its registers?** → `device_claim` + `mmio_map`: the kernel maps the
|
||||
device's physical MMIO window into your address space, and from then on it's plain
|
||||
memory. No syscall per register access.
|
||||
3. **How do I find out it wants something?** → `irq_bind`: the interrupt is delivered
|
||||
to you as an IPC notification. You block; the hardware wakes you.
|
||||
|
||||
A driver is, in one sentence, *a process that sleeps until its device has something to
|
||||
say.*
|
||||
|
||||
## How a driver gets started: discover, match, spawn
|
||||
|
||||
Nothing in the kernel decides that the HPET needs the `hpet` driver — that is policy,
|
||||
and policy lives in user space. Boot brings user space up as a three-level supervision
|
||||
hierarchy, each level owning one job:
|
||||
|
||||
```
|
||||
kernel ──spawns──► init (PID 1) ──spawns──► device-manager ──spawns──► hpet
|
||||
| | |
|
||||
spawns only init, the service supervisor: the driver supervisor: enumerates
|
||||
publishes the starts the system /system/devices, matches each device
|
||||
initial-ramdisk services (vfs, the to a driver, and system_spawn's it
|
||||
so user space can device-manager). Its
|
||||
system_spawn from it list is init policy.
|
||||
```
|
||||
|
||||
The kernel launches exactly one process — `init` — and hands it nothing but the raw
|
||||
ability to start more (`system_spawn(name, arguments)`, which loads a binary bundled
|
||||
in the initial-ramdisk as a fresh ring-3 process — `name` becoming its argv[0],
|
||||
the optional arguments its argv[1..], on a SysV entry stack, see sysv.md). Everything else is a user-space decision:
|
||||
|
||||
- **init** ([system/services/init](system/services/init/init.zig)) is the **service
|
||||
supervisor**. It spawns the system services danos brings up at boot — today `vfs` and
|
||||
the `device-manager` — from a small list. Drivers are deliberately *not* its job.
|
||||
- **device-manager** ([system/services/device-manager](system/services/device-manager/device-manager.zig))
|
||||
is the **driver supervisor**. It does the three steps a monolithic kernel would do in
|
||||
its probe path, entirely from ring 3:
|
||||
1. **Discover** — `device_enumerate` snapshots the device table the kernel built from
|
||||
ACPI/PCI ([discovery](discovery.md)).
|
||||
2. **Match** — for each device it looks up a driver by `DeviceClass`. The match policy
|
||||
is a table (`driverFor`): today a static `timer → hpet` map; a fuller system reads
|
||||
what each driver *binds* (a manifest under `/system/drivers`, or the driver
|
||||
describing its own match).
|
||||
3. **Spawn** — `system_spawn(driver_name, arguments)` starts the matched driver (the
|
||||
arguments can carry *which* device it matched), which then claims
|
||||
its device and runs the event loop below.
|
||||
|
||||
So "how is a driver discovered and configured" has two halves: **discovery** is the
|
||||
kernel's device table, read by anyone; **configuration** is two user-space policies —
|
||||
init's service list and the device-manager's match table. Both are hardcoded in their
|
||||
respective programs today; the natural next step is to move them into `/etc` (see the
|
||||
milestone notes in [driver-model.md](driver-model.md)). `system_spawn` is currently
|
||||
ungated — any process may spawn any bundled binary — because there is no spawn
|
||||
capability yet.
|
||||
|
||||
## The capability: claim before touch
|
||||
|
||||
The driver syscall numbers (`system/abi.zig`) with the device types they carry
|
||||
(`system/devices/device-abi.zig`), dispatched in `system/kernel/process.zig`:
|
||||
|
||||
| # | Call | Meaning |
|
||||
|---|------|---------|
|
||||
| 11 | `device_enumerate(buf, max) -> total` | Snapshot the device table |
|
||||
| 12 | `device_claim(id) -> ok` | Take **exclusive** ownership |
|
||||
| 13 | `mmio_map(id, res_idx) -> vaddr` | Map a claimed device's register window |
|
||||
| 14 | `irq_bind(id, res_idx, endpoint)` | Deliver that device's IRQ as a notification |
|
||||
| 15 | `irq_ack(id, res_idx)` | Re-arm the IRQ after servicing the device |
|
||||
| 16 | `device_register(parent_id, desc) -> id` | Publish a child of a device you claimed |
|
||||
|
||||
Notice that **nothing takes a physical address or an interrupt number.** Every call
|
||||
names a device by id and a resource by index. That indirection is the entire security
|
||||
model. If `mmio_map` took a physical address, any process could map the kernel's
|
||||
memory; if `irq_bind` took a GSI, any process could bind the keyboard's line and
|
||||
silently intercept it. Instead the kernel checks two things (`process.ownedGsi`, and
|
||||
the same check at the top of `sysMmioMap`):
|
||||
|
||||
- `devices_broker.ownerOf(dev_id) == me` — you claimed it, and claims are exclusive
|
||||
- the resource at `res_idx` is of the right *kind* — `memory` for `mmio_map`, `irq`
|
||||
for `irq_bind`
|
||||
|
||||
The claim is the capability. Everything else follows from it.
|
||||
|
||||
## Registers: `mmio_map`
|
||||
|
||||
`mmio_map` walks the caller's page tables and installs the device's physical frames
|
||||
with `present | user | writable | nx | pcd | pwt`
|
||||
(`arch/x86_64/paging.zig:mapUserDeviceInto`). Two of those bits are load-bearing:
|
||||
|
||||
- **`pcd | pwt`** — strong-uncacheable. A device register is not memory; a cached read
|
||||
would return a stale value and a write might never leave the CPU.
|
||||
- **`device_grant`** (bit 9, one of the PTE's available bits) — marks the leaf as MMIO
|
||||
rather than RAM, so `freeSubtree` skips `pmm.free` on it when the address space is
|
||||
destroyed. Without this, killing a driver would hand the HPET's registers back to
|
||||
the frame allocator as if they were free RAM. The `iopass` test guards it.
|
||||
|
||||
Grants land in their own arena, `0x0000_7100_0000_0000` (PML4[226]), so device pages
|
||||
never widen an existing mapping.
|
||||
|
||||
Then you just… use it:
|
||||
|
||||
```zig
|
||||
const base = dev.mmioMap(dev_id, mmio_res) orelse return;
|
||||
const counter: *volatile u64 = @ptrFromInt(base + 0xF0);
|
||||
const now = counter.*; // a load, straight to the hardware. no kernel involved.
|
||||
```
|
||||
|
||||
## Interrupts: the cycle, and why it has that shape
|
||||
|
||||
An interrupt handler in a microkernel has a problem. The code that knows how to quiet
|
||||
the device is in ring 3, in another address space, and it will not run for
|
||||
microseconds or milliseconds — after a context switch, when the scheduler gets to it.
|
||||
But the CPU wants an EOI *now*, and a **level-triggered** line stays asserted until
|
||||
the device is quieted. EOI a still-asserted line and the I/O APIC redelivers
|
||||
immediately. Forever. The driver never gets to run at all.
|
||||
|
||||
The way out is to mask the line before acknowledging it:
|
||||
|
||||
```
|
||||
kernel ISR irqMask(gsi) // line still asserted; stop it reaching a CPU
|
||||
irqEoi() // now safe to tell the LAPIC we're done
|
||||
notifyFromIsr() // wake the driver — it runs much later
|
||||
|
||||
driver replyWait() -> badge with the notify bit set
|
||||
<clear the device's status register> // NOW the line deasserts
|
||||
irq_ack(dev, res) // kernel unmasks: quiet, so it can't refire
|
||||
|
||||
```
|
||||
|
||||
`irq_ack` is not bookkeeping you could skip. **It is the unmask.** Forget it and the
|
||||
interrupt fires exactly once, ever; call it before the device is quiet and you get an
|
||||
interrupt storm. That single fact explains why `irq_bind` and `irq_ack` are two
|
||||
syscalls and not one.
|
||||
|
||||
This is also why `interruptDispatch` (`arch/x86_64/idt.zig`) no longer issues the EOI
|
||||
itself. It used to, before running the handler — correct for the LAPIC timer, and
|
||||
impossible for a routed device line. Each handler now owns its EOI, because only the
|
||||
handler knows which discipline its source needs.
|
||||
|
||||
### The driver side is an event loop, not a callback
|
||||
|
||||
`IPC_ReplyWait` returns *either* a client request *or* a notification, told apart by
|
||||
the top bit of the badge (`ipc_sync.notify_badge_bit`). So a driver is one
|
||||
single-threaded loop over both of its event sources:
|
||||
|
||||
```zig
|
||||
while (true) {
|
||||
const r = ipc.replyWait(endpoint, reply, &recv);
|
||||
if (r.isNotification()) { // r.source() is the GSI
|
||||
service_device(); // clear the status register
|
||||
_ = dev.irqAck(id, irq_res); // re-arm
|
||||
} else {
|
||||
handle_client_request(recv[0..r.len]);
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
No reentrancy, no "what am I allowed to call from an interrupt handler", no shared
|
||||
state between ISR and task context. The interrupt is just a message.
|
||||
|
||||
Two properties worth knowing:
|
||||
|
||||
- **An interrupt taken while you're elsewhere is not lost.** If the driver is off in
|
||||
an `ipc_call` to another server when the IRQ fires, `wakeLocked` finds nobody
|
||||
waiting, but the badge is already on the endpoint's notify ring. The next
|
||||
`replyWait` pops it (`ipc_sync.replyWait` checks `popNotify` before the sender FIFO).
|
||||
- **Notifications coalesce, they don't count.** The ring is 8 deep and drops on
|
||||
overflow. That's correct: an IRQ notification is a *level* ("the device wants
|
||||
attention"), not a tally. Re-read the device's status register; never assume one
|
||||
notification means exactly one event. Because the ISR masks the line until you
|
||||
`irq_ack`, at most one badge per GSI can be outstanding — so the ring can only
|
||||
overflow if you bind more than eight GSIs to a single endpoint. Don't.
|
||||
|
||||
## A whole driver
|
||||
|
||||
`system/drivers/hpet/hpet.zig` is ~150 lines and does all of it. The shape:
|
||||
|
||||
```zig
|
||||
const hpet = findHpet(buf) orelse return; // device_enumerate, look for
|
||||
// class=timer with memory + irq
|
||||
_ = dev.claim(hpet.dev_id); // the capability
|
||||
const base = dev.mmioMap(hpet.dev_id, hpet.mmio).?;
|
||||
const endpoint = ipc.createIpcEndpoint().?;
|
||||
|
||||
// program the hardware over the mapping we were just handed
|
||||
reg(base, 0x100).* = level | int_enb | (hpet.gsi << 9); // timer 0 config
|
||||
reg(base, 0x108).* = reg(base, 0xF0).* + period; // comparator
|
||||
reg(base, 0x010).* |= 1; // ENABLE
|
||||
|
||||
_ = dev.irqBind(hpet.dev_id, hpet.irq, endpoint);
|
||||
|
||||
while (...) {
|
||||
const r = ipc.replyWait(endpoint, &.{}, &recv); // blocked. not polling.
|
||||
if (r.badge & notify_bit == 0) continue;
|
||||
reg(base, 0x020).* = 1; // clear status -> deassert
|
||||
reg(base, 0x108).* = reg(base, 0xF0).* + period; // re-arm
|
||||
_ = dev.irqAck(hpet.dev_id, hpet.irq); // unmask
|
||||
}
|
||||
```
|
||||
|
||||
The HPET is a good first driver for a reason that isn't obvious. Its *counter* is a
|
||||
clocksource — the only way to use it is to read it, so it proved `mmio_map` without
|
||||
needing interrupts at all. Its *comparators* are a clockevent, and can be configured
|
||||
**level-triggered** (`Tn_INT_TYPE_CNF`), which asserts a bit in `GENERAL_INT_STATUS`
|
||||
that the driver must write-1-to-clear. That's a genuine deassert step, so the full
|
||||
mask/ack cycle above is exercised for real rather than being decoration on an
|
||||
edge-triggered line that would have been fine without it.
|
||||
|
||||
One wrinkle it also demonstrates: the ACPI HPET table carries **no interrupt number**.
|
||||
Which I/O APIC inputs a comparator may drive is a bitmask in `Tn_INT_ROUTE_CAP`, in
|
||||
the device's own registers. So discovery (`acpi.parseHpet`) maps the block, reads the
|
||||
mask, and records one concrete GSI as an `irq` resource. The driver then programs
|
||||
`Tn_INT_ROUTE_CNF` to raise exactly that line — and the kernel will only bind the one
|
||||
it recorded. Hardware that describes itself at runtime still has to fit through a
|
||||
static capability.
|
||||
|
||||
## Publishing children: `device_register`
|
||||
|
||||
A device that *contains other devices* — a PCI bridge, a USB hub, or the HPET's block
|
||||
of comparators — needs a driver that enumerates it and tells the kernel what it found.
|
||||
That's `device_register`, and it makes the device table a tree rather than a list
|
||||
(`DeviceDesc.parent`).
|
||||
|
||||
```zig
|
||||
var child = std.mem.zeroes(dev.DeviceDesc);
|
||||
child.class = @intFromEnum(dev.DeviceClass.timer);
|
||||
child.resource_count = 1;
|
||||
child.resources[0] = .{ .kind = memory, .start = bus_base + 0x100, .len = 0x20 };
|
||||
const child_id = dev.register(bus_id, &child).?;
|
||||
```
|
||||
|
||||
The child is left **unclaimed**, which is the whole point: another process claims it and
|
||||
`mmio_map`s it, and sees only that 0x20-byte window.
|
||||
|
||||
The rule the kernel enforces is **containment**: every resource of a child must lie
|
||||
inside a resource of the same kind on its parent. Ranges must nest; an IRQ must match
|
||||
exactly. This isn't bureaucracy — a `DeviceDesc` is a licence to map physical memory, so
|
||||
without containment `device_register` would be a syscall for mapping any page you like. A
|
||||
bus driver may only ever subdivide what it already owns.
|
||||
|
||||
A device with **no resources** is legal and common. A USB device is reached through its
|
||||
controller, not by MMIO, so it gets `resource_count = 0`.
|
||||
|
||||
See [`system/drivers/bus/bus.zig`](../system/drivers/bus/bus.zig) for a complete one, and
|
||||
[driver-model.md](driver-model.md) for how bus drivers, class drivers and host
|
||||
controller drivers fit together.
|
||||
|
||||
## What the kernel does not do for you
|
||||
|
||||
- **It does not quiet your device.** That's the whole reason `irq_ack` exists.
|
||||
- **It does not know your registers.** `mmio_map` hands you a base address; every
|
||||
offset in this document came from the HPET spec, not from danos.
|
||||
- **It does not serialise your driver.** Two clients calling one driver endpoint are
|
||||
serialised by `replyWait`, but nothing stops your driver from being preempted.
|
||||
|
||||
## Limits, today
|
||||
|
||||
Worth knowing before you write the second driver:
|
||||
|
||||
Several things this list used to warn about are now available (see
|
||||
[driver-model.md](driver-model.md)): **port I/O** (`io_read`/`io_write`, claim-gated by
|
||||
the device's `io_port` resource — direct ring-3 `in`/`out` is still a #GP, so a PS/2 or
|
||||
16550 driver goes through these), **DMA memory** (`dma_alloc`: contiguous, pinned,
|
||||
uncacheable, physical address exposed), and **memory barriers** (`/lib/mmio`'s
|
||||
`mb`/`rmb`/`wmb`). What remains:
|
||||
|
||||
- **Page granularity.** `mmio_map` rounds to 4 KiB. Two devices sharing a page means
|
||||
granting one grants the other. A `device_register`ed child's *resource* can be narrower
|
||||
than a page, but its *mapping* can't.
|
||||
- **DMA is not contained.** A driver that can program a bus-mastering device can make
|
||||
that device write to *any* physical address — page tables don't sit between a device
|
||||
and RAM; an IOMMU does. The IOMMU is now *detected* (M16), but no translation domains
|
||||
are programmed, so `device_claim` on a DMA-capable device is still effectively
|
||||
equivalent to granting ring 0. This is the largest gap between the design's promise and
|
||||
what it delivers; enforcement lands with the first DMA driver.
|
||||
- **No `dev_release`.** A claim is never dropped (only IRQ/MSI bindings are, on exit), so
|
||||
a device stays owned for the life of its driver — which blocks restart.
|
||||
- **One endpoint per GSI**, so shared legacy PCI INTx lines can't be split between two
|
||||
drivers. MSI/MSI-X — one vector per device, edge-triggered, unshared — is the real
|
||||
answer, and QEMU's HPET doesn't offer it (`Tn_FSB_INT_DEL_CAP = 0`).
|
||||
- **Polarity is hardcoded** active-high in `irq.bind`. A device whose MADT override
|
||||
says active-low needs that threaded through from discovery.
|
||||
- **14 device vectors** (33–46) and **24 GSIs**, bounded by the stubs `isr.s` emits and
|
||||
by a single I/O APIC.
|
||||
- **Don't bind more than 8 GSIs to one endpoint.** The notify ring is 8 deep and drops
|
||||
on overflow. With one GSI per endpoint that's unreachable — the line is masked from
|
||||
the ISR until `irq_ack`, so at most one badge is ever outstanding. Bind nine devices
|
||||
to one endpoint, though, and a dropped badge leaves that line masked with nobody
|
||||
left to ack it.
|
||||
- **A faulting driver still kills the machine.** There is no per-process kill path: a
|
||||
ring-3 page fault halts the kernel, so `releaseIrqs` runs only on a voluntary
|
||||
`exit`. Fault isolation is the whole premise ([vision](vision.md)) and it is
|
||||
[not built yet](resilience.md).
|
||||
- **A dead driver's device is not reclaimed.** `releaseIrqs` unbinds and masks the
|
||||
line on exit, but the claim is never released — restart is
|
||||
[not built](resilience.md).
|
||||
- **On real hardware, the mask/EOI cycle may need a remote-IRR flush.** Masking a
|
||||
level-triggered redirection entry with remote-IRR set doesn't clear it on some
|
||||
chipsets, and the line never fires again. QEMU clears it on EOI regardless, so the
|
||||
tests can't see this. Linux flushes remote-IRR by toggling the entry to edge and
|
||||
back. See the note at the top of `system/kernel/irq.zig`.
|
||||
|
||||
## Verifying it
|
||||
|
||||
The `hpet` test spawns `hpet` from the initial ramdisk and watches the serial log. The driver
|
||||
prints `hpet: ok` only after being woken five times, and its loop's only exit is
|
||||
through `replyWait` returning a notification — it cannot reach that line by polling.
|
||||
|
||||
The last check doesn't trust the driver's self-report at all: the kernel reads the I/O
|
||||
APIC redirection entry back and asserts the line really is routed to a device vector,
|
||||
really is level-triggered, and really was left unmasked by the driver's final
|
||||
`irq_ack`.
|
||||
|
||||
```
|
||||
$ python3 test/qemu_test.py hpet irqfree iopass
|
||||
hpet ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||
irqfree ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||
iopass ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||
```
|
||||
|
||||
Two companions cover what `hpet` can't, because it never exits:
|
||||
|
||||
- **`irqfree`** — the teardown path. Binds two owners to one shared endpoint, releases
|
||||
one, and reads the I/O APIC back: the departing owner's line is masked, the sibling's
|
||||
is not. That second half is why bindings are keyed on the owning *task* and not on
|
||||
the endpoint pointer — endpoints are shared, so releasing "everything pointing at
|
||||
this endpoint" would silently mask a live driver's device.
|
||||
- **`iopass`** — the `device_grant` teardown rule, so destroying a driver's address
|
||||
space never returns MMIO frames to the RAM pool.
|
||||
|
||||
## What's next (not done here)
|
||||
|
||||
The big driver-model pieces — capability passing (class drivers), DMA + barriers, MSI,
|
||||
and IOMMU detection — are **now done** ([driver-model.md](driver-model.md), M13–M16), as
|
||||
is **port I/O** (`io_read`/`io_write`, the claim-gated syscalls that make a PS/2 or 16550
|
||||
driver possible). What's left is IOMMU *enforcement* (per-device domains — it waits on
|
||||
the first DMA driver to protect and test against) and these smaller items:
|
||||
|
||||
- **Releasing a claim.** There is no `dev_release`, and `devices_broker` never drops a claim on
|
||||
exit — only IRQ bindings are released. A dead driver's device stays owned forever,
|
||||
which blocks restart.
|
||||
- **Unregistering children.** `device_register` only appends. A USB device that is
|
||||
unplugged cannot be removed, and a bus driver in a loop can exhaust the 64-entry
|
||||
table.
|
||||
- **Restart.** A supervisor that *spawns* drivers now exists — the device-manager starts
|
||||
them with `system_spawn` — but a supervisor that *restarts* them does not. A driver that
|
||||
dies should release its claim, have its device quiesced, and be respawned; today nothing
|
||||
notices the death. Some pieces (`releaseIrqs`, `device_grant` teardown, the claim table)
|
||||
exist, and `dev_release` (below) is the missing mechanism; the restart policy is the
|
||||
resilience track ([resilience.md](resilience.md)).
|
||||
- **Interrupt priority / threaded IRQ latency.** `notifyFromIsr` enqueues the woken
|
||||
driver but doesn't preempt (`wakeLocked` deliberately leaves that to the caller), so
|
||||
a woken driver waits for the next scheduling point.
|
||||
|
||||
## The driver contract (M17–M18)
|
||||
|
||||
Claiming and mapping is half of being a danos driver; the other half is the
|
||||
**lifecycle and protocol contract**, and the runtime makes it nearly free:
|
||||
|
||||
- Build on `runtime.service.run` — one replyWait loop folding protocol
|
||||
requests, signals, and notifications into callbacks. The harness answers the
|
||||
universal zero-length ping and turns `terminate` into a clean exit for you
|
||||
([process-lifecycle.md](process-lifecycle.md)).
|
||||
- A driver spawned with an assignment (its device id as argv[1]) sends the
|
||||
versioned `hello` to the device manager inside the deadline, and a **bus**
|
||||
driver reports what it discovers with `child_added`
|
||||
([device-manager.md](device-manager.md); usb-xhci-bus is the reference
|
||||
implementation).
|
||||
- Crash freely — that is the design. The kernel releases your claims, IRQ
|
||||
bindings, and MSI vectors at death; the manager reads your exit reason,
|
||||
prunes what you reported, restarts you with backoff, and your fresh instance
|
||||
re-claims and re-reports. Never depend on your own cleanup running
|
||||
(iron rule 1).
|
||||
+25
-20
@@ -10,7 +10,7 @@ that hands us a working CPU, a memory map, and a screen, and then gets out of th
|
||||
way.
|
||||
|
||||
The key thing to understand: **UEFI is not our OS, it's a stepping stone.** It
|
||||
exists to load *us*. Our `src/boot/efi.zig` is a UEFI *application* — a normal program
|
||||
exists to load *us*. Our `boot/efi.zig` is a UEFI *application* — a normal program
|
||||
that the firmware runs — and its entire purpose is to gather what the kernel needs
|
||||
and then jump into the kernel.
|
||||
|
||||
@@ -20,14 +20,17 @@ UEFI boots by looking for a FAT-formatted partition called the **EFI System
|
||||
Partition (ESP)** and running a file at a well-known fallback path:
|
||||
|
||||
```
|
||||
esp/EFI/BOOT/BOOTX64.efi <- the "removable media" default for x86-64
|
||||
EFI/BOOT/BOOTX64.efi <- the "removable media" default for x86-64
|
||||
```
|
||||
|
||||
That's exactly the layout `build.zig` assembles. It builds `src/boot/efi.zig` for the
|
||||
`uefi` target, installs it to `esp/EFI/BOOT/BOOTX64.efi`, and drops the kernel ELF
|
||||
at `esp/kernel`. The `run-x86-64` step then points QEMU at OVMF (UEFI firmware for
|
||||
virtual machines) and presents that `esp/` directory to the guest as a FAT drive.
|
||||
The firmware finds `BOOTX64.efi` and runs it — that's our `main()`.
|
||||
The boot volume is the **FHS-shaped `zig-out`** itself (see the repository-layout note
|
||||
in [README.md](README.md)): `build.zig` installs `boot/efi.zig` (built for the `uefi`
|
||||
target) to `zig-out/EFI/BOOT/BOOTX64.efi` — the one path UEFI firmware fixes — and lays
|
||||
the rest out by FHS path: the kernel at `zig-out/system/kernel`, init at
|
||||
`zig-out/system/services/init`, the initial-ramdisk at `zig-out/boot/`. The
|
||||
`run-x86-64` step points QEMU at OVMF (UEFI firmware for virtual machines) and presents
|
||||
`zig-out` to the guest as a FAT drive. The firmware finds `BOOTX64.efi` and runs it —
|
||||
that's our `main()`, which then loads the kernel and init from their FHS paths.
|
||||
|
||||
## Boot services: the firmware's API
|
||||
|
||||
@@ -83,9 +86,9 @@ All of this *must* happen now, because after exit there's no GOP to ask. (See
|
||||
|
||||
- Use the **LoadedImage** protocol to discover which device we booted from, then
|
||||
**SimpleFileSystem** to open that volume.
|
||||
- Open the file named `danos`, seek to the end to learn its size, rewind, and read
|
||||
the whole ELF into a firmware-allocated pool buffer. (`read` may return short, so
|
||||
we loop.)
|
||||
- Open the kernel ELF at its FHS path (`system\kernel`), seek to the end to learn its
|
||||
size, rewind, and read the whole ELF into a firmware-allocated pool buffer. (`read`
|
||||
may return short, so we loop.)
|
||||
- Parse the ELF: validate the `\x7fELF` magic and the `x86_64` machine type, then
|
||||
walk the program headers. For every `PT_LOAD` segment we:
|
||||
- reserve the exact physical pages it's linked at (`p_paddr`) via
|
||||
@@ -121,7 +124,7 @@ entirely ours.
|
||||
### 4. Jump to the kernel
|
||||
|
||||
```zig
|
||||
const kernel: *const fn (*const BootInfo) callconv(danos.kernel_abi) noreturn =
|
||||
const kernel: *const fn (*const BootInfo) callconv(boot_handoff.kernel_abi) noreturn =
|
||||
@ptrFromInt(entry);
|
||||
kernel(&boot_info);
|
||||
```
|
||||
@@ -139,17 +142,19 @@ kernel is freestanding and uses the **SysV AMD64** convention (first argument in
|
||||
read garbage.
|
||||
|
||||
So both sides pin the convention explicitly to SysV via the shared
|
||||
`danos.kernel_abi` (defined in `src/root.zig`). The loader's function-pointer type
|
||||
and the kernel's `_start` both reference it, so the pointer lands in the register
|
||||
the kernel expects. This is the whole reason `kernel_abi` lives in the shared
|
||||
`danos` module: it's a contract both binaries must agree on. See
|
||||
`boot_handoff.kernel_abi` (defined in `system/boot-handoff.zig`). The loader's
|
||||
function-pointer type and the kernel's `_start` both reference it, so the pointer lands
|
||||
in the register the kernel expects. This is the whole reason `kernel_abi` lives in the
|
||||
shared `boot-handoff` module: it's a contract both binaries must agree on. See
|
||||
[sysv.md](sysv.md) for what "SysV" means and where else it shows up.
|
||||
|
||||
## The handoff contract
|
||||
|
||||
The loader and kernel are two *separate* binaries built for two different targets,
|
||||
so everything they exchange must have an identically-defined memory layout. That's
|
||||
what `src/root.zig` provides — imported by both as the `danos` module:
|
||||
what `system/boot-handoff.zig` provides — imported by both as the `boot-handoff` module.
|
||||
It is *only* the handoff: the kernel↔user ABI (`system/abi.zig`) and the device types
|
||||
(`system/devices/device-abi.zig`) are separate contracts the bootloader never sees.
|
||||
|
||||
- `BootInfo` — the top-level struct passed to the kernel (currently just the
|
||||
framebuffer; this is where future handoff data like the memory map will go).
|
||||
@@ -164,16 +169,16 @@ the loader writes are the bytes the kernel reads.
|
||||
```
|
||||
power on
|
||||
-> UEFI firmware initialises hardware
|
||||
-> finds esp/EFI/BOOT/BOOTX64.efi, runs it (our efi.zig main)
|
||||
-> finds EFI/BOOT/BOOTX64.efi on the FHS volume, runs it (our efi.zig main)
|
||||
-> grab boot services
|
||||
-> queryFramebuffer (via GOP: EDID native res, setMode, describe fb)
|
||||
-> loadKernel (read danos ELF, load PT_LOAD segments to 0x100000)
|
||||
-> loadKernel (read system/kernel ELF, load PT_LOAD segments to 0x100000)
|
||||
-> exitBootServices (retry until the memory-map key holds)
|
||||
-> jump to e_entry, boot_info pointer in RDI
|
||||
-> kernel _start (src/kernel/main.zig: framebuffer console, then halt)
|
||||
-> kernel _start (system/kernel/kernel.zig: framebuffer console, then halt)
|
||||
```
|
||||
|
||||
Bottom line: **UEFI's job is to give us a CPU, memory, and a framebuffer, then
|
||||
disappear.** `src/boot/efi.zig` is the thin bridge that collects those gifts into a
|
||||
disappear.** `boot/efi.zig` is the thin bridge that collects those gifts into a
|
||||
`BootInfo`, tears down the firmware, and jumps into the kernel — after which we're
|
||||
on our own.
|
||||
|
||||
@@ -3,12 +3,12 @@
|
||||
Once the kernel knows what RAM exists ([memory-map.md](memory-map.md)), it needs a
|
||||
way to *hand out* that RAM: give me a free page of physical memory, and later,
|
||||
here's one back. That's the **physical frame allocator** (a "physical memory
|
||||
manager", hence `src/kernel/pmm.zig`). It deals only in fixed 4 KiB **frames** — the
|
||||
manager", hence `system/kernel/pmm.zig`). It deals only in fixed 4 KiB **frames** — the
|
||||
natural unit because that's the granularity the CPU's paging hardware maps — and
|
||||
it is the primitive everything above it stands on: page tables, the kernel heap,
|
||||
per-process memory all ultimately ask the frame allocator for pages.
|
||||
|
||||
It's **generic kernel code**: it operates on the neutral `danos.MemoryRegion`
|
||||
It's **generic kernel code**: it operates on the neutral `system.MemoryRegion`
|
||||
array, so there's no UEFI in it and nothing architecture-specific beyond the 4 KiB
|
||||
page. (Contrast [arch.md](arch.md), which is where CPU-specific code lives.)
|
||||
|
||||
@@ -33,7 +33,7 @@ RAM is 32768 frames — a **4 KiB bitmap, a single frame**. Even 64 GiB needs on
|
||||
|
||||
## How it works
|
||||
|
||||
State lives in `src/kernel/pmm.zig`: the `bitmap` slice, `total_frames`, `used_frames`,
|
||||
State lives in `system/kernel/pmm.zig`: the `bitmap` slice, `total_frames`, `used_frames`,
|
||||
and a `next_hint` marking where the next allocation scan should start.
|
||||
|
||||
### init(map) — building it from the memory map
|
||||
|
||||
+2
-2
@@ -9,10 +9,10 @@ write a 32-bit value to the right address, and a pixel changes color. That's
|
||||
exactly what `Console.pixel` does:
|
||||
|
||||
```zig
|
||||
self.rowPtr(y)[x] = color; // src/kernel/console.zig
|
||||
self.rowPtr(y)[x] = color; // system/kernel/console.zig
|
||||
```
|
||||
|
||||
Our `Framebuffer` struct (`src/root.zig`) is the four facts you need to
|
||||
Our `Framebuffer` struct (`system/boot-handoff.zig`) is the four facts you need to
|
||||
address it:
|
||||
|
||||
| Field | Meaning |
|
||||
|
||||
+3
-3
@@ -16,7 +16,7 @@ safely, until the machine is reset or powered off.
|
||||
## The core of it: `hlt`
|
||||
|
||||
Everything comes down to one x86 instruction. It's CPU-specific, so it lives in
|
||||
the arch module, `src/kernel/arch/x86_64/cpu.zig` (see [arch.md](arch.md)), and the
|
||||
the arch module, `system/kernel/architecture/x86_64/cpu.zig` (see [arch.md](arch.md)), and the
|
||||
generic kernel calls it as `arch.halt()`:
|
||||
|
||||
```zig
|
||||
@@ -83,7 +83,7 @@ treats the call:
|
||||
signature for a kernel entry point — the bootloader jumps in and nothing ever
|
||||
jumps back out.
|
||||
|
||||
You can see the chain in `src/kernel/main.zig`: `_start` is `noreturn`, it calls
|
||||
You can see the chain in `system/kernel/kernel.zig`: `_start` is `noreturn`, it calls
|
||||
`kmain` which is `noreturn`, which ends by calling `arch.halt()` which is
|
||||
`noreturn`. The "never returns" property is threaded all the way down.
|
||||
|
||||
@@ -106,7 +106,7 @@ There are three halt sites, and they're all the same idea:
|
||||
`arch.halt()`. A panic is unrecoverable here, so stopping the machine — rather
|
||||
than limping on with corrupted state — is the safe response.
|
||||
|
||||
3. **Bootloader failure** — in `src/boot/efi.zig`, if `boot()` fails *before* handing
|
||||
3. **Bootloader failure** — in `boot/efi.zig`, if `boot()` fails *before* handing
|
||||
off to the kernel, `main` logs the error and parks the machine with the same
|
||||
loop so the message stays on screen:
|
||||
|
||||
|
||||
+1
-1
@@ -7,7 +7,7 @@ top of both to provide what the rest of the kernel actually wants: `alloc(n)` /
|
||||
the thing that unlocks dynamic data structures — lists, hash maps, driver state,
|
||||
eventually a process table.
|
||||
|
||||
It's generic kernel code (`src/kernel/heap.zig`): the allocator logic is
|
||||
It's generic kernel code (`system/kernel/heap.zig`): the allocator logic is
|
||||
architecture-neutral, using `arch.mapPage` and the frame allocator underneath.
|
||||
|
||||
## A growable free-list allocator
|
||||
|
||||
+161
@@ -0,0 +1,161 @@
|
||||
# The input module: broadcasting input events
|
||||
|
||||
A keyboard driver has one keystroke and *many* programs that might want it — a shell, a
|
||||
window server, a logger. None of them owns the hardware, and the driver should not know
|
||||
who is listening. So between the drivers and the listeners sits the **input service**
|
||||
(`system/services/input/`): drivers **publish** events to it, programs **subscribe**, and
|
||||
it fans each event out to every interested subscriber. It is an ordinary ring-3 process
|
||||
reached over IPC, like the [VFS server](../system/services/vfs/vfs.zig) — no kernel knows
|
||||
what a key is.
|
||||
|
||||
## One service, several device classes
|
||||
|
||||
The service carries three device classes today — **keyboard**, **mouse**, and
|
||||
**joystick/gamepad** — and is built to take more
|
||||
([protocol.zig](../system/services/input/protocol.zig)). Each class has its own typed
|
||||
event:
|
||||
|
||||
- `KeyEvent` — `key_down`/`key_up` (physical make/break) and `key_press` (a character was
|
||||
produced, carrying the Unicode scalar); plus a layout-independent `keycode` and a
|
||||
`modifiers` bitmask.
|
||||
- `MouseEvent` — relative `motion` (`dx`/`dy`), `button_down`/`button_up`, and `scroll`.
|
||||
- `JoystickEvent` — `axis` moves (a signed value on a `control` index) and
|
||||
`button_down`/`button_up`.
|
||||
|
||||
All three travel in one **`InputEvent` envelope** tagged with a `DeviceKind`, so the
|
||||
fan-out is a single code path and a subscriber can take a mix of classes on one stream.
|
||||
Decode an envelope with `asKeyboard()` / `asMouse()` / `asJoystick()` (each returns null
|
||||
unless the tag matches). A subscriber names the classes it wants with a **`device_mask`**,
|
||||
and the service routes each event only to subscribers whose mask includes its class — so a
|
||||
mouse-only listener never wakes for keystrokes.
|
||||
|
||||
## Why this needed a new kernel primitive
|
||||
|
||||
The interesting part is delivery, and it runs straight into the shape of danos IPC.
|
||||
[ipc.md](ipc.md) describes a **synchronous rendezvous**: a server holds exactly one
|
||||
pending reply (`Task.ipc_client`) and *must* answer it on its next `replyWait`. Two
|
||||
consequences decide the whole design:
|
||||
|
||||
1. **You cannot block N subscribers waiting for "the next event".** A server can hold only
|
||||
one caller at a time, so the natural "subscriber calls `next_event()` and blocks" API
|
||||
is impossible for more than one subscriber. Delivery therefore has to be **push** — the
|
||||
service reaching out to subscribers — not pull.
|
||||
|
||||
2. **A synchronous push can hang the whole service.** If the service delivered with
|
||||
`ipc_call`, it would block until each subscriber replied. `ipc_call` has no timeout, and
|
||||
the kernel does **not** wake a caller parked on a *dead* peer's endpoint (it only fails a
|
||||
peer that was mid-reply — see [process.zig](../system/kernel/process.zig)
|
||||
`releaseTaskResourcesLocked`). One subscriber that exits mid-delivery would wedge input
|
||||
for everyone. That is the opposite of the resilience the microkernel is for.
|
||||
|
||||
The fix is the asynchronous send that [ipc.md](ipc.md) had already earmarked as future
|
||||
work ("asynchronous / buffered send … for notifications between servers"):
|
||||
|
||||
```
|
||||
ipc_send(handle, message_ptr, message_len) -> 0 / -errno
|
||||
```
|
||||
|
||||
`ipc_send` copies a small payload into the endpoint's **bounded queue** and wakes a
|
||||
receiver, then returns immediately — it never blocks and so can never hang on a dead or
|
||||
slow subscriber. The receiver picks it up through the same `replyWait` it already runs:
|
||||
the wake arrives as a **buffered message** — `notify_badge_bit | notify_message_bit` set in
|
||||
the badge (distinguishing it from a bare IRQ/child-exit notification), the sender's task id
|
||||
in the low bits, and the payload in the receive buffer, with no reply owed. The queue holds
|
||||
16 messages per endpoint; a full queue **drops the oldest**, because a buffered message is
|
||||
discrete data, not a coalescing "level" like an interrupt. See
|
||||
[ipc-synchronous.zig](../system/kernel/ipc-synchronous.zig) (`sendLocked`, `popPost`, and
|
||||
the `replyWait` receive loop).
|
||||
|
||||
This is the async counterpart of `ipc_call`, and the input service is its first consumer.
|
||||
|
||||
## How the pieces fit
|
||||
|
||||
```
|
||||
keyboard/mouse driver, input-source input service subscriber(s)
|
||||
----------------------------------- ------------- -------------
|
||||
connectSource(); loop: replyWait: subscribeKeyboard()/…All:
|
||||
publishKeyboardEvent(k) ─ ipc_call ─▶ publish → broadcast: createIpcEndpoint()
|
||||
publishMouseEvent(m) for each sub whose callCap(subscribe,
|
||||
publishJoystickEvent(j) mask matches event.device: send_cap = ep,
|
||||
ipc_send(sub_ep) ──────▶ device_mask)
|
||||
reply ok loop: next()
|
||||
subscribe → store {ep cap, └─ replyWait(ep)
|
||||
task id, device_mask} → InputEvent
|
||||
```
|
||||
|
||||
- A **subscriber** calls `input.subscribe(mask)` — or a typed helper: `subscribeKeyboard()`,
|
||||
`subscribeMouse()`, `subscribeJoystick()` (one class, `next()` returns the decoded event),
|
||||
or `subscribeAll()` (every class, `next()` returns a tagged `InputEvent`)
|
||||
([library/runtime/input.zig](../library/runtime/input.zig)). It creates its own endpoint
|
||||
and hands it to the service as a **capability** (M13 capability passing — the input
|
||||
service is that feature's first real user), along with its `device_mask`. Then it loops on
|
||||
`next()`, a `replyWait` on that endpoint returning each pushed event.
|
||||
- A **source** (a keyboard, mouse, or joystick driver) calls `input.connectSource()` and the
|
||||
method for its class: `publishKeyboardEvent`, `publishMouseEvent`, or
|
||||
`publishJoystickEvent`. Publishing is a short synchronous `ipc_call` the service answers at
|
||||
once; the service's own fan-out is asynchronous, so publishing never blocks on a slow
|
||||
subscriber.
|
||||
- The **service** ([input.zig](../system/services/input/input.zig)) keeps a small subscriber
|
||||
table (endpoint handle + owning task id + `device_mask`). On `publish` it `ipc_send`s the
|
||||
event to every subscriber whose mask includes the event's device class. On `subscribe` it
|
||||
stores the passed capability and mask and, as housekeeping, prunes any slot whose owning
|
||||
process has exited (checked against `process_enumerate`) — not for correctness (an async
|
||||
send to an orphaned endpoint is harmless) but to reclaim the slot.
|
||||
|
||||
Publisher and subscriber must be **separate processes**: a single thread that both
|
||||
published and serviced its own subscription would deadlock (its `publish` call blocks until
|
||||
the service delivers to its endpoint, which only the same thread could receive).
|
||||
|
||||
## Status and follow-ups
|
||||
|
||||
- **The keyboard is real.** The `ps2-bus` driver owns PNP0303, which carries *both* the
|
||||
0x60/0x64 ports and IRQ1, so reading the hardware lives in the bus, not in
|
||||
[keyboard.zig](../system/drivers/ps2-bus/keyboard.zig): the bus binds IRQ1 and, on each
|
||||
interrupt, drains port 0x60, routing every byte by the status register's
|
||||
auxiliary-output bit to whichever child driver **attached** for that device (an
|
||||
`AttachRequest` to the well-known `ps2_bus` service, carrying the child's endpoint as a
|
||||
capability; the bytes then arrive as asynchronous `ForwardedByte` messages, so the IRQ
|
||||
path never blocks on a child). The keyboard driver decodes the stream — scancode **set 2**,
|
||||
what the keyboard sends with the 8042's legacy translation off, decoded by
|
||||
[scancode.zig](../system/drivers/ps2-bus/scancode.zig) into USB HID usage keycodes with
|
||||
make/break, typematic-repeat, and modifier tracking (host-tested under `zig build test`) —
|
||||
and publishes real `key_down`/`key_press`/`key_up` events.
|
||||
- **Keycode → character** is wired in: the keyboard driver fills a `key_press` event's
|
||||
`character` through [`library/xkeyboard-config`](../library/xkeyboard-config/README.md)
|
||||
(`xkb.map(layout, keycode, mods)` → keysym + Unicode character), synthesizing the ASCII
|
||||
control characters for Enter/Tab/Backspace/Escape, whose keysyms map to no Unicode. The
|
||||
layout defaults to `us`; the bus can pass another as the driver's argv[2] — the seam for
|
||||
a future settings source.
|
||||
- **The mouse is real too.** IRQ12 is enumerated on the auxiliary device's own ACPI node
|
||||
(PNP0F13), so the bus claims that node alongside the controller and routes both IRQs to
|
||||
its one endpoint, acking whichever line the notification's badge names.
|
||||
[mouse.zig](../system/drivers/ps2-bus/mouse.zig) attaches the way the keyboard does and
|
||||
assembles the forwarded bytes with
|
||||
[mouse-packet.zig](../system/drivers/ps2-bus/mouse-packet.zig) (three-byte stream-mode
|
||||
packets: sync/overflow handling, nine-bit movement, screen-convention `dy` — host-tested
|
||||
under `zig build test`) into `button_down`/`button_up` transitions and `motion` events.
|
||||
**Follow-up:** the IntelliMouse magic-knock for a scroll wheel (four-byte packets) and
|
||||
`scroll` events. The hardware-free `input-source` still rotates through all three classes
|
||||
synthetically (including a joystick, which has no driver yet) via the
|
||||
`input.synthetic*Event` helpers.
|
||||
- **Drop-oldest under overflow** is a defined loss; the 16-slot ring absorbs normal bursts.
|
||||
Real backpressure/flow-control is future work.
|
||||
- **`publish` is unauthenticated** — any process may publish, consistent with the current
|
||||
bring-up trust model (see [driver-model.md](driver-model.md)). A source capability is
|
||||
future work.
|
||||
|
||||
## Verifying it
|
||||
|
||||
The `input` case (`python3 test/qemu_test.py input`, in
|
||||
[tests.zig](../system/kernel/tests.zig) `inputTest`) boots the real kernel and spawns the
|
||||
service, the synthetic source (which cycles keyboard, mouse, and joystick events), and a
|
||||
subscriber that took all three classes. It passes only when the subscriber heartbeats
|
||||
`input-test: ok` — proof that an event travelled source → service → subscriber over IPC,
|
||||
exercising `ipc_send`, capability-passing subscription, and per-device routing. Each
|
||||
serial line names the class received, so the log shows all three arriving on one stream.
|
||||
|
||||
## See also
|
||||
|
||||
- [ipc.md](ipc.md) — the synchronous rendezvous and the notification path `ipc_send` extends.
|
||||
- [syscall.md](syscall.md) — the system-call surface, including `ipc_send`.
|
||||
- [driver-model.md](driver-model.md) — class drivers, capability passing (M13), the trust model.
|
||||
+20
-9
@@ -9,7 +9,7 @@ reboot is miserable.
|
||||
|
||||
This is the machinery that catches those faults and prints what happened instead.
|
||||
It's all x86_64-specific, so it lives behind the [arch](arch.md) boundary in
|
||||
`src/kernel/arch/x86_64/`. Only the 32 CPU-defined exception vectors are wired up so far;
|
||||
`system/kernel/architecture/x86_64/`. Only the 32 CPU-defined exception vectors are wired up so far;
|
||||
device interrupts (timer, keyboard, via the APIC) come later, on the same IDT.
|
||||
|
||||
## First the GDT
|
||||
@@ -20,7 +20,7 @@ IDT gate names a code-segment *selector* that must resolve in the current GDT. T
|
||||
firmware left a GDT in place, but we don't control it, so we install our own with
|
||||
known selectors: `0x08` kernel code, `0x10` kernel data.
|
||||
|
||||
`src/kernel/arch/x86_64/gdt.zig` holds three flat descriptors — a required null entry,
|
||||
`system/kernel/architecture/x86_64/gdt.zig` holds three flat descriptors — a required null entry,
|
||||
plus code and data — where the only bits that matter in long mode are the access
|
||||
byte and the code segment's long-mode (`L`) flag. Loading it (`gdt_flush` in
|
||||
`isr.s`) does two things: `lgdt`, then reload the segment registers. The data
|
||||
@@ -33,7 +33,7 @@ into CS:RIP.
|
||||
The **Interrupt Descriptor Table** maps each of 256 vectors to a handler. Each
|
||||
entry is a 16-byte *gate* holding the handler's address (split across three
|
||||
fields, a quirk of the format), the code selector (`0x08`), and flags: `0x8E`
|
||||
means present, ring 0, 64-bit interrupt gate. `src/kernel/arch/x86_64/idt.zig` builds the
|
||||
means present, ring 0, 64-bit interrupt gate. `system/kernel/architecture/x86_64/idt.zig` builds the
|
||||
table, points the first 32 vectors at their stubs, and loads it with `lidt`
|
||||
(`idt_flush`).
|
||||
|
||||
@@ -49,7 +49,7 @@ hit a fault *while trying to deliver another fault* — very often because the
|
||||
current stack pointer is bad, so pushing the exception frame itself faulted. If
|
||||
the #DF handler then tried to push onto that same bad stack, it would fault a
|
||||
third time and **triple-fault** — an instant reset. So the #DF gate is pointed at
|
||||
**IST1**, a small dedicated stack (`src/kernel/arch/x86_64/tss.zig`) that's always valid.
|
||||
**IST1**, a small dedicated stack (`system/kernel/architecture/x86_64/tss.zig`) that's always valid.
|
||||
|
||||
Bringing it up: fill in the TSS's IST1 pointer, publish the TSS through a
|
||||
descriptor in the GDT (`gdt.setTss`), and load it into the task register with
|
||||
@@ -60,7 +60,7 @@ which is why the GDT grew from three entries to five.
|
||||
|
||||
On an exception the CPU pushes a small frame (SS, RSP, RFLAGS, CS, RIP) and, for
|
||||
*some* vectors, an **error code**. That inconsistency is a nuisance, so each stub
|
||||
in `src/kernel/arch/x86_64/isr.s` normalises it: vectors that don't get a hardware error
|
||||
in `system/kernel/architecture/x86_64/isr.s` normalises it: vectors that don't get a hardware error
|
||||
code push a dummy `0`, then every stub pushes its **vector number** and jumps to a
|
||||
shared tail, `isr_common`. The tail pushes all the general registers and calls the
|
||||
Zig handler with a pointer to the whole thing.
|
||||
@@ -80,11 +80,22 @@ inline). `build.zig` adds `isr.s` to the arch module.
|
||||
## Reporting a fault
|
||||
|
||||
`isr_common` calls `exceptionHandler`, which forwards to a swappable `on_fault`
|
||||
hook. The generic kernel installs a reporter (`onException` in `main.zig`) that
|
||||
prints, in red, the exception name and vector, the error code, the faulting RIP
|
||||
hook. The generic kernel installs a reporter (`onException` in `kernel.zig`) that
|
||||
prints the exception name and vector, the error code, the faulting RIP
|
||||
and RSP, and — for a page fault (#PF, vector 14) — the faulting address from
|
||||
**CR2**. Then it halts. There's no fault *recovery* yet, so every exception is
|
||||
terminal; the point is that it's now **visible** instead of a silent reset.
|
||||
**CR2**. What happens next depends on where the fault came from:
|
||||
|
||||
- **User mode (CPL 3): kill the process, keep the machine.** The kernel is intact
|
||||
(the CPU trapped onto the task's kernel stack), so the faulting process is
|
||||
killed — address space, IRQ bindings, and IPC handles reclaimed; a client it
|
||||
owed a reply to is failed with `-EPEER` — and the core reschedules. A crashing
|
||||
driver takes itself down, never the OS. This is fault recovery step 2 of
|
||||
[resilience.md](resilience.md). NMI, double fault, and machine check are
|
||||
excluded: they report machine trouble regardless of what was running.
|
||||
- **Kernel mode: halt this core.** The trusted base itself is broken, so there is
|
||||
nothing safe to kill; the fault is still *contained* to the core (an
|
||||
application-processor fault leaves the rest of the system running), and the
|
||||
report makes it **visible** instead of a silent reset.
|
||||
|
||||
The hook is set before `arch.init()` in `kmain`, so a fault during setup is still
|
||||
caught.
|
||||
|
||||
+79
-12
@@ -6,12 +6,20 @@ just call each other — a request becomes a **message**. In a microkernel, what
|
||||
was a function call across a monolithic kernel is IPC, so it's a first-class
|
||||
concern, not an afterthought.
|
||||
|
||||
This first form is a **bounded blocking channel** (`src/kernel/ipc.zig`): a fixed-size
|
||||
ring buffer of messages with a producer/consumer rendezvous, built on the
|
||||
scheduler's [wait queues](scheduling.md).
|
||||
There are two layers, built a milestone apart:
|
||||
|
||||
- **`system/kernel/ipc.zig`** — a bounded blocking channel between *kernel threads*,
|
||||
described below. The primitive, and where the blocking discipline was worked out.
|
||||
- **`system/kernel/ipc-synchronous.zig`** — synchronous call/reply between *processes*, across
|
||||
address spaces. What user-space servers and drivers actually talk over. It's the
|
||||
second half of this document.
|
||||
|
||||
## The channel
|
||||
|
||||
The first form is a **bounded blocking channel** (`system/kernel/ipc.zig`): a fixed-size
|
||||
ring buffer of messages with a producer/consumer rendezvous, built on the
|
||||
scheduler's [wait queues](scheduling.md).
|
||||
|
||||
`Channel(T, capacity)` is generic over the message type and buffer size. It holds a
|
||||
ring buffer, a count, and two wait queues:
|
||||
|
||||
@@ -42,16 +50,75 @@ full and empty over and over, so both the blocking-send and blocking-recv paths
|
||||
exercised heavily. The messages arrive intact and in order (their sum is the
|
||||
expected `5050`), and neither task busy-waits — they block and wake each other.
|
||||
|
||||
## Endpoints: call/reply across address spaces
|
||||
|
||||
A channel connects two kernel threads sharing one address space. Real servers are
|
||||
*processes*, so the payload has to cross an address-space boundary. That's
|
||||
`system/kernel/ipc-synchronous.zig`, and its shape is L4's: a synchronous **rendezvous** at an
|
||||
`Endpoint`, with the message copied directly from the sender's pages to the receiver's
|
||||
(`copyAcross` walks both sets of page tables through the physmap — no CR3 switch, no
|
||||
bounce buffer).
|
||||
|
||||
Two syscalls carry it:
|
||||
|
||||
- **`ipc_call(h, msg, reply)`** — copy `msg` to the server, block until it replies.
|
||||
- **`ipc_reply_wait(h, reply, recv)`** — reply to the client you're still holding (if
|
||||
any), then block for the next request. One syscall, because a server's steady state
|
||||
is *always* "finish the last one, wait for the next".
|
||||
|
||||
An endpoint is reached by **handle** — a small integer index into the process's handle
|
||||
table (`Task.handles`), exactly like a file descriptor, and just as unforgeable. The
|
||||
bootstrap problem (how do you get the first handle?) is solved by a tiny name registry:
|
||||
a server calls `ipc_register(service_id, h)` under a well-known small integer, and a
|
||||
client calls `ipc_lookup(service_id)`.
|
||||
|
||||
The server never learns the client's identity beyond a **badge**, delivered alongside
|
||||
the message: the caller's task id.
|
||||
|
||||
### Interrupts are messages too
|
||||
|
||||
`notifyFromIsr` posts an *asynchronous* notification to an endpoint — no payload, no
|
||||
reply owed — and wakes whoever is blocked in `reply_wait`. Its badge has the top bit
|
||||
set (`notify_badge_bit`), which is how a driver's single event loop distinguishes "a
|
||||
client wants something" from "the hardware wants something". Notifications sit in a
|
||||
small coalescing ring on the endpoint, so an interrupt taken while the driver was busy
|
||||
elsewhere is not lost.
|
||||
|
||||
This is what makes a user-space driver possible at all, and it's the subject of
|
||||
[drivers.md](drivers.md).
|
||||
|
||||
## What's next (not done here)
|
||||
|
||||
- **Across address spaces.** Today both endpoints are kernel threads sharing the
|
||||
kernel's memory, so the message is copied within one address space. When user
|
||||
mode arrives, the same channel carries messages between *isolated* processes,
|
||||
copying the payload across the boundary — which is where IPC earns its place as
|
||||
the microkernel's backbone.
|
||||
- **Synchronous call/reply.** A request/response pattern (send-and-wait-for-reply)
|
||||
on top of channels, the shape most driver/service calls take.
|
||||
- **Interrupts as messages.** A hardware interrupt delivered to the driver task
|
||||
that owns the device, as an IPC message.
|
||||
- **Priority inheritance** through IPC, so a high-priority client blocked on a
|
||||
low-priority server doesn't suffer unbounded priority inversion.
|
||||
- **Handle transfer.** A server can't hand a client a handle to a third endpoint, so
|
||||
every capability is either well-known (the registry) or inherited — there's no way
|
||||
to delegate one.
|
||||
- **Asynchronous / buffered send** for the cases where a rendezvous is the wrong
|
||||
shape (logging, notifications between servers). *Landed as `ipc_send`* — a
|
||||
non-blocking post to an endpoint's bounded payload queue, delivered through
|
||||
`reply_wait` as a buffered message (badge bit `notify_message_bit`). Built for, and
|
||||
first used by, the [input service](input.md)'s keyboard-event broadcast, where a
|
||||
synchronous push would let one dead subscriber hang the fan-out. A full queue drops
|
||||
the oldest (discrete messages, not a coalescing level like the notification ring).
|
||||
- **A bounded reply.** `MSG_MAX` is 256 bytes and the copy runs under the big kernel
|
||||
lock; a bulk transfer wants shared pages, not a copy.
|
||||
|
||||
## Lifecycle conventions over IPC (M17)
|
||||
|
||||
Three conventions from [process-lifecycle.md](process-lifecycle.md) ride the
|
||||
notification mechanism:
|
||||
|
||||
- **Signals** arrive as notifications on the endpoint a process nominated with
|
||||
`signal_bind` (`runtime.process.bindSignals`): badge = the signal bit plus the
|
||||
coalesced pending mask (`runtime.process.signalsFrom` decodes). Statements,
|
||||
never questions; no payload, no reply.
|
||||
- **One-shot timers** (`timer_bind`, `runtime.system.timerOnce`) land as a
|
||||
timer-bit notification — the timed wait: a service arms a deadline and keeps
|
||||
serving, instead of blocking in sleep.
|
||||
- **The universal ping**: a **zero-length request is the liveness probe**,
|
||||
answered with a zero-length reply by the service harness itself
|
||||
(`runtime.service.run`). No protocol's requests start at length zero, so the
|
||||
encoding cannot collide, and a wedged service simply fails to answer — which
|
||||
is the diagnosis. Deep health ("can I reach my hardware?") stays a per-service
|
||||
protocol message.
|
||||
|
||||
+104
@@ -0,0 +1,104 @@
|
||||
# Logging: the diagnostic log vs. the display
|
||||
|
||||
danos separates two things that are easy to conflate: the **diagnostic log** — the
|
||||
machine-readable stream of *what the kernel is doing* — and the **display**, the
|
||||
framebuffer surface the OS draws on. They are different concerns with different
|
||||
lifetimes, so they're different code paths.
|
||||
|
||||
The guiding rule: **output is a diagnostic convenience, never a correctness
|
||||
dependency.** The kernel must boot and run correctly with *zero* output channels —
|
||||
no serial, no screen. Logging that can take the kernel down isn't robust; it's a
|
||||
liability. This is the same [resilience](resilience.md) posture the rest of the
|
||||
kernel follows.
|
||||
|
||||
## The log is multi-sink
|
||||
|
||||
`system/kernel/log.zig` is the diagnostic log. It fans a message out to a set of
|
||||
registered **sinks**, each best-effort and self-guarding:
|
||||
|
||||
```zig
|
||||
log.addSink(arch.serialWrite); // the serial UART
|
||||
if (arch.debugconPresent()) log.addSink(arch.debugconWrite); // 0xE9 debug console
|
||||
// later: log.addSink(fs.logWrite); // a file on a ramdisk / USB / SSD
|
||||
log.write("…"); log.print("x={d}\n", .{x});
|
||||
```
|
||||
|
||||
Properties that matter:
|
||||
|
||||
- **No allocation.** The sink table is a fixed array, so the log works before the
|
||||
heap is up and inside a panic.
|
||||
- **Best-effort.** A sink whose device is absent is a no-op (e.g. writing to a
|
||||
missing UART just goes nowhere — the TX-wait is bounded so it can't hang). A
|
||||
message reaches whatever channels exist; if none do, the kernel runs on, silent.
|
||||
- **Order-independent.** Every registered sink gets every message. Adding the file
|
||||
logger later is one `addSink` call and **zero** changes to call sites.
|
||||
|
||||
## The framebuffer is *not* a log sink
|
||||
|
||||
The framebuffer is a general graphics surface, **not inherently a text terminal**.
|
||||
Today `system/kernel/console.zig` paints a text grid on it as a *bootstrap* console, but
|
||||
that's a stop-gap: once the driver machinery exists the framebuffer becomes a proper
|
||||
**graphics device driver**, and the text crutch goes away. So the log must not assume
|
||||
it — routing the verbose log through a pixel console would bake in "the OS is text".
|
||||
|
||||
Instead the two paths are explicit:
|
||||
|
||||
```
|
||||
verbose diagnostics ──► log ──► serial, debugcon, (file later)
|
||||
user status / panics ──► status() ──► log (above) + framebuffer (if present)
|
||||
```
|
||||
|
||||
A handful of user-facing lines (`kernel initialised`, a panic) go through
|
||||
`main.zig`'s `status()` / `statusPrint()`, which write to the log **and** paint the
|
||||
framebuffer when one is present. Everything else uses `log.*` and never touches the
|
||||
screen. `console.write` is a no-op when the firmware gave us no framebuffer.
|
||||
|
||||
## Optional framebuffer (headless machines)
|
||||
|
||||
A framebuffer is not guaranteed — a headless server exposes no UEFI Graphics Output
|
||||
Protocol. That used to be *fatal* (the loader failed the boot). Now the loader hands
|
||||
over a "no framebuffer" descriptor (`base == 0`) rather than failing, and
|
||||
`Framebuffer.present()` (in `system/boot-handoff.zig`) gates every on-screen path. A headless,
|
||||
serial-less machine boots and runs correctly — it just goes quiet.
|
||||
|
||||
## Last-resort channels (no text output at all)
|
||||
|
||||
Two signals bypass the sink list, because they must survive even a total
|
||||
output-channel failure:
|
||||
|
||||
- **`log.checkpoint(code)`** — a one-byte **POST code** to I/O port `0x80` (a POST
|
||||
card or BMC shows it). `main.zig` emits one at each boot milestone (`cp_paging`,
|
||||
`cp_heap`, …) and on a fault/panic, so "where did it hang?" is answerable with no
|
||||
text output whatsoever. Writing `0x80` is universally safe.
|
||||
- **`log.recordPanic(msg)`** — stamps the panic message into a fixed record
|
||||
(`log.panic_record`, with a `magic` written last). A post-mortem — an attached
|
||||
debugger, a RAM dump, or a future file/pstore reader — recovers *what killed it*
|
||||
even though nothing was on screen.
|
||||
|
||||
The panic and CPU-exception handlers fan out to every sink, emit a POST code, and
|
||||
drop the breadcrumb — they never assume a console.
|
||||
|
||||
## The 0xE9 debug console
|
||||
|
||||
Port `0xE9` is the Bochs/QEMU debug console. It's detected safely: the port returns
|
||||
`0xE9` when read if present, and `0xFF` on real hardware, so `debugconPresent()`
|
||||
only enables the sink when it's really there. Under QEMU it's captured with
|
||||
`-debugcon file:…`, giving CI a log channel independent of `-serial`.
|
||||
|
||||
## The robustness spectrum
|
||||
|
||||
The result handles every combination — framebuffer only, serial only, both, or
|
||||
**neither**. With no channels at all the kernel still boots and runs; port-`0x80`
|
||||
checkpoints track progress and the panic breadcrumb captures failures. *Runs blind
|
||||
but correct* is the goal, not *always has output*.
|
||||
|
||||
## Related
|
||||
|
||||
- [framebuffer.md](framebuffer.md) — the display surface itself (pitch, format), the
|
||||
thing that becomes a graphics device driver.
|
||||
- [efi.md](efi.md) — where the loader captures (or, headless, doesn't capture) the
|
||||
framebuffer before `ExitBootServices`.
|
||||
- [device-interrupts.md](device-interrupts.md) — the serial UART bring-up the log's
|
||||
primary sink rides on.
|
||||
- [resilience.md](resilience.md) — why "never let a missing peripheral take the
|
||||
kernel down" is a core design stance.
|
||||
@@ -0,0 +1,210 @@
|
||||
# M17–M18 execution plan: process lifecycle + device manager
|
||||
|
||||
**Archived — completed 2026-07-13** (every item checked; suite ended 54/54).
|
||||
Kept as the record of how M17–M18 landed; the successor is
|
||||
[m19-m20-plan.md](m19-m20-plan.md).
|
||||
|
||||
The operational plan for building [process-lifecycle.md](process-lifecycle.md)
|
||||
(M17) and [device-manager.md](device-manager.md) increments 5–7 (M18). Design is
|
||||
settled in those documents; this file is the build order — one phase at a time,
|
||||
each phase green before the next starts. Delete or archive this file when M18
|
||||
lands.
|
||||
|
||||
**Definition of green, every phase:** `zig build` clean, `zig build test` clean,
|
||||
`python3 test/qemu_test.py` passes (existing scenarios plus the phase's new one),
|
||||
and the relevant design doc's "known gaps" / status lines updated. Commit per
|
||||
green phase (no co-author trailers).
|
||||
|
||||
**Workflow (settled 2026-07-12):** work happens in a dedicated git worktree, on
|
||||
feature branches cut from `main` — `feat/process-lifecycle` (M17.1–17.4),
|
||||
`feat/device-manager` (M18.1), `feat/usb-xhci-bus` (M18.2–18.3). When a branch's
|
||||
phases are all green it is **auto-merged into `main`**; branches are kept after
|
||||
merge, not deleted. Merges and branches are pushed to origin. Phase 0 (once):
|
||||
commit the design docs, merge the outstanding `feat/usb` work into `main`, and
|
||||
run the existing QEMU suite green before any new work starts.
|
||||
|
||||
**Numbering note:** continues the milestone sequence (driver track ended at M16).
|
||||
|
||||
## Status
|
||||
|
||||
The loop marks a phase `[x]` in the same commit that lands it. A phase is marked
|
||||
only when its definition of green holds.
|
||||
|
||||
- [x] **Phase 0** — baseline: docs committed, feat/usb merged to main, pushed;
|
||||
`usb-xhci-libary.zig` renamed to `usb-xhci-library.zig`; existing QEMU
|
||||
suite green from the worktree (48/48, 2026-07-12).
|
||||
- [x] **M17.1** — kernel releases claims/MSI on death (claims: `releaseAllOwnedBy`
|
||||
in the reap; MSI was already swept by `irq.releaseOwner`; `claim-release`
|
||||
test; suite 49/49)
|
||||
- [x] **M17.2** — exit reasons (`ExitReason` recorded at exit/fault/kill before
|
||||
the notification; `process_exit_reason` supervisor-gated;
|
||||
`runtime.process.exitReason`; kernel + ring-3 assertions; suite 49/49)
|
||||
- [x] **M17.3** — published exit events + VFS subscriber (`process_subscribe`,
|
||||
bounded ref-counted table, publish on every death;
|
||||
`runtime.process.subscribeExits`; VFS handles carry owners and are swept on
|
||||
the owner's death; `vfs-client-death` test; suite 50/50)
|
||||
- [x] **M17.4** — signals, timer notifications, `runtime.process`, the service
|
||||
harness (signal_bind/process_signal + coalescing pending mask; timer_bind
|
||||
on the tick; bindSignals/signalsFrom/sendSignal/stop + timerOnce;
|
||||
runtime.service.run with the zero-length ping; VFS converted; `signals`
|
||||
scenario; suite 51/51)
|
||||
- [x] **merge** `feat/process-lifecycle` → main, push (merged 2026-07-13)
|
||||
- [x] **M18.1** — device-manager protocol: hello + restart policy
|
||||
(device-manager-protocol module; the manager as a harness service:
|
||||
supervised spawns, hello deadline via timer sweep, restart with
|
||||
300/600/1200ms backoff, exit reasons deciding restart-vs-stopped,
|
||||
crash-loop cap; usb-xhci-bus first conforming driver; crash-test fixture
|
||||
re-proving claim release each respawn; `driver-restart` scenario;
|
||||
maximum_tasks 16→32 — the sweep was overflowing the pool; suite 52/52)
|
||||
- [x] **merge** `feat/device-manager` → main, push (merged 2026-07-13)
|
||||
- [x] **M18.2** — xHCI port scan + tree reports (child_added/child_removed in
|
||||
the protocol; the manager's child mirror with death-pruning; xHCI maps the
|
||||
register BAR — resource 0 is ECAM — reads CAPLENGTH/HCSPARAMS1, scans
|
||||
PORTSC, reports connected ports with speed-class identity; `usb-report`
|
||||
scenario proves report → prune → respawn → re-report; suite 53/53)
|
||||
- [x] **M18.3** — app surface: enumerate/subscribe over IPC (subscriber
|
||||
endpoint rides as the call's capability; events are the same structs the
|
||||
buses send); device-list first client; protocol capped at the kernel's
|
||||
IPC MESSAGE_MAXIMUM (256); the startUserTask debug print removed — it
|
||||
sheared concurrent serial lines and was the scenario-flake root cause;
|
||||
`device-list` scenario; suite 54/54)
|
||||
- [x] **merge** `feat/usb-xhci-bus` → main, push (merged 2026-07-13) — **plan complete**
|
||||
|
||||
---
|
||||
|
||||
## M17.1 — the kernel releases a dead process's claims
|
||||
|
||||
The cleanup half of iron rule 1; the prerequisite for every restart story.
|
||||
|
||||
- `system/kernel/devices-broker.zig`: `releaseAllOwnedBy(owner: u32)` — clear
|
||||
every `claimed[]` slot holding this task id.
|
||||
- `system/kernel/process.zig`: call it from the reap path, alongside the existing
|
||||
IRQ-binding release (the ordering comment there says why IRQs go first — claims
|
||||
slot in after them, before the exit notification).
|
||||
- MSI vectors: find where `msi_bind` records per-device vectors (interrupts
|
||||
module) and release those by owner in the same pass.
|
||||
- Docs: remove the claims bullet from process-management.md "Known gaps".
|
||||
|
||||
**Test:** new QEMU scenario `claim-release` — a test child claims an unclaimed
|
||||
device, is killed, is respawned, and claims the same device again successfully;
|
||||
assert both claims in the serial log. Kernel-side unit coverage in
|
||||
`system/kernel/tests.zig` for `releaseAllOwnedBy` (claim two devices as two owners,
|
||||
release one owner, verify exactly its claims freed).
|
||||
|
||||
## M17.2 — exit reasons
|
||||
|
||||
- `system/abi.zig`: `ExitReason` (exited, aborted, segmentation_fault,
|
||||
illegal_instruction, arithmetic_fault, killed).
|
||||
- Kernel: record the reason at every death site — clean exit path, each fault
|
||||
class in `onException`, the kill path. Bounded recent-exits table (ids are never
|
||||
reused, so a small ring keyed by id is enough).
|
||||
- New system call `process_exit_reason(id)` — supervisor-gated, like kill; returns
|
||||
the recorded reason or `-ESRCH` once evicted.
|
||||
- `library/runtime/process.zig`: `ExitReason` + `exitReason(id: u32)`.
|
||||
- Docs: remove the no-exit-status bullet from process-management.md.
|
||||
|
||||
**Test:** extend the `supervision` scenario — three children: one exits cleanly,
|
||||
one faults (the fault-recovery pattern), one is killed; the supervisor asserts all
|
||||
three reasons.
|
||||
|
||||
## M17.3 — published exit events
|
||||
|
||||
- Kernel: bounded subscriber table (endpoints); new system call
|
||||
`process_subscribe(endpoint)` (ungated, like `process_enumerate`); every death
|
||||
posts `notify_exit_bit | id` to each subscriber — the same post the supervisor
|
||||
path already uses.
|
||||
- `library/runtime/process.zig`: `subscribeExits(endpoint)`.
|
||||
- VFS becomes the first subscriber: on an exit event, release every handle keyed
|
||||
by that task id (badges already are task ids). Log the release.
|
||||
- Docs: note the convention in ipc.md (exit events reuse the exit-notification
|
||||
badge encoding).
|
||||
|
||||
**Test:** new QEMU scenario `vfs-client-death` — a client opens a file and is
|
||||
killed without closing; assert the VFS logs the handle release and its open-handle
|
||||
count returns to baseline.
|
||||
|
||||
## M17.4 — signals and the service harness
|
||||
|
||||
- Kernel: per-task pending mask + bound endpoint; system calls
|
||||
`signal_bind(endpoint)` and `process_signal(id, signal)` (supervisor-or-self
|
||||
gated); delivery posts `notify_signal_bit | pending mask`, coalescing; pending
|
||||
signals with no bound endpoint pend silently.
|
||||
- `library/runtime/process.zig`: `Signal`, `SignalSet`, `bindSignals`,
|
||||
`signalsFrom`, `sendSignal`, `stop(id, deadline_ms)` (terminate → wait for exit
|
||||
notification → kill). Implement `terminate`, `reload`, `user_1`, `user_2`;
|
||||
`interrupt`/`quit` are enum members with no sender yet; `alarm` stays unbuilt.
|
||||
- Kernel: **one-shot timer notifications** — `timer_bind(endpoint, ms)` posts a
|
||||
notification badge when the deadline lands (IRQ-as-IPC again, on the timer
|
||||
wheel `sleep` already uses). This is the missing timed-wait primitive:
|
||||
`replyWait` blocks forever and `sleep` blocks the whole process, but `stop()`'s
|
||||
escalation, the device manager's `hello` deadline (M18.1), and restart backoff
|
||||
all need a deadline while staying responsive. It is also the mechanism `alarm`
|
||||
gets for free later.
|
||||
- New `library/runtime/service.zig`: the harness — `run(callbacks)` owning the
|
||||
replyWait loop, folding protocol messages, signals, and child-exit notifications
|
||||
into `init` / `on_message` / `on_reload` / `on_terminate`; answers the common
|
||||
`ping` automatically. Define the reserved `ping` request encoding here and
|
||||
document it in ipc.md (one obvious encoding; smallest that cannot collide with
|
||||
existing protocols).
|
||||
- Convert one existing service (input-source or hpet) to the harness as proof it
|
||||
subtracts code rather than adding it.
|
||||
|
||||
**Test:** extend `supervision` — a harness-built child: `sendSignal(reload)`
|
||||
observed in its log, `ping` answered, `stop()` produces a clean exit with reason
|
||||
`exited`; a second child that ignores signals (no bind) is killed by `stop()`'s
|
||||
deadline with reason `killed`.
|
||||
|
||||
## M18.1 — device-manager protocol: hello + restart policy
|
||||
|
||||
- New `system/services/device-manager/device-manager-protocol.zig` module
|
||||
(vfs-protocol pattern): `hello { version, role, device_id }`; version constant;
|
||||
reserved fields.
|
||||
- Device manager: register the `.device_manager` endpoint; spawn drivers with its
|
||||
exit endpoint; enforce the hello deadline; restart policy — backoff, crash-loop
|
||||
cap (three fast deaths → mark failed, log, stop), reasons from M17.2 deciding
|
||||
restart vs not.
|
||||
- usb-xhci-bus: adopt the harness + send hello. hpet/ps2-bus follow only if the
|
||||
conversion is mechanical; otherwise they keep working unconverted (the manager
|
||||
only enforces hello on drivers spawned with an assignment).
|
||||
- build.zig: test-loop entry for the protocol module if it grows pure logic.
|
||||
|
||||
**Test:** new QEMU scenario `driver-restart` — the xHCI driver takes a test-only
|
||||
argv flag to fault after hello on its first run; assert: fault, exit reason
|
||||
recorded, manager respawns with backoff, second run claims the controller
|
||||
(M17.1) and hellos clean. Assert the crash-loop cap by a driver that always
|
||||
faults (a tiny test driver, not xhci).
|
||||
|
||||
## M18.2 — bus tree reports
|
||||
|
||||
- Protocol: `child_added { parent, identity, resources }` / `child_removed { id }`.
|
||||
- usb-xhci-bus: bring-up to **port scan only** — map the MMIO window (claimed in
|
||||
M16-era work), controller reset/start per xHCI spec, walk the port registers,
|
||||
report one `child_added` per connected port with speed + port number as
|
||||
identity. **No transfer rings, no descriptors** — reading device/interface
|
||||
descriptors (and therefore USB class triples for matching) is the follow-on USB
|
||||
track, not this plan.
|
||||
- Device manager: mirror reports into its tree; prune the subtree (emitting
|
||||
`child_removed`) when a bus driver dies; assert re-report on restart.
|
||||
|
||||
**Test:** QEMU already attaches usb-kbd + usb-mouse on xhci.0 — assert two
|
||||
`child_added` events reach the manager and appear in its tree dump; kill the
|
||||
driver, assert two `child_removed` then two fresh `child_added` after respawn.
|
||||
|
||||
## M18.3 — the application surface
|
||||
|
||||
- Protocol: `enumerate` (tree snapshot) + `subscribe` (published add/remove
|
||||
events, input-service pattern).
|
||||
- A small client (`device-list`, the `ps` analog) exercising both; the manager
|
||||
becomes the one answer to "what devices exist" for user space.
|
||||
`device_enumerate` stays for drivers/kernel seeding — its retreat is tied to the
|
||||
discovery migration, out of this plan.
|
||||
|
||||
**Test:** QEMU scenario — `device-list` shows the tree including USB children;
|
||||
during a driver restart the subscribing client logs remove + add events.
|
||||
|
||||
---
|
||||
|
||||
**Explicitly out of scope** (own tracks, after M18): discovery migration (pci-bus
|
||||
driver, acpi service, retiring the kernel scan), USB control transfers +
|
||||
descriptors + class-driver matching, the musl layer, `interrupt`/`quit` senders
|
||||
(needs a console), job control.
|
||||
@@ -0,0 +1,212 @@
|
||||
# M19–M20 execution plan: discovery migration
|
||||
|
||||
The operational plan for [device-manager.md](device-manager.md)'s increment 8:
|
||||
discovery leaves the kernel — a **pci-bus driver** (M19) and an **acpi service**
|
||||
(M20), with the kernel's device enumeration retired behind them. Same rules as
|
||||
[m17-m18-plan.md](m17-m18-plan.md): one phase at a time, each green before the
|
||||
next; this file is the build order and the checklist.
|
||||
|
||||
**Definition of green, every phase:** `zig build` clean, `zig build test` clean,
|
||||
`python3 test/qemu_test.py` passes (existing scenarios plus the phase's new
|
||||
one), and the relevant design doc updated. Commit per green phase (no co-author
|
||||
trailers). The full suite is the regression net — the existing
|
||||
`driver-restart` / `usb-report` / `device-list` / `input` scenarios must stay
|
||||
green *through* the migration, which is the whole point: the system must not be
|
||||
able to tell who enumerated it.
|
||||
|
||||
**Workflow:** dedicated worktree; branches off `main` — `feat/pci-bus`
|
||||
(M19.0–19.3), `feat/acpi-service` (M20.1–20.3); auto-merge to main when a
|
||||
branch is green; keep branches; push everything.
|
||||
|
||||
## Settled decisions (2026-07-13 — veto before the loop starts)
|
||||
|
||||
1. **What "retiring the kernel scan" means.** The kernel keeps, forever, the
|
||||
parses it needs before user space exists: RSDP/XSDT location, MADT (SMP),
|
||||
the HPET table (the tick), FADT + the AML `\_S5` evaluation (poweroff — the
|
||||
power tests prove it), and MCFG (the host bridge node). What retires is
|
||||
**device enumeration**: the ECAM function walk (M19.3) and the DSDT/SSDT
|
||||
namespace walk that builds device nodes (M20.3). The AML module stays a
|
||||
shared build module compiled into both the kernel (for `\_S5`) and the acpi
|
||||
service (for everything else) — same source, two builds, no fork.
|
||||
2. **Bridge apertures come from the firmware memory map, not AML.** Registered
|
||||
PCI functions carry BAR resources, and containment demands the bridge own
|
||||
windows that cover them. The apertures are derived kernel-side from the
|
||||
boot memory map's MMIO holes (regions that are neither RAM nor tables) —
|
||||
mechanical, AML-free, and available at boot regardless of what later moved
|
||||
to user space. (The bridge today carries only ECAM + bus range; this is the
|
||||
prerequisite M19.0 exists for.)
|
||||
3. **`device_register` becomes idempotent on exact match.** A re-registration
|
||||
with identical (parent, class, resources) returns the existing id instead
|
||||
of appending. The kernel table has no unregister, so without this a
|
||||
restarted registering bus would duplicate its children on every respawn —
|
||||
idempotence makes restart-and-re-report safe for every future bus, not just
|
||||
PCI.
|
||||
4. **The manager matches from reports.** `ChildAdded` gains a `device_id`
|
||||
field (the kernel-registered id, `no_device` for unregistered leaves like
|
||||
USB ports). After the M19.3 flip, PCI driver matching keys off reported
|
||||
identity (the class triple) instead of the manager's boot-time snapshot —
|
||||
the snapshot match remains only for what the kernel still seeds. One flip
|
||||
phase changes both sides at once so no device is ever matched twice.
|
||||
5. **The acpi service's authority is one node.** The kernel publishes an
|
||||
`acpi-tables` device: memory resources covering the table blobs plus a
|
||||
broad `io_port` resource — the documented trust grant to exactly one
|
||||
process (AML OperationRegions reach EC/PM ports; the claim-gated
|
||||
io_read/io_write calls already exist). The service claims it, maps the
|
||||
tables, and runs the shared AML module in ring 3 behind a `Hal` backed by
|
||||
`mmio_map` + `io_read`/`io_write`.
|
||||
6. **Both new processes are protocol drivers** under the manager: hello,
|
||||
supervision, restart with backoff — all inherited from M18.1 for free.
|
||||
Registration idempotence (decision 3) is what makes their restarts sound.
|
||||
7. **Firmware neutrality is the contract** (2026-07-13). The generic layer is
|
||||
everything at and above the device-manager protocol — descriptors,
|
||||
containment, reports, matching, supervision — and none of it may become
|
||||
x86-specific. Discovery is one swappable process per firmware: the acpi
|
||||
service on x86; an **fdt service** on the Raspberry Pis (claims a
|
||||
`devicetree-blob` node, reports children from the flattened device tree —
|
||||
pure data, no bytecode, no port grant, strictly simpler than ACPI). The
|
||||
manager owns the tree as *data* and touches no hardware, ever — AML runs in
|
||||
a crashable, supervised discoverer precisely so a firmware-bytecode fault
|
||||
can never take down the supervisor. Two consequences recorded now:
|
||||
`DeviceDescriptor`'s 8-byte `hid` cannot hold an FDT `compatible` string
|
||||
("brcm,bcm2835-aux-uart") — identity widens before the fdt service exists;
|
||||
and cross-firmware surfaces are named by **domain, not firmware** (M21
|
||||
defines a *power* protocol, not an "ACPI events" protocol — PSCI/mailbox
|
||||
sources feed the same subscribers on ARM). **Landed early (2026-07-13):**
|
||||
both services exist as placeholders (system/services/acpi, system/services/
|
||||
fdt) and the build's `-Ddiscovery=acpi|fdt` option fills the ramdisk's
|
||||
neutral `discovery` slot — the manager will spawn "discovery" by that name
|
||||
in M20.3 and never learn which firmware it is on.
|
||||
|
||||
## Status
|
||||
|
||||
- [x] **M19.0** — prerequisites (bridge apertures from the memory map's
|
||||
*gaps* — the single-hole rule died on OVMF's flash at the top of 4 GiB,
|
||||
caught by the new every-BAR-contained assert in `discovery`; idempotent
|
||||
`device_register` proven in `bus`; `ChildAdded.device_id`;
|
||||
m17-m18-plan.md archived; suite 54/54).
|
||||
- [x] **M19.1** — pci-bus driver, scan only (claims the bridge, maps ECAM
|
||||
through its grant, brute-force walk with the multifunction rule; the
|
||||
manager matches pci_host_bridge → pci-bus per device with the full
|
||||
protocol contract; `pci-scan` builds its expected marker from the
|
||||
kernel's own count — equivalence on the first run; suite 55/55).
|
||||
- [x] **M19.2** — register + report (BAR probe mirrored byte-for-byte from the
|
||||
kernel's addBars so dedupe returns the kernel's node ids during
|
||||
coexistence; the bridge gained the io_port aperture I/O BARs need;
|
||||
reports carry the registered device_id; pci-scan drills a forced restart
|
||||
and asserts the PCI node count never grows — plus harness hardening: a
|
||||
failing case now preserves its serial as <case>-failed-serial.log, and
|
||||
the heavy scenarios run at 150s; suite 55/55).
|
||||
- [x] **M19.3** — the flip: kernel `enumeratePci`/`addBars`/`PciHeader` all
|
||||
deleted (bridge node stays); manager matches PCI drivers from reported
|
||||
identity, deduped by registered id. Surfaced and fixed a real SMP race the
|
||||
flip created — ring-3 device_register made the broker table concurrent, so
|
||||
mmio_map's lock-free read intermittently tore hpet's resource length
|
||||
(user fault) and overflowed `r.len-1` into a kernel panic; now the broker
|
||||
read is under the big lock and the arithmetic is guarded, and pci-bus
|
||||
skips size-0 BARs. discovery.md updated; suite 55/55 (driver-restart
|
||||
hammered 6×).
|
||||
- [x] **merge** `feat/pci-bus` → main, push (merged 2026-07-13).
|
||||
- [x] **M20.1** — acpi service, parse only: the AML interpreter is now a build
|
||||
module compiled into both kernel and service; the kernel publishes the
|
||||
`acpi-tables` node (AML blobs as memory resources, the broad io_port grant,
|
||||
the SCI); the service claims it, maps the blobs, runs the shared parser in
|
||||
ring 3, and self-verifies its Device count against the kernel's (34 = 34,
|
||||
deterministic via argv, no log-scraping); the manager spawns `discovery`
|
||||
at startup. Parse-only touches no hardware. Suite 56/56.
|
||||
|
||||
- [x] **M20.2** — register + report: the service evaluates `_STA`/`_CRS` in
|
||||
ring 3 (interpreter Hal = port I/O over the claimed node; a scratch page
|
||||
backs SystemMemory maps so a stray region can't fault it) and registers +
|
||||
reports each present `_HID` device under `acpi-tables`. Containment: the
|
||||
broker's irq check became range-based (len-1 == the old equality) so the
|
||||
node's broad irq window covers children's legacy lines; io ports fall in
|
||||
the broad io grant. ChildAdded gained `hid`. Matching stays off. The
|
||||
`acpi-report` scenario asserts the PS/2 keyboard (3 resources) and mouse
|
||||
(1 resource) among the reports. Suite 57/57.
|
||||
- [x] **M20.3** — the flip: the kernel's `wireAcpiDevices` call is gone (the
|
||||
device-building helpers are retained-but-dead pending a focused sweep,
|
||||
spawned as a task; static tables + `\_S5` + the acpi-tables node stay).
|
||||
The manager matches ps2-bus from ACPI `_HID` reports; the service
|
||||
registers all devices before reporting any (no keyboard-before-mouse
|
||||
race). The `acpi-ps2` scenario proves report → spawn → ps2-bus attaches
|
||||
its keyboard; `ioport` retargeted to the acpi-tables I/O window (the
|
||||
kernel-built PS/2 node is gone). Suite 58/58.
|
||||
- [ ] **merge** `feat/acpi-service` → main, push — **loop ends here**.
|
||||
|
||||
---
|
||||
|
||||
## Phase notes
|
||||
|
||||
**M19.0 apertures:** the boot memory map already crosses the handoff
|
||||
([boot-handoff]), but discovery never sees it today — expect a small
|
||||
pass-through (kernel init hands the map to the platform layer) before the
|
||||
holes computation, which belongs where the bridge node is built
|
||||
(`parseMcfg`). Sanity-check on QEMU q35: the xHCI BAR (`0xc0000000`-region
|
||||
values seen in the M18 logs) must land inside a derived aperture, asserted in
|
||||
the kernel unit test.
|
||||
|
||||
**M19.1 scanning without owning config access twice:** the driver reads config
|
||||
space through its ECAM mmio_map grant of the *bridge* window — the same bytes
|
||||
the kernel walk read. Vendor-id `0xFFFF` skip, header-type multifunction rule,
|
||||
no bridge recursion (matches the kernel's current single-segment walk).
|
||||
|
||||
**M19.2 BAR sizing:** the classic size probe (write all-ones, read mask,
|
||||
restore) is deferred — the BARs' current programmed values and types are
|
||||
enough for containment-checked registration at bring-up; sizing lands with the
|
||||
first driver that needs to *move* a BAR. Log what is registered so the
|
||||
scenario can assert it.
|
||||
|
||||
**M19.3 what the manager still seeds from the snapshot:** everything the
|
||||
kernel still enumerates (timers, ACPI nodes until M20.3). The PCI arm of
|
||||
`pciDriverFor` switches source; `driverFor` doesn't move until M20.3.
|
||||
|
||||
**M20.1 spawn and identity (pre-settled 2026-07-13):** the manager spawns
|
||||
`discovery` by its neutral ramdisk name at startup, as an ordinary protocol
|
||||
driver (hello, supervision) — from M20.1 on, on every boot. For reporting ACPI
|
||||
devices, `ChildAdded` gains `hid: [8]u8` (EISA ids fit; zero = none):
|
||||
firmware *string* identity travels beside the numeric `identity` field until
|
||||
the FDT-driven widening replaces both (decision 7).
|
||||
|
||||
**M20.1 Hal in ring 3:** `mapMmio` → `device.mmioMap` over the claimed
|
||||
acpi-tables node (plus a table-offset map for blobs); `pioRead`/`pioWrite` →
|
||||
`device.ioRead`/`ioWrite` against its io_port resource. The interpreter cannot
|
||||
tell it moved — that is the assertion of `acpi-parse`.
|
||||
|
||||
**M20.2 containment for `_CRS`:** io ports fall inside the node's broad
|
||||
io_port resource; MMIO windows (HPET, LAPIC ranges some firmwares list) fall
|
||||
inside the memory-map holes added to the node in M20.1. Anything that doesn't
|
||||
fit is logged and skipped, loudly — bring-up honesty over silent drops.
|
||||
|
||||
**M20.3 ps2 ordering:** ps2-bus binds nodes the acpi service now reports, so
|
||||
its spawn moves behind the report (the manager's matching handles this once
|
||||
the source flips); the `input` scenario proves the keyboard still types.
|
||||
|
||||
**Explicitly out of scope:** PCI bridge recursion (single segment, flat bus
|
||||
walk stays); BAR reprogramming/sizing; disk/PCIe hotplug; interrupt routing
|
||||
changes (`_PRT` stays wherever it is today); the USB descriptor track;
|
||||
multi-segment ECAM; per-device power states (D-states, `_PSx`/`_PRx`,
|
||||
suspend/resume — a future *lifecycle-vocabulary* extension, since "suspend"
|
||||
has the shape of a signal every driver must answer, and it has no consumer
|
||||
until laptop sleep); CPU P/C-states.
|
||||
|
||||
## M21 preview — ACPI events + system power (planned next, not in this loop)
|
||||
|
||||
The acpi service grows the event side (settled direction 2026-07-13; detailed
|
||||
phases when M20 lands):
|
||||
|
||||
- **21.1 SCI + fixed events**: irq_bind the SCI (the resource M20.1 already
|
||||
records), read/clear PM1 status, publish the power-button event to
|
||||
subscribers (the same pub/sub shape the manager uses).
|
||||
- **21.2 GPE + Notify**: Notify dispatch in the shared AML interpreter, GPE
|
||||
block handling, `Notify(device, code)` published per reported node. The
|
||||
acpi service is a **bus** here: battery (PNP0C0A), AC (ACPI0003), and lid
|
||||
(PNP0C0D) nodes are reported children; small class drivers bind them and
|
||||
speak an evaluate/subscribe protocol to the service — the xHCI split,
|
||||
repeated. The embedded controller (`_Qxx` queries) rides this phase;
|
||||
QEMU emulates no battery/EC, so those paths are interface-complete and
|
||||
validated on real hardware (the laptop is the win condition), while the
|
||||
plumbing is proven by the power button.
|
||||
- **21.3 the capstone**: QEMU `system_powerdown` → acpi service event → init
|
||||
runs the M17 stop sequence over its children → kernel `\_S5` — orderly
|
||||
shutdown as the scenario that proves lifecycle + events compose. (The
|
||||
harness grows a QMP poke to inject the event.)
|
||||
+3
-3
@@ -27,7 +27,7 @@ danos's own neutral format, and the kernel only ever sees that.**
|
||||
|
||||
## The neutral format
|
||||
|
||||
Defined in `src/root.zig`, the shared loader↔kernel contract:
|
||||
Defined in `system/boot-handoff.zig`, the shared loader↔kernel contract:
|
||||
|
||||
```zig
|
||||
pub const MemoryKind = enum(u32) {
|
||||
@@ -68,7 +68,7 @@ pub const BootInfo = extern struct {
|
||||
|
||||
## The loader side (UEFI)
|
||||
|
||||
Two functions in `src/boot/efi.zig`, called from `exitBootServices`:
|
||||
Two functions in `boot/efi.zig`, called from `exitBootServices`:
|
||||
|
||||
- **`classify`** maps each UEFI descriptor to a `MemoryKind`:
|
||||
`conventional_memory` **and** `boot_services_code`/`boot_services_data → usable`;
|
||||
@@ -121,7 +121,7 @@ The kernel receives a plain array and reads it with zero UEFI knowledge:
|
||||
|
||||
```zig
|
||||
const mm = boot_info.memory_map;
|
||||
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
||||
const regions = @as([*]const system.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
||||
for (regions) |r| {
|
||||
if (r.kind == .usable) usable_pages += r.pages;
|
||||
}
|
||||
|
||||
+49
-14
@@ -7,7 +7,7 @@ which live in memory we'd like to reclaim and don't control), switches CR3 onto
|
||||
them, and — crucially — maps with **real permissions**.
|
||||
|
||||
It's x86_64-specific (the 4-level table format is an Intel/AMD thing), so it lives
|
||||
behind the [arch](arch.md) boundary in `src/kernel/arch/x86_64/paging.zig`.
|
||||
behind the [arch](arch.md) boundary in `system/kernel/architecture/x86_64/paging.zig`.
|
||||
|
||||
## The format
|
||||
|
||||
@@ -17,20 +17,54 @@ the final 4 KiB page. Each entry holds a physical address plus flag bits —
|
||||
present, writable, and (bit 63) **no-execute**. danos maps everything with 4 KiB
|
||||
pages: precise, and the extra table memory is negligible against available RAM.
|
||||
|
||||
## Higher half: the address-space layout
|
||||
|
||||
danos is a **higher-half kernel**. The kernel is linked to run at
|
||||
`0xFFFF_FFFF_8000_0000` but loaded low (the linker script's `AT()` gives each
|
||||
segment a physical load address at 1 MiB up; the bootloader maps the high link
|
||||
address to the low load address in its bootstrap tables and jumps in). The entire
|
||||
**low canonical half is reserved for user space**; the kernel lives in the top half
|
||||
alongside a **physmap** — a straight window onto all of physical memory at
|
||||
`physmap_base + phys`. Wherever the kernel needs to touch a physical address (a
|
||||
page-table frame, an ACPI table, a device register), it adds that constant:
|
||||
`system.physToVirt(phys)`. The layout constants live in `system/boot-handoff.zig`:
|
||||
|
||||
| region | virtual base | PML4 slot |
|
||||
|--------|--------------|-----------|
|
||||
| user image + stack | `0x0000_7000_0000_0000` | 224 (low half) |
|
||||
| kernel heap | `0xFFFF_8000_0000_0000` | 256 |
|
||||
| physmap (all RAM + MMIO windows) | `0xFFFF_8800_0000_0000` + phys | 272 |
|
||||
| kernel image | `0xFFFF_FFFF_8000_0000` | 511 |
|
||||
|
||||
The bootloader builds temporary **bootstrap tables** (identity + a 4 GiB physmap +
|
||||
the high kernel) so it can switch CR3 and jump to the high entry; the kernel then
|
||||
builds its own precise tables below and abandons them. Because both use the same
|
||||
`physmap_base`, any physmap pointer minted before the switch stays valid after it.
|
||||
|
||||
## What gets mapped, and with what permissions
|
||||
|
||||
The address space is built in three passes (`init`):
|
||||
The address space is built in four passes (`init`):
|
||||
|
||||
1. **All RAM, identity-mapped RW + NX.** Every non-MMIO region from the
|
||||
[memory map](memory-map.md) is mapped virtual == physical, read-write and
|
||||
*non-executable*. Identity mapping keeps everything already running valid across
|
||||
the CR3 switch (the frame allocator addresses frames by physical address, page
|
||||
tables are reached the same way, the stack stays put).
|
||||
2. **The framebuffer and the Local APIC**, the device memory we actually touch,
|
||||
also RW + NX. Everything else — unbacked address space, other MMIO — is simply
|
||||
left unmapped, so a stray access faults instead of silently succeeding.
|
||||
3. **The kernel's own segments, overlaid with their true ELF permissions.** This is
|
||||
1. **All RAM in the physmap, RW + NX.** Every non-MMIO region from the
|
||||
[memory map](memory-map.md) is mapped at `physToVirt(phys)`, read-write and
|
||||
*non-executable*. There is **no low/identity mapping** — the low half is user
|
||||
space. (Frames the kernel touches while still building these tables are reached
|
||||
through the loader's bootstrap physmap, which covers the low 4 GiB; both the
|
||||
frame allocator and the table builder scan low-address-up, so those frames stay
|
||||
under that limit.)
|
||||
2. **The framebuffer and the Local APIC**, the device memory the kernel touches
|
||||
directly, as physmap windows (RW + NX). Other MMIO is mapped on demand by
|
||||
`mapMmio`, also into the physmap; everything else is left unmapped, so a stray
|
||||
access faults instead of silently succeeding.
|
||||
3. **The kernel's own segments, overlaid with their true ELF permissions**, at
|
||||
their high link addresses mapped to their low physical load addresses. This is
|
||||
the interesting part.
|
||||
4. **Every higher-half PML4 entry pre-created** (an empty PDPT where none exists
|
||||
yet). The kernel half is then a fixed set of top-level slots, so a per-process
|
||||
address space can share it by copying `PML4[256..512)` once — growth beneath
|
||||
those slots (heap, on-demand MMIO) propagates to every address space because
|
||||
they share the PDPTs. `init` asserts no new higher-half PML4 entry appears
|
||||
afterward.
|
||||
|
||||
### W^X from the ELF program headers
|
||||
|
||||
@@ -55,9 +89,10 @@ reserved bit and fault.
|
||||
|
||||
### The null guard
|
||||
|
||||
Page 0 is deliberately left unmapped. A null (or near-null) pointer dereference now
|
||||
takes a page fault instead of quietly reading or writing real memory — turning a
|
||||
whole class of silent bugs into an immediate, located crash.
|
||||
The whole low half is unmapped except for explicit user mappings, so page 0 (and
|
||||
every near-null address) is unmapped by construction. A null (or near-null) pointer
|
||||
dereference in the kernel takes a page fault instead of quietly reading or writing
|
||||
real memory — turning a whole class of silent bugs into an immediate, located crash.
|
||||
|
||||
## Switching on, and the on-demand API
|
||||
|
||||
|
||||
@@ -0,0 +1,327 @@
|
||||
# Process lifecycle: signals over IPC
|
||||
|
||||
**Status: increments 1–4 built** (2026-07-12): claim release on death, exit
|
||||
reasons, published exit events, and signals + one-shot timers + the service
|
||||
harness are all in — the interface below is as-built. The primitives underneath
|
||||
predate this design ([process-management.md](process-management.md):
|
||||
spawn, the supervision link, kill, child-exit notifications); this document designs
|
||||
the layer above them — the standard vocabulary a danos process speaks about its own
|
||||
life, and the stable `runtime.process` interface that carries it. Nothing here is
|
||||
device- or driver-specific: a driver, the VFS, and a user application all stop,
|
||||
reload, and die the same way. The device manager is simply this design's first
|
||||
serious customer ([device-manager.md](device-manager.md)).
|
||||
|
||||
**"POSIX" in this document means the concepts, never the letter of the standard.**
|
||||
danos borrows the ideas and the hard-won lessons (what SIGTERM *means*, why SIGPIPE
|
||||
was a mistake) without inheriting the mechanism, the API, or the names. The naming
|
||||
rule is danos's own and it is strict: plain words that communicate intent
|
||||
(`terminate`, `reload`, `exited`) and the IPC vocabulary the system already speaks
|
||||
(`bind`, `subscribe`, `publish`, `endpoint`) — never `SIG*`, never a second word for
|
||||
a concept that already has one. Literal POSIX arrives later and lives elsewhere: a
|
||||
**musl-based C layer** (growing out of library/posix) that wires C programs to the
|
||||
danos runtime — musl's syscall surface retargeted at danos system calls and IPC
|
||||
protocols (files onto the VFS protocol, `sigaction`/`wait` onto this lifecycle,
|
||||
sockets onto whatever networking becomes). Ported programs see POSIX; the system
|
||||
underneath never does.
|
||||
|
||||
## Why a standard vocabulary
|
||||
|
||||
A supervisor can only manage processes it has never heard of if "please exit" means
|
||||
the same thing to all of them. That is the one thing POSIX signals got deeply right:
|
||||
`SIGTERM` means the same thing to nginx and to a five-line script, which is why
|
||||
process supervision on Unix (init systems, container runtimes) is possible at all.
|
||||
danos wants that property from day one, because supervision-and-restart is the
|
||||
system's core motivation ([resilience.md](resilience.md)).
|
||||
|
||||
What POSIX got wrong — for a system like this — is the **delivery mechanism**:
|
||||
asynchronous control-flow hijack. A Unix handler runs on a stolen stack at an
|
||||
arbitrary instruction boundary, which is why the async-signal-safe function list
|
||||
exists, why `errno` must be saved, and why the canonical signal bug is a SIGTERM
|
||||
handler innocently calling `printf` mid-`malloc`. That entire bug class comes from
|
||||
the mechanism, not the vocabulary, and none of it is worth importing.
|
||||
|
||||
A microkernel already has the right channel: **a signal is a message.** QNX delivers
|
||||
POSIX signals over its message passing; seL4 has notification objects; Erlang turned
|
||||
"death is a message to whoever linked" into a reliability philosophy. danos has
|
||||
already done it once without naming it: a child's death arrives as a notification
|
||||
badge on the supervisor's endpoint — the microkernel's SIGCHLD, the IRQ-as-IPC
|
||||
pattern reused. Signals are the same pattern reused a third time.
|
||||
|
||||
## The mechanism
|
||||
|
||||
- **`signal_bind(endpoint)`** — a process nominates the endpoint its signals arrive
|
||||
on, exactly as `irq_bind` nominates where a device's interrupts land. The runtime
|
||||
does this at startup for any program that opts in.
|
||||
- **`process_signal(id, signal)`** — posts the signal as an asynchronous
|
||||
notification to the target's bound endpoint: badge = `notify_badge_bit |
|
||||
notify_signal_bit | pending signals`. Non-blocking for the sender, always.
|
||||
- **Pending signals coalesce** in a per-process bitmask until the target next waits
|
||||
— exactly like interrupt notifications, and exactly POSIX's own semantics for
|
||||
non-realtime signals (two pending SIGTERMs are one SIGTERM). The bitmask *is* the
|
||||
design: signals carry no payload. Anything with a payload is a protocol message.
|
||||
- **Authority**: the supervisor may signal its children — the same link that is
|
||||
already the kill authority. A process may signal itself. Anything broader waits
|
||||
for transferable process handles.
|
||||
- **No binding, no problem**: a process that never calls `signal_bind` is not
|
||||
broken — its signals pend unread and only `process_kill` works on it. Simple
|
||||
programs stay simple; the vocabulary is opt-in, the kill authority is not.
|
||||
|
||||
Because delivery is a message into the process's own event loop, there is no
|
||||
async-signal-safe list in danos: a handler is ordinary code running at a point the
|
||||
process chose. The bug class is gone by construction, not by discipline.
|
||||
|
||||
## The vocabulary: POSIX.1-1990, sorted honestly
|
||||
|
||||
The full 1990 set, and what each becomes. Two intrinsically problematic cases get a
|
||||
defense below the table.
|
||||
|
||||
| POSIX.1-1990 | danos disposition | Notes |
|
||||
|---|---|---|
|
||||
| SIGTERM | signal `terminate` | finish up and exit; the supervisor's polite half |
|
||||
| SIGHUP | signal `reload` | re-read configuration / re-scan |
|
||||
| SIGINT | signal `interrupt` | interactive interrupt; meaningful once a console can send it, in the vocabulary now so numbering is stable |
|
||||
| SIGQUIT | signal `quit` | as SIGINT, without the core-dump baggage |
|
||||
| SIGALRM | signal `alarm` | timer expiry as a message; the Unix SIGALRM+`longjmp` timeout hacks are impossible here. In the vocabulary, unbuilt: no consumer yet, and when one appears it is runtime sugar over the existing timer — zero kernel work |
|
||||
| SIGUSR1, SIGUSR2 | signals `user_1`, `user_2` | service-defined |
|
||||
| SIGCHLD | **already exists** — the exit notification | the badge carries the child id, dodging the classic coalescing bug (Unix code must loop `waitpid`) |
|
||||
| SIGKILL | `process_kill` — kernel mechanism | its definition is "cannot be handled"; it was never really a signal |
|
||||
| SIGABRT | exit reason `abort` | `abort()` is synchronous self-termination, not an event |
|
||||
| SIGSEGV, SIGILL, SIGFPE | exit reasons, **never delivered** | see below |
|
||||
| SIGPIPE | **an error return**, not a signal | see below |
|
||||
| SIGSTOP, SIGTSTP, SIGTTIN, SIGTTOU, SIGCONT | deferred | job control needs terminals, sessions, and process groups; stop/continue is scheduler territory |
|
||||
|
||||
**The fault signals (SIGSEGV, SIGILL, SIGFPE) are intrinsically wrong for messages.**
|
||||
They are *synchronous* — raised at a specific faulting instruction, not "sometime
|
||||
soon". A message cannot be delivered to a process whose next instruction re-faults;
|
||||
it never reaches its event loop to read it. POSIX only makes fault handlers "work"
|
||||
via the async hijack (run the handler *instead of* the instruction), and even there,
|
||||
returning from a SIGSEGV handler without curing the cause is undefined behavior.
|
||||
danos's architecture already has the better answer: fault → the kernel kills the
|
||||
process ([resilience.md](resilience.md) step 2, built) → the supervisor reads the
|
||||
reason → restart. Recovery is restart, not a handler. This is also truer to the 1990
|
||||
standard than handling is: the standard's default action for all three was
|
||||
"terminate the process".
|
||||
|
||||
**SIGPIPE deserves special contempt.** Its default kills a process that writes to a
|
||||
closed pipe — which is why "the whole server died because one client disconnected"
|
||||
is roughly every network daemon's first production bug, and why every mature codebase
|
||||
contains the same fix: ignore SIGPIPE, handle the `EPIPE` error return. danos made
|
||||
the right choice natively already — a reply owed to a dead peer fails with `-EPEER`.
|
||||
Errors from operations are error returns from those operations. The posix layer can
|
||||
synthesize SIGPIPE for ported code that expects it.
|
||||
|
||||
### Statements, not questions
|
||||
|
||||
A signal and a protocol message both travel over IPC — the difference is the
|
||||
**contract**, not the transport. danos IPC has two primitives, both already in
|
||||
daily use: the **asynchronous notification** (a badge — bits that coalesce into a
|
||||
pending mask; the sender never blocks; no payload, *no reply path*; how IRQs and
|
||||
exit events arrive) and the **synchronous call** (a rendezvous — payload both
|
||||
ways, the caller waits for the reply; how VFS requests work). A signal is the
|
||||
first kind: a *statement*. `terminate` wants no reply — the exit notification is
|
||||
its acknowledgement.
|
||||
|
||||
A health probe is the second kind: a *question*, worthless without its answer —
|
||||
and the answer's absence within a deadline is the very thing being measured.
|
||||
Asked as a signal it has no reply channel (a coalescing bit can't carry an answer,
|
||||
and the authority rule forbids a child signalling its supervisor back); asked as a
|
||||
call, the timeout-is-the-diagnosis semantics come free. So there is no `health`
|
||||
signal. Liveness is the common **`ping`**: a reserved request every harness-run
|
||||
service answers automatically on its main endpoint — still free for the service
|
||||
author, still one obvious way — and a supervisor's probe is a `ping` call with a
|
||||
deadline.
|
||||
|
||||
## The two iron rules
|
||||
|
||||
1. **Cleanup is the kernel's job.** A process can die with no warning — fault,
|
||||
kill, power. Correctness must never depend on a `terminate` handler running. On
|
||||
any death the kernel releases the address space, IPC handles, IRQ bindings, and
|
||||
owed replies (built), and must also release **device, I/O-port, and interrupt
|
||||
claims and MSI vectors** (the known gap in
|
||||
[process-management.md](process-management.md); increment 1). A signal handler is
|
||||
for *graceful* work — flushing, deregistering, saving — never for *necessary*
|
||||
work.
|
||||
2. **Kill is not a signal, and exit reasons are load-bearing.** The standard stop
|
||||
sequence is *terminate → deadline → `process_kill`*; the unhandleable kill stays
|
||||
a kernel mechanism. And a supervisor deciding whether to restart must know *how*
|
||||
the child died: clean exit (meant to — don't restart), fault (restart with
|
||||
backoff), killed (the supervisor did it). The exit notification today carries
|
||||
only the id; it grows a reason. Restart policy cannot be written without it.
|
||||
|
||||
## Who learns of a death
|
||||
|
||||
A death has three audiences, and conflating them is how systems end up with either
|
||||
zombie state or privileged snooping:
|
||||
|
||||
1. **The supervisor** — gets the exit notification on the endpoint it gave at spawn
|
||||
(built), which grows the `ExitReason` (increment 2). The supervisor is the only
|
||||
audience that needs the *reason*, because it is the only one deciding whether to
|
||||
restart.
|
||||
2. **The peer owed a reply** — already built: a client that dies mid-request fails
|
||||
the server's reply with `-EPEER`; a server that dies fails its waiting clients
|
||||
the same way. This covers the *synchronous* case only.
|
||||
3. **The subscribers** — the new piece, and it is the input service's
|
||||
publish/subscribe shape ([input.md](input.md)) applied to exits. A stateful
|
||||
service accumulates per-client state across many requests: the VFS holds a dead
|
||||
client's open file handles, the input service holds its subscriptions, a future
|
||||
network stack holds its sockets. None of these are the client's supervisor, and
|
||||
none learn anything from a failed reply if the client simply never calls again.
|
||||
So the kernel **publishes every exit** to whoever subscribed:
|
||||
`process_subscribe(endpoint)` adds a subscriber, and each death posts a
|
||||
notification to every subscriber (badge = `notify_exit_bit | process id` — the
|
||||
same encoding supervisors already decode, the IRQ-as-IPC pattern once more). The
|
||||
subscriber filters for ids it holds state for and releases what the dead client
|
||||
held. Correlating is free of bookkeeping: an IPC sender's badge already *is* its
|
||||
task id (`runtime.ipc.Received`), so the id a service has been keying client
|
||||
state by all along is the id the exit event carries.
|
||||
|
||||
Subscription, not broadcast-to-everyone: only processes that asked receive
|
||||
events, the kernel keeps a bounded subscriber table, and delivery is the same
|
||||
non-blocking coalescing notification as everything else — a dying process never
|
||||
waits on its mourners. Subscribing is ungated, like `process_enumerate`: what is
|
||||
running (and dying) is not a secret between cooperating processes. Subscribers
|
||||
do not receive the exit reason — the VFS does not care *why* the client died.
|
||||
|
||||
This is the service-side mirror of iron rule 1: **a service must never depend on
|
||||
its clients cleaning up after themselves.** Handle release on client death is the
|
||||
service's job, triggered by the published exit event — never by a courtesy
|
||||
"closing now" message that a crashed client will never send.
|
||||
|
||||
## The stable interface: `runtime.process`
|
||||
|
||||
`runtime.process` already owns what a process receives at birth (`Init`, the
|
||||
argv contract). It grows to own the other end of life.
|
||||
|
||||
**The runtime is the stable interface; the numbers are not.** danos applications do
|
||||
not make system calls — they call the runtime library, and the system-call numbers,
|
||||
notification bits, and signal bit positions beneath it are a **private kernel ↔
|
||||
runtime contract** that may change at any time (settled 2026-07-12). This is why
|
||||
the runtime exists. Today kernel and runtime ship from one tree in one image, so
|
||||
"stability" is simply building them together. When driver binaries start shipping
|
||||
as separately-versioned applications — the whole point of the restart design — the
|
||||
binary's embedded runtime version becomes compatibility metadata (the same idea as
|
||||
the protocol version in the device manager's `hello`), and the kernel refuses what
|
||||
it cannot serve. Signals therefore need no reserved numbering scheme: the enum
|
||||
below is vocabulary, not ABI.
|
||||
|
||||
```zig
|
||||
/// The signal vocabulary. The value is the bit position in the pending mask — a
|
||||
/// private kernel/runtime detail, free to change while they ship together.
|
||||
pub const Signal = enum(u5) {
|
||||
terminate = 0, // SIGTERM: finish up and exit
|
||||
reload = 1, // SIGHUP: re-read configuration
|
||||
interrupt = 2, // SIGINT
|
||||
quit = 3, // SIGQUIT
|
||||
alarm = 4, // SIGALRM
|
||||
user_1 = 5, // SIGUSR1
|
||||
user_2 = 6, // SIGUSR2
|
||||
};
|
||||
|
||||
/// A decoded pending mask: the coalesced set of signals a notification delivered.
|
||||
pub const SignalSet = struct {
|
||||
pending: u32,
|
||||
pub fn has(set: SignalSet, signal: Signal) bool { ... }
|
||||
pub fn iterate(set: SignalSet) Iterator { ... }
|
||||
};
|
||||
|
||||
/// Nominate `endpoint` as this process's signal endpoint (signal_bind). The
|
||||
/// runtime's service harness calls this; a bare program may call it directly and
|
||||
/// fold signals into its own replyWait loop.
|
||||
pub fn bindSignals(endpoint: usize) bool { ... }
|
||||
|
||||
/// Decode a received badge into signals, or null if the badge is not a signal
|
||||
/// notification (mirrors ipc.Received.isChildExit).
|
||||
pub fn signalsFrom(badge: usize) ?SignalSet { ... }
|
||||
|
||||
/// Send `signal` to process `id`. Supervisor-gated, like kill; non-blocking.
|
||||
pub fn sendSignal(id: u32, signal: Signal) bool { ... }
|
||||
|
||||
/// The standard stop sequence: terminate, wait up to `deadline_ms` for the exit
|
||||
/// notification, then process_kill. The one call a supervisor needs.
|
||||
pub fn stop(id: u32, deadline_ms: u64) void { ... }
|
||||
|
||||
/// Subscribe `endpoint` to published exit events (process_subscribe). Every
|
||||
/// process death posts an asynchronous notification: badge = notify_exit_bit |
|
||||
/// process id — the same encoding a supervisor's exit notification uses, decoded
|
||||
/// by the same ipc.Received helpers. For stateful services: release what the dead
|
||||
/// client held (file handles, subscriptions, sockets). Ungated, like
|
||||
/// process_enumerate.
|
||||
pub fn subscribeExits(endpoint: usize) bool { ... }
|
||||
|
||||
/// How a process ended — queried after the exit notification (the kernel records
|
||||
/// it first, so the two never race). What restart policy reads. (Built in M17.2.)
|
||||
pub const ExitReason = enum(u8) {
|
||||
exited, // returned from main / clean exit
|
||||
aborted, // abort() — deliberate self-termination (SIGABRT's ghost; reserved)
|
||||
segmentation_fault, // SIGSEGV's ghost
|
||||
illegal_instruction, // SIGILL's ghost
|
||||
arithmetic_fault, // SIGFPE's ghost
|
||||
protection_fault, // general protection fault
|
||||
fault, // any other CPU exception
|
||||
killed, // process_kill
|
||||
};
|
||||
```
|
||||
|
||||
Two deliberate absences. There is no `mask`/`block` API — a process that is not
|
||||
ready for a signal simply has not waited on its endpoint yet; the pending mask *is*
|
||||
the blocked set. And there is no per-signal handler registration at this layer —
|
||||
dispatch is the process's own `switch` over `SignalSet`, or the service harness's
|
||||
callbacks (`on_terminate`, `on_reload`) for programs that want defaults.
|
||||
|
||||
### The service harness
|
||||
|
||||
`runtime.service` owns the `replyWait` loop and folds every event source — signals,
|
||||
child exits, protocol messages — into callbacks, with the vocabulary's defaults:
|
||||
`terminate` returns from the loop (clean exit), the common `ping` is answered automatically,
|
||||
`reload` is ignored unless overridden. One loop, no locking, nothing reentrant. A
|
||||
service author writes domain logic; the lifecycle contract is satisfied by the
|
||||
harness. A process that bypasses the harness and ignores its signals meets the
|
||||
deadline-then-kill escalation — you cannot force a process to implement an
|
||||
interface, but you can make compliance free and non-compliance fatal.
|
||||
|
||||
### The musl layer later
|
||||
|
||||
The POSIX C layer is a **musl port**: musl's arch/syscall layer retargeted so that
|
||||
what musl believes are kernel syscalls become danos runtime calls and IPC — `open`
|
||||
and `read` onto the VFS protocol, `kill`/`sigaction`/`waitpid` onto this document's
|
||||
vocabulary, `exit` onto the runtime's exit path. `sigaction` handlers registered
|
||||
through it are invoked by the runtime's loop when the signal message arrives —
|
||||
synchronous underneath, async-looking to ported code, delivered at wait boundaries
|
||||
the way most Unix programs already experience signals (at syscalls). No stack hijack
|
||||
ever happens, `SA_RESTART` semantics come free because nothing was interrupted, and
|
||||
SIGPIPE can be synthesized from `-EPEER` for the programs that expect it. C programs
|
||||
get POSIX; danos-native programs never pay for it.
|
||||
|
||||
## Increments
|
||||
|
||||
1. **Kernel: release device/port/IRQ claims and MSI vectors on death** — the
|
||||
cleanup half of iron rule 1, and the prerequisite for any restart story. Test:
|
||||
kill a claiming driver, spawn it again, the claim succeeds.
|
||||
2. **Exit reason in the death notification** (`ExitReason` above).
|
||||
3. **Exit events**: `process_subscribe` in the kernel (bounded subscriber table,
|
||||
publishes on every death), `runtime.process.subscribeExits`; the VFS becomes the
|
||||
first subscriber — releasing a dead client's handles is its proof test.
|
||||
4. **Signals**: `signal_bind` + `process_signal` + the pending mask in the kernel;
|
||||
`runtime.process` grows the interface above; the service harness handles
|
||||
`terminate` and answers the common `ping`; `stop()` for supervisors.
|
||||
|
||||
[device-manager.md](device-manager.md) builds directly on all four.
|
||||
|
||||
## Settled questions (2026-07-12)
|
||||
|
||||
- **Signal numbering is not ABI**: the runtime is the stable interface; the numbers
|
||||
beneath it are a private kernel ↔ runtime contract (see "The stable interface").
|
||||
- **Liveness is a `ping` call, not a signal**: signals are statements, questions
|
||||
are synchronous calls (see "Statements, not questions"). A service wanting *deep*
|
||||
health ("can I reach my hardware?") defines its own protocol message on top.
|
||||
- **Process handles: deferred.** Pids + the supervisor gate cover everything
|
||||
planned; transferable handles (Fuchsia-style, delegating signalling without
|
||||
delegating kill) wait for the capability table to grow types beyond endpoints.
|
||||
- **`alarm`: in the vocabulary, unbuilt.** No consumer yet; when one appears it is
|
||||
runtime sugar over the existing timer (arm a timer that posts your own signal) —
|
||||
zero kernel work, so deferring costs nothing.
|
||||
- **Subscription granularity: all exits**, subscriber-side filtering — one
|
||||
subscription per service, a bounded kernel table. Per-id subscriptions only if
|
||||
event volume ever matters (hundreds of processes, not before).
|
||||
- **Client identity across the exit boundary: no convention needed** — an IPC
|
||||
sender's badge already is its task id (see "Who learns of a death").
|
||||
@@ -0,0 +1,120 @@
|
||||
# Process Management
|
||||
|
||||
How danos lists, supervises, and kills processes — the microkernel answer to
|
||||
`ps`, `kill`, and `SIGCHLD`/`wait`.
|
||||
|
||||
## Why system calls, not `/proc`
|
||||
|
||||
Unix systems sit on a spectrum. Classic BSD/macOS list processes through
|
||||
syscalls (`sysctl(KERN_PROC)`) and kill through `kill(2)`; Linux renders the
|
||||
process table as `/proc` for *reading* but still kills through a syscall; Plan 9
|
||||
made the file tree the whole interface (`echo kill > /proc/n/ctl`). Microkernels
|
||||
mostly abandon ambient PIDs: Minix and QNX route everything through a user-space
|
||||
process-manager server, and Fuchsia/seL4 control processes only through handles.
|
||||
|
||||
danos rules out `/proc` **as the primitive**: here a `/proc` would be served by
|
||||
the VFS server — a user process — which would put the VFS in the path of process
|
||||
control. If the VFS (or anything under it) hangs, nothing could be listed or
|
||||
killed, *including the hung VFS*. The control plane for processes must not
|
||||
depend on a process. So the primitives are kernel system calls; a read-only
|
||||
`/proc` rendering can be layered on later, and a POSIX-style process-manager
|
||||
server can be built *from* these primitives when one is needed.
|
||||
|
||||
## The three primitives
|
||||
|
||||
### `process_enumerate(buffer, maximum) -> total`
|
||||
|
||||
A snapshot of the task table into a caller buffer of `abi.ProcessDescriptor`
|
||||
(id, supervisor, state, priority, name) — the exact shape of
|
||||
`device_enumerate`, so `ps` is a user program over a snapshot, not a kernel
|
||||
service. The total may exceed what fit; call again with a larger buffer. Kernel
|
||||
tasks are included with an empty name — an honest listing shows the idle tasks
|
||||
too. Ungated and read-only: what is running is not a secret between cooperating
|
||||
bring-up processes.
|
||||
|
||||
### `system_spawn(..., exit_endpoint) -> child id`, and the supervision link
|
||||
|
||||
`system_spawn` records the caller as the child's **supervisor** and returns the
|
||||
child's process id (ids are monotonic, never reused — a stale id can only miss).
|
||||
That link is the kill authority: it answers "who may kill process 7?" without
|
||||
inventing users or permissions, the same way a device *claim* is the capability
|
||||
for `mmio_map`. It composes with the supervision hierarchy the device manager
|
||||
already forms: init supervises the services it starts, the device manager
|
||||
supervises the drivers it matches. (A transferable process *handle* — Fuchsia
|
||||
style — can replace the id once the handle table grows types beyond endpoints.)
|
||||
|
||||
`exit_endpoint` (a handle, or `abi.no_cap`) is the supervisor's death-watch: when
|
||||
the child ends — clean exit, CPU fault, or `process_kill` — the kernel posts an
|
||||
asynchronous notification to that endpoint, exactly like a bound IRQ. The badge
|
||||
carries `abi.notify_badge_bit | abi.notify_exit_bit | child_id`, so one endpoint
|
||||
supervises many children and can even share with IRQ notifications. This is the
|
||||
microkernel's SIGCHLD: no new mechanism, just the IRQ-as-IPC pattern reused, and
|
||||
a supervisor's event loop (`ipc.replyWait`) already knows how to receive it. The
|
||||
child holds a reference to the endpoint from birth, so the notification cannot
|
||||
dangle even if the supervisor dies first.
|
||||
|
||||
### `process_kill(id) -> 0 / -ESRCH / -EPERM`
|
||||
|
||||
Only the supervisor may kill; kernel tasks are not killable processes. Like a
|
||||
signal, delivery is prompt but asynchronous — 0 means the kill is accepted and
|
||||
irrevocable; the exit notification confirms completion.
|
||||
|
||||
## How a kill lands (the kernel mechanics)
|
||||
|
||||
Everything below runs under the big kernel lock, where task states cannot move.
|
||||
|
||||
- **Target ready or blocked** (not on any core): reaped on the killer's own
|
||||
call. The reap releases what death always releases (IRQ bindings first, then
|
||||
a client the target still owed a reply to is failed with `-EPEER`, IPC handles
|
||||
closed, the exit notification posted last) — plus the unlinking only a
|
||||
*remote* death needs: out of the ready queue, out of an endpoint's sender FIFO
|
||||
(`Task.ipc_wait_endpoint`), out of a receive wait queue (`Task.wait_queue`),
|
||||
and out of any server's owed-reply slot, so nothing ever dequeues a dangling
|
||||
pointer. Destroying the address space is safe because no core can have it
|
||||
loaded: every switch away from a task loads the next task's tables.
|
||||
- **Target running on another core**: it cannot be torn down mid-instruction,
|
||||
so it is condemned (`Task.kill_pending`) and dies at whichever comes first:
|
||||
- its next **system_call entry** — checked before dispatch, so a condemned
|
||||
process cannot spawn, claim, or message anything on its way out;
|
||||
- its core's next **timer tick** — but only when the task is not inside one
|
||||
of its own system calls (`Task.in_system_call`): the tick may have
|
||||
interrupted kernel code mid-operation, where teardown would leak whatever
|
||||
the operation held. User-mode execution is always a safe kill point. The
|
||||
tick-time terminate abandons the interrupt frame exactly like the fault
|
||||
path (the LAPIC is acknowledged before the tick hook runs);
|
||||
- any core's tick finding it **blocked or ready** (it entered a syscall and
|
||||
parked after being condemned) — reaped by the same remote-reap path.
|
||||
|
||||
A pure user-mode spin loop that never makes a system call therefore dies
|
||||
within one tick; nothing a process does can outrun the kill.
|
||||
|
||||
The scheduler stays below the process layer: finishing a kill (IRQ bindings,
|
||||
handles, the notification) is called *up* through two hooks process.zig
|
||||
registers at boot (`terminate_current_hook`, `reap_task_hook`), mirroring how
|
||||
the architecture layer calls up into `tick`.
|
||||
|
||||
## Known gaps (bring-up honesty)
|
||||
|
||||
- ~~Device claims are not released on death~~ Closed (M17.1): every path out of a
|
||||
process releases its device claims alongside its IRQ and MSI bindings
|
||||
(`releaseTaskResourcesLocked`), so a restarted driver can claim its hardware
|
||||
again — the cleanup half of [process-lifecycle.md](process-lifecycle.md)'s iron
|
||||
rule 1. The `claim-release` test proves the kill → release → re-claim cycle.
|
||||
- Kernel stacks of dead tasks are leaked, as on every exit path (no reaper yet).
|
||||
- ~~There is no exit status in the notification~~ Closed (M17.2): the kernel
|
||||
records how every process ends — exited, a fault class, or killed — before it
|
||||
posts the exit notification, and the supervisor reads it with
|
||||
`process_exit_reason` (`runtime.process.exitReason`). This is the input to
|
||||
restart policy ([process-lifecycle.md](process-lifecycle.md)); an exit *code*
|
||||
for the clean case can still ride alongside later.
|
||||
- Enumerate writes through the caller's raw pointer under the bring-up trust
|
||||
model, like `device_enumerate` (an unmapped page is a self-DoS, not an
|
||||
isolation break).
|
||||
|
||||
## Tests
|
||||
|
||||
`process-list` (enumerate), `process-kill` (kernel-level kill paths, refusals,
|
||||
notifications), `supervision` (the whole user-side surface via the process-test
|
||||
service: spawn supervised → enumerate → kill blocked and spinning children →
|
||||
notifications → gone), `claim-release` (a killed claim-holder's device is
|
||||
claimable again). See test/qemu_test.py.
|
||||
+16
-3
@@ -1,6 +1,17 @@
|
||||
# Resilience: fault isolation and live restart
|
||||
|
||||
A design/research note, not built yet. This is the property danos is really chasing:
|
||||
Steps 1–4 of the ordering below are **built** (M17–M18, 2026-07-13): user-mode
|
||||
isolation; fault → kill the process → keep the core (`onException`; the
|
||||
`fault-recovery` test); the supervisor notification **with exit reasons**
|
||||
([process-lifecycle.md](process-lifecycle.md) — clean exit, fault class, or
|
||||
killed, recorded before the notice posts); and the **restart policy itself**
|
||||
([device-manager.md](device-manager.md)): the device manager supervises every
|
||||
driver, restarts crashes with backoff, caps crash loops, and re-claims work
|
||||
because the kernel releases a dead process's claims. The `driver-restart` and
|
||||
`usb-report` scenarios prove kill → release → respawn → re-claim → re-report
|
||||
end to end. What remains of this document's ladder is scope, not mechanism:
|
||||
more of the system moved into restartable processes (the discovery migration,
|
||||
[m19-m20-plan.md](m19-m20-plan.md), is the next rung). This is the property danos is really chasing:
|
||||
**if a part of the OS breaks, isolate it, and re-initialise it — without rebooting.**
|
||||
A crashed driver gets restarted; a wedged service gets killed and brought back. It's
|
||||
the reason the [microkernel](vision.md) shape was chosen, and it's a *separate* goal
|
||||
@@ -111,9 +122,11 @@ Honest boundaries:
|
||||
## Suggested ordering
|
||||
|
||||
1. **User mode + address-space isolation** — the shared prerequisite (also on the
|
||||
path for everything else).
|
||||
path for everything else). **Done.**
|
||||
2. **Kernel: fault → kill process → notify.** Turn today's "halt on fault" into
|
||||
"confine to the process and report it."
|
||||
"confine to the process and report it." **Done** (the kill and reclaim; the
|
||||
supervisor notification waits for step 3's supervisor). A killed server's
|
||||
pending client is unblocked with `-EPEER` rather than hung.
|
||||
3. **A minimal supervisor server** that can (re)start a process.
|
||||
4. **Resource cleanup on death** — reclaim memory/MMIO/IPC/IRQ, via caps or a grant
|
||||
table.
|
||||
|
||||
+23
-2
@@ -6,8 +6,8 @@ ready task always runs, and tasks at the same priority take turns. That model is
|
||||
chosen for [real-time](vision.md) — it's predictable (you can reason about which
|
||||
task runs when) and its decisions are O(1), unlike a fair-share scheduler.
|
||||
|
||||
The scheduler proper (`src/kernel/sched.zig`) is generic; the context switch and new-task
|
||||
stack setup are architecture-specific (`src/kernel/arch/x86_64/`, see [arch](arch.md)).
|
||||
The scheduler proper (`system/kernel/sched.zig`) is generic; the context switch and new-task
|
||||
stack setup are architecture-specific (`system/kernel/architecture/x86_64/`, see [arch](arch.md)).
|
||||
|
||||
## Tasks
|
||||
|
||||
@@ -67,6 +67,27 @@ exist, which is what a real-time scheduler needs.
|
||||
- **Round-robin within a level.** When a task is descheduled it goes to the *back*
|
||||
of its level's queue, so equal-priority tasks share the CPU fairly.
|
||||
|
||||
## Affinity: pinning a task to a core
|
||||
|
||||
By default a task runs on **any** core — the ready queue above is global, and any
|
||||
idle core pulls the highest-priority task from it (work-conserving; see
|
||||
[smp.md](smp.md)). A task can instead be **pinned** to one core with
|
||||
`spawnOn(entry, priority, cpu)`, giving it an *affinity*: it will only ever run
|
||||
there, never migrating.
|
||||
|
||||
Mechanically, each core has its **own** pinned queue (same 8-level FIFO + bitmap)
|
||||
alongside the global one. A pinned task is enqueued only into its core's pinned
|
||||
queue; selection compares the top of the global queue and the running core's pinned
|
||||
queue and takes the higher priority (still O(1) — two bit-scans and a compare), with
|
||||
a pinned task winning an equal-priority tie so it can't be starved by global work.
|
||||
Because every queue is mutated under the [big kernel lock](smp.md), one core enqueuing
|
||||
into another core's pinned queue is safe.
|
||||
|
||||
This is the *explicit-affinity* model (no surprise migration mid-deadline), which is
|
||||
the more real-time-predictable direction. `spawnOn` refuses to pin to an offline or
|
||||
out-of-range core — it creates the task unpinned instead, so it still runs somewhere
|
||||
rather than stranding in a queue no core services, and returns whether the pin took.
|
||||
|
||||
## Sleeping and the idle task
|
||||
|
||||
A task can **block** — give up the CPU until an event, rather than busy-wait
|
||||
|
||||
+114
-7
@@ -16,10 +16,14 @@ danos specifics):
|
||||
- **Real-time** — whether timing is *predictable*. Comes from bounded operations
|
||||
(our O(1) scheduler), not from core count.
|
||||
|
||||
danos is uniprocessor today: one global `current` task, one set of ready queues, one
|
||||
timer. Even on an 8-core CPU, the firmware starts only the **bootstrap processor
|
||||
(BSP)**; the other cores (**application processors**, APs) sit parked until the
|
||||
kernel wakes them, which it doesn't yet.
|
||||
danos now runs on multiple cores. The firmware starts only the **bootstrap processor
|
||||
(BSP)**; the kernel wakes the other cores (**application processors**, APs) with
|
||||
INIT–SIPI–SIPI, brings each up into 64-bit long mode with its own descriptor tables,
|
||||
LAPIC and timer, and drops it into the scheduler. Tasks run **genuinely in parallel** —
|
||||
the `smp` self-test confirms worker tasks executing on all four cores at once under
|
||||
QEMU `-smp 4`. Shared kernel state (scheduler queues, IPC) is serialised behind a big
|
||||
kernel lock. What's left is refinement, not first-light: per-core run queues, IPIs,
|
||||
and thread-to-core affinity (see [Implementation status](#implementation-status)).
|
||||
|
||||
## The common microkernel instinct: don't share kernel state
|
||||
|
||||
@@ -114,14 +118,20 @@ active reconsideration in favour of resilience — see [vision.md](vision.md).)
|
||||
Whatever the top goal, the *sequence* is the same and seL4 validates starting simple:
|
||||
|
||||
1. **Enumerate cores** — needs [device discovery](discovery.md) (ACPI MADT on x86,
|
||||
device tree on ARM). SMP is a concrete consumer of that work.
|
||||
device tree on ARM). SMP is a concrete consumer of that work. **Done on x86:** the
|
||||
MADT parse records every usable Local APIC — with the `apic_id` an AP wake targets —
|
||||
and `platform.cpus()` returns the list (see [discovery.md](discovery.md)). The boot
|
||||
log reports the count; the ARM (device-tree) path still needs it.
|
||||
2. **Wake the APs** — INIT–SIPI–SIPI on x86; PSCI/spin-tables on ARM. Each core brings
|
||||
up its own tables, timer, and idle task.
|
||||
up its own tables, timer, and idle task. **Done on x86** — cores climb to long mode,
|
||||
set up their own GDT/TSS, and enter the scheduler; tasks run in parallel across all
|
||||
cores ([status](#implementation-status)).
|
||||
3. **Start with a big kernel lock.** It's a legitimate first design, not a shortcut —
|
||||
philosophically aligned with a tiny kernel, and it lets the single-core correctness
|
||||
model you already have (the interrupt-flag discipline in
|
||||
[scheduling.md](scheduling.md)) stay largely intact: one lock around kernel entry
|
||||
instead of rethinking every critical section.
|
||||
instead of rethinking every critical section. **Done** — see
|
||||
`system/kernel/sync.zig`.
|
||||
4. **Later, if contention bites,** evolve toward **per-core run queues + explicit
|
||||
affinity** (the Fiasco.OC direction) — also the more real-time-predictable model.
|
||||
5. **Placement stays a user-space policy** — the kernel runs a thread on the core it's
|
||||
@@ -130,6 +140,103 @@ Whatever the top goal, the *sequence* is the same and seL4 validates starting si
|
||||
Big-lock-first → per-core-later. The affinity/MCS depth is only worth it if real-time
|
||||
turns out to be the actual goal.
|
||||
|
||||
## Implementation status
|
||||
|
||||
The "wake + schedule" build (real parallel task execution) is going in as a sequence
|
||||
of green checkpoints — each step keeps the single-core test suite passing before the
|
||||
next lands.
|
||||
|
||||
**Done:**
|
||||
|
||||
- **Core enumeration** — the MADT parse records every usable Local APIC (with its
|
||||
`apic_id`, which an AP wake targets); `platform.cpus()` returns the list. See
|
||||
[discovery.md](discovery.md).
|
||||
- **The big kernel lock** (`system/kernel/sync.zig`) — one coarse spinlock guarding the
|
||||
scheduler queues and IPC, always held with local interrupts disabled. It is held
|
||||
*across* a context switch and released by whichever task resumes (the hand-off
|
||||
rule); `task_trampoline` releases it for a freshly-spawned task. `scheduler.zig` and
|
||||
`ipc.zig` run every critical section under it. Uncontended on one core, so behaviour
|
||||
is identical to the old interrupt-flag model.
|
||||
- **Per-CPU state** — a `PerCpu` struct (running task, idle task, APIC id) per core,
|
||||
its pointer kept in the x86 **GS base** (`IA32_GS_BASE`; no `swapgs`, since there's
|
||||
no user mode yet). The old global `current` is now `thisCpu().current`. The ready
|
||||
queues stay **global** under the lock — work-conserving, so any idle core will pull
|
||||
the highest-priority ready task; per-core queues are a later optimisation.
|
||||
- **AP wake to long mode** — `arch.startSecondary` drives INIT–SIPI–SIPI (via the
|
||||
LAPIC ICR) to wake each parked core one at a time. A woken core starts in 16-bit
|
||||
real mode at a low page and runs the [trampoline](../system/kernel/architecture/x86_64/trampoline.s)
|
||||
up through protected mode into 64-bit long mode, then lands in `smp.zig:apEntry`,
|
||||
publishes its per-CPU pointer, and reports in. Verified in QEMU with `-smp 4`:
|
||||
all four cores report `online`.
|
||||
|
||||
The trampoline earns its complexity from four hardware facts:
|
||||
- a STARTUP IPI vectors a core to physical `vector << 12` (a *byte* vector), so the
|
||||
trampoline must live **below 1 MiB** — the kernel reserves that page from the frame
|
||||
allocator at boot, before paging/heap draw down the scarce low frames;
|
||||
- the blanket RAM identity map is **NX** (W^X), but the AP fetches the trampoline
|
||||
from it under paging, so that one page is made executable for bring-up;
|
||||
- the blob is copied to a page whose address isn't known at link time, so it is
|
||||
**position-independent**: it derives its own base from `CS` and, crucially,
|
||||
addresses data *segment-relative in real mode* (where the segment base already
|
||||
supplies the page base) but *base-register-relative in protected/long mode* (flat
|
||||
segments, base 0). Getting that distinction wrong was the first bug found;
|
||||
- an AP starts with a bare `CR0`/`CR4`, but the kernel is built **with SSE** (the
|
||||
x86_64 baseline) and the compiler emits SSE for things as ordinary as a struct
|
||||
copy — so the trampoline must set `CR4.OSFXSR`/`OSXMMEXCPT` and fix `CR0.EM`/`MP`,
|
||||
or the first SSE instruction on the AP `#UD`s. The BSP inherited those bits from
|
||||
UEFI; the AP has to set them itself. This was the second bug — it masqueraded as a
|
||||
fault in `lgdt` (the first kernel code after entry that the compiler vectorised).
|
||||
|
||||
- **Per-core tables + scheduler entry** — each AP loads **its own GDT** (with its own
|
||||
TSS descriptor) and **its own TSS** (its own IST/`rsp0` stack), loads the shared
|
||||
IDT, enables its LAPIC and timer, then calls the generic `secondaryMain`: it turns
|
||||
its bring-up context into the core's idle task (as task 0 is for the BSP), marks the
|
||||
core online, and enters the run loop. With interrupts on, each core's own timer tick
|
||||
preempts its idle context into whatever the global ready queue offers — so all cores
|
||||
pull real work in parallel. The `smp` test spawns CPU-bound workers and confirms they
|
||||
execute on all four cores at once, and `fault-ap-df` pins a #DF to an AP and checks
|
||||
that core catches it on **its own** IST (a broken per-core TSS would triple-fault) —
|
||||
reported as "core N: …", so a fault is always attributed to the core it happened on,
|
||||
and is contained to that core (the rest of the system keeps running).
|
||||
- **Thread affinity** — `spawnOn(entry, priority, cpu)` pins a task to a core (its own
|
||||
per-core pinned queue, merged with the global queue at selection; see
|
||||
[scheduling.md](scheduling.md#affinity-pinning-a-task-to-a-core)). The `affinity`
|
||||
test confirms a pinned task never migrates. This is the mechanism the fault-on-AP
|
||||
test rides on, and the *explicit-affinity* real-time-predictable model.
|
||||
- **Right-sized footprint** — the per-CPU ceiling (`system.max_cpus`, one constant
|
||||
shared by discovery, the scheduler, and the per-core GDT/TSS) is generous (128), but
|
||||
the *large* per-core resources — the kernel and IST (double-fault) stacks — are
|
||||
**heap-allocated at bring-up**, only for cores that actually come online. Only the
|
||||
BSP's IST stack is static, because it must exist before the frame allocator does.
|
||||
This kept the kernel image small (a static `[128][16 KiB]` IST array would have been
|
||||
2 MiB of `.bss`); it's a few tens of KiB instead.
|
||||
- **`single_threaded` off** — the kernel was built `single_threaded = true`, which
|
||||
compiles `std.atomic` down to plain non-atomic ops. Harmless on one core, but it
|
||||
quietly breaks the big kernel lock across cores; it's now `false`.
|
||||
- **Re-armable wake + retry** — the trampoline frame is reserved for the system's
|
||||
life, but kept **inert between wakes**: zeroed and non-executable, armed (blob
|
||||
copied in, page made executable) only for the moment a core is actually climbing,
|
||||
then disarmed again. So there's never a dormant executable page, and a core can be
|
||||
(re)woken at any time — `arch.startSecondary` is one self-contained attempt (arm →
|
||||
INIT–SIPI–SIPI → disarm), and its `INIT` resets a wedged core, so retrying just
|
||||
works. Boot retries a non-responding core up to three times; the same primitive is
|
||||
the groundwork a future **power manager** would drive to bring cores up (and,
|
||||
eventually, its counterpart to take them offline — which additionally needs the
|
||||
core's tasks migrated off first).
|
||||
|
||||
**Next (refinement, not first-light):**
|
||||
|
||||
- **IPIs** — cross-core wake/preempt. Not needed for correctness: an idle core wakes
|
||||
on its own timer tick and pulls ready work then; IPIs only cut that latency from
|
||||
≤1 ms to near-instant.
|
||||
- **Per-core run queues** — the Fiasco.OC direction, if the single global queue's lock
|
||||
contention ever bites. (Thread *affinity* already exists — see above; this is the
|
||||
further step of giving each core its own primary run queue for load distribution.)
|
||||
- **Fault recovery** — today a fault halts (only) the faulting core. Turning that into
|
||||
"kill the task, keep the core running" is the [resilience](resilience.md) track (it
|
||||
needs the task's lock/resource state handled), and for taking a core fully offline,
|
||||
its tasks migrated first.
|
||||
|
||||
## Further reading
|
||||
|
||||
**Microkernel SMP & scheduling**
|
||||
|
||||
+13
-1
@@ -1,5 +1,15 @@
|
||||
# System Calls
|
||||
System calls (syscalls) are the bridge between your programs and the operating system's restricted core (kernel).
|
||||
System calls (syscalls) are the bridge between your programs and the operating system's restricted core (kernel).
|
||||
|
||||
> **Status:** danos has real user processes (M3). User programs enter the kernel
|
||||
> via the `syscall` instruction (STAR/LSTAR/SFMASK set per core; the entry stub in
|
||||
> `isr.s` does the `swapgs` + kernel-stack switch and reuses the interrupt
|
||||
> dispatcher). The `int 0x80` gate is kept alongside as a minimal test path. The
|
||||
> current call set is still a placeholder — `0 = exit(code)`, `1 = ping`,
|
||||
> `2 = write(ptr, len)`, `3 = sleep(ms)` (see `system/kernel/process.zig`); the
|
||||
> handler dispatches on whether the caller is a scheduled process (its own address
|
||||
> space) or a borrowed test thread. The microkernel set below (IPC_Call /
|
||||
> IPC_ReplyWait / Yield) replaces it once a second user server exists.
|
||||
|
||||
## The Mechanism of a Syscall
|
||||
|
||||
@@ -37,6 +47,8 @@ Everything else---including`read()`,`write()`,`malloc()`, and`fork()`---will run
|
||||
- **What it does:**Used strictly by your background user-space servers (like your disk driver or filesystem). It sends a reply to the last client that called it, and immediately puts the server to sleep until the next request arrives.[[1](https://news.ycombinator.com/item?id=33078441)]
|
||||
3. **`Yield()`/`Thread_Ctrl()`**
|
||||
- **What it does:**Allows a thread to voluntarily give up its CPU time slice, or allows a root task to spawn/kill threads.
|
||||
4. **`ipc_send(endpoint, message_buffer)`(Asynchronous Send)**
|
||||
- **What it does:**Posts a small payload to an endpoint's bounded queue and returns *without* blocking — no rendezvous, no reply. The receiver picks it up through the same `IPC_ReplyWait`, as a buffered message. It is the async counterpart of `IPC_Call`, for one-to-many broadcasts where a synchronous rendezvous would let one dead or slow receiver hang the sender. The [input service](input.md) — keyboard-event fan-out — is its first user. A full queue drops the oldest message (a buffered message is discrete data, unlike a coalescing interrupt notification).
|
||||
|
||||
* * * * *
|
||||
|
||||
|
||||
+36
-4
@@ -1,6 +1,6 @@
|
||||
# SysV: the kernel's calling convention
|
||||
|
||||
Several places in danos say "the kernel is SysV" — most visibly `src/root.zig`:
|
||||
Several places in danos say "the kernel is SysV" — most visibly `system/boot-handoff.zig`:
|
||||
|
||||
```zig
|
||||
pub const kernel_abi: std.builtin.CallingConvention = .{ .x86_64_sysv = .{} };
|
||||
@@ -52,19 +52,51 @@ argument arrives in **RCX**, not RDI.
|
||||
|
||||
danos's two binaries default to different conventions:
|
||||
|
||||
- `src/boot/efi.zig` is built for the UEFI target, so its default C convention is
|
||||
- `boot/efi.zig` is built for the UEFI target, so its default C convention is
|
||||
Microsoft x64 (first argument → RCX).
|
||||
- The kernel is freestanding, so its convention is SysV (first argument → RDI).
|
||||
|
||||
When the loader jumps to the kernel passing the `BootInfo` pointer, both sides have
|
||||
to agree *which register that pointer lands in*. Left to their defaults, the loader
|
||||
would place it in RCX while the kernel looked in RDI — and the kernel would read
|
||||
garbage. So both sides reference the same `danos.kernel_abi` (SysV): the loader's
|
||||
garbage. So both sides reference the same `system.kernel_abi` (SysV): the loader's
|
||||
function-pointer type and the kernel's `_start` both carry
|
||||
`callconv(danos.kernel_abi)`, and the pointer reliably arrives in RDI. That is the
|
||||
`callconv(system.kernel_abi)`, and the pointer reliably arrives in RDI. That is the
|
||||
whole reason `kernel_abi` lives in the shared contract — see [efi.md](efi.md) for
|
||||
the handoff it governs.
|
||||
|
||||
## The process-entry stack (argc/argv)
|
||||
|
||||
The SysV ABI also fixes what a *fresh process* finds on its stack — and danos
|
||||
follows it, so its own runtime and any future C libc read arguments the same way.
|
||||
At the first user instruction, `rsp` is 16-byte aligned and points at (addresses
|
||||
growing upward):
|
||||
|
||||
```
|
||||
rsp → argc u64
|
||||
argv[0] … argv[argc-1] pointers into the strings area below
|
||||
NULL argv terminator
|
||||
NULL envp terminator (no environment yet)
|
||||
{AT_PAGESZ, page size} auxiliary vector
|
||||
{AT_NULL, 0} auxiliary-vector terminator
|
||||
argv string bytes NUL-terminated
|
||||
───────────────────────── stack top (stack_top_virtual)
|
||||
```
|
||||
|
||||
The kernel builds this block at the top of the process's stack — 8 pages (32 KiB,
|
||||
`parameters.user_stack_pages`) mapped RW+NX below a fixed top, with the page below
|
||||
them left unmapped as a **guard**, so a stack overflow faults (killing only that
|
||||
process) instead of silently corrupting the image
|
||||
(`buildEntryStack` in `system/kernel/process.zig`); `argv[0]` is always the path
|
||||
or initial-ramdisk name the process was spawned as, and `system_spawn`'s optional
|
||||
argument blob becomes `argv[1..]`. The runtime's `_start`
|
||||
(`library/runtime/start.zig`) hands the block to `rt_start`, which builds a
|
||||
`runtime.process.Init` from it and passes that to the program's `main`
|
||||
(`pub fn main(init: runtime.process.Init)`; a parameterless `main()` is also
|
||||
accepted). A C runtime's `crt0` would walk
|
||||
the identical layout unmodified — that's the compatibility being bought. The
|
||||
`args` test proves the round trip.
|
||||
|
||||
## Where else it surfaces
|
||||
|
||||
- **The red zone → `red_zone = false`.** `build.zig` disables the red zone for the
|
||||
|
||||
+7
-6
@@ -8,8 +8,9 @@ without a human staring at the screen.
|
||||
There are two layers:
|
||||
|
||||
- **Host unit tests** (`zig build test`) — for pure, platform-independent logic in
|
||||
the shared `danos` module (the handoff layout in `src/root.zig`). These compile
|
||||
for the host and run natively.
|
||||
the shared contracts (`system/boot-handoff.zig`, `system/abi.zig`,
|
||||
`system/devices/device-abi.zig`), which also compile-checks the three-way split
|
||||
stays self-consistent. These compile for the host and run natively.
|
||||
- **QEMU integration tests** (`python3 test/qemu_test.py`) — boot the real kernel
|
||||
and check its behaviour. This is the interesting part.
|
||||
|
||||
@@ -17,7 +18,7 @@ There are two layers:
|
||||
|
||||
The framebuffer console draws pixels, which a test can't read without
|
||||
screen-scraping. So the kernel also writes everything to a **serial port**
|
||||
(`src/kernel/arch/x86_64/serial.zig`, a 16550 UART on COM1). `Console.write` mirrors every
|
||||
(`system/kernel/architecture/x86_64/serial.zig`, a 16550 UART on COM1). `Console.write` mirrors every
|
||||
byte to it, so all kernel output — boot log, memory summary, exception reports —
|
||||
appears on serial as plain text.
|
||||
|
||||
@@ -29,7 +30,7 @@ a new architecture's UART is what makes the same tests run there.
|
||||
## In-kernel test cases
|
||||
|
||||
Building with `-Dtest-case=<name>` makes the kernel, after normal bring-up, run one
|
||||
self-test from `src/kernel/tests.zig` instead of idling. Each case writes structured
|
||||
self-test from `system/kernel/tests.zig` instead of idling. Each case writes structured
|
||||
markers to serial:
|
||||
|
||||
```
|
||||
@@ -108,7 +109,7 @@ firmware, boot method, serial device). The cases are architecture-neutral —
|
||||
So bringing up a second architecture — an AArch64 Raspberry Pi is the motivating
|
||||
one — means:
|
||||
|
||||
1. implement `src/kernel/arch/aarch64/` (CPU ops, its UART, exception vectors, page
|
||||
1. implement `system/kernel/arch/aarch64/` (CPU ops, its UART, exception vectors, page
|
||||
tables) behind the same `arch` interface,
|
||||
2. add an `aarch64` entry to `ARCHES` with its `qemu-system-aarch64` invocation,
|
||||
|
||||
@@ -118,7 +119,7 @@ architectures".
|
||||
|
||||
## Writing a new case
|
||||
|
||||
1. Add a function to `src/kernel/tests.zig` and dispatch it in `run` on its name.
|
||||
1. Add a function to `system/kernel/tests.zig` and dispatch it in `run` on its name.
|
||||
2. Emit `[PASS]/[FAIL]` lines and a `DANOS-TEST-RESULT:` line (non-faulting cases),
|
||||
or trigger the condition and rely on the handler's output (faulting cases).
|
||||
3. Add an entry to `CASES` in `test/qemu_test.py` with the regex that proves it.
|
||||
|
||||
+13
-4
@@ -81,11 +81,20 @@ prerequisites.
|
||||
(with boot-services memory reclaimed), [paging](paging.md) with W^X, [exceptions and
|
||||
interrupts](interrupts.md), a [calibrated timer + ns clock](device-interrupts.md), a
|
||||
[heap](heap.md), a [fixed-priority preemptive scheduler](scheduling.md) with blocking,
|
||||
and in-kernel [IPC channels](ipc.md) — plus a [test harness](testing.md).
|
||||
in-kernel [IPC channels](ipc.md), SMP (all cores scheduling, with affinity), a
|
||||
**higher-half kernel** with a physmap, and **user space**: per-process address
|
||||
spaces, `syscall`/`sysret` with the `swapgs` discipline, a user-ELF loader, and
|
||||
`/system/services/init` — a real user ELF built from `system/services/init/`, running at CPL 3 as PID 1 on its
|
||||
own page tables — plus a [test harness](testing.md).
|
||||
|
||||
- **Isolation track** — **user mode + address-space isolation** (higher-half kernel,
|
||||
ring 3, per-process page tables). The substrate everything else needs. *Next, and a
|
||||
prerequisite for the resilience and driver tracks.*
|
||||
- **Isolation track** — **user mode + address-space isolation**. *Done: a
|
||||
higher-half kernel with a physmap (the low half is user space), per-process
|
||||
address spaces with CR3 switched on context switch, the `swapgs` discipline,
|
||||
`syscall`/`sysret`, a user-ELF loader, and `/system/services/init` running as a real
|
||||
preemptive ring-3 process (PID 1). Remaining polish: an address-space/stack
|
||||
reaper for exited tasks, SMAP + fault-recovering copy-in/out, the real IPC
|
||||
syscalls (IPC_Call/IPC_ReplyWait — they arrive with the second user server),
|
||||
and TLB shootdown once a process has more than one thread.*
|
||||
- **Resilience track** — fault → kill → notify, a supervisor/reincarnation server,
|
||||
resource cleanup on death, then a restartable driver as proof. Needs isolation.
|
||||
See [resilience.md](resilience.md).
|
||||
|
||||
@@ -0,0 +1,80 @@
|
||||
//! /lib/mmio — typed volatile MMIO register access, plus the memory-ordering
|
||||
//! barriers a device driver needs. Used by drivers on top of an `mmio_map` grant.
|
||||
//!
|
||||
//! **`volatile` is not a barrier.** In Zig it means only: don't elide this access, and
|
||||
//! don't reorder it against *other volatile* accesses. It says nothing about ordinary
|
||||
//! stores — the DMA descriptor you just filled in write-back RAM — which the compiler
|
||||
//! (and, on weakly-ordered hardware, the CPU) may freely move past a volatile MMIO
|
||||
//! write. The canonical bug:
|
||||
//!
|
||||
//! ring[i] = descriptor; // ordinary store to WB RAM
|
||||
//! doorbell.* = i; // volatile store to UC MMIO
|
||||
//! // nothing orders these; the device can read a stale descriptor
|
||||
//!
|
||||
//! Put a `wmb()` between them. The barriers lower per-architecture — which is the whole
|
||||
//! reason they are a named primitive and not scattered `asm volatile`:
|
||||
//!
|
||||
//! x86_64 aarch64
|
||||
//! mb() mfence dsb sy
|
||||
//! rmb() lfence dsb ld
|
||||
//! wmb() sfence dsb st
|
||||
//!
|
||||
//! x86 is forgiving (TSO + strong-uncacheable MMIO), so a compiler barrier usually
|
||||
//! suffices; ARM is not, and ARM is the win condition (docs/vision.md) — so the
|
||||
//! abstraction exists now, while there is one caller (hpet) to get right. See
|
||||
//! docs/driver-model.md (M14) for the full ordering contract.
|
||||
|
||||
const builtin = @import("builtin");
|
||||
|
||||
/// Read a register of type `T` at absolute virtual address `addr` — a location inside
|
||||
/// a device's `mmio_map` grant. `volatile`: never elided, never reordered against
|
||||
/// another volatile access.
|
||||
pub inline fn read(comptime T: type, addr: usize) T {
|
||||
return @as(*const volatile T, @ptrFromInt(addr)).*;
|
||||
}
|
||||
|
||||
/// Write `value` of type `T` to the register at absolute virtual address `addr`.
|
||||
pub inline fn write(comptime T: type, addr: usize, value: T) void {
|
||||
@as(*volatile T, @ptrFromInt(addr)).* = value;
|
||||
}
|
||||
|
||||
/// Full barrier: all loads and stores before it are globally visible before any after
|
||||
/// it. Use when an MMIO write must complete before a following read.
|
||||
pub inline fn mb() void {
|
||||
switch (builtin.target.cpu.arch) {
|
||||
.x86_64 => asm volatile ("mfence" ::: .{ .memory = true }),
|
||||
.aarch64 => asm volatile ("dsb sy" ::: .{ .memory = true }),
|
||||
else => @compileError("mmio.mb: unsupported architecture"),
|
||||
}
|
||||
}
|
||||
|
||||
/// Read barrier: loads before it complete before loads after it. Use after an IRQ
|
||||
/// wake, before reading what the device wrote to shared memory.
|
||||
pub inline fn rmb() void {
|
||||
switch (builtin.target.cpu.arch) {
|
||||
.x86_64 => asm volatile ("lfence" ::: .{ .memory = true }),
|
||||
.aarch64 => asm volatile ("dsb ld" ::: .{ .memory = true }),
|
||||
else => @compileError("mmio.rmb: unsupported architecture"),
|
||||
}
|
||||
}
|
||||
|
||||
/// Write barrier: stores before it become visible before stores after it. Use between
|
||||
/// filling a DMA descriptor in RAM and ringing the device's doorbell.
|
||||
pub inline fn wmb() void {
|
||||
switch (builtin.target.cpu.arch) {
|
||||
.x86_64 => asm volatile ("sfence" ::: .{ .memory = true }),
|
||||
.aarch64 => asm volatile ("dsb st" ::: .{ .memory = true }),
|
||||
else => @compileError("mmio.wmb: unsupported architecture"),
|
||||
}
|
||||
}
|
||||
|
||||
test "barriers emit and registers round-trip through a RAM cell" {
|
||||
// The barriers must at least assemble for the host arch; ordering can't be unit
|
||||
// tested, but a missing/mistyped mnemonic is caught here.
|
||||
wmb();
|
||||
rmb();
|
||||
mb();
|
||||
var cell: u64 = 0;
|
||||
write(u64, @intFromPtr(&cell), 0xDEAD_BEEF);
|
||||
try @import("std").testing.expectEqual(@as(u64, 0xDEAD_BEEF), read(u64, @intFromPtr(&cell)));
|
||||
}
|
||||
@@ -0,0 +1,13 @@
|
||||
//! DanOS's POSIX / C compatibility layer — `unistd`, `stdio`, and (later) the C
|
||||
//! `errno` / `struct stat` / `extern "C"` surface. This is the *one* place POSIX and
|
||||
//! C spellings are allowed to appear verbatim (see docs/coding-standards.md): a file
|
||||
//! under library/posix/ *is* the foreign ABI, so it keeps the ABI's names. Everything
|
||||
//! it touches on the danos side (the VFS protocol, the runtime) uses danos names,
|
||||
//! which this layer translates to at the boundary.
|
||||
//!
|
||||
//! It is layered strictly *over* the runtime: it calls the runtime's IPC and heap,
|
||||
//! never the kernel's system calls directly. danos-native applications use the
|
||||
//! runtime; this exists so *POSIX* software can too.
|
||||
|
||||
pub const unistd = @import("unistd.zig");
|
||||
pub const stdio = @import("stdio.zig");
|
||||
@@ -0,0 +1,115 @@
|
||||
//! A small C stdio layer over the POSIX-style file API (unistd.zig). Unbuffered
|
||||
//! for now — each fread/fwrite is one VFS round trip; an internal buffer (fewer
|
||||
//! IPC calls) is a later optimisation. Both a Zig-callable API and `extern "C"`
|
||||
//! symbols are provided, so Zig and future C programs share it.
|
||||
|
||||
const std = @import("std");
|
||||
const unistd = @import("unistd.zig");
|
||||
const heap = @import("runtime").heap;
|
||||
|
||||
pub const SEEK_SET = unistd.SEEK_SET;
|
||||
pub const SEEK_CURRENT = unistd.SEEK_CURRENT;
|
||||
pub const SEEK_END = unistd.SEEK_END;
|
||||
|
||||
/// A C `FILE`: an fd plus sticky end-of-file / error flags. Allocated on the
|
||||
/// heap; `fclose` frees it.
|
||||
pub const FILE = extern struct {
|
||||
fd: i32,
|
||||
eof: c_int = 0,
|
||||
err: c_int = 0,
|
||||
};
|
||||
|
||||
fn flagsFor(mode: []const u8) u32 {
|
||||
if (mode.len == 0) return 0;
|
||||
return switch (mode[0]) {
|
||||
'w', 'a' => unistd.O_CREAT,
|
||||
else => 0,
|
||||
};
|
||||
}
|
||||
|
||||
/// Open `path` in `mode` ("r"/"w"/"a", '+' ignored for now). Returns null on error.
|
||||
pub fn fopen(path: []const u8, mode: []const u8) ?*FILE {
|
||||
const fd = unistd.open(path, flagsFor(mode));
|
||||
if (fd < 0) return null;
|
||||
const f = heap.allocator().create(FILE) catch {
|
||||
unistd.close(fd);
|
||||
return null;
|
||||
};
|
||||
f.* = .{ .fd = fd };
|
||||
if (mode.len > 0 and mode[0] == 'a') _ = unistd.lseek(fd, 0, unistd.SEEK_END);
|
||||
return f;
|
||||
}
|
||||
|
||||
pub fn fclose(f: *FILE) c_int {
|
||||
unistd.close(f.fd);
|
||||
heap.allocator().destroy(f);
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// Read `size*nmemb` bytes; returns the number of whole items read.
|
||||
pub fn fread(buffer: []u8, size: usize, nmemb: usize, f: *FILE) usize {
|
||||
const total = size * nmemb;
|
||||
if (total == 0) return 0;
|
||||
const n = unistd.read(f.fd, buffer[0..@min(buffer.len, total)]);
|
||||
if (n <= 0) {
|
||||
f.eof = 1;
|
||||
return 0;
|
||||
}
|
||||
return @as(usize, @intCast(n)) / size;
|
||||
}
|
||||
|
||||
/// Write `size*nmemb` bytes; returns the number of whole items written.
|
||||
pub fn fwrite(data: []const u8, size: usize, nmemb: usize, f: *FILE) usize {
|
||||
const total = @min(data.len, size * nmemb);
|
||||
if (total == 0) return 0;
|
||||
const n = unistd.write(f.fd, data[0..total]);
|
||||
if (n <= 0) {
|
||||
f.err = 1;
|
||||
return 0;
|
||||
}
|
||||
return @as(usize, @intCast(n)) / size;
|
||||
}
|
||||
|
||||
pub fn fseek(f: *FILE, off: i64, whence: u32) c_int {
|
||||
f.eof = 0;
|
||||
return if (unistd.lseek(f.fd, off, whence) < 0) -1 else 0;
|
||||
}
|
||||
|
||||
pub fn ftell(f: *FILE) i64 {
|
||||
return unistd.lseek(f.fd, 0, unistd.SEEK_CURRENT);
|
||||
}
|
||||
|
||||
pub fn rewind(f: *FILE) void {
|
||||
_ = fseek(f, 0, SEEK_SET);
|
||||
}
|
||||
|
||||
pub fn feof(f: *FILE) c_int {
|
||||
return f.eof;
|
||||
}
|
||||
|
||||
pub fn ferror(f: *FILE) c_int {
|
||||
return f.err;
|
||||
}
|
||||
|
||||
pub fn fputs(s: []const u8, f: *FILE) c_int {
|
||||
return if (unistd.write(f.fd, s) < 0) -1 else 0;
|
||||
}
|
||||
|
||||
pub fn fputc(c: u8, f: *FILE) c_int {
|
||||
const b = [_]u8{c};
|
||||
return if (unistd.write(f.fd, &b) == 1) c else -1;
|
||||
}
|
||||
|
||||
pub fn fgetc(f: *FILE) c_int {
|
||||
var b: [1]u8 = undefined;
|
||||
const n = unistd.read(f.fd, &b);
|
||||
if (n <= 0) {
|
||||
f.eof = 1;
|
||||
return -1; // EOF
|
||||
}
|
||||
return b[0];
|
||||
}
|
||||
|
||||
// Real `extern "C"` symbols (fopen/fread/fseek/...) — with a C-string signature
|
||||
// distinct from the Zig slice API above — land with the first C program, wired
|
||||
// via @export so they don't collide with these Zig names.
|
||||
@@ -0,0 +1,148 @@
|
||||
//! POSIX-style file API for user programs — the low level under C stdio. Files
|
||||
//! are named objects served by the user-space VFS server (system/services/vfs/vfs.zig); each
|
||||
//! call marshals a request, IPC_Calls the VFS, and unmarshals the reply. The
|
||||
//! kernel knows nothing of files or fds — the fd table lives here, per process.
|
||||
|
||||
const std = @import("std");
|
||||
const protocol = @import("vfs-protocol");
|
||||
const ipc = @import("runtime").ipc;
|
||||
|
||||
pub const O_CREAT = protocol.create;
|
||||
pub const SEEK_SET: u32 = 0;
|
||||
pub const SEEK_CURRENT: u32 = 1;
|
||||
pub const SEEK_END: u32 = 2;
|
||||
|
||||
// Resolve (and cache) the VFS server endpoint, looked up by well-known id.
|
||||
var vfs_handle: usize = 0;
|
||||
var vfs_resolved = false;
|
||||
fn vfs() ?usize {
|
||||
if (!vfs_resolved) {
|
||||
vfs_handle = ipc.lookup(.vfs) orelse return null;
|
||||
vfs_resolved = true;
|
||||
}
|
||||
return vfs_handle;
|
||||
}
|
||||
|
||||
const maximum_fds = 32;
|
||||
const Fd = struct { used: bool = false, node: u64 = 0, offset: u64 = 0 };
|
||||
var fds = [_]Fd{.{}} ** maximum_fds;
|
||||
|
||||
fn allocFd() ?usize {
|
||||
for (&fds, 0..) |*f, i| {
|
||||
if (!f.used) {
|
||||
f.* = .{ .used = true };
|
||||
return i;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
const Result = struct { reply: protocol.Reply, payload: []u8 };
|
||||
|
||||
/// One request/reply round trip: [Request header][send payload] -> VFS ->
|
||||
/// [Reply header][receive payload]. The receive payload is written into `out`.
|
||||
fn transact(request: protocol.Request, send: []const u8, out: []u8) ?Result {
|
||||
const h = vfs() orelse return null;
|
||||
var message: [protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||
const slen = @min(send.len, protocol.maximum_payload);
|
||||
@memcpy(message[protocol.request_size..][0..slen], send[0..slen]);
|
||||
|
||||
var rbuf: [protocol.message_maximum]u8 = undefined;
|
||||
const n = ipc.call(h, message[0 .. protocol.request_size + slen], &rbuf) catch return null;
|
||||
if (n < protocol.reply_size) return null;
|
||||
const reply = std.mem.bytesToValue(protocol.Reply, rbuf[0..protocol.reply_size]);
|
||||
const rpl = @min(n - protocol.reply_size, out.len);
|
||||
@memcpy(out[0..rpl], rbuf[protocol.reply_size..][0..rpl]);
|
||||
return .{ .reply = reply, .payload = out[0..rpl] };
|
||||
}
|
||||
|
||||
/// Open (or create, with O_CREAT) `path`; returns an fd or -1.
|
||||
pub fn open(path: []const u8, flags: u32) i32 {
|
||||
const fd = allocFd() orelse return -1;
|
||||
const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = flags };
|
||||
const r = transact(request, path, &.{}) orelse {
|
||||
fds[fd].used = false;
|
||||
return -1;
|
||||
};
|
||||
if (r.reply.status != 0) {
|
||||
fds[fd].used = false;
|
||||
return -1;
|
||||
}
|
||||
fds[fd] = .{ .used = true, .node = r.reply.node, .offset = 0 };
|
||||
return @intCast(fd);
|
||||
}
|
||||
|
||||
fn fdPtr(fd: i32) ?*Fd {
|
||||
if (fd < 0 or fd >= maximum_fds) return null;
|
||||
const f = &fds[@intCast(fd)];
|
||||
return if (f.used) f else null;
|
||||
}
|
||||
|
||||
/// Read up to `buffer.len` bytes at the current offset; returns the count or -1.
|
||||
pub fn read(fd: i32, buffer: []u8) isize {
|
||||
const f = fdPtr(fd) orelse return -1;
|
||||
const want: u32 = @intCast(@min(buffer.len, protocol.maximum_payload));
|
||||
const request = protocol.Request{ .operation = .read, .node = f.node, .offset = f.offset, .len = want, .flags = 0 };
|
||||
const r = transact(request, &.{}, buffer) orelse return -1;
|
||||
if (r.reply.status != 0) return -1;
|
||||
f.offset += r.reply.len;
|
||||
return @intCast(r.reply.len);
|
||||
}
|
||||
|
||||
/// Write `data` at the current offset; returns the count or -1.
|
||||
pub fn write(fd: i32, data: []const u8) isize {
|
||||
const f = fdPtr(fd) orelse return -1;
|
||||
const want: u32 = @intCast(@min(data.len, protocol.maximum_payload));
|
||||
const request = protocol.Request{ .operation = .write, .node = f.node, .offset = f.offset, .len = want, .flags = 0 };
|
||||
const r = transact(request, data[0..want], &.{}) orelse return -1;
|
||||
if (r.reply.status != 0) return -1;
|
||||
f.offset += r.reply.len;
|
||||
return @intCast(r.reply.len);
|
||||
}
|
||||
|
||||
/// Reposition the fd's offset. Returns the new offset or -1. (SEEK_END needs the
|
||||
/// file size, which `stat` provides; handled by fetching it here.)
|
||||
pub fn lseek(fd: i32, off: i64, whence: u32) i64 {
|
||||
const f = fdPtr(fd) orelse return -1;
|
||||
const base: i64 = switch (whence) {
|
||||
SEEK_SET => 0,
|
||||
SEEK_CURRENT => @intCast(f.offset),
|
||||
SEEK_END => blk: {
|
||||
const request = protocol.Request{ .operation = .status, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
var sbuf: [@sizeOf(protocol.FileStatus)]u8 = undefined;
|
||||
const r = transact(request, &.{}, &sbuf) orelse return -1;
|
||||
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.FileStatus)) return -1;
|
||||
const st = std.mem.bytesToValue(protocol.FileStatus, sbuf[0..@sizeOf(protocol.FileStatus)]);
|
||||
break :blk @intCast(st.size);
|
||||
},
|
||||
else => return -1,
|
||||
};
|
||||
const pos = base + off;
|
||||
if (pos < 0) return -1;
|
||||
f.offset = @intCast(pos);
|
||||
return pos;
|
||||
}
|
||||
|
||||
/// Stat `path`. Returns 0 or -1.
|
||||
pub fn stat(path: []const u8, out: *protocol.FileStatus) i32 {
|
||||
// Open, stat by node, close — simple and enough for now.
|
||||
const fd = open(path, 0);
|
||||
if (fd < 0) return -1;
|
||||
defer close(fd);
|
||||
const f = fdPtr(fd).?;
|
||||
const request = protocol.Request{ .operation = .status, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
var sbuf: [@sizeOf(protocol.FileStatus)]u8 = undefined;
|
||||
const r = transact(request, &.{}, &sbuf) orelse return -1;
|
||||
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.FileStatus)) return -1;
|
||||
out.* = std.mem.bytesToValue(protocol.FileStatus, sbuf[0..@sizeOf(protocol.FileStatus)]);
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// Close an fd (best effort — tells the VFS to release the open file).
|
||||
pub fn close(fd: i32) void {
|
||||
const f = fdPtr(fd) orelse return;
|
||||
const request = protocol.Request{ .operation = .close, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
_ = transact(request, &.{}, &.{});
|
||||
f.used = false;
|
||||
}
|
||||
@@ -0,0 +1,131 @@
|
||||
//! User-space device access: enumerate the kernel's device table, claim a device,
|
||||
//! map its MMIO, and bind its interrupt. A driver uses these to find and take
|
||||
//! ownership of its hardware; the claim is the capability the kernel checks before
|
||||
//! mapping registers or routing an IRQ.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const device_abi = @import("device-abi");
|
||||
const sc = @import("system-call.zig");
|
||||
|
||||
pub const DeviceDescriptor = device_abi.DeviceDescriptor;
|
||||
pub const ResourceDescriptor = device_abi.ResourceDescriptor;
|
||||
pub const DeviceClass = device_abi.DeviceClass;
|
||||
pub const ResourceKind = device_abi.ResourceKind;
|
||||
|
||||
inline fn failed(r: usize) bool {
|
||||
return r > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
|
||||
/// Copy up to `buffer.len` device descriptors into `buffer`; returns the total count.
|
||||
pub fn enumerate(buffer: []DeviceDescriptor) usize {
|
||||
return sc.systemCall2(.device_enumerate, @intFromPtr(buffer.ptr), buffer.len);
|
||||
}
|
||||
|
||||
/// Take exclusive ownership of device `id`. Returns false if taken or invalid.
|
||||
pub fn claim(id: u64) bool {
|
||||
return !failed(sc.systemCall1(.device_claim, id));
|
||||
}
|
||||
|
||||
/// Map resource `resource_index` (which must be an MMIO window) of claimed device
|
||||
/// `device_id` into this address space; returns the register base virtual address.
|
||||
pub fn mmioMap(device_id: u64, resource_index: u64) ?usize {
|
||||
const r = sc.systemCall2(.mmio_map, device_id, resource_index);
|
||||
return if (failed(r)) null else r;
|
||||
}
|
||||
|
||||
/// `DeviceDescriptor.parent` for a device with no parent.
|
||||
pub const no_parent = device_abi.no_parent;
|
||||
|
||||
/// `DeviceDescriptor.pci_class` for a device that is not a PCI function. Set this on
|
||||
/// descriptors passed to `register` unless the child really is one.
|
||||
pub const no_pci_class = device_abi.no_pci_class;
|
||||
|
||||
/// Publish `descriptor` as a child of `parent_id`, which this process must have claimed.
|
||||
/// Returns the new device id. The child is left unclaimed, so whichever driver owns
|
||||
/// that class of device can `claim` it — that is how a bus hands off a device.
|
||||
///
|
||||
/// Every resource in `descriptor` must be **contained** in a parent resource of the same
|
||||
/// kind: a sub-window of the parent's MMIO, or one of its IRQs. The kernel refuses
|
||||
/// anything else, because a device descriptor is a licence to map physical memory and
|
||||
/// a bus driver may only subdivide what it already owns. `descriptor.id` and `descriptor.parent`
|
||||
/// are ignored. A device with no resources at all is fine — a USB device is reached
|
||||
/// through its controller, not by MMIO.
|
||||
pub fn register(parent_id: u64, descriptor: *const DeviceDescriptor) ?u64 {
|
||||
const r = sc.systemCall2(.device_register, parent_id, @intFromPtr(descriptor));
|
||||
return if (failed(r)) null else r;
|
||||
}
|
||||
|
||||
/// Bind resource `resource_index` (which must be an IRQ) of claimed device `device_id` to
|
||||
/// `endpoint`. From then on the interrupt arrives as an asynchronous notification:
|
||||
/// `ipc.replyWait` on that endpoint returns with the high bit set in `badge` and the
|
||||
/// low bits carrying the GSI. The kernel masks the line before waking you.
|
||||
pub fn irqBind(device_id: u64, resource_index: u64, endpoint: usize) bool {
|
||||
return !failed(sc.systemCall3(.irq_bind, device_id, resource_index, endpoint));
|
||||
}
|
||||
|
||||
/// Re-arm a bound IRQ. Call this **after** quieting the device (clearing whatever
|
||||
/// status register holds its line asserted) — the kernel left the line masked
|
||||
/// precisely because it could not do that for you. Skip it and the interrupt never
|
||||
/// fires again; call it before the device is quiet and a level-triggered line storms.
|
||||
pub fn irqAck(device_id: u64, resource_index: u64) bool {
|
||||
return !failed(sc.systemCall2(.irq_ack, device_id, resource_index));
|
||||
}
|
||||
|
||||
/// The Message-Signalled Interrupt address/data a driver programs into its device's
|
||||
/// MSI capability. The device raises the interrupt by writing `data` to `address`.
|
||||
pub const Msi = struct { address: u64, data: u32 };
|
||||
|
||||
/// Set up MSI for a claimed device: the kernel allocates a per-device edge-triggered
|
||||
/// vector, binds it to `endpoint` (delivered like `irqBind`, but with no mask and no
|
||||
/// `irqAck` cycle), and returns the (address, data) to write into the device's MSI
|
||||
/// capability — found by mmio_mapping the device's ECAM config space (resource 0) and
|
||||
/// walking its capability list. Returns null on failure. Two return values (address in
|
||||
/// rax, data in rdx), so a hand-written stub.
|
||||
pub fn msiBind(device_id: u64, endpoint: usize) ?Msi {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = undefined;
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "={rdx}" (rdx),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.msi_bind)),
|
||||
[a0] "{rdi}" (device_id),
|
||||
[a1] "{rsi}" (endpoint),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (failed(rax)) return null;
|
||||
return .{ .address = rax, .data = @intCast(rdx) };
|
||||
}
|
||||
|
||||
/// Read `width` bytes (1, 2, or 4) from a port in a claimed device's `io_port`
|
||||
/// resource, at byte `offset` within it. Ring 3 has no direct `in`/`out`, so a legacy
|
||||
/// driver (PS/2, 16550 UART) reaches its ports through this claim-gated call — each
|
||||
/// access is a syscall, which is fine for the low-rate hardware that needs it. Returns
|
||||
/// null if the capability check fails (device not claimed, wrong resource, out of
|
||||
/// range). A device that decodes no data returns all-ones, which is a valid value, not
|
||||
/// a failure.
|
||||
pub fn ioRead(device_id: u64, resource_index: u64, offset: u64, width: u8) ?u32 {
|
||||
const r = sc.systemCall4(.io_read, device_id, resource_index, offset, width);
|
||||
return if (failed(r)) null else @intCast(r);
|
||||
}
|
||||
|
||||
/// Write `value` (its low `width` bytes, 1/2/4) to a port in a claimed device's
|
||||
/// `io_port` resource, at byte `offset`. Same capability gate as `ioRead`.
|
||||
pub fn ioWrite(device_id: u64, resource_index: u64, offset: u64, width: u8, value: u32) bool {
|
||||
return !failed(sc.systemCall5(.io_write, device_id, resource_index, offset, width, value));
|
||||
}
|
||||
|
||||
/// Find DeviceDescription by hid
|
||||
///
|
||||
/// Utility function for driver development
|
||||
pub fn findDeviceDescriptorByHid(buffer: []DeviceDescriptor, hid_needle: []const u8) ?DeviceDescriptor {
|
||||
const total = enumerate(buffer);
|
||||
const n = @min(total, buffer.len);
|
||||
for (@as([]DeviceDescriptor, buffer[0..n])) |d| {
|
||||
const hid_haystack = d.hid[0..@intCast(d.hid_len)];
|
||||
if (std.mem.eql(u8, hid_haystack, hid_needle)) {
|
||||
return d;
|
||||
}
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
//! User-space DMA memory: `dma_alloc` / `dma_free`. A driver that programs a
|
||||
//! bus-mastering engine needs a descriptor ring the device can read — memory that is
|
||||
//! physically contiguous, at a physical address the driver knows, uncacheable, and
|
||||
//! pinned. `mmap` gives none of those; this does. Pair it with the barriers in
|
||||
//! `/lib/mmio` (fill the ring, `wmb()`, ring the doorbell). See docs/driver-model.md.
|
||||
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
|
||||
/// Allocation flags. `coherent` (uncacheable) is the portable default; the rest are
|
||||
/// opt-in for specific hardware — see `abi`.
|
||||
pub const coherent: usize = abi.dma_coherent;
|
||||
pub const write_combining: usize = abi.dma_write_combining;
|
||||
pub const below_4g: usize = abi.dma_below_4g;
|
||||
|
||||
/// A DMA allocation: the `virtual` address the CPU touches, and the `physical` address
|
||||
/// to program into the device's descriptor-ring / base registers.
|
||||
pub const Region = struct {
|
||||
virtual: usize,
|
||||
physical: usize,
|
||||
};
|
||||
|
||||
inline fn failed(r: usize) bool {
|
||||
return r > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
|
||||
/// Allocate `len` bytes of DMA-capable memory with `flags` (e.g. `coherent`, or
|
||||
/// `coherent | below_4g`). Returns the virtual/physical pair, or null on failure. Two
|
||||
/// return values — the virtual address in rax, the physical address in rdx — so it
|
||||
/// needs a hand-written stub.
|
||||
pub fn alloc(len: usize, flags: usize) ?Region {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = undefined; // out: physical address
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "={rdx}" (rdx),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.dma_alloc)),
|
||||
[a0] "{rdi}" (len),
|
||||
[a1] "{rsi}" (flags),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (failed(rax)) return null;
|
||||
return .{ .virtual = rax, .physical = rdx };
|
||||
}
|
||||
|
||||
/// Release a region from a prior `alloc` (`virtual` and the same `len`).
|
||||
pub fn free(virtual: usize, len: usize) void {
|
||||
_ = sc.systemCall2(.dma_free, virtual, len);
|
||||
}
|
||||
@@ -0,0 +1,193 @@
|
||||
//! The user-space heap: C-convention dynamic allocation (`malloc`/`free`/…) plus
|
||||
//! a `std.mem.Allocator` adapter over the same free list, so both C-style code
|
||||
//! and Zig `std` containers share one heap.
|
||||
//!
|
||||
//! The algorithm is a straight port of the kernel's first-fit free list
|
||||
//! (system/kernel/heap.zig): an address-ordered singly linked list of free blocks,
|
||||
//! split on allocation and coalesced with neighbours on free. The only thing
|
||||
//! that changes on this side of the system_call boundary is where memory comes from
|
||||
//! — `grow` asks the kernel for pages via `mmap` instead of mapping frames
|
||||
//! itself, and the kernel picks the base address.
|
||||
//!
|
||||
//! Single-threaded and 16-byte maximum alignment, exactly like the kernel heap; a
|
||||
//! lock and larger alignments come when user programs gain threads.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const system_calls = @import("system.zig");
|
||||
|
||||
const page_size = abi.page_size;
|
||||
|
||||
/// A block header, at the start of every block; while free it also links the
|
||||
/// free list via `next`.
|
||||
const Block = extern struct {
|
||||
size: usize, // total block size in bytes, including this header; a multiple of 16
|
||||
next: ?*Block, // free-list link (only meaningful while free)
|
||||
};
|
||||
|
||||
const header_size = @sizeOf(Block); // 16
|
||||
const minimum_block = header_size + 16; // smallest block worth splitting off
|
||||
/// Grow granularity: one `mmap` per 64 KiB amortises the system_call.
|
||||
const chunk = 64 * 1024;
|
||||
|
||||
var free_list: ?*Block = null;
|
||||
|
||||
fn alignUp(value: usize, alignment: usize) usize {
|
||||
return (value + alignment - 1) & ~(alignment - 1);
|
||||
}
|
||||
|
||||
fn payloadOf(block: *Block) [*]u8 {
|
||||
return @ptrFromInt(@intFromPtr(block) + header_size);
|
||||
}
|
||||
|
||||
/// Ask the kernel for more pages and add them as a free block. Because each
|
||||
/// `mmap` is an independent grant, cross-grant coalescing happens only when the
|
||||
/// kernel returns adjacent bases (its arena is a bump allocator, so consecutive
|
||||
/// grants usually are adjacent). Returns false if the kernel is out of memory.
|
||||
fn grow(minimum_bytes: usize) bool {
|
||||
const bytes = alignUp(@max(minimum_bytes, chunk), page_size);
|
||||
const ret = system_calls.mmap(bytes, system_calls.PROT_READ | system_calls.PROT_WRITE);
|
||||
if (system_calls.mmapFailed(ret)) return false;
|
||||
|
||||
const block: *Block = @ptrFromInt(ret);
|
||||
block.size = bytes;
|
||||
insertFree(block); // coalesces if this grant is adjacent to a prior one
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Insert a block into the address-ordered free list, coalescing with the
|
||||
/// physically adjacent free blocks on either side.
|
||||
fn insertFree(block: *Block) void {
|
||||
var previous: ?*Block = null;
|
||||
var current = free_list;
|
||||
while (current) |c| : (current = c.next) {
|
||||
if (@intFromPtr(c) > @intFromPtr(block)) break;
|
||||
previous = c;
|
||||
}
|
||||
|
||||
block.next = current;
|
||||
if (previous) |p| p.next = block else free_list = block;
|
||||
|
||||
// Merge forward into `current` if they're contiguous.
|
||||
if (current) |c| {
|
||||
if (@intFromPtr(block) + block.size == @intFromPtr(c)) {
|
||||
block.size += c.size;
|
||||
block.next = c.next;
|
||||
}
|
||||
}
|
||||
// Merge `previous` forward into `block` if they're contiguous.
|
||||
if (previous) |p| {
|
||||
if (@intFromPtr(p) + p.size == @intFromPtr(block)) {
|
||||
p.size += block.size;
|
||||
p.next = block.next;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Allocate `len` bytes (16-byte aligned), or null if out of memory.
|
||||
fn rawAlloc(len: usize) ?[*]u8 {
|
||||
const need = alignUp(header_size + len, 16);
|
||||
|
||||
var attempts: u32 = 0;
|
||||
while (attempts < 2) : (attempts += 1) {
|
||||
var previous: ?*Block = null;
|
||||
var current = free_list;
|
||||
while (current) |block| : ({
|
||||
previous = block;
|
||||
current = block.next;
|
||||
}) {
|
||||
if (block.size < need) continue;
|
||||
|
||||
if (block.size >= need + minimum_block) {
|
||||
// Split: carve `need` off the front, leave the rest free.
|
||||
const rest: *Block = @ptrFromInt(@intFromPtr(block) + need);
|
||||
rest.size = block.size - need;
|
||||
rest.next = block.next;
|
||||
if (previous) |p| p.next = rest else free_list = rest;
|
||||
block.size = need;
|
||||
} else {
|
||||
// Take the whole block.
|
||||
if (previous) |p| p.next = block.next else free_list = block.next;
|
||||
}
|
||||
return payloadOf(block);
|
||||
}
|
||||
|
||||
// Nothing fit: grow and try once more.
|
||||
if (!grow(need)) return null;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
fn rawFree(ptr: [*]u8) void {
|
||||
const block: *Block = @ptrFromInt(@intFromPtr(ptr) - header_size);
|
||||
insertFree(block);
|
||||
}
|
||||
|
||||
// --- C ABI: the global implicit heap ---------------------------------------
|
||||
// `extern "C"` symbols so future C code links the same malloc/free directly.
|
||||
|
||||
export fn malloc(size: usize) callconv(.c) ?*anyopaque {
|
||||
if (size == 0) return null;
|
||||
const p = rawAlloc(size) orelse return null;
|
||||
return @ptrCast(p);
|
||||
}
|
||||
|
||||
export fn free(ptr: ?*anyopaque) callconv(.c) void {
|
||||
const p = ptr orelse return;
|
||||
rawFree(@ptrCast(p));
|
||||
}
|
||||
|
||||
export fn calloc(nmemb: usize, size: usize) callconv(.c) ?*anyopaque {
|
||||
const total = std.math.mul(usize, nmemb, size) catch return null; // overflow-safe
|
||||
if (total == 0) return null;
|
||||
const p = rawAlloc(total) orelse return null;
|
||||
@memset(p[0..total], 0);
|
||||
return @ptrCast(p);
|
||||
}
|
||||
|
||||
export fn realloc(ptr: ?*anyopaque, size: usize) callconv(.c) ?*anyopaque {
|
||||
const p = ptr orelse return malloc(size);
|
||||
if (size == 0) {
|
||||
rawFree(@ptrCast(p));
|
||||
return null;
|
||||
}
|
||||
const block: *Block = @ptrFromInt(@intFromPtr(p) - header_size);
|
||||
const old_payload = block.size - header_size;
|
||||
if (size <= old_payload) return p; // shrink/same: keep the block
|
||||
const np = rawAlloc(size) orelse return null; // grow: alloc + copy + free
|
||||
@memcpy(np[0..old_payload], @as([*]u8, @ptrCast(p))[0..old_payload]);
|
||||
rawFree(@ptrCast(p));
|
||||
return @ptrCast(np);
|
||||
}
|
||||
|
||||
// --- std.mem.Allocator interface (same free list) --------------------------
|
||||
|
||||
pub fn allocator() std.mem.Allocator {
|
||||
return .{ .ptr = undefined, .vtable = &vtable };
|
||||
}
|
||||
|
||||
const vtable = std.mem.Allocator.VTable{
|
||||
.alloc = allocImpl,
|
||||
.resize = resizeImpl,
|
||||
.remap = remapImpl,
|
||||
.free = freeImpl,
|
||||
};
|
||||
|
||||
fn allocImpl(_: *anyopaque, len: usize, alignment: std.mem.Alignment, _: usize) ?[*]u8 {
|
||||
if (alignment.toByteUnits() > 16) return null; // blocks are 16-byte aligned
|
||||
return rawAlloc(len);
|
||||
}
|
||||
|
||||
fn resizeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, new_len: usize, _: usize) bool {
|
||||
// In-place iff the new payload still fits the current block.
|
||||
const block: *Block = @ptrFromInt(@intFromPtr(memory.ptr) - header_size);
|
||||
return new_len + header_size <= block.size;
|
||||
}
|
||||
|
||||
fn remapImpl(_: *anyopaque, _: []u8, _: std.mem.Alignment, _: usize, _: usize) ?[*]u8 {
|
||||
return null;
|
||||
}
|
||||
|
||||
fn freeImpl(_: *anyopaque, memory: []u8, _: std.mem.Alignment, _: usize) void {
|
||||
rawFree(memory.ptr);
|
||||
}
|
||||
@@ -0,0 +1,220 @@
|
||||
//! User-space input helpers: the client and publisher sides of the input service, so a
|
||||
//! program listening for input events — or a driver broadcasting them — doesn't hand-roll
|
||||
//! the IPC. Layered over `ipc` (endpoints, capability passing, `send`) and the shared
|
||||
//! `input-protocol` wire format, the way `device.zig` layers over the raw `device_*` calls.
|
||||
//! See system/services/input/input.zig.
|
||||
//!
|
||||
//! The service carries several device classes (keyboard, mouse, joystick/gamepad). A
|
||||
//! **source** publishes its class with the matching method:
|
||||
//! var source = input.connectSource() orelse return;
|
||||
//! _ = source.publishKeyboardEvent(.{ .kind = ..., .keycode = ..., ... });
|
||||
//! _ = source.publishMouseEvent(.{ ... });
|
||||
//! _ = source.publishJoystickEvent(.{ ... });
|
||||
//!
|
||||
//! A **subscriber** either takes one class with a typed helper —
|
||||
//! var keys = input.subscribeKeyboard() orelse return;
|
||||
//! while (true) { const key = keys.next() orelse continue; ... }
|
||||
//! — or takes several at once and inspects the tagged envelope:
|
||||
//! var listener = input.subscribeAll() orelse return;
|
||||
//! while (true) {
|
||||
//! const event = listener.next() orelse continue;
|
||||
//! if (event.asKeyboard()) |k| { ... } else if (event.asMouse()) |m| { ... }
|
||||
//! }
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const protocol = @import("input-protocol");
|
||||
|
||||
pub const DeviceKind = protocol.DeviceKind;
|
||||
pub const InputEvent = protocol.InputEvent;
|
||||
pub const KeyEvent = protocol.KeyEvent;
|
||||
pub const MouseEvent = protocol.MouseEvent;
|
||||
pub const JoystickEvent = protocol.JoystickEvent;
|
||||
pub const EventKind = protocol.EventKind;
|
||||
pub const MouseEventKind = protocol.MouseEventKind;
|
||||
pub const JoystickEventKind = protocol.JoystickEventKind;
|
||||
pub const Keycode = protocol.Keycode;
|
||||
|
||||
/// Interest masks re-exported so a caller can `subscribe(input.device_keyboard |
|
||||
/// input.device_mouse)`.
|
||||
pub const device_keyboard = protocol.device_keyboard;
|
||||
pub const device_mouse = protocol.device_mouse;
|
||||
pub const device_joystick = protocol.device_joystick;
|
||||
pub const device_all = protocol.device_all;
|
||||
|
||||
/// Look up the input service, retrying while it is still coming up. Both a subscriber and
|
||||
/// a source race the service's registration at boot, so both wait for it here rather than
|
||||
/// failing. Returns the service endpoint handle, or null if it never appears.
|
||||
fn lookupService() ?ipc.Handle {
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 100) : (attempts += 1) {
|
||||
if (ipc.lookup(.input)) |handle| return handle;
|
||||
system.sleep(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- subscribing ------------------------------------------------------------
|
||||
|
||||
/// A subscription to the input service: our own endpoint, which the service pushes events
|
||||
/// to. `next` returns each event as a tagged `InputEvent`; use `asKeyboard`/`asMouse`/
|
||||
/// `asJoystick` to decode. Created with `subscribe`/`subscribeAll`; for a single device
|
||||
/// class prefer the typed helpers (`subscribeKeyboard`, ...), which return decoded events.
|
||||
pub const Subscriber = struct {
|
||||
/// The endpoint the service delivers events to (created and owned by us; its handle
|
||||
/// was handed to the service as a capability at subscribe time).
|
||||
endpoint: ipc.Handle,
|
||||
receive: [protocol.event_size]u8 = undefined,
|
||||
|
||||
/// Block until the next event is pushed, and return it. Events arrive as asynchronous
|
||||
/// buffered messages (`ipc_send` from the service), so nothing is owed in reply — the
|
||||
/// empty reply this issues is a harmless no-op. Returns null for any non-event wake-up
|
||||
/// (there should be none), so callers can loop.
|
||||
pub fn next(self: *Subscriber) ?InputEvent {
|
||||
const got = ipc.replyWait(self.endpoint, &.{}, &self.receive, null);
|
||||
if (!got.isMessage() or got.len < protocol.event_size) return null;
|
||||
return std.mem.bytesToValue(InputEvent, self.receive[0..protocol.event_size]);
|
||||
}
|
||||
};
|
||||
|
||||
/// Subscribe to the input classes named in `device_mask` (an OR of `device_*`, or
|
||||
/// `device_all`). Creates an endpoint for the service to push to and hands it over as a
|
||||
/// capability. Returns a `Subscriber` to loop `next` on, or null on failure.
|
||||
pub fn subscribe(device_mask: u32) ?Subscriber {
|
||||
const service = lookupService() orelse return null;
|
||||
const endpoint = ipc.createIpcEndpoint() orelse return null;
|
||||
|
||||
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.subscribe), .device_mask = device_mask };
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
const result = ipc.callCap(service, std.mem.asBytes(&request), &reply, endpoint) catch return null;
|
||||
if (result.len < protocol.reply_size) return null;
|
||||
if (std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status != 0) return null;
|
||||
return .{ .endpoint = endpoint };
|
||||
}
|
||||
|
||||
/// Subscribe to every input class (keyboard, mouse, joystick) on one stream.
|
||||
pub fn subscribeAll() ?Subscriber {
|
||||
return subscribe(device_all);
|
||||
}
|
||||
|
||||
/// A subscriber filtered to keyboard events, whose `next` returns a decoded `KeyEvent`.
|
||||
pub const KeyboardSubscriber = struct {
|
||||
inner: Subscriber,
|
||||
pub fn next(self: *KeyboardSubscriber) ?KeyEvent {
|
||||
return (self.inner.next() orelse return null).asKeyboard();
|
||||
}
|
||||
};
|
||||
|
||||
/// A subscriber filtered to mouse events, whose `next` returns a decoded `MouseEvent`.
|
||||
pub const MouseSubscriber = struct {
|
||||
inner: Subscriber,
|
||||
pub fn next(self: *MouseSubscriber) ?MouseEvent {
|
||||
return (self.inner.next() orelse return null).asMouse();
|
||||
}
|
||||
};
|
||||
|
||||
/// A subscriber filtered to joystick/gamepad events, whose `next` returns a decoded
|
||||
/// `JoystickEvent`.
|
||||
pub const JoystickSubscriber = struct {
|
||||
inner: Subscriber,
|
||||
pub fn next(self: *JoystickSubscriber) ?JoystickEvent {
|
||||
return (self.inner.next() orelse return null).asJoystick();
|
||||
}
|
||||
};
|
||||
|
||||
/// Subscribe to keyboard events only; `next` returns decoded `KeyEvent`s.
|
||||
pub fn subscribeKeyboard() ?KeyboardSubscriber {
|
||||
return .{ .inner = subscribe(device_keyboard) orelse return null };
|
||||
}
|
||||
|
||||
/// Subscribe to mouse events only; `next` returns decoded `MouseEvent`s.
|
||||
pub fn subscribeMouse() ?MouseSubscriber {
|
||||
return .{ .inner = subscribe(device_mouse) orelse return null };
|
||||
}
|
||||
|
||||
/// Subscribe to joystick/gamepad events only; `next` returns decoded `JoystickEvent`s.
|
||||
pub fn subscribeJoystick() ?JoystickSubscriber {
|
||||
return .{ .inner = subscribe(device_joystick) orelse return null };
|
||||
}
|
||||
|
||||
// --- publishing -------------------------------------------------------------
|
||||
|
||||
/// A connection to the input service for a source (a keyboard/mouse/joystick driver) that
|
||||
/// publishes events. Each `publish*Event` is a short synchronous call the service answers
|
||||
/// at once; its own fan-out to subscribers is asynchronous, so publishing never blocks on
|
||||
/// a slow subscriber.
|
||||
pub const Publisher = struct {
|
||||
service: ipc.Handle,
|
||||
|
||||
fn publish(self: Publisher, event: InputEvent) bool {
|
||||
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.publish), .event = event };
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
const len = ipc.call(self.service, std.mem.asBytes(&request), &reply) catch return false;
|
||||
if (len < protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status == 0;
|
||||
}
|
||||
|
||||
/// Broadcast a keyboard event to every subscriber that took keyboard events.
|
||||
pub fn publishKeyboardEvent(self: Publisher, event: KeyEvent) bool {
|
||||
return self.publish(InputEvent.fromKeyboard(event));
|
||||
}
|
||||
/// Broadcast a mouse event to every subscriber that took mouse events.
|
||||
pub fn publishMouseEvent(self: Publisher, event: MouseEvent) bool {
|
||||
return self.publish(InputEvent.fromMouse(event));
|
||||
}
|
||||
/// Broadcast a joystick/gamepad event to every subscriber that took joystick events.
|
||||
pub fn publishJoystickEvent(self: Publisher, event: JoystickEvent) bool {
|
||||
return self.publish(InputEvent.fromJoystick(event));
|
||||
}
|
||||
};
|
||||
|
||||
/// Connect to the input service as an event source, waiting for it to come up. Returns a
|
||||
/// `Publisher`, or null if the service never registered.
|
||||
pub fn connectSource() ?Publisher {
|
||||
return .{ .service = lookupService() orelse return null };
|
||||
}
|
||||
|
||||
// --- synthetic scaffolding --------------------------------------------------
|
||||
|
||||
/// Synthetic key events, shared by the demo source and the keyboard driver's placeholder
|
||||
/// stream while real scancode decoding is still a follow-up. `step` rolls through A..E,
|
||||
/// emitting for each key a `key_down`, then a `key_press` carrying the character, then a
|
||||
/// `key_up`. Scaffolding, not wire protocol — hence it lives with the helpers.
|
||||
pub fn syntheticKeyEvent(step: usize) KeyEvent {
|
||||
const Key = struct { code: Keycode, character: u32 };
|
||||
const keys = [_]Key{
|
||||
.{ .code = .a, .character = 'A' },
|
||||
.{ .code = .b, .character = 'B' },
|
||||
.{ .code = .c, .character = 'C' },
|
||||
.{ .code = .d, .character = 'D' },
|
||||
.{ .code = .e, .character = 'E' },
|
||||
};
|
||||
const key = keys[(step / 3) % keys.len];
|
||||
return switch (step % 3) {
|
||||
0 => .{ .kind = @intFromEnum(EventKind.key_down), .keycode = @intFromEnum(key.code), .character = 0, .modifiers = 0 },
|
||||
1 => .{ .kind = @intFromEnum(EventKind.key_press), .keycode = @intFromEnum(key.code), .character = key.character, .modifiers = 0 },
|
||||
else => .{ .kind = @intFromEnum(EventKind.key_up), .keycode = @intFromEnum(key.code), .character = 0, .modifiers = 0 },
|
||||
};
|
||||
}
|
||||
|
||||
/// Synthetic mouse events (placeholder until real PS/2 packet decoding). `step` alternates
|
||||
/// a small diagonal motion with a left-button click.
|
||||
pub fn syntheticMouseEvent(step: usize) MouseEvent {
|
||||
return switch (step % 3) {
|
||||
0 => .{ .kind = @intFromEnum(MouseEventKind.motion), .button = 0, .dx = 1, .dy = 1, .scroll_x = 0, .scroll_y = 0, .buttons = 0 },
|
||||
1 => .{ .kind = @intFromEnum(MouseEventKind.button_down), .button = protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = protocol.mouse_button_left },
|
||||
else => .{ .kind = @intFromEnum(MouseEventKind.button_up), .button = protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = 0 },
|
||||
};
|
||||
}
|
||||
|
||||
/// Synthetic joystick/gamepad events (placeholder until a real controller driver). `step`
|
||||
/// sweeps axis 0 and toggles button 0.
|
||||
pub fn syntheticJoystickEvent(step: usize) JoystickEvent {
|
||||
return switch (step % 3) {
|
||||
0 => .{ .kind = @intFromEnum(JoystickEventKind.axis), .control = 0, .value = 16384, .buttons = 0 },
|
||||
1 => .{ .kind = @intFromEnum(JoystickEventKind.button_down), .control = 0, .value = 0, .buttons = 1 },
|
||||
else => .{ .kind = @intFromEnum(JoystickEventKind.button_up), .control = 0, .value = 0, .buttons = 0 },
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,194 @@
|
||||
//! User-space IPC helpers over the kernel's synchronous IPC syscalls. A client
|
||||
//! `call`s an endpoint (send + block for reply); the VFS server and drivers are
|
||||
//! reached this way. The server side (`replyWait`, which returns two values) is
|
||||
//! added with the first server binary.
|
||||
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
|
||||
/// A small-int handle into the calling process's handle table.
|
||||
pub const Handle = usize;
|
||||
|
||||
/// A fixed-size, register-friendly message payload. Server protocols (VFS, driver)
|
||||
/// layer their own wire format on top of the bytes a call carries.
|
||||
pub const Message = extern struct {
|
||||
tag: u64 = 0,
|
||||
a: u64 = 0,
|
||||
b: u64 = 0,
|
||||
c: u64 = 0,
|
||||
};
|
||||
|
||||
/// Whether a system_call return value is a wrapped -errno (lands in the top page).
|
||||
inline fn failed(r: usize) bool {
|
||||
return r > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
|
||||
/// Create a new endpoint owned by this process; returns its handle.
|
||||
pub fn createIpcEndpoint() ?Handle {
|
||||
const r = sc.systemCall0(.create_ipc_endpoint);
|
||||
return if (failed(r)) null else r;
|
||||
}
|
||||
|
||||
/// Publish endpoint `h` under a well-known service id so other processes find it.
|
||||
pub fn register(id: abi.ServiceId, h: Handle) bool {
|
||||
return !failed(sc.systemCall2(.ipc_register, @intFromEnum(id), h));
|
||||
}
|
||||
|
||||
/// Find the endpoint published under `id`, installing a handle to it in this
|
||||
/// process.
|
||||
pub fn lookup(id: abi.ServiceId) ?Handle {
|
||||
const r = sc.systemCall1(.ipc_lookup, @intFromEnum(id));
|
||||
return if (failed(r)) null else r;
|
||||
}
|
||||
|
||||
pub const CallError = error{Failed};
|
||||
|
||||
/// The result of a capability-passing `callCap`: the reply length, and the handle of
|
||||
/// an endpoint the server sent back (e.g. a per-device channel), or null.
|
||||
pub const Reply = struct {
|
||||
len: usize,
|
||||
cap: ?Handle,
|
||||
};
|
||||
|
||||
/// Send `message` to endpoint `h` and block until the server replies into `reply`,
|
||||
/// optionally handing the server a capability (`send_cap`) and receiving one back.
|
||||
/// This is the class-driver "open" primitive: call a bus with `send_cap = null`, get a
|
||||
/// private per-device endpoint back in `.cap`. Two return values (reply length in rax,
|
||||
/// received handle in r8) need a hand-written stub — r8 is read-write (in: reply
|
||||
/// capacity, arg #4; out: the received handle).
|
||||
pub fn callCap(h: Handle, message: []const u8, reply: []u8, send_cap: ?Handle) CallError!Reply {
|
||||
var rax: usize = undefined;
|
||||
var r8: usize = reply.len; // in: reply capacity (arg #4); out: received capability handle
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[r8] "+{r8}" (r8),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.ipc_call)),
|
||||
[a0] "{rdi}" (h),
|
||||
[a1] "{rsi}" (@intFromPtr(message.ptr)),
|
||||
[a2] "{rdx}" (message.len),
|
||||
[a3] "{r10}" (@intFromPtr(reply.ptr)),
|
||||
[a5] "{r9}" (send_cap orelse abi.no_cap),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (failed(rax)) return error.Failed;
|
||||
return .{ .len = rax, .cap = if (r8 == abi.no_cap) null else r8 };
|
||||
}
|
||||
|
||||
/// Send `message` to endpoint `h` and block until the server replies into `reply`.
|
||||
/// Returns the reply length. The common case: no capability passed either way.
|
||||
pub fn call(h: Handle, message: []const u8, reply: []u8) CallError!usize {
|
||||
return (try callCap(h, message, reply, null)).len;
|
||||
}
|
||||
|
||||
/// Post `message` to endpoint `h`'s asynchronous queue and return immediately — no
|
||||
/// rendezvous, no reply, no blocking. The receiver picks it up through `replyWait` as a
|
||||
/// buffered message (`Received.isMessage`). Unlike `call`, this **cannot hang on a dead
|
||||
/// or slow peer**, which is why a broadcaster (the input service) delivers events this
|
||||
/// way. The payload must fit an endpoint slot (64 bytes); a full queue drops the oldest
|
||||
/// message. Returns false on failure (bad handle, oversized payload, bad buffer).
|
||||
pub fn send(h: Handle, message: []const u8) bool {
|
||||
return !failed(sc.systemCall3(.ipc_send, h, @intFromPtr(message.ptr), message.len));
|
||||
}
|
||||
|
||||
/// Set in `Received.badge` when what arrived is an asynchronous notification — a
|
||||
/// bound device interrupt — rather than a client's message. The low bits carry the
|
||||
/// GSI. See `isNotification`.
|
||||
pub const notify_badge_bit: u64 = abi.notify_badge_bit;
|
||||
|
||||
/// Set alongside `notify_badge_bit` when the notification is a **signal** — the
|
||||
/// lifecycle vocabulary of docs/process-lifecycle.md, delivered to the endpoint
|
||||
/// nominated with `process.bindSignals`. Decode with `process.signalsFrom`.
|
||||
pub const notify_signal_bit: u64 = abi.notify_signal_bit;
|
||||
|
||||
/// Set alongside `notify_badge_bit` when the notification is a **one-shot timer**
|
||||
/// landing (`system.timerOnce`).
|
||||
pub const notify_timer_bit: u64 = abi.notify_timer_bit;
|
||||
|
||||
/// Set alongside `notify_badge_bit` when the notification is a **child-exit
|
||||
/// notice** — a process this one spawned (with an exit endpoint) has ended —
|
||||
/// rather than a device interrupt. The low bits carry the child's process id.
|
||||
pub const notify_exit_bit: u64 = abi.notify_exit_bit;
|
||||
|
||||
/// Set alongside `notify_badge_bit` when the wake-up is a **buffered message** — a payload
|
||||
/// posted with `send` (`ipc_send`) — rather than a bare device interrupt or child-exit
|
||||
/// notice. The payload is in the `replyWait` receive buffer (`Received.len` bytes); the
|
||||
/// low bits of the badge carry the sender's task id. See `Received.isMessage`.
|
||||
pub const notify_message_bit: u64 = abi.notify_message_bit;
|
||||
|
||||
/// The result of a `replyWait`: the request length, the sender's badge (a task id, or
|
||||
/// an IRQ notification if the high bit is set), and any capability the request carried.
|
||||
pub const Received = struct {
|
||||
len: usize,
|
||||
badge: u64,
|
||||
cap: ?Handle,
|
||||
|
||||
/// True if this wake-up was an asynchronous notification (a device interrupt
|
||||
/// or a child-exit notice), not a client request. An event loop branches on
|
||||
/// this; there is no reply owed on the notification path.
|
||||
pub fn isNotification(self: Received) bool {
|
||||
return self.badge & notify_badge_bit != 0;
|
||||
}
|
||||
|
||||
/// True if this wake-up tells of a supervised child's end — the notification
|
||||
/// requested by passing an exit endpoint to `system.spawnSupervised`.
|
||||
pub fn isChildExit(self: Received) bool {
|
||||
return self.isNotification() and self.badge & notify_exit_bit != 0;
|
||||
}
|
||||
|
||||
/// True if this wake-up is a **buffered message** posted with `send` (`ipc_send`):
|
||||
/// there is a payload in the receive buffer (`self.len` bytes) and no reply is owed.
|
||||
/// The subscriber side of a broadcast branches on this.
|
||||
pub fn isMessage(self: Received) bool {
|
||||
return self.isNotification() and self.badge & notify_message_bit != 0;
|
||||
}
|
||||
|
||||
/// The task id of whoever posted a buffered message, meaningful only when
|
||||
/// Whether this arrival is a signal notification — decode the set with
|
||||
/// `process.signalsFrom(badge)`.
|
||||
pub fn isSignal(self: Received) bool {
|
||||
return self.isNotification() and self.badge & notify_signal_bit != 0;
|
||||
}
|
||||
|
||||
/// Whether this arrival is a one-shot timer landing (`system.timerOnce`).
|
||||
pub fn isTimer(self: Received) bool {
|
||||
return self.isNotification() and self.badge & notify_timer_bit != 0;
|
||||
}
|
||||
|
||||
/// `isMessage`. (The badge's low bits, with the three high marker bits masked off.)
|
||||
pub fn senderTaskId(self: Received) u32 {
|
||||
return @intCast(self.badge & ~(notify_badge_bit | notify_exit_bit | notify_message_bit));
|
||||
}
|
||||
|
||||
/// The interrupt source (a GSI), meaningful only when `isNotification` and
|
||||
/// not `isChildExit`.
|
||||
pub fn source(self: Received) u64 {
|
||||
return self.badge & ~notify_badge_bit;
|
||||
}
|
||||
|
||||
/// The ended child's process id, meaningful only when `isChildExit`.
|
||||
pub fn childProcessId(self: Received) u32 {
|
||||
return @intCast(self.badge & ~(notify_badge_bit | notify_exit_bit));
|
||||
}
|
||||
};
|
||||
|
||||
/// Server side of IPC_ReplyWait: deliver `reply` to the client last received (if any,
|
||||
/// optionally handing it `send_cap`), then block until the next request arrives in
|
||||
/// `receive`. Returns its length, the sender badge, and any capability the request
|
||||
/// carried (in `.cap`). Three return values — length in rax, badge in rdx, received
|
||||
/// handle in r8 — so it needs a hand-written stub: rdx is read-write (in: reply length,
|
||||
/// arg #3; out: badge) and r8 is read-write (in: receive capacity, arg #4; out: handle).
|
||||
pub fn replyWait(h: Handle, reply: []const u8, receive: []u8, send_cap: ?Handle) Received {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = reply.len; // in: reply_len (arg #3); out: badge
|
||||
var r8: usize = receive.len; // in: receive capacity (arg #4); out: received capability handle
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "+{rdx}" (rdx),
|
||||
[r8] "+{r8}" (r8),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.ipc_reply_wait)),
|
||||
[a0] "{rdi}" (h),
|
||||
[a1] "{rsi}" (@intFromPtr(reply.ptr)),
|
||||
[a3] "{r10}" (@intFromPtr(receive.ptr)),
|
||||
[a5] "{r9}" (send_cap orelse abi.no_cap),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
return .{ .len = rax, .badge = rdx, .cap = if (r8 == abi.no_cap) null else r8 };
|
||||
}
|
||||
@@ -0,0 +1,137 @@
|
||||
//! Process-level runtime types: what a user program receives at entry (`Init`,
|
||||
//! the argv contract) and the process end of the lifecycle
|
||||
//! (docs/process-lifecycle.md) — today the exit reason a supervisor reads to
|
||||
//! decide restart; signals and the stop sequence land here with M17.4. Mirrors
|
||||
//! the spirit of `std.process.Init.Minimal` in danos terms — std's `Args` holds
|
||||
//! no data on freestanding targets, so the type is danos's own.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
|
||||
/// Everything a program receives at entry. Passed to
|
||||
/// `pub fn main(init: runtime.process.Init)`; programs that need nothing keep
|
||||
/// `pub fn main() void`. An `environment` field is added here once the kernel
|
||||
/// passes a non-empty envp (today it is always empty — see docs/sysv.md).
|
||||
pub const Init = struct {
|
||||
arguments: Arguments,
|
||||
};
|
||||
|
||||
/// The process arguments (argc/argv), parsed from the kernel-built System V
|
||||
/// entry block. The bytes live in the entry block at the top of the stack page,
|
||||
/// NUL-terminated, valid for the process's lifetime.
|
||||
pub const Arguments = struct {
|
||||
/// argc — at least 1: argument 0 is the path or name this binary was
|
||||
/// spawned as.
|
||||
count: usize,
|
||||
/// The argv pointers in the entry block (NULL-terminated after `count`
|
||||
/// entries).
|
||||
vector: [*]const [*:0]const u8,
|
||||
|
||||
/// Argument `index` (0 = the program's own path/name), or null if out of
|
||||
/// range.
|
||||
pub fn get(arguments: Arguments, index: usize) ?[:0]const u8 {
|
||||
if (index >= arguments.count) return null;
|
||||
return std.mem.span(arguments.vector[index]);
|
||||
}
|
||||
|
||||
pub fn iterate(arguments: Arguments) Iterator {
|
||||
return .{ .arguments = arguments };
|
||||
}
|
||||
|
||||
pub const Iterator = struct {
|
||||
arguments: Arguments,
|
||||
index: usize = 0,
|
||||
|
||||
pub fn next(iterator: *Iterator) ?[:0]const u8 {
|
||||
const argument = iterator.arguments.get(iterator.index) orelse return null;
|
||||
iterator.index += 1;
|
||||
return argument;
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
/// How a process ended — what a supervisor's restart policy reads: a clean exit
|
||||
/// meant to stop, a fault wants a restart with backoff, killed means the
|
||||
/// supervisor did it itself (docs/process-lifecycle.md).
|
||||
pub const ExitReason = abi.ExitReason;
|
||||
|
||||
/// How dead child `id` ended. Ask after the exit notification arrives — the
|
||||
/// kernel records the reason before it posts the notification, so this never
|
||||
/// races it. Returns null for an id that never lived, is still alive, was
|
||||
/// evicted from the kernel's bounded record, or is not this process's child
|
||||
/// (the same authority gate as `kill`).
|
||||
pub fn exitReason(id: u32) ?ExitReason {
|
||||
const r = sc.systemCall1(.process_exit_reason, id);
|
||||
if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno
|
||||
return @enumFromInt(r);
|
||||
}
|
||||
|
||||
/// The signal vocabulary (docs/process-lifecycle.md): POSIX's concepts, danos's
|
||||
/// names, message delivery. A signal is a one-way coalescing statement — never a
|
||||
/// question (liveness is the zero-length ping call) and never kill (that is
|
||||
/// `system.kill`, unhandleable by definition).
|
||||
pub const Signal = abi.Signal;
|
||||
|
||||
/// The coalesced set of signals one notification delivered: two pending
|
||||
/// terminates arrive as one. Decode a received badge with `signalsFrom`.
|
||||
pub const SignalSet = struct {
|
||||
pending: u32,
|
||||
|
||||
pub fn has(set: SignalSet, signal: Signal) bool {
|
||||
return set.pending & (@as(u32, 1) << @intFromEnum(signal)) != 0;
|
||||
}
|
||||
};
|
||||
|
||||
/// Nominate `endpoint` as this process's signal endpoint. Signals posted while
|
||||
/// unbound have pended; they are delivered immediately on bind, coalesced.
|
||||
pub fn bindSignals(endpoint: usize) bool {
|
||||
return sc.systemCall1(.signal_bind, endpoint) == 0;
|
||||
}
|
||||
|
||||
/// Decode a received badge into the signals it delivered, or null if it is not
|
||||
/// a signal notification.
|
||||
pub fn signalsFrom(badge: u64) ?SignalSet {
|
||||
if (badge & abi.notify_badge_bit == 0 or badge & abi.notify_signal_bit == 0) return null;
|
||||
return .{ .pending = @truncate(badge & ~(abi.notify_badge_bit | abi.notify_signal_bit)) };
|
||||
}
|
||||
|
||||
/// Post `signal` to child `id` (or to yourself). Supervisor-gated, like kill;
|
||||
/// non-blocking, always — a statement, not a conversation.
|
||||
pub fn sendSignal(id: u32, signal: Signal) bool {
|
||||
return sc.systemCall2(.process_signal, id, @intFromEnum(signal)) == 0;
|
||||
}
|
||||
|
||||
/// The standard stop sequence (docs/process-lifecycle.md): terminate, wait up to
|
||||
/// `deadline_ms` for the exit notification on `exit_endpoint` (the endpoint the
|
||||
/// child was spawned with), then kill. Any *other* notifications arriving on
|
||||
/// that endpoint while stopping are consumed and dropped — a supervisor with
|
||||
/// concurrent traffic implements the same sequence inside its own event loop
|
||||
/// (arm `system.timerOnce`, keep serving) instead of calling this.
|
||||
pub fn stop(id: u32, deadline_ms: u64, exit_endpoint: usize) void {
|
||||
_ = sendSignal(id, .terminate);
|
||||
_ = system.timerOnce(exit_endpoint, deadline_ms);
|
||||
var receive: [8]u8 = undefined;
|
||||
while (true) {
|
||||
const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null);
|
||||
if (got.isChildExit() and got.childProcessId() == id) return;
|
||||
if (got.isTimer()) break; // the deadline passed first — escalate
|
||||
}
|
||||
_ = system.kill(id);
|
||||
while (true) {
|
||||
const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null);
|
||||
if (got.isChildExit() and got.childProcessId() == id) return;
|
||||
}
|
||||
}
|
||||
|
||||
/// Subscribe `endpoint` to published exit events: every process death posts an
|
||||
/// asynchronous notification with the same badge encoding as a supervisor's exit
|
||||
/// notice (decode with `ipc.Received.isChildExit`/`childProcessId`). For stateful
|
||||
/// services: release what the dead client held — file handles, subscriptions —
|
||||
/// because a service must never depend on clients cleaning up after themselves
|
||||
/// (docs/process-lifecycle.md). Ungated, like `system.processes`.
|
||||
pub fn subscribeExits(endpoint: usize) bool {
|
||||
return sc.systemCall1(.process_subscribe, endpoint) == 0;
|
||||
}
|
||||
@@ -0,0 +1,46 @@
|
||||
//! danos user-space runtime library — a nascent libc. Every user binary (init,
|
||||
//! and later the VFS server + device drivers) imports this as `@import("runtime")`:
|
||||
//! system_call wrappers, the C-convention heap, IPC helpers, and the process start
|
||||
//! shim. It is compiled into each binary (inheriting its `.large` code model and
|
||||
//! freestanding target), so all user programs share one implementation.
|
||||
//!
|
||||
//! A user binary needs three lines:
|
||||
//! const runtime = @import("runtime");
|
||||
//! pub const panic = runtime.panic;
|
||||
//! comptime { _ = &runtime.start._start; } // pull the entry shim in
|
||||
//! and a `pub fn main() void` or `pub fn main(init: runtime.process.Init) void`
|
||||
//! (arguments arrive via `init`).
|
||||
|
||||
pub const system = @import("system.zig");
|
||||
pub const heap = @import("heap.zig");
|
||||
pub const ipc = @import("ipc.zig");
|
||||
pub const start = @import("start.zig");
|
||||
/// The VFS wire protocol (shared with the VFS server).
|
||||
pub const vfs_protocol = @import("vfs-protocol");
|
||||
|
||||
/// The device-manager protocol: hello + tree reports (docs/device-manager.md).
|
||||
pub const device_manager_protocol = @import("device-manager-protocol");
|
||||
/// Keyboard-event listening (subscribe/next) and broadcasting (publish), over the input
|
||||
/// service. See library/runtime/input.zig and system/services/input/.
|
||||
pub const input = @import("input.zig");
|
||||
/// The input wire protocol (shared with the input service and its clients).
|
||||
pub const input_protocol = @import("input-protocol");
|
||||
/// POSIX-style file API: open/read/write/lseek/stat/close.
|
||||
/// C stdio: fopen/fread/fwrite/fseek/ftell/fclose over unistd.
|
||||
/// Device access for drivers: enumerate/claim/mmioMap.
|
||||
pub const device = @import("device.zig");
|
||||
/// DMA-capable memory for drivers: contiguous, pinned, uncacheable buffers.
|
||||
pub const dma = @import("dma.zig");
|
||||
|
||||
/// Re-exported so a user binary can `pub const panic = runtime.panic;`.
|
||||
pub const panic = start.panic;
|
||||
|
||||
/// Process entry types: the `Init` handed to `main`, and its `Arguments`.
|
||||
pub const process = @import("process.zig");
|
||||
|
||||
/// The service harness: one replyWait loop folding requests, signals, and
|
||||
/// notifications into callbacks (docs/process-lifecycle.md).
|
||||
pub const service = @import("service.zig");
|
||||
|
||||
/// The heap as a `std.mem.Allocator`, for Zig `std` containers in user code.
|
||||
pub const allocator = heap.allocator;
|
||||
@@ -0,0 +1,83 @@
|
||||
//! The service harness (docs/process-lifecycle.md): one replyWait loop that
|
||||
//! folds protocol requests, signals, and subscribed notifications into
|
||||
//! callbacks — so the lifecycle contract ("answers ping, exits on terminate")
|
||||
//! is satisfied by construction and a service author writes domain logic only.
|
||||
//! Nothing is asynchronous inside the process: a callback runs at a point the
|
||||
//! loop chose, never on a hijacked stack — the whole reason signals are
|
||||
//! messages.
|
||||
//!
|
||||
//! The liveness probe: a **zero-length request is the universal ping**, answered
|
||||
//! with a zero-length reply by the harness itself. No protocol's requests start
|
||||
//! at length zero, so the encoding cannot collide, and there is nothing for a
|
||||
//! service author to implement — a wedged service simply fails to answer, which
|
||||
//! is the diagnosis (see docs/ipc.md).
|
||||
|
||||
const abi = @import("abi");
|
||||
const ipc = @import("ipc.zig");
|
||||
const process = @import("process.zig");
|
||||
|
||||
pub const Callbacks = struct {
|
||||
/// Called once with the service's endpoint before the loop starts — the
|
||||
/// place to subscribe to exit events, bind IRQs, or announce readiness.
|
||||
/// Return false to abort startup (the process exits).
|
||||
init: ?*const fn (endpoint: ipc.Handle) bool = null,
|
||||
/// One protocol request from `sender` (a task id): write the reply into
|
||||
/// `reply`, return its length. `capability` is the handle the request
|
||||
/// carried, if any (M13 cap passing — how a subscriber hands over its
|
||||
/// endpoint). The zero-length ping never reaches this.
|
||||
on_message: *const fn (message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize,
|
||||
/// A notification that is not a signal — a subscribed exit event, a bound
|
||||
/// IRQ, a timer landing. The raw badge; decode with the ipc helpers.
|
||||
on_notification: ?*const fn (badge: u64) void = null,
|
||||
/// The reload signal. Default: ignored.
|
||||
on_reload: ?*const fn () void = null,
|
||||
/// The terminate signal, called before the loop returns. The clean exit is
|
||||
/// the return itself — never put *necessary* work here (iron rule 1: a kill
|
||||
/// arrives with no warning; this is for graceful extras only).
|
||||
on_terminate: ?*const fn () void = null,
|
||||
/// Publish the endpoint under a well-known service id at startup.
|
||||
service: ?abi.ServiceId = null,
|
||||
};
|
||||
|
||||
/// Run the service: create and (optionally) register the endpoint, bind signals
|
||||
/// to it, call `init`, then serve until `terminate` arrives — at which point the
|
||||
/// loop returns and main's return is the clean exit the supervisor reads as
|
||||
/// `ExitReason.exited`. `maximum_message` sizes the receive and reply buffers
|
||||
/// (a service passes its protocol's message maximum).
|
||||
pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
|
||||
const endpoint = ipc.createIpcEndpoint() orelse return;
|
||||
if (callbacks.service) |id| {
|
||||
if (!ipc.register(id, endpoint)) return;
|
||||
}
|
||||
_ = process.bindSignals(endpoint);
|
||||
if (callbacks.init) |initialise| {
|
||||
if (!initialise(endpoint)) return;
|
||||
}
|
||||
|
||||
var reply_buffer: [maximum_message]u8 = undefined;
|
||||
var reply_len: usize = 0;
|
||||
var receive: [maximum_message]u8 = undefined;
|
||||
while (true) {
|
||||
const got = ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive, null);
|
||||
if (got.isNotification()) {
|
||||
reply_len = 0; // nothing owed for a notification
|
||||
if (process.signalsFrom(got.badge)) |signals| {
|
||||
if (signals.has(.reload)) {
|
||||
if (callbacks.on_reload) |onReload| onReload();
|
||||
}
|
||||
if (signals.has(.terminate)) {
|
||||
if (callbacks.on_terminate) |onTerminate| onTerminate();
|
||||
return; // the loop's return IS the clean exit
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (callbacks.on_notification) |onNotification| onNotification(got.badge);
|
||||
continue;
|
||||
}
|
||||
if (got.len == 0) {
|
||||
reply_len = 0; // the universal ping: a zero-length reply, from the harness
|
||||
continue;
|
||||
}
|
||||
reply_len = callbacks.on_message(receive[0..got.len], &reply_buffer, got.senderTaskId(), got.cap);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,86 @@
|
||||
//! The user-space process entry shim. Every user binary roots `_start` here (via
|
||||
//! `entry = _start` in build.zig) and forces this file to be analysed with
|
||||
//! `comptime { _ = &runtime.start._start; }`, so the whole runtime is linked in.
|
||||
|
||||
const std = @import("std");
|
||||
const system = @import("system.zig");
|
||||
const process = @import("process.zig");
|
||||
|
||||
/// The kernel enters at `_start` with rsp 16-aligned, pointing at the System V
|
||||
/// process-entry block it built: argc, argv pointers, NULL, envp terminator, the
|
||||
/// auxiliary vector, then the strings (see system/kernel/process.zig,
|
||||
/// `buildEntryStack`). Capture that address in rdi — the first SysV argument —
|
||||
/// before `call` disturbs the stack; the call's pushed return address also puts
|
||||
/// rsp ≡ 8 (mod 16), satisfying the ABI before any Zig frame runs. The `ud2` is a
|
||||
/// safety net if `rt_start` ever returns.
|
||||
pub export fn _start() callconv(.naked) noreturn {
|
||||
asm volatile (
|
||||
\\mov %%rsp, %%rdi
|
||||
\\call rt_start
|
||||
\\ud2
|
||||
);
|
||||
}
|
||||
|
||||
/// The first Zig frame, entered with `stack` pointing at the kernel-built entry
|
||||
/// block. Build the `process.Init` from it and dispatch to the program's `main`,
|
||||
/// whose signature is inspected at comptime. The heap is lazy (first alloc grows
|
||||
/// it), so there is no other runtime init to order here.
|
||||
export fn rt_start(stack: [*]const u64) callconv(.c) noreturn {
|
||||
const init: process.Init = .{ .arguments = .{
|
||||
.count = stack[0],
|
||||
.vector = @ptrCast(stack + 1),
|
||||
} };
|
||||
system.exit(callMain(init));
|
||||
}
|
||||
|
||||
/// Comptime-dispatch on root.main's signature, in the spirit of std's start.zig:
|
||||
/// zero parameters or one `process.Init`; returns void, noreturn, u8, !void, or !u8.
|
||||
fn callMain(init: process.Init) u8 {
|
||||
const root = @import("root"); // the user binary's root source file
|
||||
const main_information = @typeInfo(@TypeOf(root.main)).@"fn";
|
||||
|
||||
const call_arguments = switch (main_information.params.len) {
|
||||
0 => .{},
|
||||
1 => arguments: {
|
||||
const Parameter = main_information.params[0].type orelse
|
||||
@compileError("main's parameter must be runtime.process.Init (not anytype)");
|
||||
if (Parameter != process.Init)
|
||||
@compileError("main's parameter must be runtime.process.Init, found " ++ @typeName(Parameter));
|
||||
break :arguments .{init};
|
||||
},
|
||||
else => @compileError("main takes no parameters or a single runtime.process.Init"),
|
||||
};
|
||||
|
||||
const ReturnType = main_information.return_type.?;
|
||||
switch (@typeInfo(ReturnType)) {
|
||||
.noreturn => @call(.auto, root.main, call_arguments),
|
||||
.void => {
|
||||
@call(.auto, root.main, call_arguments);
|
||||
return 0;
|
||||
},
|
||||
.int => {
|
||||
if (ReturnType != u8)
|
||||
@compileError("main's integer return type must be u8, found " ++ @typeName(ReturnType));
|
||||
return @call(.auto, root.main, call_arguments);
|
||||
},
|
||||
.error_union => {
|
||||
const payload = @call(.auto, root.main, call_arguments) catch |err| {
|
||||
var buffer: [128]u8 = undefined;
|
||||
const line = std.fmt.bufPrint(&buffer, "main returned error: {s}\n", .{@errorName(err)}) catch "main returned an error\n";
|
||||
_ = system.write(line);
|
||||
return 1; // distinct from panic's 127
|
||||
};
|
||||
if (@TypeOf(payload) == void) return 0;
|
||||
if (@TypeOf(payload) == u8) return payload;
|
||||
@compileError("main's error-union payload must be void or u8, found " ++ @typeName(@TypeOf(payload)));
|
||||
},
|
||||
else => @compileError("main must return void, noreturn, u8, !void, or !u8, found " ++ @typeName(ReturnType)),
|
||||
}
|
||||
}
|
||||
|
||||
/// No runtime to unwind into — report a panic as a nonzero exit code.
|
||||
pub const panic = std.debug.FullPanic(struct {
|
||||
fn panic(_: []const u8, _: ?usize) noreturn {
|
||||
system.exit(127);
|
||||
}
|
||||
}.panic);
|
||||
@@ -0,0 +1,67 @@
|
||||
//! Raw `system_call` instruction wrappers for user space — one per arity.
|
||||
//!
|
||||
//! ABI: number in rax, arguments in rdi, rsi, rdx, r10, r8, r9, result in rax.
|
||||
//! The `system_call` instruction itself clobbers rcx (it holds the return rip) and
|
||||
//! r11 (the saved rflags); the kernel entry stub preserves everything else.
|
||||
//! Note argument #3 goes in **r10, not rcx** — rcx is unavailable across the
|
||||
//! instruction, so the kernel reads the 4th argument from r10.
|
||||
|
||||
const abi = @import("abi");
|
||||
const SystemCall = abi.SystemCall;
|
||||
|
||||
pub inline fn systemCall0(n: SystemCall) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
|
||||
pub inline fn systemCall1(n: SystemCall, a0: usize) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)),
|
||||
[a0] "{rdi}" (a0),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
|
||||
pub inline fn systemCall2(n: SystemCall, a0: usize, a1: usize) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)),
|
||||
[a0] "{rdi}" (a0),
|
||||
[a1] "{rsi}" (a1),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
|
||||
pub inline fn systemCall3(n: SystemCall, a0: usize, a1: usize, a2: usize) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)),
|
||||
[a0] "{rdi}" (a0),
|
||||
[a1] "{rsi}" (a1),
|
||||
[a2] "{rdx}" (a2),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
|
||||
pub inline fn systemCall4(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: usize) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)),
|
||||
[a0] "{rdi}" (a0),
|
||||
[a1] "{rsi}" (a1),
|
||||
[a2] "{rdx}" (a2),
|
||||
[a3] "{r10}" (a3),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
|
||||
pub inline fn systemCall5(n: SystemCall, a0: usize, a1: usize, a2: usize, a3: usize, a4: usize) usize {
|
||||
return asm volatile ("syscall"
|
||||
: [ret] "={rax}" (-> usize),
|
||||
: [n] "{rax}" (@intFromEnum(n)),
|
||||
[a0] "{rdi}" (a0),
|
||||
[a1] "{rsi}" (a1),
|
||||
[a2] "{rdx}" (a2),
|
||||
[a3] "{r10}" (a3),
|
||||
[a4] "{r8}" (a4),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
}
|
||||
@@ -0,0 +1,147 @@
|
||||
//! Typed system_call surface for user space — thin wrappers over the raw `system_call`
|
||||
//! stubs, one per kernel call. Numbers come from `abi.SystemCall`, the single
|
||||
//! source of truth shared with the kernel dispatcher.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
|
||||
/// `mmap` protection flags (matching the usual C bit values). Grants are always
|
||||
/// readable+writable today; the kernel does not yet honour finer prot.
|
||||
pub const PROT_READ: usize = abi.prot_read;
|
||||
pub const PROT_WRITE: usize = abi.prot_write;
|
||||
pub const PROT_EXEC: usize = abi.prot_exec;
|
||||
|
||||
/// One `processes` entry — re-exported from the shared ABI so a user program can
|
||||
/// declare its snapshot buffer without importing `abi` itself.
|
||||
pub const ProcessDescriptor = abi.ProcessDescriptor;
|
||||
|
||||
/// Give up the rest of this quantum.
|
||||
pub fn yield() void {
|
||||
_ = sc.systemCall0(.yield);
|
||||
}
|
||||
|
||||
/// Write raw bytes to the kernel log (a bring-up diagnostic; real output goes
|
||||
/// through the console/VFS later). Returns the byte count, or a wrapped -1.
|
||||
pub fn write(message: []const u8) usize {
|
||||
return sc.systemCall2(.debug_write, @intFromPtr(message.ptr), message.len);
|
||||
}
|
||||
|
||||
/// Block the caller for `ms` milliseconds.
|
||||
pub fn sleep(ms: usize) void {
|
||||
_ = sc.systemCall1(.sleep, ms);
|
||||
}
|
||||
|
||||
/// Arm a one-shot timer: after `ms` milliseconds the kernel posts a timer
|
||||
/// notification (`ipc.Received.isTimer`) to `endpoint`. The timed wait of
|
||||
/// docs/process-lifecycle.md — a service arms a deadline and keeps serving,
|
||||
/// instead of blocking in sleep; what stop-sequence escalation, hello deadlines,
|
||||
/// and restart backoff are built from.
|
||||
pub fn timerOnce(endpoint: usize, ms: u64) bool {
|
||||
return sc.systemCall2(.timer_bind, endpoint, ms) == 0;
|
||||
}
|
||||
|
||||
/// Monotonic nanoseconds since boot — a time source for timeouts and short delays. It
|
||||
/// only ever moves forward. This is *not* wall-clock time (no date, no timezone — that
|
||||
/// is a user-space service layered on top). Deadline pattern for a bounded poll loop:
|
||||
///
|
||||
/// const deadline = clock() + timeout_ns;
|
||||
/// while (clock() < deadline) { ... }
|
||||
pub fn clock() u64 {
|
||||
return @intCast(sc.systemCall0(.clock));
|
||||
}
|
||||
|
||||
/// End the process. Never returns.
|
||||
pub fn exit(code: usize) noreturn {
|
||||
_ = sc.systemCall1(.exit, code);
|
||||
unreachable; // the kernel never returns from exit
|
||||
}
|
||||
|
||||
/// Start the binary bundled in the initial-ramdisk under `name` as a new ring-3
|
||||
/// process, returning the child's process id (or null on failure). The child's
|
||||
/// argv[0] is `name`, and the caller becomes its **supervisor** — the only process
|
||||
/// allowed to `kill` it. This is how a supervisor (the device manager) launches a
|
||||
/// driver it matched — danos-native, not POSIX (a spawn/exec family comes with the
|
||||
/// POSIX layer later).
|
||||
pub fn spawn(name: []const u8) ?u32 {
|
||||
return spawnSupervised(name, &.{}, null);
|
||||
}
|
||||
|
||||
/// Like `spawn`, but hands the child command-line arguments: they arrive as
|
||||
/// argv[1..] on its System V entry stack (argv[0] is still `name`).
|
||||
pub fn spawnWithArguments(name: []const u8, arguments: []const []const u8) ?u32 {
|
||||
return spawnSupervised(name, arguments, null);
|
||||
}
|
||||
|
||||
/// The full spawn: command-line arguments for the child, and an optional endpoint
|
||||
/// (a handle from `ipc.createIpcEndpoint`) the kernel notifies when the child ends
|
||||
/// — any way it ends: clean exit, fault, or `kill`. The notification arrives via
|
||||
/// `ipc.replyWait` as a badge with the child-exit bit set and the child's id in
|
||||
/// the low bits (`ipc.Received.isChildExit`/`childProcessId`), so one endpoint can
|
||||
/// supervise many children. Arguments are marshalled to the kernel as one
|
||||
/// NUL-separated blob; the combined arguments must fit `blob` (the kernel caps the
|
||||
/// blob at 256 bytes and argc at 8 anyway). Returns the child's process id, or
|
||||
/// null on failure.
|
||||
pub fn spawnSupervised(name: []const u8, arguments: []const []const u8, exit_endpoint: ?usize) ?u32 {
|
||||
var blob: [256]u8 = undefined;
|
||||
var len: usize = 0;
|
||||
for (arguments, 0..) |argument, i| {
|
||||
if (i != 0) {
|
||||
if (len >= blob.len) return null;
|
||||
blob[len] = 0;
|
||||
len += 1;
|
||||
}
|
||||
if (len + argument.len > blob.len) return null;
|
||||
@memcpy(blob[len..][0..argument.len], argument);
|
||||
len += argument.len;
|
||||
}
|
||||
const r = sc.systemCall5(.system_spawn, @intFromPtr(name.ptr), name.len, if (len == 0) 0 else @intFromPtr(&blob), len, exit_endpoint orelse abi.no_cap);
|
||||
if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno
|
||||
return @intCast(r);
|
||||
}
|
||||
|
||||
/// Snapshot the process table into `out` (up to its length) and return the total
|
||||
/// number of live processes — which may exceed `out.len`; call again with a larger
|
||||
/// buffer for the full listing. Kernel tasks are included, with an empty name.
|
||||
/// The primitive `ps` is built on.
|
||||
pub fn processes(out: []abi.ProcessDescriptor) usize {
|
||||
return sc.systemCall2(.process_enumerate, @intFromPtr(out.ptr), out.len);
|
||||
}
|
||||
|
||||
/// Whether a process spawned under `name` (its argv[0]) is currently alive.
|
||||
pub fn isProcessRunning(name: []const u8) bool {
|
||||
var table: [32]ProcessDescriptor = undefined;
|
||||
const total = processes(&table);
|
||||
for (table[0..@min(total, table.len)]) |descriptor| {
|
||||
if (std.mem.eql(u8, descriptor.name[0..descriptor.name_length], name)) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// End process `id`. Only its supervisor — the process that spawned it — may;
|
||||
/// anyone else gets false, as does a stale or unknown id (ids are never reused).
|
||||
/// Delivery is prompt but asynchronous, like a signal: a target caught running on
|
||||
/// another core dies at its next system call or timer tick. True means the kill
|
||||
/// is accepted and irrevocable; the exit notification (if an endpoint was given
|
||||
/// at spawn) confirms completion.
|
||||
pub fn kill(id: u32) bool {
|
||||
return sc.systemCall1(.process_kill, id) == 0;
|
||||
}
|
||||
|
||||
/// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable
|
||||
/// memory and return the base virtual address. On failure returns a value in the
|
||||
/// top page (see `mmapFailed`). The user heap grows through this call.
|
||||
pub fn mmap(len: usize, prot: usize) usize {
|
||||
return sc.systemCall2(.mmap, len, prot);
|
||||
}
|
||||
|
||||
/// Release a range previously handed out by `mmap`.
|
||||
pub fn munmap(base: usize, len: usize) usize {
|
||||
return sc.systemCall2(.munmap, base, len);
|
||||
}
|
||||
|
||||
/// Whether an `mmap` return value is an error (the kernel returns a wrapped
|
||||
/// -errno, which lands in the top page — no real grant base is ever that high).
|
||||
pub inline fn mmapFailed(ret: usize) bool {
|
||||
return ret > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
@@ -0,0 +1,52 @@
|
||||
/* Shared link layout for every user binary (init, servers, drivers).
|
||||
*
|
||||
* Linked at a fixed user-space virtual base (set by `image_base` in build.zig,
|
||||
* inside the kernel's user region). Same discipline as the kernel's script:
|
||||
* one PT_LOAD per permission set, every section page-aligned, so the kernel's
|
||||
* user-ELF loader can map each segment with exact W^X permissions. Note the
|
||||
* linker also emits a read-only PT_LOAD covering the ELF headers at the image
|
||||
* base, so the entry point comes from e_entry, not the base address.
|
||||
*/
|
||||
|
||||
ENTRY(_start)
|
||||
|
||||
/* FLAGS bits: 1=X, 2=W, 4=R. */
|
||||
PHDRS {
|
||||
text PT_LOAD FLAGS(5); /* R + X */
|
||||
rodata PT_LOAD FLAGS(4); /* R */
|
||||
data PT_LOAD FLAGS(6); /* R + W */
|
||||
}
|
||||
|
||||
SECTIONS {
|
||||
/* The `.large` code model (needed for the >4 GiB image base) emits code and
|
||||
* data into .ltext/.lrodata/.ldata/.lbss; fold those into the matching
|
||||
* permission segment alongside the normal names. */
|
||||
.text ALIGN(4K) : {
|
||||
*(.text .text.*)
|
||||
*(.ltext .ltext.*)
|
||||
} :text
|
||||
|
||||
.rodata ALIGN(4K) : {
|
||||
*(.rodata .rodata.*)
|
||||
*(.lrodata .lrodata.*)
|
||||
} :rodata
|
||||
|
||||
.data ALIGN(4K) : {
|
||||
*(.data .data.*)
|
||||
*(.ldata .ldata.*)
|
||||
} :data
|
||||
|
||||
/* .bss occupies memory but not file space; the loader zeroes the
|
||||
* filesz..memsz gap. */
|
||||
.bss ALIGN(4K) : {
|
||||
*(.bss .bss.*)
|
||||
*(.lbss .lbss.*)
|
||||
*(COMMON)
|
||||
} :data
|
||||
|
||||
/DISCARD/ : {
|
||||
*(.comment)
|
||||
*(.note .note.*)
|
||||
*(.eh_frame .eh_frame_hdr)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
# xkeyboard-config — X11 keyboard layouts, compiled to Zig
|
||||
|
||||
This module turns a physical key (a **USB HID usage**, as the [input module](../../docs/input.md)
|
||||
delivers in `KeyEvent.keycode`) plus a modifier state into a **keysym** and, when the key
|
||||
produces one, a **character** (a Unicode scalar). It is what lets a `keycode` become a
|
||||
`character` — a keymap — without danos shipping an X11 runtime.
|
||||
|
||||
The layout data comes from the X11 [xkeyboard-config](https://gitlab.freedesktop.org/xkeyboard-config/xkeyboard-config)
|
||||
database, but it is **compiled to native Zig at build time** rather than parsed at runtime.
|
||||
`tools/make-xkeyboard-config.py` reads the vendored xkb data and emits pure-data tables into
|
||||
`generated/layouts.zig`; `xkeyboard-config.zig` is the hand-written API over them. This is
|
||||
the same build-time-codegen pattern as `tools/make-initial-ramdisk.py`.
|
||||
|
||||
## Using it
|
||||
|
||||
```zig
|
||||
const xkb = @import("xkeyboard-config");
|
||||
|
||||
const m = xkb.map(xkb.us, key_event.keycode, .{ .shift = shift_held, .caps_lock = caps });
|
||||
if (m.character) |ch| { /* a printable Unicode scalar */ }
|
||||
// m.keysym is always set (e.g. an X11 keysym for Return / F1 / a dead key).
|
||||
|
||||
const layout = xkb.byName("gb") orelse xkb.us; // choose a layout by name
|
||||
for (xkb.all) |l| { /* enumerate available layouts */ }
|
||||
```
|
||||
|
||||
`Modifiers` carries `shift`, `caps_lock`, `level3` (AltGr), and `control`. `map` selects the
|
||||
level from the key's XKB *type* (the generated data) and those modifiers (the policy, in
|
||||
`xkeyboard-config.zig`), so data and semantics stay separable.
|
||||
|
||||
Layouts: **us, gb, de, fr, es, dvorak**.
|
||||
|
||||
## Regenerating
|
||||
|
||||
```sh
|
||||
python3 tools/make-xkeyboard-config.py fetch # network: download + vendor the data subset
|
||||
python3 tools/make-xkeyboard-config.py generate # offline: emit generated/layouts.zig
|
||||
# or, from the build:
|
||||
zig build gen-xkeyboard-config
|
||||
```
|
||||
|
||||
- **`fetch`** downloads the pinned xkeyboard-config release (version + sha256 in the script),
|
||||
resolves the `include` graph for the configured layouts, and vendors *only* the symbols
|
||||
files actually reached (plus `keysymdef.h`, `COPYING`, and `PROVENANCE.md`) into `vendor/`.
|
||||
Run it when bumping the version or adding a layout.
|
||||
- **`generate`** is deterministic and offline — same vendored input produces byte-identical
|
||||
output. To add a layout, extend `TARGETS` (and `HID_TO_NAME` if a new physical key is
|
||||
involved), then re-run `fetch` (to vendor any new includes) and `generate`.
|
||||
|
||||
## Scope
|
||||
|
||||
A pragmatic subset, enough for real Latin-script typing:
|
||||
|
||||
- **Group 1 only** — no multi-layout group switching.
|
||||
- **No dead-key / compose composition** — a dead key returns its keysym with no `character`
|
||||
(composing `´` + `e` → `é` is a higher layer's job).
|
||||
- **Curated key types** — the common XKB types (one/two-level, alphabetic, four-level, …);
|
||||
unmapped keys and unknown types fall back to level-by-shift.
|
||||
- **6 layouts** — extend via `TARGETS` as above.
|
||||
|
||||
## Licensing
|
||||
|
||||
xkeyboard-config and `keysymdef.h` (xorgproto) are MIT/X11 licensed. The vendored data
|
||||
subset carries the upstream `vendor/COPYING`, and `vendor/PROVENANCE.md` records the exact
|
||||
version, source URL, and sha256. The generated tables are a derived work under the same terms.
|
||||
File diff suppressed because it is too large
Load Diff
+190
@@ -0,0 +1,190 @@
|
||||
Copyright 1996 by Joseph Moss
|
||||
Copyright (C) 2002-2007 Free Software Foundation, Inc.
|
||||
Copyright (C) Dmitry Golubev <lastguru@mail.ru>, 2003-2004
|
||||
Copyright (C) 2004, Gregory Mokhin <mokhin@bog.msu.ru>
|
||||
Copyright (C) 2006 Erdal Ronahî
|
||||
|
||||
Permission to use, copy, modify, distribute, and sell this software and its
|
||||
documentation for any purpose is hereby granted without fee, provided that
|
||||
the above copyright notice appear in all copies and that both that
|
||||
copyright notice and this permission notice appear in supporting
|
||||
documentation, and that the name of the copyright holder(s) not be used in
|
||||
advertising or publicity pertaining to distribution of the software without
|
||||
specific, written prior permission. The copyright holder(s) makes no
|
||||
representations about the suitability of this software for any purpose. It
|
||||
is provided "as is" without express or implied warranty.
|
||||
|
||||
THE COPYRIGHT HOLDER(S) DISCLAIMS ALL WARRANTIES WITH REGARD TO THIS SOFTWARE,
|
||||
INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS, IN NO
|
||||
EVENT SHALL THE COPYRIGHT HOLDER(S) BE LIABLE FOR ANY SPECIAL, INDIRECT OR
|
||||
CONSEQUENTIAL DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE,
|
||||
DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER
|
||||
TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR
|
||||
PERFORMANCE OF THIS SOFTWARE.
|
||||
|
||||
|
||||
Copyright (c) 1996 Digital Equipment Corporation
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining
|
||||
a copy of this software and associated documentation files (the
|
||||
"Software"), to deal in the Software without restriction, including
|
||||
without limitation the rights to use, copy, modify, merge, publish,
|
||||
distribute, sublicense, and sell copies of the Software, and to
|
||||
permit persons to whom the Software is furnished to do so, subject to
|
||||
the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included
|
||||
in all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS
|
||||
OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||
IN NO EVENT SHALL DIGITAL EQUIPMENT CORPORATION BE LIABLE FOR ANY CLAIM,
|
||||
DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR
|
||||
OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR
|
||||
THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
|
||||
Except as contained in this notice, the name of the Digital Equipment
|
||||
Corporation shall not be used in advertising or otherwise to promote
|
||||
the sale, use or other dealings in this Software without prior written
|
||||
authorization from Digital Equipment Corporation.
|
||||
|
||||
|
||||
Copyright 1996, 1998 The Open Group
|
||||
|
||||
Permission to use, copy, modify, distribute, and sell this software and its
|
||||
documentation for any purpose is hereby granted without fee, provided that
|
||||
the above copyright notice appear in all copies and that both that
|
||||
copyright notice and this permission notice appear in supporting
|
||||
documentation.
|
||||
|
||||
The above copyright notice and this permission notice shall be
|
||||
included in all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||
IN NO EVENT SHALL THE OPEN GROUP BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
OTHER DEALINGS IN THE SOFTWARE.
|
||||
|
||||
Except as contained in this notice, the name of The Open Group shall
|
||||
not be used in advertising or otherwise to promote the sale, use or
|
||||
other dealings in this Software without prior written authorization
|
||||
from The Open Group.
|
||||
|
||||
|
||||
Copyright 2004-2005 Sun Microsystems, Inc. All rights reserved.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a
|
||||
copy of this software and associated documentation files (the "Software"),
|
||||
to deal in the Software without restriction, including without limitation
|
||||
the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
and/or sell copies of the Software, and to permit persons to whom the
|
||||
Software is furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice (including the next
|
||||
paragraph) shall be included in all copies or substantial portions of the
|
||||
Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
|
||||
FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
DEALINGS IN THE SOFTWARE.
|
||||
|
||||
|
||||
Copyright (c) 1996 by Silicon Graphics Computer Systems, Inc.
|
||||
|
||||
Permission to use, copy, modify, and distribute this
|
||||
software and its documentation for any purpose and without
|
||||
fee is hereby granted, provided that the above copyright
|
||||
notice appear in all copies and that both that copyright
|
||||
notice and this permission notice appear in supporting
|
||||
documentation, and that the name of Silicon Graphics not be
|
||||
used in advertising or publicity pertaining to distribution
|
||||
of the software without specific prior written permission.
|
||||
Silicon Graphics makes no representation about the suitability
|
||||
of this software for any purpose. It is provided "as is"
|
||||
without any express or implied warranty.
|
||||
|
||||
SILICON GRAPHICS DISCLAIMS ALL WARRANTIES WITH REGARD TO THIS
|
||||
SOFTWARE, INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY
|
||||
AND FITNESS FOR A PARTICULAR PURPOSE. IN NO EVENT SHALL SILICON
|
||||
GRAPHICS BE LIABLE FOR ANY SPECIAL, INDIRECT OR CONSEQUENTIAL
|
||||
DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM LOSS OF USE,
|
||||
DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE
|
||||
OR OTHER TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH
|
||||
THE USE OR PERFORMANCE OF THIS SOFTWARE.
|
||||
|
||||
|
||||
Copyright (c) 1996 X Consortium
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining
|
||||
a copy of this software and associated documentation files (the
|
||||
"Software"), to deal in the Software without restriction, including
|
||||
without limitation the rights to use, copy, modify, merge, publish,
|
||||
distribute, sublicense, and/or sell copies of the Software, and to
|
||||
permit persons to whom the Software is furnished to do so, subject to
|
||||
the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be
|
||||
included in all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||
IN NO EVENT SHALL THE X CONSORTIUM BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
OTHER DEALINGS IN THE SOFTWARE.
|
||||
|
||||
Except as contained in this notice, the name of the X Consortium shall
|
||||
not be used in advertising or otherwise to promote the sale, use or
|
||||
other dealings in this Software without prior written authorization
|
||||
from the X Consortium.
|
||||
|
||||
|
||||
Copyright (C) 2004, 2006 Ævar Arnfjörð Bjarmason <avarab@gmail.com>
|
||||
|
||||
Permission to use, copy, modify, distribute, and sell this software and its
|
||||
documentation for any purpose is hereby granted without fee, provided that
|
||||
the above copyright notice appear in all copies and that both that
|
||||
copyright notice and this permission notice appear in supporting
|
||||
documentation.
|
||||
|
||||
The above copyright notice and this permission notice shall be
|
||||
included in all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||
IN NO EVENT SHALL THE OPEN GROUP BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
OTHER DEALINGS IN THE SOFTWARE.
|
||||
|
||||
Except as contained in this notice, the name of a copyright holder shall
|
||||
not be used in advertising or otherwise to promote the sale, use or
|
||||
other dealings in this Software without prior written authorization of
|
||||
the copyright holder.
|
||||
|
||||
|
||||
Copyright (C) 1999, 2000 by Anton Zinoviev <anton@lml.bas.bg>
|
||||
|
||||
This software may be used, modified, copied, distributed, and sold,
|
||||
in both source and binary form provided that the above copyright
|
||||
and these terms are retained. Under no circumstances is the author
|
||||
responsible for the proper functioning of this software, nor does
|
||||
the author assume any responsibility for damages incurred with its
|
||||
use.
|
||||
|
||||
Permission is granted to anyone to use, distribute and modify
|
||||
this file in any way, provided that the above copyright notice
|
||||
is left intact and the author of the modification summarizes
|
||||
the changes in this header.
|
||||
|
||||
This file is distributed without any expressed or implied warranty.
|
||||
+21
@@ -0,0 +1,21 @@
|
||||
# Vendored xkeyboard-config subset
|
||||
|
||||
- **Package**: xkeyboard-config 2.44
|
||||
- **Source**: https://gitlab.freedesktop.org/xkeyboard-config/xkeyboard-config/-/archive/xkeyboard-config-2.44/xkeyboard-config-2.44.tar.gz
|
||||
- **sha256**: `35e34edeaf4e8da8d0696ff6b241ee11ddb1b8c6730bac7252d4d0a88ea5f05b`
|
||||
- **keysymdef.h**: xorgproto, copied from `/opt/homebrew/include/X11/keysymdef.h`
|
||||
- **License**: MIT/X11 (see COPYING)
|
||||
|
||||
Only the symbols files reachable from the generated layouts (tools/make-xkeyboard-config.py `TARGETS`) are vendored; regenerate with
|
||||
`python3 tools/make-xkeyboard-config.py fetch` then `... generate`.
|
||||
|
||||
Vendored symbols files:
|
||||
|
||||
- `symbols/de`
|
||||
- `symbols/es`
|
||||
- `symbols/fr`
|
||||
- `symbols/gb`
|
||||
- `symbols/kpdl`
|
||||
- `symbols/latin`
|
||||
- `symbols/level3`
|
||||
- `symbols/us`
|
||||
+2584
File diff suppressed because it is too large
Load Diff
+1232
File diff suppressed because it is too large
Load Diff
+250
@@ -0,0 +1,250 @@
|
||||
// Keyboard layouts for Spain.
|
||||
|
||||
// Modified for a real Spanish keyboard by Jon Tombs.
|
||||
default partial alphanumeric_keys
|
||||
xkb_symbols "basic" {
|
||||
|
||||
include "latin(type4)"
|
||||
|
||||
name[Group1]="Spanish";
|
||||
|
||||
key <TLDE> { [ masculine, ordfeminine, backslash, backslash ] };
|
||||
key <AE01> { [ 1, exclam, bar, exclamdown ] };
|
||||
key <AE03> { [ 3, periodcentered, numbersign, sterling ] };
|
||||
key <AE04> { [ 4, dollar, asciitilde, dollar ] };
|
||||
key <AE11> { [apostrophe, question, backslash, questiondown ] };
|
||||
key <AE12> { [exclamdown, questiondown, dead_cedilla, dead_ogonek] };
|
||||
|
||||
key <AD11> { [dead_grave, dead_circumflex, bracketleft, dead_abovering ] };
|
||||
key <AD12> { [ plus, asterisk, bracketright, dead_macron ] };
|
||||
|
||||
key <AC10> { [ ntilde, Ntilde, dead_tilde, dead_doubleacute ] };
|
||||
key <AC11> { [dead_acute, dead_diaeresis, braceleft, dead_caron ] };
|
||||
key <BKSL> { [ ccedilla, Ccedilla, braceright, dead_breve ] };
|
||||
|
||||
include "level3(ralt_switch)"
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "winkeys" {
|
||||
|
||||
include "es(basic)"
|
||||
name[Group1]="Spanish (Windows)";
|
||||
include "eurosign(5)"
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "nodeadkeys" {
|
||||
|
||||
include "es(basic)"
|
||||
|
||||
name[Group1]="Spanish (no dead keys)";
|
||||
|
||||
key <AE12> { [exclamdown, questiondown, cedilla, ogonek ] };
|
||||
key <AD11> { [ grave, asciicircum, bracketleft, degree ] };
|
||||
key <AD12> { [ plus, asterisk, bracketright, macron ] };
|
||||
key <AC07> { [ j, J, ezh, EZH ] };
|
||||
key <AC10> { [ ntilde, Ntilde, asciitilde, doubleacute ] };
|
||||
key <AC11> { [ acute, diaeresis, braceleft, caron ] };
|
||||
key <BKSL> { [ ccedilla, Ccedilla, braceright, breve ] };
|
||||
key <AB10> { [ minus, underscore, ellipsis, abovedot ] };
|
||||
};
|
||||
|
||||
// Spanish Dvorak mapping (note R-H exchange)
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "dvorak" {
|
||||
|
||||
name[Group1]="Spanish (Dvorak)";
|
||||
|
||||
key <TLDE> {[ masculine, ordfeminine, backslash, degree ]};
|
||||
key <AE01> {[ 1, exclam, bar, onesuperior ]};
|
||||
key <AE02> {[ 2, quotedbl, at, twosuperior ]};
|
||||
key <AE03> {[ 3, periodcentered, numbersign, threesuperior ]};
|
||||
key <AE04> {[ 4, dollar, asciitilde, onequarter ]};
|
||||
key <AE05> {[ 5, percent, brokenbar, fiveeighths ]};
|
||||
key <AE06> {[ 6, ampersand, notsign, threequarters ]};
|
||||
key <AE07> {[ 7, slash, onehalf, seveneighths ]};
|
||||
key <AE08> {[ 8, parenleft, oneeighth, threeeighths ]};
|
||||
key <AE09> {[ 9, parenright, asciicircum ]};
|
||||
key <AE10> {[ 0, equal, grave, dead_doubleacute ]};
|
||||
key <AE11> {[ apostrophe, question, dead_macron, dead_ogonek ]};
|
||||
key <AE12> {[ exclamdown, questiondown, dead_breve, dead_abovedot ]};
|
||||
|
||||
key <AD01> {[ period, colon, less, guillemotleft ]};
|
||||
key <AD02> {[ comma, semicolon, greater, guillemotright ]};
|
||||
key <AD03> {[ ntilde, Ntilde, lstroke, Lstroke ]};
|
||||
key <AD04> {[ p, P, paragraph ]};
|
||||
key <AD05> {[ y, Y, yen ]};
|
||||
key <AD06> {[ f, F, tslash, Tslash ]};
|
||||
key <AD07> {[ g, G, dstroke, Dstroke ]};
|
||||
key <AD08> {[ c, C, cent, copyright ]};
|
||||
key <AD09> {[ h, H, hstroke, Hstroke ]};
|
||||
key <AD10> {[ l, L, sterling ]};
|
||||
key <AD11> {[ dead_grave, dead_circumflex, bracketleft, dead_caron ]};
|
||||
key <AD12> {[ plus, asterisk, bracketright, plusminus ]};
|
||||
|
||||
key <AC01> {[ a, A, ae, AE ]};
|
||||
key <AC02> {[ o, O, oslash, Oslash ]};
|
||||
key <AC03> {[ e, E, EuroSign ]};
|
||||
key <AC04> {[ u, U, aring, Aring ]};
|
||||
key <AC05> {[ i, I, oe, OE ]};
|
||||
key <AC06> {[ d, D, eth, ETH ]};
|
||||
key <AC07> {[ r, R, registered, trademark ]};
|
||||
key <AC08> {[ t, T, thorn, THORN ]};
|
||||
key <AC09> {[ n, N, eng, ENG ]};
|
||||
key <AC10> {[ s, S, ssharp, section ]};
|
||||
key <AC11> {[ dead_acute, dead_diaeresis, braceleft, dead_tilde ]};
|
||||
key <BKSL> {[ ccedilla, Ccedilla, braceright, dead_cedilla ]};
|
||||
|
||||
key <LSGT> {[ less, greater, guillemotleft, guillemotright ]};
|
||||
key <AB01> {[ minus, underscore, hyphen, macron ]};
|
||||
key <AB02> {[ q, Q, currency ]};
|
||||
key <AB03> {[ j, J ]};
|
||||
key <AB04> {[ k, K, kra ]};
|
||||
key <AB05> {[ x, X, multiply, division ]};
|
||||
key <AB06> {[ b, B ]};
|
||||
key <AB07> {[ m, M, mu ]};
|
||||
key <AB08> {[ w, W ]};
|
||||
key <AB09> {[ v, V ]};
|
||||
key <AB10> {[ z, Z ]};
|
||||
|
||||
include "level3(ralt_switch)"
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "cat" {
|
||||
|
||||
include "es(basic)"
|
||||
|
||||
name[Group1]="Catalan (Spain, with middle-dot L)";
|
||||
|
||||
key <AC09> { [ l, L, 0x1000140, 0x100013F ] };
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "ast" {
|
||||
|
||||
include "es(basic)"
|
||||
|
||||
name[Group1]="Asturian (Spain, with bottom-dot H and L)";
|
||||
|
||||
key <AC06> { [ h, H, 0x1001E25, 0x1001E24 ] };
|
||||
key <AC09> { [ l, L, 0x1001E37, 0x1001E36 ] };
|
||||
};
|
||||
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "olpc" {
|
||||
|
||||
// #HW-SPECIFIC
|
||||
|
||||
// http://wiki.laptop.org/go/OLPC_Spanish_Keyboard
|
||||
|
||||
include "us(basic)"
|
||||
name[Group1]="Spanish";
|
||||
|
||||
key <AE00> { [ masculine, ordfeminine ] };
|
||||
key <AE01> { [ 1, exclam, bar ] };
|
||||
key <AE02> { [ 2, quotedbl, at ] };
|
||||
key <AE03> { [ 3, dead_grave, numbersign, grave ] };
|
||||
key <AE05> { [ 5, percent, asciicircum, dead_circumflex ] };
|
||||
key <AE06> { [ 6, ampersand, notsign ] };
|
||||
key <AE07> { [ 7, slash, backslash ] };
|
||||
key <AE08> { [ 8, parenleft ] };
|
||||
key <AE09> { [ 9, parenright ] };
|
||||
key <AE10> { [ 0, equal ] };
|
||||
key <AE11> { [ apostrophe, question ] };
|
||||
key <AE12> { [ exclamdown, questiondown ] };
|
||||
|
||||
key <AD03> { [ e, E, EuroSign ] };
|
||||
key <AD11> { [ dead_acute, dead_diaeresis, acute, dead_abovering ] };
|
||||
key <AD12> { [ bracketleft, braceleft ] };
|
||||
|
||||
key <AC10> { [ ntilde, Ntilde ] };
|
||||
key <AC11> { [ plus, asterisk, dead_tilde ] };
|
||||
key <AC12> { [ bracketright, braceright, section ] };
|
||||
|
||||
key <AB08> { [ comma, semicolon ] };
|
||||
key <AB09> { [ period, colon ] };
|
||||
key <AB10> { [ minus, underscore ] };
|
||||
|
||||
key <I219> { [ less, greater, ISO_Next_Group ] };
|
||||
|
||||
include "level3(ralt_switch)"
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "olpcm" {
|
||||
|
||||
// #HW-SPECIFIC
|
||||
|
||||
// Mechanical (non-membrane) OLPC Spanish keyboard layout.
|
||||
// See: http://wiki.laptop.org/go/OLPC_Spanish_Non-membrane_Keyboard
|
||||
|
||||
include "us(basic)"
|
||||
name[Group1]="Spanish";
|
||||
|
||||
key <AE00> { [ questiondown, exclamdown, backslash ] };
|
||||
key <AE01> { [ 1, exclam, bar ] };
|
||||
key <AE02> { [ 2, quotedbl, at ] };
|
||||
key <AE03> { [ 3, dead_grave, numbersign, grave ] };
|
||||
key <AE04> { [ 4, dollar, asciitilde, dead_tilde ] };
|
||||
key <AE05> { [ 5, percent, asciicircum, dead_circumflex ] };
|
||||
key <AE06> { [ 6, ampersand, notsign ] };
|
||||
key <AE07> { [ 7, slash, backslash ] }; // no '\' label on olpcm, leave for compatibility
|
||||
key <AE08> { [ 8, parenleft, masculine ] };
|
||||
key <AE09> { [ 9, parenright, ordfeminine ] };
|
||||
key <AE10> { [ 0, equal ] };
|
||||
key <AE11> { [ apostrophe, question ] };
|
||||
|
||||
key <AD03> { [ e, E, EuroSign ] };
|
||||
key <AD11> { [ dead_acute, dead_diaeresis, dead_abovering, acute ] };
|
||||
key <AD12> { [ plus, asterisk ] };
|
||||
|
||||
key <AC10> { [ ntilde, Ntilde ] };
|
||||
// no AC11 or AC12 on olpcm
|
||||
|
||||
key <AB08> { [ comma, semicolon ] };
|
||||
key <AB09> { [ period, colon ] };
|
||||
key <AB10> { [ minus, underscore ] };
|
||||
|
||||
key <AA02> { [ less, greater ] };
|
||||
key <AA06> { [ bracketleft, braceleft, ccedilla, Ccedilla ] };
|
||||
key <AA07> { [ bracketright, braceright ] };
|
||||
|
||||
include "level3(ralt_switch)"
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "deadtilde" {
|
||||
|
||||
include "es(basic)"
|
||||
|
||||
name[Group1]="Spanish (dead tilde)";
|
||||
|
||||
key <AE04> { [ 4, dollar, dead_tilde, dollar ] };
|
||||
key <AC10> { [ ntilde, Ntilde, asciitilde, dead_doubleacute ] };
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "olpc2" {
|
||||
// #HW-SPECIFIC
|
||||
|
||||
// Modified variant of US International layout, specifically for Peru
|
||||
// Contact: Sayamindu Dasgupta <sayamindu@laptop.org>
|
||||
|
||||
include "us(olpc)"
|
||||
name[Group1]="Spanish";
|
||||
|
||||
key <AE03> { [ 3, numbersign, dead_grave, dead_grave] }; // combining grave
|
||||
key <I236> { [ XF86Start ] };
|
||||
|
||||
include "level3(ralt_switch)"
|
||||
};
|
||||
|
||||
// EXTRAS:
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "sun_type6" {
|
||||
include "sun_vndr/es(sun_type6)"
|
||||
};
|
||||
+1404
File diff suppressed because it is too large
Load Diff
+249
@@ -0,0 +1,249 @@
|
||||
// Keyboard layouts for Great Britain.
|
||||
|
||||
default partial alphanumeric_keys
|
||||
xkb_symbols "basic" {
|
||||
|
||||
// The basic UK layout, also known as the IBM 166 layout,
|
||||
// but with the useless brokenbar pushed two levels up.
|
||||
|
||||
include "latin"
|
||||
|
||||
name[Group1]="English (UK)";
|
||||
|
||||
key <TLDE> { [ grave, notsign, bar, bar ] };
|
||||
key <AE02> { [ 2, quotedbl, twosuperior, oneeighth ] };
|
||||
key <AE03> { [ 3, sterling, threesuperior, sterling ] };
|
||||
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||
|
||||
key <AC11> { [apostrophe, at, dead_circumflex, dead_caron] };
|
||||
key <BKSL> { [numbersign, asciitilde, dead_grave, dead_breve ] };
|
||||
|
||||
key <LSGT> { [ backslash, bar, bar, brokenbar ] };
|
||||
|
||||
include "level3(ralt_switch)"
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "intl" {
|
||||
|
||||
// A UK layout but with five accents made into dead keys:
|
||||
// grave, diaeresis, circumflex, acute, and tilde.
|
||||
// By Phil Jones <philjones1 at blueyonder.co.uk>.
|
||||
|
||||
include "latin"
|
||||
|
||||
name[Group1]="English (UK, intl., with dead keys)";
|
||||
|
||||
key <TLDE> { [ dead_grave, notsign, bar, bar ] };
|
||||
key <AE02> { [ 2, dead_diaeresis, twosuperior, onehalf ] };
|
||||
key <AE03> { [ 3, sterling, threesuperior, onethird ] };
|
||||
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||
key <AE06> { [ 6, dead_circumflex, threequarters, onesixth ] };
|
||||
|
||||
key <AC11> { [ dead_acute, at, apostrophe, bar ] };
|
||||
key <BKSL> { [ numbersign, dead_tilde, bar, bar ] };
|
||||
|
||||
key <LSGT> { [ backslash, bar, bar, bar ] };
|
||||
key <AB08> { [ comma, less, ccedilla, Ccedilla ] };
|
||||
|
||||
include "level3(ralt_switch)"
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "extd" {
|
||||
// Clone of the Microsoft "United Kingdom Extended" layout, which
|
||||
// includes dead keys for: grave; diaeresis; circumflex; tilde; and
|
||||
// accute. It also enables direct access to accute characters using
|
||||
// the Multi_key (Alt Gr).
|
||||
//
|
||||
// Taken from...
|
||||
// "Windows Keyboard Layouts"
|
||||
// https://docs.microsoft.com/en-gb/globalization/windows-keyboard-layouts#U
|
||||
//
|
||||
// -- Jonathan Miles <jon@cybah.co.uk>
|
||||
|
||||
include "latin"
|
||||
|
||||
name[Group1]="English (UK, extended, Windows)";
|
||||
|
||||
key <TLDE> { [ dead_grave, notsign, brokenbar, NoSymbol ] };
|
||||
key <AE02> { [ 2, quotedbl, dead_diaeresis, onehalf ] };
|
||||
key <AE03> { [ 3, sterling, threesuperior, onethird ] };
|
||||
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||
key <AE06> { [ 6, asciicircum, dead_circumflex, NoSymbol ] };
|
||||
|
||||
key <AD02> { [ w, W, wacute, Wacute ] };
|
||||
key <AD03> { [ e, E, eacute, Eacute ] };
|
||||
key <AD06> { [ y, Y, yacute, Yacute ] };
|
||||
key <AD07> { [ u, U, uacute, Uacute ] };
|
||||
key <AD08> { [ i, I, iacute, Iacute ] };
|
||||
key <AD09> { [ o, O, oacute, Oacute ] };
|
||||
key <AD12> { [ bracketright, braceright, NoSymbol, bar ] };
|
||||
|
||||
key <AC01> { [ a, A, aacute, Aacute ] };
|
||||
key <AC11> { [ apostrophe, at, dead_acute, grave ] };
|
||||
key <BKSL> { [ numbersign, asciitilde, dead_tilde, backslash ] };
|
||||
|
||||
key <LSGT> { [ backslash, bar, NoSymbol, NoSymbol ] };
|
||||
key <AB03> { [ c, C, ccedilla, Ccedilla ] };
|
||||
|
||||
include "level3(ralt_switch)"
|
||||
};
|
||||
|
||||
// Describe the differences between the US Colemak layout
|
||||
// and a UK variant. By Andy Buckley (andy@insectnation.org)
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "colemak" {
|
||||
include "us(colemak)"
|
||||
|
||||
name[Group1]="English (UK, Colemak)";
|
||||
|
||||
key <TLDE> { [ grave, notsign, bar, asciitilde ] };
|
||||
key <AE02> { [ 2, quotedbl, twosuperior, oneeighth ] };
|
||||
key <AE03> { [ 3, sterling, threesuperior, sterling ] };
|
||||
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||
|
||||
key <AC11> { [apostrophe, at, dead_circumflex, dead_caron] };
|
||||
key <BKSL> { [numbersign, asciitilde, dead_grave, dead_breve ] };
|
||||
|
||||
key <LSGT> { [ backslash, bar, asciitilde, brokenbar ] };
|
||||
};
|
||||
|
||||
// Colemak-DH (ISO) layout, UK Variant, https://colemakmods.github.io/mod-dh/
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "colemak_dh" {
|
||||
include "us(colemak_dh)"
|
||||
|
||||
name[Group1]="English (UK, Colemak-DH)";
|
||||
|
||||
key <TLDE> { [ grave, notsign, bar, asciitilde ] };
|
||||
key <AE02> { [ 2, quotedbl, twosuperior, oneeighth ] };
|
||||
key <AE03> { [ 3, sterling, threesuperior, sterling ] };
|
||||
key <AE04> { [ 4, dollar, EuroSign, onequarter ] };
|
||||
|
||||
key <AC11> { [apostrophe, at, dead_circumflex, dead_caron] };
|
||||
key <BKSL> { [numbersign, asciitilde, dead_grave, dead_breve ] };
|
||||
|
||||
key <AB05> { [ backslash, bar, asciitilde, brokenbar ] };
|
||||
};
|
||||
|
||||
|
||||
// Dvorak (UK) keymap (by odaen) allowing the usage of
|
||||
// the £ and ? key and swapping the @ and " keys.
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "dvorak" {
|
||||
include "us(dvorak-alt-intl)"
|
||||
|
||||
name[Group1]="English (UK, Dvorak)";
|
||||
|
||||
key <TLDE> { [ grave, notsign, bar, bar ] };
|
||||
key <AE02> { [ 2, quotedbl, twosuperior, NoSymbol ] };
|
||||
key <AE03> { [ 3, sterling, threesuperior, NoSymbol ] };
|
||||
key <AD01> { [ apostrophe, at ] };
|
||||
key <BKSL> { [ numbersign, asciitilde ] };
|
||||
key <LSGT> { [ backslash, bar ] };
|
||||
};
|
||||
|
||||
// Dvorak letter positions, but punctuation all in the normal UK positions.
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "dvorakukp" {
|
||||
include "gb(dvorak)"
|
||||
|
||||
name[Group1]="English (UK, Dvorak, with UK punctuation)";
|
||||
|
||||
key <AE11> { [ minus, underscore ] };
|
||||
key <AE12> { [ equal, plus ] };
|
||||
key <AD11> { [ bracketleft, braceleft ] };
|
||||
key <AD12> { [ bracketright, braceright ] };
|
||||
key <AD01> { [ slash, question ] };
|
||||
key <AC11> { [apostrophe, at, dead_circumflex, dead_caron] };
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "mac" {
|
||||
|
||||
include "latin"
|
||||
|
||||
name[Group1]= "English (UK, Macintosh)";
|
||||
|
||||
key <TLDE> { [ section, plusminus ] };
|
||||
key <AE02> { [ 2, at, EuroSign ] };
|
||||
key <AE03> { [ 3, sterling, numbersign ] };
|
||||
key <LSGT> { [ grave, asciitilde ] };
|
||||
|
||||
include "level3(ralt_switch)"
|
||||
include "level3(enter_switch)"
|
||||
};
|
||||
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "mac_intl" {
|
||||
|
||||
include "latin"
|
||||
|
||||
name[Group1]="English (UK, Macintosh, intl.)";
|
||||
|
||||
key <TLDE> { [ section, plusminus, notsign, notsign ] }; //dead_grave
|
||||
key <AE02> { [ 2, at, EuroSign, onehalf ] };
|
||||
key <AE03> { [ 3, sterling, twosuperior, onethird ] };
|
||||
key <AE04> { [ 4, dollar, threesuperior, onequarter ] };
|
||||
key <AE06> { [ 6, dead_circumflex, NoSymbol, onesixth ] };
|
||||
key <AD09> { [ o, O, oe, OE ] };
|
||||
|
||||
key <AC11> { [ dead_acute, dead_diaeresis, dead_diaeresis, bar ] }; //dead_doubleacute
|
||||
key <BKSL> { [ backslash, bar, numbersign, bar ] };
|
||||
|
||||
key <LSGT> { [ dead_grave, dead_tilde, brokenbar, bar ] };
|
||||
|
||||
include "level3(ralt_switch)"
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "pl" {
|
||||
|
||||
// Polish accented letters on upper levels of corresponding base letters.
|
||||
// Idea from Wawrzyniec Niewodniczański, adapted by Aleksander Kowalski.
|
||||
|
||||
include "gb(basic)"
|
||||
|
||||
name[Group1]="Polish (British keyboard)";
|
||||
|
||||
key <AD03> { [ e, E, eogonek, Eogonek ] };
|
||||
key <AD09> { [ o, O, oacute, Oacute ] };
|
||||
|
||||
key <AC01> { [ a, A, aogonek, Aogonek ] };
|
||||
key <AC02> { [ s, S, sacute, Sacute ] };
|
||||
|
||||
key <AB01> { [ z, Z, zabovedot, Zabovedot ] };
|
||||
key <AB02> { [ x, X, zacute, Zacute ] };
|
||||
key <AB03> { [ c, C, cacute, Cacute ] };
|
||||
key <AB06> { [ n, N, nacute, Nacute ] };
|
||||
};
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "gla" {
|
||||
|
||||
// Grave-accented letters on the upper levels of the relevant vowels.
|
||||
|
||||
include "gb(basic)"
|
||||
|
||||
name[Group1]="Scottish Gaelic";
|
||||
|
||||
key <AD03> { [ e, E, egrave, Egrave ] };
|
||||
key <AD07> { [ u, U, ugrave, Ugrave ] };
|
||||
key <AD08> { [ i, I, igrave, Igrave ] };
|
||||
key <AD09> { [ o, O, ograve, Ograve ] };
|
||||
|
||||
key <AC01> { [ a, A, agrave, Agrave ] };
|
||||
};
|
||||
|
||||
// EXTRAS:
|
||||
|
||||
partial alphanumeric_keys
|
||||
xkb_symbols "sun_type6" {
|
||||
include "sun_vndr/gb(sun_type6)"
|
||||
};
|
||||
+102
@@ -0,0 +1,102 @@
|
||||
// The <KPDL> key is a mess.
|
||||
// It was probably originally meant to be a decimal separator.
|
||||
// Except since it was declared by USA people it didn't use the original
|
||||
// SI separator "," but a "." (since then the USA managed to f-up the SI
|
||||
// by making "." an accepted alternative, but standards still use "," as
|
||||
// default)
|
||||
// As a result users of SI-abiding countries expect either a "." or a ","
|
||||
// or a "decimal_separator" which may or may not be translated in one of the
|
||||
// above depending on applications.
|
||||
// It's not possible to define a default per-country since user expectations
|
||||
// depend on the conflicting choices of their most-used applications,
|
||||
// operating system, etc. Therefore it needs to be a configuration setting
|
||||
// Copyright © 2007 Nicolas Mailhot <nicolas.mailhot @ laposte.net>
|
||||
|
||||
|
||||
// Legacy <KPDL> #1
|
||||
// This assumes KP_Decimal will be translated in a dot
|
||||
partial keypad_keys
|
||||
xkb_symbols "dot" {
|
||||
|
||||
key.type[Group1]="KEYPAD" ;
|
||||
|
||||
key <KPDL> { [ KP_Delete, KP_Decimal ] }; // <delete> <separator>
|
||||
};
|
||||
|
||||
|
||||
// Legacy <KPDL> #2
|
||||
// This assumes KP_Separator will be translated in a comma
|
||||
partial keypad_keys
|
||||
xkb_symbols "comma" {
|
||||
|
||||
key.type[Group1]="KEYPAD" ;
|
||||
|
||||
key <KPDL> { [ KP_Delete, KP_Separator ] }; // <delete> <separator>
|
||||
};
|
||||
|
||||
|
||||
// Period <KPDL>, usual keyboard serigraphy in most countries
|
||||
partial keypad_keys
|
||||
xkb_symbols "dotoss" {
|
||||
|
||||
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||
|
||||
key <KPDL> { [ KP_Delete, period, comma, 0x100202F ] }; // <delete> . , ⍽ (narrow no-break space)
|
||||
};
|
||||
|
||||
|
||||
// Period <KPDL>, usual keyboard serigraphy in most countries, latin-9 restriction
|
||||
partial keypad_keys
|
||||
xkb_symbols "dotoss_latin9" {
|
||||
|
||||
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||
|
||||
key <KPDL> { [ KP_Delete, period, comma, nobreakspace ] }; // <delete> . , ⍽ (no-break space)
|
||||
};
|
||||
|
||||
|
||||
// Comma <KPDL>, what most non anglo-saxon people consider the real separator
|
||||
partial keypad_keys
|
||||
xkb_symbols "commaoss" {
|
||||
|
||||
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||
|
||||
key <KPDL> { [ KP_Delete, comma, period, 0x100202F ] }; // <delete> , . ⍽ (narrow no-break space)
|
||||
};
|
||||
|
||||
|
||||
// Momayyez <KPDL>: Bahrain, Iran, Iraq, Kuwait, Oman, Qatar, Saudi Arabia, Syria, UAE
|
||||
partial keypad_keys
|
||||
xkb_symbols "momayyezoss" {
|
||||
|
||||
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||
|
||||
key <KPDL> { [ KP_Delete, 0x100066B, comma, 0x100202F ] }; // <delete> ? , ⍽ (narrow no-break space)
|
||||
};
|
||||
|
||||
|
||||
// Abstracted <KPDL>, pray everything will work out (it usually does not)
|
||||
partial keypad_keys
|
||||
xkb_symbols "kposs" {
|
||||
|
||||
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||
|
||||
key <KPDL> { [ KP_Delete, KP_Decimal, KP_Separator, 0x100202F ] }; // <delete> ? ? ⍽ (narrow no-break space)
|
||||
};
|
||||
|
||||
// Spreadsheets may be configured to use the dot as decimal
|
||||
// punctuation, comma as a thousands separator and then semi-colon as
|
||||
// the list separator. Of these, dot and semi-colon is most important
|
||||
// when entering data by the keyboard; the comma can then be inferred
|
||||
// and added to the presentation afterwards. Using semi-colon as a
|
||||
// general separator may in fact be preferred to avoid ambiguities
|
||||
// in data files. Most times a decimal separator is hard-coded, it
|
||||
// seems to be period, probably since this is the syntax used in
|
||||
// (most) programming languages.
|
||||
partial keypad_keys
|
||||
xkb_symbols "semi" {
|
||||
|
||||
key.type[Group1]="FOUR_LEVEL_MIXED_KEYPAD" ;
|
||||
|
||||
key <KPDL> { [ NoSymbol, NoSymbol, semicolon ] };
|
||||
};
|
||||
+255
@@ -0,0 +1,255 @@
|
||||
// Common Latin alphabet layout
|
||||
|
||||
default partial
|
||||
xkb_symbols "basic" {
|
||||
|
||||
key <AE01> { [ 1, exclam, onesuperior, exclamdown ] };
|
||||
key <AE02> { [ 2, at, twosuperior, oneeighth ] };
|
||||
key <AE03> { [ 3, numbersign, threesuperior, sterling ] };
|
||||
key <AE04> { [ 4, dollar, onequarter, dollar ] };
|
||||
key <AE05> { [ 5, percent, onehalf, threeeighths ] };
|
||||
key <AE06> { [ 6, asciicircum, threequarters, fiveeighths ] };
|
||||
key <AE07> { [ 7, ampersand, braceleft, seveneighths ] };
|
||||
key <AE08> { [ 8, asterisk, bracketleft, trademark ] };
|
||||
key <AE09> { [ 9, parenleft, bracketright, plusminus ] };
|
||||
key <AE10> { [ 0, parenright, braceright, degree ] };
|
||||
key <AE11> { [ minus, underscore, backslash, questiondown ] };
|
||||
key <AE12> { [ equal, plus, dead_cedilla, dead_ogonek ] };
|
||||
|
||||
key <AD01> { [ q, Q, at, Greek_OMEGA ] };
|
||||
key <AD02> { [ w, W, U017F, section ] };
|
||||
key <AD03> { [ e, E, e, E ] };
|
||||
key <AD04> { [ r, R, paragraph, registered ] };
|
||||
key <AD05> { [ t, T, tslash, Tslash ] };
|
||||
key <AD06> { [ y, Y, leftarrow, yen ] };
|
||||
key <AD07> { [ u, U, downarrow, uparrow ] };
|
||||
key <AD08> { [ i, I, rightarrow, idotless ] };
|
||||
key <AD09> { [ o, O, oslash, Oslash ] };
|
||||
key <AD10> { [ p, P, thorn, THORN ] };
|
||||
key <AD11> { [bracketleft, braceleft, dead_diaeresis, dead_abovering ] };
|
||||
key <AD12> { [bracketright, braceright, dead_tilde, dead_macron ] };
|
||||
|
||||
key <AC01> { [ a, A, ae, AE ] };
|
||||
key <AC02> { [ s, S, ssharp, U1E9E ] };
|
||||
key <AC03> { [ d, D, eth, ETH ] };
|
||||
key <AC04> { [ f, F, dstroke, ordfeminine ] };
|
||||
key <AC05> { [ g, G, eng, ENG ] };
|
||||
key <AC06> { [ h, H, hstroke, Hstroke ] };
|
||||
key <AC07> { [ j, J, dead_hook, dead_horn ] };
|
||||
key <AC08> { [ k, K, kra, ampersand ] };
|
||||
key <AC09> { [ l, L, lstroke, Lstroke ] };
|
||||
key <AC10> { [ semicolon, colon, dead_acute, dead_doubleacute ] };
|
||||
key <AC11> { [apostrophe, quotedbl, dead_circumflex, dead_caron ] };
|
||||
key <TLDE> { [ grave, asciitilde, notsign, notsign ] };
|
||||
|
||||
key <BKSL> { [ backslash, bar, dead_grave, dead_breve ] };
|
||||
key <AB01> { [ z, Z, guillemotleft, less ] };
|
||||
key <AB02> { [ x, X, guillemotright, greater ] };
|
||||
key <AB03> { [ c, C, cent, copyright ] };
|
||||
key <AB04> { [ v, V, doublelowquotemark, singlelowquotemark ] };
|
||||
key <AB05> { [ b, B, leftdoublequotemark, leftsinglequotemark ] };
|
||||
key <AB06> { [ n, N, rightdoublequotemark, rightsinglequotemark ] };
|
||||
key <AB07> { [ m, M, mu, masculine ] };
|
||||
key <AB08> { [ comma, less, U2022, multiply ] }; // bullet
|
||||
key <AB09> { [ period, greater, periodcentered, division ] };
|
||||
key <AB10> { [ slash, question, dead_belowdot, dead_abovedot ] };
|
||||
};
|
||||
|
||||
// Northern Europe ( Danish, Finnish, Norwegian, Swedish) common layout
|
||||
|
||||
partial
|
||||
xkb_symbols "type2" {
|
||||
|
||||
include "latin"
|
||||
|
||||
key <AE01> { [ 1, exclam, exclamdown, onesuperior ] };
|
||||
key <AE02> { [ 2, quotedbl, at, twosuperior ] };
|
||||
key <AE03> { [ 3, numbersign, sterling, threesuperior] };
|
||||
key <AE04> { [ 4, currency, dollar, onequarter ] };
|
||||
key <AE05> { [ 5, percent, onehalf, cent ] };
|
||||
key <AE06> { [ 6, ampersand, yen, fiveeighths ] };
|
||||
key <AE07> { [ 7, slash, braceleft, division ] };
|
||||
key <AE08> { [ 8, parenleft, bracketleft, guillemotleft] };
|
||||
key <AE09> { [ 9, parenright, bracketright, guillemotright] };
|
||||
key <AE10> { [ 0, equal, braceright, degree ] };
|
||||
|
||||
key <AD03> { [ e, E, EuroSign, cent ] };
|
||||
key <AD04> { [ r, R, registered, registered ] };
|
||||
key <AD05> { [ t, T, thorn, THORN ] };
|
||||
key <AD09> { [ o, O, oe, OE ] };
|
||||
key <AD11> { [ aring, Aring, dead_diaeresis, dead_abovering ] };
|
||||
key <AD12> { [dead_diaeresis, dead_circumflex, dead_tilde, dead_caron ] };
|
||||
|
||||
key <AC01> { [ a, A, ordfeminine, masculine ] };
|
||||
|
||||
key <AB03> { [ c, C, copyright, copyright ] };
|
||||
key <AB08> { [ comma, semicolon, dead_cedilla, dead_ogonek ] };
|
||||
key <AB09> { [ period, colon, periodcentered, dead_abovedot ] };
|
||||
key <AB10> { [ minus, underscore, dead_belowdot, dead_abovedot ] };
|
||||
};
|
||||
|
||||
// Slavic Latin ( Albanian, Croatian, Polish, Slovene, Yugoslav)
|
||||
// common layout
|
||||
|
||||
partial
|
||||
xkb_symbols "type3" {
|
||||
|
||||
include "latin"
|
||||
|
||||
key <AD01> { [ q, Q, backslash, Greek_OMEGA ] };
|
||||
key <AD02> { [ w, W, bar, section ] };
|
||||
key <AD06> { [ z, Z, leftarrow, yen ] };
|
||||
|
||||
key <AC04> { [ f, F, bracketleft, ordfeminine ] };
|
||||
key <AC05> { [ g, G, bracketright, ENG ] };
|
||||
key <AC08> { [ k, K, lstroke, ampersand ] };
|
||||
|
||||
key <AB01> { [ y, Y, guillemotleft, less ] };
|
||||
key <AB04> { [ v, V, at, grave ] };
|
||||
key <AB05> { [ b, B, braceleft, apostrophe ] };
|
||||
key <AB06> { [ n, N, braceright, acute ] };
|
||||
key <AB07> { [ m, M, section, masculine ] };
|
||||
key <AB08> { [ comma, semicolon, less, multiply ] };
|
||||
key <AB09> { [ period, colon, greater, division ] };
|
||||
};
|
||||
|
||||
// Another common Latin layout
|
||||
// (German, Estonian, Spanish, Icelandic, Italian, Latin American, Portuguese)
|
||||
|
||||
partial
|
||||
xkb_symbols "type4" {
|
||||
|
||||
include "latin"
|
||||
|
||||
key <AE02> { [ 2, quotedbl, at, oneeighth ] };
|
||||
key <AE06> { [ 6, ampersand, notsign, fiveeighths ] };
|
||||
key <AE07> { [ 7, slash, braceleft, seveneighths ] };
|
||||
key <AE08> { [ 8, parenleft, bracketleft, trademark ] };
|
||||
key <AE09> { [ 9, parenright, bracketright, plusminus ] };
|
||||
key <AE10> { [ 0, equal, braceright, degree ] };
|
||||
|
||||
key <AD03> { [ e, E, EuroSign, cent ] };
|
||||
|
||||
key <AB08> { [ comma, semicolon, U2022, multiply ] }; // bullet
|
||||
key <AB09> { [ period, colon, periodcentered, division ] };
|
||||
key <AB10> { [ minus, underscore, dead_belowdot, dead_abovedot ] };
|
||||
};
|
||||
|
||||
partial
|
||||
xkb_symbols "nodeadkeys" {
|
||||
|
||||
key <AE12> { [ equal, plus, cedilla, ogonek ] };
|
||||
key <AD11> { [bracketleft, braceleft, diaeresis, degree ] };
|
||||
key <AD12> { [bracketright, braceright, asciitilde, macron ] };
|
||||
key <AC07> { [ j, J, ezh, EZH ] };
|
||||
key <AC10> { [ semicolon, colon, acute, doubleacute ] };
|
||||
key <AC11> { [apostrophe, quotedbl, asciicircum, caron ] };
|
||||
key <BKSL> { [ backslash, bar, grave, breve ] };
|
||||
key <AB10> { [ slash, question, ellipsis, abovedot ] };
|
||||
};
|
||||
|
||||
partial
|
||||
xkb_symbols "type2_nodeadkeys" {
|
||||
|
||||
include "latin(nodeadkeys)"
|
||||
|
||||
key <AD11> { [ aring, Aring, diaeresis, degree ] };
|
||||
key <AD12> { [ diaeresis, asciicircum, asciitilde, caron ] };
|
||||
key <AB08> { [ comma, semicolon, cedilla, ogonek ] };
|
||||
key <AB09> { [ period, colon, periodcentered, abovedot ] };
|
||||
key <AB10> { [ minus, underscore, ellipsis, abovedot ] };
|
||||
};
|
||||
|
||||
partial
|
||||
xkb_symbols "type3_nodeadkeys" {
|
||||
|
||||
include "latin(nodeadkeys)"
|
||||
};
|
||||
|
||||
partial
|
||||
xkb_symbols "type4_nodeadkeys" {
|
||||
|
||||
include "latin(nodeadkeys)"
|
||||
|
||||
key <AB10> { [ minus, underscore, ellipsis, abovedot ] };
|
||||
};
|
||||
|
||||
// Added 2008.03.05 by Marcin Woliński
|
||||
// See http://marcinwolinski.pl/keyboard/ for a description.
|
||||
// Used by pl(intl)
|
||||
//
|
||||
// ┌─────┐
|
||||
// │ 2 4 │ 2 = Shift, 4 = Level3 + Shift
|
||||
// │ 1 3 │ 1 = Normal, 3 = Level3
|
||||
// └─────┘
|
||||
// ┌─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┬─────┲━━━━━━━━━┓
|
||||
// │ ~ ~ │ ! ' │ @ " │ # ˝ │ $ ¸ │ % ˇ │ ^ ^ │ & ˘ │ * ̇ │ ( ̣ │ ) ° │ _ ¯ │ + ˛ ┃ ⌫ Back- ┃
|
||||
// │ ` ` │ 1 ¡ │ 2 © │ 3 • │ 4 § │ 5 € │ 6 ¢ │ 7 − │ 8 × │ 9 ÷ │ 0 ° │ - – │ = — ┃ space ┃
|
||||
// ┢━━━━━┷━┱───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┴─┬───┺━┳━━━━━━━┫
|
||||
// ┃ ┃ Q │ W │ E │ R │ T │ Y │ U │ I │ O │ P │ { « │ } » ┃ Enter ┃
|
||||
// ┃Tab ↹ ┃ q │ w │ e │ r │ t │ y │ u │ i │ o │ p │ [ ‹ │ ] › ┃ ⏎ ┃
|
||||
// ┣━━━━━━━┻┱────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┴┬────┺┓ ┃
|
||||
// ┃ ┃ A │ S │ D │ F │ G │ H │ J │ K │ L │ : “ │ " ” │ | ¶ ┃ ┃
|
||||
// ┃Caps ⇬ ┃ a │ s │ d │ f │ g │ h │ j │ k │ l │ ; ‘ │ ' ’ │ \ ┃ ┃
|
||||
// ┣━━━━━━━━┹────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┬┴────┲┷━━━━━┻━━━━━━┫
|
||||
// ┃ │ Z │ X │ C │ V │ B │ N │ M │ < „ │ > · │ ? ¿ ┃ ┃
|
||||
// ┃Shift ⇧ │ z │ x │ c │ v │ b │ n │ m │ , ‚ │ . … │ / ⁄ ┃Shift ⇧ ┃
|
||||
// ┣━━━━━━━┳━━━━━┷━┳━━━┷━━━┱─┴─────┴─────┴─────┴─────┴─────┴───┲━┷━━━━━╈━━━━━┻━┳━━━━━━━┳━━━┛
|
||||
// ┃ ┃ ┃ ┃ ␣ ⍽ ┃ ┃ ┃ ┃
|
||||
// ┃Ctrl ┃Meta ┃Alt ┃ ␣ Space ⍽ ┃AltGr ⇮┃Menu ┃Ctrl ┃
|
||||
// ┗━━━━━━━┻━━━━━━━┻━━━━━━━┹───────────────────────────────────┺━━━━━━━┻━━━━━━━┻━━━━━━━┛
|
||||
|
||||
partial
|
||||
xkb_symbols "intl" {
|
||||
|
||||
key <TLDE> { [ grave, asciitilde, dead_grave, dead_tilde ] };
|
||||
key <AE01> { [ 1, exclam, exclamdown, dead_acute ] };
|
||||
key <AE02> { [ 2, at, copyright, dead_diaeresis ] };
|
||||
key <AE03> { [ 3, numbersign, U2022, dead_doubleacute ] }; // U+2022 is bullet (the name bullet does not work)
|
||||
key <AE04> { [ 4, dollar, section, dead_cedilla ] };
|
||||
key <AE05> { [ 5, percent, EuroSign, dead_caron ] };
|
||||
key <AE06> { [ 6, asciicircum, cent, dead_circumflex ] };
|
||||
key <AE07> { [ 7, ampersand, U2212, dead_breve ] }; // U+2212 is MINUS SIGN
|
||||
key <AE08> { [ 8, asterisk, multiply, dead_abovedot ] };
|
||||
key <AE09> { [ 9, parenleft, division, dead_belowdot ] };
|
||||
key <AE10> { [ 0, parenright, degree, dead_abovering ] };
|
||||
key <AE11> { [ minus, underscore, endash, dead_macron ] };
|
||||
key <AE12> { [ equal, plus, emdash, dead_ogonek ] };
|
||||
|
||||
key <AD01> { [ q, Q ] };
|
||||
key <AD02> { [ w, W ] };
|
||||
key <AD03> { [ e, E ] };
|
||||
key <AD04> { [ r, R ] };
|
||||
key <AD05> { [ t, T ] };
|
||||
key <AD06> { [ y, Y ] };
|
||||
key <AD07> { [ u, U ] };
|
||||
key <AD08> { [ i, I ] };
|
||||
key <AD09> { [ o, O ] };
|
||||
key <AD10> { [ p, P ] };
|
||||
key <AD11> { [bracketleft, braceleft, U2039, guillemotleft ] };
|
||||
key <AD12> { [bracketright, braceright, U203A, guillemotright ] };
|
||||
|
||||
key <AC01> { [ a, A ] };
|
||||
key <AC02> { [ s, S ] };
|
||||
key <AC03> { [ d, D ] };
|
||||
key <AC04> { [ f, F ] };
|
||||
key <AC05> { [ g, G ] };
|
||||
key <AC06> { [ h, H ] };
|
||||
key <AC07> { [ j, J ] };
|
||||
key <AC08> { [ k, K ] };
|
||||
key <AC09> { [ l, L ] };
|
||||
key <AC10> { [ semicolon, colon, leftsinglequotemark, leftdoublequotemark ] };
|
||||
key <AC11> { [apostrophe, quotedbl, rightsinglequotemark, rightdoublequotemark ] };
|
||||
|
||||
key <BKSL> { [ backslash, bar, NoSymbol, paragraph ] };
|
||||
key <AB01> { [ z, Z ] };
|
||||
key <AB02> { [ x, X ] };
|
||||
key <AB03> { [ c, C ] };
|
||||
key <AB04> { [ v, V ] };
|
||||
key <AB05> { [ b, B ] };
|
||||
key <AB06> { [ n, N ] };
|
||||
key <AB07> { [ m, M ] };
|
||||
key <AB08> { [ comma, less, singlelowquotemark, doublelowquotemark ] };
|
||||
key <AB09> { [ period, greater, ellipsis, periodcentered ] };
|
||||
key <AB10> { [ slash, question, U2044, questiondown ] }; // U+2044 is FRACTION SLASH
|
||||
};
|
||||
+156
@@ -0,0 +1,156 @@
|
||||
// These variants assign ISO_Level3_Shift to various keys
|
||||
// so that levels 3 and 4 can be reached.
|
||||
|
||||
// The default behaviour:
|
||||
// the right Alt key (AltGr) chooses the third symbol engraved on a key.
|
||||
default partial modifier_keys
|
||||
xkb_symbols "ralt_switch" {
|
||||
key <RALT> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||
};
|
||||
|
||||
// The right Alt key never chooses the third level.
|
||||
// This option attempts to undo the effect of a layout's inclusion of
|
||||
// 'ralt_switch'. You may want to also select another level3 option
|
||||
// to map the level3 shift to some other key.
|
||||
partial modifier_keys
|
||||
xkb_symbols "ralt_alt" {
|
||||
key <RALT> {[ Alt_R, Meta_R ], type[group1]="TWO_LEVEL" };
|
||||
modifier_map Mod1 { <RALT> };
|
||||
};
|
||||
|
||||
// The right Alt key (while pressed) chooses the third shift level,
|
||||
// and Compose is mapped to its second level.
|
||||
partial modifier_keys
|
||||
xkb_symbols "ralt_switch_multikey" {
|
||||
key <RALT> {[ ISO_Level3_Shift, Multi_key ], type[group1]="TWO_LEVEL" };
|
||||
};
|
||||
|
||||
// Either Alt key (while pressed) chooses the third shift level.
|
||||
// (To be used mostly to imitate Mac OS functionality.)
|
||||
partial modifier_keys
|
||||
xkb_symbols "alt_switch" {
|
||||
include "level3(lalt_switch)"
|
||||
include "level3(ralt_switch)"
|
||||
};
|
||||
|
||||
// The left Alt key (while pressed) chooses the third shift level.
|
||||
partial modifier_keys
|
||||
xkb_symbols "lalt_switch" {
|
||||
key <LALT> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||
};
|
||||
|
||||
// The right Ctrl key (while pressed) chooses the third shift level.
|
||||
partial modifier_keys
|
||||
xkb_symbols "switch" {
|
||||
key <RCTL> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||
};
|
||||
|
||||
// The Menu key (while pressed) chooses the third shift level.
|
||||
partial modifier_keys
|
||||
xkb_symbols "menu_switch" {
|
||||
key <MENU> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||
};
|
||||
|
||||
// Either Win key (while pressed) chooses the third shift level.
|
||||
partial modifier_keys
|
||||
xkb_symbols "win_switch" {
|
||||
include "level3(lwin_switch)"
|
||||
include "level3(rwin_switch)"
|
||||
};
|
||||
|
||||
// The left Win key (while pressed) chooses the third shift level.
|
||||
partial modifier_keys
|
||||
xkb_symbols "lwin_switch" {
|
||||
key <LWIN> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||
};
|
||||
|
||||
// The right Win key (while pressed) chooses the third shift level.
|
||||
partial modifier_keys
|
||||
xkb_symbols "rwin_switch" {
|
||||
key <RWIN> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||
};
|
||||
|
||||
// The Enter key on the kepypad (while pressed) chooses the third shift level.
|
||||
// (This is especially useful for Mac laptops which miss the right Alt key.)
|
||||
partial modifier_keys
|
||||
xkb_symbols "enter_switch" {
|
||||
key <KPEN> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||
};
|
||||
|
||||
// The CapsLock key (while pressed) chooses the third shift level.
|
||||
partial modifier_keys
|
||||
xkb_symbols "caps_switch" {
|
||||
key <CAPS> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||
};
|
||||
|
||||
// The CapsLock key (while pressed) chooses the third shift level and
|
||||
// Ctrl + CapsLock has the original CapsLock function.
|
||||
// The 2023 DIN standard for German keyboards recommends it as an option:
|
||||
// - https://de.wikipedia.org/wiki/E1_(Tastaturbelegung)#Feststelltaste/Umschaltsperre
|
||||
// - https://en.wikipedia.org/wiki/Caps_Lock#Abolition
|
||||
partial modifier_keys
|
||||
xkb_symbols "caps_switch_capslock_with_ctrl" {
|
||||
virtual_modifiers LevelThree;
|
||||
|
||||
key <CAPS> {
|
||||
type[Group1] = "PC_CONTROL_LEVEL2",
|
||||
symbols[Group1] = [ ISO_Level3_Shift, Caps_Lock ],
|
||||
// Explicit actions are preferred over modMap None/Mod5 { Caps_Lock }
|
||||
// because they have no side effect
|
||||
actions[Group1] = [ SetMods(modifiers = LevelThree), LockMods(modifiers = Lock) ]
|
||||
};
|
||||
};
|
||||
|
||||
// The Backslash key (while pressed) chooses the third shift level.
|
||||
partial modifier_keys
|
||||
xkb_symbols "bksl_switch" {
|
||||
key <BKSL> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||
};
|
||||
|
||||
// The AC11 key (while pressed) chooses the third shift level.
|
||||
partial modifier_keys
|
||||
xkb_symbols "ac11_switch" {
|
||||
key <AC11> {[ ISO_Level3_Shift ], type[Group1]="ONE_LEVEL" };
|
||||
};
|
||||
|
||||
// The Less/Greater key (while pressed) chooses the third shift level.
|
||||
partial modifier_keys
|
||||
xkb_symbols "lsgt_switch" {
|
||||
key <LSGT> {[ ISO_Level3_Shift ], type[group1]="ONE_LEVEL" };
|
||||
};
|
||||
|
||||
// The CapsLock key (while pressed) chooses the third shift level,
|
||||
// and latches when pressed together with another third-level chooser.
|
||||
partial modifier_keys
|
||||
xkb_symbols "caps_switch_latch" {
|
||||
key <CAPS> {[ ISO_Level3_Shift, ISO_Level3_Shift, ISO_Level3_Latch ],
|
||||
type[group1]="THREE_LEVEL" };
|
||||
};
|
||||
|
||||
// The Backslash key (while pressed) chooses the third shift level,
|
||||
// and latches when pressed together with another third-level chooser.
|
||||
partial modifier_keys
|
||||
xkb_symbols "bksl_switch_latch" {
|
||||
key <BKSL> {[ ISO_Level3_Shift, ISO_Level3_Shift, ISO_Level3_Latch ],
|
||||
type[group1]="THREE_LEVEL" };
|
||||
};
|
||||
|
||||
// The Less/Greater key (while pressed) chooses the third shift level,
|
||||
// and latches when pressed together with another third-level chooser.
|
||||
partial modifier_keys
|
||||
xkb_symbols "lsgt_switch_latch" {
|
||||
key <LSGT> {[ ISO_Level3_Shift, ISO_Level3_Shift, ISO_Level3_Latch ],
|
||||
type[group1]="THREE_LEVEL" };
|
||||
};
|
||||
|
||||
// Top-row digit key 4 chooses third shift level when pressed alone.
|
||||
partial modifier_keys
|
||||
xkb_symbols "4_switch_isolated" {
|
||||
override key <AE04> {[ ISO_Level3_Shift ]};
|
||||
};
|
||||
|
||||
// Top-row digit key 9 chooses third shift level when pressed alone.
|
||||
partial modifier_keys
|
||||
xkb_symbols "9_switch_isolated" {
|
||||
override key <AE09> {[ ISO_Level3_Shift ]};
|
||||
};
|
||||
+2238
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,153 @@
|
||||
//! xkeyboard-config — keyboard layouts, compiled from the X11 xkeyboard-config database
|
||||
//! into native Zig. It turns a physical key (a USB HID usage, as the input module delivers)
|
||||
//! plus a modifier state into a **keysym** and, when the key produces one, a **character**
|
||||
//! (a Unicode scalar). This is the piece that lets a `KeyEvent.keycode` become a
|
||||
//! `KeyEvent.character`, without shipping an X11 runtime.
|
||||
//!
|
||||
//! The layout tables in `generated/layouts.zig` are produced by
|
||||
//! `tools/make-xkeyboard-config.py` (see ./README.md to regenerate). Those tables are
|
||||
//! deliberately pure data — each key carries its up-to-four levels and an XKB *type*. The
|
||||
//! type -> level selection semantics (which modifier picks which level) live here, so the
|
||||
//! data and the policy are separable.
|
||||
//!
|
||||
//! Scope (documented in README.md): group 1 only, no dead-key/compose composition (a dead
|
||||
//! key returns its keysym with no character), and a curated set of key types. Layouts:
|
||||
//! us, gb, de, fr, es, dvorak.
|
||||
//!
|
||||
//! Upstream xkeyboard-config and keysymdef.h are MIT/X11 licensed; see vendor/COPYING and
|
||||
//! vendor/PROVENANCE.md.
|
||||
|
||||
const std = @import("std");
|
||||
const generated = @import("layouts");
|
||||
|
||||
pub const Level = generated.Level;
|
||||
pub const KeyType = generated.KeyType;
|
||||
pub const Key = generated.Key;
|
||||
pub const Layout = generated.Layout;
|
||||
|
||||
/// The generated layouts, by name — as pointers, so they share identity with `all` and
|
||||
/// `byName` (and match the `*const Layout` that `map` takes).
|
||||
pub const us: *const Layout = &generated.us;
|
||||
pub const gb: *const Layout = &generated.gb;
|
||||
pub const de: *const Layout = &generated.de;
|
||||
pub const fr: *const Layout = &generated.fr;
|
||||
pub const es: *const Layout = &generated.es;
|
||||
pub const dvorak: *const Layout = &generated.dvorak;
|
||||
|
||||
/// Every generated layout, for enumeration (e.g. a settings UI).
|
||||
pub const all = generated.all;
|
||||
|
||||
/// The modifier state that selects a key's level. `level3` is AltGr (ISO Level3 Shift);
|
||||
/// `control` is accepted for completeness but does not affect level selection here.
|
||||
pub const Modifiers = struct {
|
||||
shift: bool = false,
|
||||
caps_lock: bool = false,
|
||||
level3: bool = false,
|
||||
control: bool = false,
|
||||
};
|
||||
|
||||
/// The result of a lookup: the X11 `keysym`, and the `character` it produces (a Unicode
|
||||
/// scalar) when it is a printable key — null for keys that produce none (Return, F1, a
|
||||
/// bare dead key, an unmapped key).
|
||||
pub const Mapping = struct {
|
||||
keysym: u32,
|
||||
character: ?u21,
|
||||
};
|
||||
|
||||
/// Which level (0..3) a key of `kind` selects under `mods`. XKB's canonical semantics:
|
||||
/// Shift picks the odd level, AltGr (level3) adds 2, and Caps acts like Shift for the
|
||||
/// alphabetic types. See the XKB "key types" — this covers the ones the vendored layouts
|
||||
/// use; anything else falls back to shift-or-not.
|
||||
fn selectLevel(kind: KeyType, mods: Modifiers) usize {
|
||||
const shift_or_caps = mods.shift != mods.caps_lock; // XOR: Caps behaves like Shift
|
||||
const low: usize = if (mods.shift) 1 else 0;
|
||||
const high: usize = if (mods.level3) 2 else 0;
|
||||
return switch (kind) {
|
||||
.one_level => 0,
|
||||
.two_level, .keypad, .other => low,
|
||||
.alphabetic => if (shift_or_caps) 1 else 0,
|
||||
.four_level => low + high,
|
||||
.four_level_alphabetic => (if (shift_or_caps) @as(usize, 1) else 0) + high,
|
||||
// Caps affects only the base pair, not the AltGr pair.
|
||||
.four_level_semialphabetic => if (mods.level3) 2 + low else (if (shift_or_caps) @as(usize, 1) else 0),
|
||||
};
|
||||
}
|
||||
|
||||
/// Map a physical key (`hid_usage`, a USB HID keyboard-page usage) under `mods` on
|
||||
/// `layout` to its keysym and character. Falls back gracefully when the selected level is
|
||||
/// undefined for the key: it drops the AltGr component, then the shift component, so a key
|
||||
/// with only a base/shift pair still yields something sensible under AltGr.
|
||||
pub fn map(layout: *const Layout, hid_usage: u8, mods: Modifiers) Mapping {
|
||||
const key = &layout.keys[hid_usage];
|
||||
var level = selectLevel(key.kind, mods);
|
||||
// Fall back to a defined level: full -> without AltGr -> base.
|
||||
if (key.levels[level].keysym == 0 and key.levels[level].unicode == 0) {
|
||||
const candidates = [_]usize{ level & 1, 0 };
|
||||
for (candidates) |candidate| {
|
||||
if (key.levels[candidate].keysym != 0 or key.levels[candidate].unicode != 0) {
|
||||
level = candidate;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
const chosen = key.levels[level];
|
||||
return .{
|
||||
.keysym = chosen.keysym,
|
||||
.character = if (chosen.unicode != 0) @intCast(chosen.unicode) else null,
|
||||
};
|
||||
}
|
||||
|
||||
/// Look up a layout by its name (`"us"`, `"gb"`, ...), or null if unknown.
|
||||
pub fn byName(name: []const u8) ?*const Layout {
|
||||
for (all) |layout| {
|
||||
if (std.mem.eql(u8, layout.name, name)) return layout;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- tests (host-run via `zig build test`) ---------------------------------
|
||||
|
||||
const testing = std.testing;
|
||||
|
||||
// USB HID usages used in the tests (keyboard page 0x07).
|
||||
const hid_a: u8 = 0x04;
|
||||
const hid_1: u8 = 0x1e;
|
||||
const hid_3: u8 = 0x20;
|
||||
|
||||
test "us: letters obey shift and caps" {
|
||||
try testing.expectEqual(@as(?u21, 'a'), map(us, hid_a, .{}).character);
|
||||
try testing.expectEqual(@as(?u21, 'A'), map(us, hid_a, .{ .shift = true }).character);
|
||||
try testing.expectEqual(@as(?u21, 'A'), map(us, hid_a, .{ .caps_lock = true }).character);
|
||||
// Shift + Caps cancels for an alphabetic key.
|
||||
try testing.expectEqual(@as(?u21, 'a'), map(us, hid_a, .{ .shift = true, .caps_lock = true }).character);
|
||||
}
|
||||
|
||||
test "us: digits and their shifted symbols" {
|
||||
try testing.expectEqual(@as(?u21, '1'), map(us, hid_1, .{}).character);
|
||||
try testing.expectEqual(@as(?u21, '!'), map(us, hid_1, .{ .shift = true }).character);
|
||||
try testing.expectEqual(@as(?u21, '3'), map(us, hid_3, .{}).character);
|
||||
try testing.expectEqual(@as(?u21, '#'), map(us, hid_3, .{ .shift = true }).character);
|
||||
// A digit is not alphabetic: Caps alone must not shift it.
|
||||
try testing.expectEqual(@as(?u21, '3'), map(us, hid_3, .{ .caps_lock = true }).character);
|
||||
}
|
||||
|
||||
test "layouts differ: GB pound vs US hash on shift+3" {
|
||||
try testing.expectEqual(@as(?u21, '#'), map(us, hid_3, .{ .shift = true }).character);
|
||||
try testing.expectEqual(@as(?u21, '£'), map(gb, hid_3, .{ .shift = true }).character);
|
||||
}
|
||||
|
||||
test "french azerty places q where us has a" {
|
||||
try testing.expectEqual(@as(?u21, 'q'), map(fr, hid_a, .{}).character);
|
||||
try testing.expectEqual(@as(?u21, 'Q'), map(fr, hid_a, .{ .shift = true }).character);
|
||||
}
|
||||
|
||||
test "byName resolves and rejects" {
|
||||
try testing.expect(byName("us") == us);
|
||||
try testing.expect(byName("gb") == gb);
|
||||
try testing.expect(byName("nonsense") == null);
|
||||
}
|
||||
|
||||
test "unmapped key yields no character" {
|
||||
// HID 0x00 is not a key; every level is empty.
|
||||
try testing.expectEqual(@as(?u21, null), map(us, 0x00, .{}).character);
|
||||
}
|
||||
@@ -1,737 +0,0 @@
|
||||
//! A tree-walking AML interpreter — the evaluation stage on top of the parser's
|
||||
//! structural namespace. It executes control methods (their bodies captured by
|
||||
//! the parser) far enough to serve device discovery: device status (`_STA`, is a
|
||||
//! device present), current resource settings (`_CRS`), and the operators, control
|
||||
//! flow, locals/args, and
|
||||
//! OperationRegion field access those methods reach for.
|
||||
//!
|
||||
//! Scope: integers, buffers, strings, packages, and references; If/Else/While/
|
||||
//! Return; the arithmetic/logic operators; method invocation; Name/Local/Arg
|
||||
//! access; CreateField buffer patching (the common current-resource-settings
|
||||
//! (`_CRS`) idiom); and field
|
||||
//! reads/writes against SystemMemory and SystemIO regions. Opcodes outside this
|
||||
//! set return `error.Unsupported`, which callers treat as "couldn't evaluate" and
|
||||
//! fall back — never a hard failure.
|
||||
|
||||
const std = @import("std");
|
||||
const op = @import("opcodes.zig");
|
||||
const nsp = @import("namespace.zig");
|
||||
const Node = nsp.Node;
|
||||
const Namespace = nsp.Namespace;
|
||||
|
||||
/// Injected hardware access for OperationRegion reads/writes (the arch VMM + pio).
|
||||
pub const Hal = struct {
|
||||
mapMmio: *const fn (virt: u64, phys: u64, writable: bool) void,
|
||||
pioRead: *const fn (width: u8, port: u16) u32,
|
||||
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
||||
};
|
||||
|
||||
pub const Error = error{ Unsupported, Truncated, DivByZero } || std.mem.Allocator.Error;
|
||||
|
||||
/// A runtime AML value.
|
||||
pub const Object = union(enum) {
|
||||
uninitialized,
|
||||
integer: u64,
|
||||
buffer: []u8,
|
||||
string: []u8,
|
||||
package: []Object,
|
||||
reference: *Node,
|
||||
|
||||
pub fn asInt(self: Object) Error!u64 {
|
||||
return switch (self) {
|
||||
.integer => |v| v,
|
||||
.buffer => |b| blk: {
|
||||
var v: u64 = 0;
|
||||
for (b, 0..) |byte, i| {
|
||||
if (i >= 8) break;
|
||||
v |= @as(u64, byte) << @intCast(i * 8);
|
||||
}
|
||||
break :blk v;
|
||||
},
|
||||
else => error.Unsupported,
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
const max_segs = 16;
|
||||
const NamePath = struct {
|
||||
rooted: bool = false,
|
||||
parents: u8 = 0,
|
||||
segs: [max_segs][4]u8 = undefined,
|
||||
count: usize = 0,
|
||||
fn slice(self: *const NamePath) []const [4]u8 {
|
||||
return self.segs[0..self.count];
|
||||
}
|
||||
};
|
||||
|
||||
const Cursor = struct {
|
||||
b: []const u8,
|
||||
i: usize = 0,
|
||||
|
||||
fn eof(self: *Cursor) bool {
|
||||
return self.i >= self.b.len;
|
||||
}
|
||||
fn peek(self: *Cursor) ?u8 {
|
||||
return if (self.eof()) null else self.b[self.i];
|
||||
}
|
||||
fn byte(self: *Cursor) Error!u8 {
|
||||
if (self.eof()) return error.Truncated;
|
||||
const v = self.b[self.i];
|
||||
self.i += 1;
|
||||
return v;
|
||||
}
|
||||
fn take(self: *Cursor, n: usize) Error![]const u8 {
|
||||
if (self.i + n > self.b.len) return error.Truncated;
|
||||
const s = self.b[self.i .. self.i + n];
|
||||
self.i += n;
|
||||
return s;
|
||||
}
|
||||
fn pkgLen(self: *Cursor) Error!usize {
|
||||
const lead = try self.byte();
|
||||
const follow: usize = lead >> 6;
|
||||
if (follow == 0) return lead & 0x3F;
|
||||
var value: usize = lead & 0x0F;
|
||||
var k: usize = 0;
|
||||
while (k < follow) : (k += 1) value |= @as(usize, try self.byte()) << @intCast(4 + k * 8);
|
||||
return value;
|
||||
}
|
||||
fn nameString(self: *Cursor) Error!NamePath {
|
||||
var np = NamePath{};
|
||||
if (self.peek() == op.root_char) {
|
||||
np.rooted = true;
|
||||
self.i += 1;
|
||||
} else {
|
||||
while (self.peek() == op.parent_prefix_char) : (self.i += 1) np.parents += 1;
|
||||
}
|
||||
const lead = self.peek() orelse return np;
|
||||
switch (lead) {
|
||||
0x00 => self.i += 1,
|
||||
op.dual_name_prefix => {
|
||||
self.i += 1;
|
||||
try self.seg(&np);
|
||||
try self.seg(&np);
|
||||
},
|
||||
op.multi_name_prefix => {
|
||||
self.i += 1;
|
||||
const cnt = try self.byte();
|
||||
var k: usize = 0;
|
||||
while (k < cnt) : (k += 1) try self.seg(&np);
|
||||
},
|
||||
else => try self.seg(&np),
|
||||
}
|
||||
return np;
|
||||
}
|
||||
fn seg(self: *Cursor, np: *NamePath) Error!void {
|
||||
const s = try self.take(4);
|
||||
if (np.count < max_segs) {
|
||||
np.segs[np.count] = s[0..4].*;
|
||||
np.count += 1;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
const Frame = struct {
|
||||
args: [7]Object = .{.uninitialized} ** 7,
|
||||
locals: [8]Object = .{.uninitialized} ** 8,
|
||||
scope: *Node,
|
||||
ret: Object = .uninitialized,
|
||||
returned: bool = false,
|
||||
broke: bool = false,
|
||||
};
|
||||
|
||||
/// A CreateField binding: a name that indexes into a buffer object.
|
||||
const BufField = struct { buf: *Node, byte_off: usize, bit_width: u32 };
|
||||
|
||||
pub const Interp = struct {
|
||||
ns: *Namespace,
|
||||
hal: Hal,
|
||||
arena: std.mem.Allocator,
|
||||
/// Runtime object overrides for Name nodes (Store targets, patched buffers).
|
||||
dyn: std.AutoHashMapUnmanaged(*Node, Object) = .{},
|
||||
/// CreateField bindings active for the current evaluation.
|
||||
fields: std.AutoHashMapUnmanaged(*Node, BufField) = .{},
|
||||
|
||||
pub fn init(ns: *Namespace, hal: Hal, arena: std.mem.Allocator) Interp {
|
||||
return .{ .ns = ns, .hal = hal, .arena = arena };
|
||||
}
|
||||
|
||||
/// Evaluate a namespace object: invoke a Method, read a Name's value, or read a
|
||||
/// Field. Resets per-evaluation runtime state first.
|
||||
pub fn evaluate(self: *Interp, node: *Node, args: []const Object) Error!Object {
|
||||
self.dyn.clearRetainingCapacity();
|
||||
self.fields.clearRetainingCapacity();
|
||||
return self.invoke(node, args);
|
||||
}
|
||||
|
||||
fn invoke(self: *Interp, node: *Node, args: []const Object) Error!Object {
|
||||
switch (node.kind) {
|
||||
.method => {
|
||||
var frame = Frame{ .scope = node };
|
||||
for (args, 0..) |a, i| {
|
||||
if (i < frame.args.len) frame.args[i] = a;
|
||||
}
|
||||
var cur = Cursor{ .b = node.value };
|
||||
try self.execList(&cur, &frame);
|
||||
return frame.ret;
|
||||
},
|
||||
.name => {
|
||||
if (self.dyn.get(node)) |o| return o;
|
||||
var cur = Cursor{ .b = node.value };
|
||||
var frame = Frame{ .scope = node.parent orelse self.ns.root };
|
||||
return self.term(&cur, &frame);
|
||||
},
|
||||
.field => return .{ .integer = try self.readField(node) },
|
||||
else => return .{ .reference = node },
|
||||
}
|
||||
}
|
||||
|
||||
/// Execute a TermList until it ends or the frame returns/breaks.
|
||||
fn execList(self: *Interp, cur: *Cursor, frame: *Frame) Error!void {
|
||||
while (!cur.eof() and !frame.returned and !frame.broke) {
|
||||
_ = try self.term(cur, frame);
|
||||
}
|
||||
}
|
||||
|
||||
/// Evaluate/execute one term, returning its value (`.uninitialized` for pure
|
||||
/// statements).
|
||||
fn term(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const lead = cur.peek() orelse return error.Truncated;
|
||||
if (isNameStart(lead)) return self.nameRef(cur, frame);
|
||||
_ = try cur.byte();
|
||||
|
||||
return switch (lead) {
|
||||
op.zero_op => Object{ .integer = 0 },
|
||||
op.one_op => Object{ .integer = 1 },
|
||||
op.ones_op => Object{ .integer = ~@as(u64, 0) },
|
||||
op.byte_prefix => Object{ .integer = try self.readConst(cur, 1) },
|
||||
op.word_prefix => Object{ .integer = try self.readConst(cur, 2) },
|
||||
op.dword_prefix => Object{ .integer = try self.readConst(cur, 4) },
|
||||
op.qword_prefix => Object{ .integer = try self.readConst(cur, 8) },
|
||||
op.string_prefix => try self.readString(cur),
|
||||
op.buffer_op => try self.buffer(cur, frame),
|
||||
op.package_op, op.var_package_op => try self.package(cur, frame, lead == op.var_package_op),
|
||||
|
||||
op.local0_op...op.local7_op => frame.locals[lead - op.local0_op],
|
||||
op.arg0_op...op.arg6_op => frame.args[lead - op.arg0_op],
|
||||
|
||||
op.return_op => blk: {
|
||||
frame.ret = try self.term(cur, frame);
|
||||
frame.returned = true;
|
||||
break :blk .uninitialized;
|
||||
},
|
||||
op.break_op => blk: {
|
||||
frame.broke = true;
|
||||
break :blk .uninitialized;
|
||||
},
|
||||
op.continue_op, op.noop_op => .uninitialized,
|
||||
|
||||
op.if_op => try self.ifElse(cur, frame),
|
||||
op.while_op => try self.whileLoop(cur, frame),
|
||||
op.store_op => try self.store(cur, frame),
|
||||
op.increment_op => try self.incDec(cur, frame, 1),
|
||||
op.decrement_op => try self.incDec(cur, frame, -1),
|
||||
|
||||
op.add_op => try self.binary(cur, frame, .add),
|
||||
op.subtract_op => try self.binary(cur, frame, .sub),
|
||||
op.multiply_op => try self.binary(cur, frame, .mul),
|
||||
op.mod_op => try self.binary(cur, frame, .mod),
|
||||
op.and_op => try self.binary(cur, frame, .band),
|
||||
op.or_op => try self.binary(cur, frame, .bor),
|
||||
op.xor_op => try self.binary(cur, frame, .bxor),
|
||||
op.nand_op => try self.binary(cur, frame, .nand),
|
||||
op.nor_op => try self.binary(cur, frame, .nor),
|
||||
op.shift_left_op => try self.binary(cur, frame, .shl),
|
||||
op.shift_right_op => try self.binary(cur, frame, .shr),
|
||||
op.divide_op => try self.divide(cur, frame),
|
||||
|
||||
op.land_op => try self.logic2(cur, frame, .land),
|
||||
op.lor_op => try self.logic2(cur, frame, .lor),
|
||||
op.lequal_op => try self.logic2(cur, frame, .eq),
|
||||
op.lgreater_op => try self.logic2(cur, frame, .gt),
|
||||
op.lless_op => try self.logic2(cur, frame, .lt),
|
||||
op.lnot_op => try self.lnot(cur, frame),
|
||||
|
||||
op.not_op => blk: {
|
||||
const v = try self.evalInt(cur, frame);
|
||||
const r = ~v;
|
||||
try self.storeTarget(cur, frame, .{ .integer = r });
|
||||
break :blk .{ .integer = r };
|
||||
},
|
||||
|
||||
op.size_of_op => try self.sizeOf(cur, frame),
|
||||
op.index_op => try self.index(cur, frame),
|
||||
op.deref_of_op => try self.derefOf(cur, frame),
|
||||
op.to_integer_op => blk: {
|
||||
const v = try self.evalInt(cur, frame);
|
||||
try self.storeTarget(cur, frame, .{ .integer = v });
|
||||
break :blk .{ .integer = v };
|
||||
},
|
||||
op.to_buffer_op => try self.passThroughUnary(cur, frame),
|
||||
|
||||
op.ext_op_prefix => try self.ext(cur, frame),
|
||||
|
||||
// CreateXField: source, index, name (bit widths differ by op)
|
||||
op.create_bit_field_op => try self.createField(cur, frame, 1),
|
||||
op.create_byte_field_op => try self.createField(cur, frame, 8),
|
||||
op.create_word_field_op => try self.createField(cur, frame, 16),
|
||||
op.create_dword_field_op => try self.createField(cur, frame, 32),
|
||||
op.create_qword_field_op => try self.createField(cur, frame, 64),
|
||||
|
||||
else => error.Unsupported,
|
||||
};
|
||||
}
|
||||
|
||||
// --- name references ----------------------------------------------------
|
||||
|
||||
fn nameRef(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const np = try cur.nameString();
|
||||
const node = self.ns.resolve(frame.scope, np.rooted, np.parents, np.slice()) orelse
|
||||
return .uninitialized; // unknown name -> treat as uninitialised
|
||||
switch (node.kind) {
|
||||
.method => {
|
||||
var argbuf: [7]Object = undefined;
|
||||
var i: usize = 0;
|
||||
while (i < node.arg_count and i < argbuf.len) : (i += 1) argbuf[i] = try self.term(cur, frame);
|
||||
return self.invoke(node, argbuf[0..@min(node.arg_count, argbuf.len)]);
|
||||
},
|
||||
.field => return .{ .integer = try self.readField(node) },
|
||||
.name => return self.invoke(node, &.{}),
|
||||
else => return .{ .reference = node },
|
||||
}
|
||||
}
|
||||
|
||||
// --- data objects -------------------------------------------------------
|
||||
|
||||
fn readConst(self: *Interp, cur: *Cursor, n: usize) Error!u64 {
|
||||
_ = self;
|
||||
const bytes = try cur.take(n);
|
||||
var v: u64 = 0;
|
||||
for (bytes, 0..) |b, i| v |= @as(u64, b) << @intCast(i * 8);
|
||||
return v;
|
||||
}
|
||||
|
||||
fn readString(self: *Interp, cur: *Cursor) Error!Object {
|
||||
const start = cur.i;
|
||||
while (cur.peek()) |c| {
|
||||
cur.i += 1;
|
||||
if (c == 0) break;
|
||||
}
|
||||
const raw = cur.b[start .. cur.i - 1];
|
||||
const s = try self.arena.dupe(u8, raw);
|
||||
return .{ .string = s };
|
||||
}
|
||||
|
||||
fn buffer(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const start = cur.i;
|
||||
const len = try cur.pkgLen();
|
||||
const end = @min(start + len, cur.b.len);
|
||||
const size = try self.evalInt(cur, frame);
|
||||
const data = cur.b[@min(cur.i, end)..end];
|
||||
const buf = try self.arena.alloc(u8, @intCast(size));
|
||||
@memset(buf, 0);
|
||||
@memcpy(buf[0..@min(buf.len, data.len)], data[0..@min(buf.len, data.len)]);
|
||||
cur.i = end;
|
||||
return .{ .buffer = buf };
|
||||
}
|
||||
|
||||
fn package(self: *Interp, cur: *Cursor, frame: *Frame, variable: bool) Error!Object {
|
||||
const start = cur.i;
|
||||
const len = try cur.pkgLen();
|
||||
const end = @min(start + len, cur.b.len);
|
||||
const count: usize = if (variable) @intCast(try self.evalInt(cur, frame)) else try cur.byte();
|
||||
const elems = try self.arena.alloc(Object, count);
|
||||
var i: usize = 0;
|
||||
while (i < count and cur.i < end) : (i += 1) elems[i] = try self.term(cur, frame);
|
||||
while (i < count) : (i += 1) elems[i] = .uninitialized;
|
||||
cur.i = end;
|
||||
return .{ .package = elems };
|
||||
}
|
||||
|
||||
// --- operators ----------------------------------------------------------
|
||||
|
||||
const BinOp = enum { add, sub, mul, mod, band, bor, bxor, nand, nor, shl, shr };
|
||||
|
||||
fn binary(self: *Interp, cur: *Cursor, frame: *Frame, kind: BinOp) Error!Object {
|
||||
const a = try self.evalInt(cur, frame);
|
||||
const b = try self.evalInt(cur, frame);
|
||||
const r: u64 = switch (kind) {
|
||||
.add => a +% b,
|
||||
.sub => a -% b,
|
||||
.mul => a *% b,
|
||||
.mod => if (b == 0) return error.DivByZero else a % b,
|
||||
.band => a & b,
|
||||
.bor => a | b,
|
||||
.bxor => a ^ b,
|
||||
.nand => ~(a & b),
|
||||
.nor => ~(a | b),
|
||||
.shl => if (b >= 64) 0 else a << @intCast(b),
|
||||
.shr => if (b >= 64) 0 else a >> @intCast(b),
|
||||
};
|
||||
try self.storeTarget(cur, frame, .{ .integer = r });
|
||||
return .{ .integer = r };
|
||||
}
|
||||
|
||||
fn divide(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const a = try self.evalInt(cur, frame);
|
||||
const b = try self.evalInt(cur, frame);
|
||||
if (b == 0) return error.DivByZero;
|
||||
try self.storeTarget(cur, frame, .{ .integer = a % b }); // remainder target
|
||||
try self.storeTarget(cur, frame, .{ .integer = a / b }); // quotient target
|
||||
return .{ .integer = a / b };
|
||||
}
|
||||
|
||||
const LogicOp = enum { land, lor, eq, gt, lt };
|
||||
|
||||
fn logic2(self: *Interp, cur: *Cursor, frame: *Frame, kind: LogicOp) Error!Object {
|
||||
const a = try self.evalInt(cur, frame);
|
||||
const b = try self.evalInt(cur, frame);
|
||||
const r = switch (kind) {
|
||||
.land => a != 0 and b != 0,
|
||||
.lor => a != 0 or b != 0,
|
||||
.eq => a == b,
|
||||
.gt => a > b,
|
||||
.lt => a < b,
|
||||
};
|
||||
return .{ .integer = if (r) ~@as(u64, 0) else 0 };
|
||||
}
|
||||
|
||||
fn lnot(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
// 0x92 0x93/94/95 are the compound comparisons.
|
||||
const b = cur.peek() orelse return error.Truncated;
|
||||
switch (b) {
|
||||
op.lnot.not_equal => {
|
||||
cur.i += 1;
|
||||
const x = try self.evalInt(cur, frame);
|
||||
const y = try self.evalInt(cur, frame);
|
||||
return .{ .integer = if (x != y) ~@as(u64, 0) else 0 };
|
||||
},
|
||||
op.lnot.less_equal => {
|
||||
cur.i += 1;
|
||||
const x = try self.evalInt(cur, frame);
|
||||
const y = try self.evalInt(cur, frame);
|
||||
return .{ .integer = if (x <= y) ~@as(u64, 0) else 0 };
|
||||
},
|
||||
op.lnot.greater_equal => {
|
||||
cur.i += 1;
|
||||
const x = try self.evalInt(cur, frame);
|
||||
const y = try self.evalInt(cur, frame);
|
||||
return .{ .integer = if (x >= y) ~@as(u64, 0) else 0 };
|
||||
},
|
||||
else => {
|
||||
const x = try self.evalInt(cur, frame);
|
||||
return .{ .integer = if (x == 0) ~@as(u64, 0) else 0 };
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
fn incDec(self: *Interp, cur: *Cursor, frame: *Frame, delta: i64) Error!Object {
|
||||
// Operand is a SuperName that is both read and written.
|
||||
const save = cur.i;
|
||||
const cur_val = try self.term(cur, frame);
|
||||
const v = try cur_val.asInt();
|
||||
const r = if (delta > 0) v +% 1 else v -% 1;
|
||||
var tcur = Cursor{ .b = cur.b, .i = save };
|
||||
try self.storeInto(&tcur, frame, .{ .integer = r });
|
||||
return .{ .integer = r };
|
||||
}
|
||||
|
||||
fn sizeOf(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const o = try self.term(cur, frame);
|
||||
return .{ .integer = switch (o) {
|
||||
.buffer => |b| b.len,
|
||||
.string => |s| s.len,
|
||||
.package => |p| p.len,
|
||||
else => 0,
|
||||
} };
|
||||
}
|
||||
|
||||
fn passThroughUnary(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const o = try self.term(cur, frame);
|
||||
try self.storeTarget(cur, frame, o);
|
||||
return o;
|
||||
}
|
||||
|
||||
fn index(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const src = try self.term(cur, frame);
|
||||
const idx: usize = @intCast(try self.evalInt(cur, frame));
|
||||
// Optional target (a reference); we don't materialise references, so store
|
||||
// the indexed value if a target is present.
|
||||
const val: Object = switch (src) {
|
||||
.buffer => |b| .{ .integer = if (idx < b.len) b[idx] else 0 },
|
||||
.package => |p| if (idx < p.len) p[idx] else .uninitialized,
|
||||
.string => |s| .{ .integer = if (idx < s.len) s[idx] else 0 },
|
||||
else => .uninitialized,
|
||||
};
|
||||
try self.storeTarget(cur, frame, val);
|
||||
return val;
|
||||
}
|
||||
|
||||
fn derefOf(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const o = try self.term(cur, frame);
|
||||
return switch (o) {
|
||||
.reference => |n| self.invoke(n, &.{}),
|
||||
else => o,
|
||||
};
|
||||
}
|
||||
|
||||
// --- control flow -------------------------------------------------------
|
||||
|
||||
fn ifElse(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const start = cur.i;
|
||||
const end = @min(start + try cur.pkgLen(), cur.b.len);
|
||||
const cond = try self.evalInt(cur, frame);
|
||||
if (cond != 0) {
|
||||
var body = Cursor{ .b = cur.b[0..end], .i = cur.i };
|
||||
try self.execList(&body, frame);
|
||||
cur.i = end;
|
||||
// Skip a trailing Else.
|
||||
if (cur.peek() == op.else_op) {
|
||||
cur.i += 1;
|
||||
const es = cur.i;
|
||||
const ee = @min(es + try cur.pkgLen(), cur.b.len);
|
||||
cur.i = ee;
|
||||
}
|
||||
} else {
|
||||
cur.i = end;
|
||||
if (cur.peek() == op.else_op) {
|
||||
cur.i += 1;
|
||||
const es = cur.i;
|
||||
const ee = @min(es + try cur.pkgLen(), cur.b.len);
|
||||
var body = Cursor{ .b = cur.b[0..ee], .i = cur.i };
|
||||
try self.execList(&body, frame);
|
||||
cur.i = ee;
|
||||
}
|
||||
}
|
||||
return .uninitialized;
|
||||
}
|
||||
|
||||
fn whileLoop(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const start = cur.i;
|
||||
const end = @min(start + try cur.pkgLen(), cur.b.len);
|
||||
const pred_at = cur.i;
|
||||
var guard: usize = 0;
|
||||
while (guard < 100_000) : (guard += 1) {
|
||||
var pc = Cursor{ .b = cur.b[0..end], .i = pred_at };
|
||||
const cond = try self.evalInt(&pc, frame);
|
||||
if (cond == 0) break;
|
||||
var body = Cursor{ .b = cur.b[0..end], .i = pc.i };
|
||||
try self.execList(&body, frame);
|
||||
if (frame.returned) break;
|
||||
if (frame.broke) {
|
||||
frame.broke = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
cur.i = end;
|
||||
return .uninitialized;
|
||||
}
|
||||
|
||||
// --- store --------------------------------------------------------------
|
||||
|
||||
fn store(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const value = try self.term(cur, frame);
|
||||
try self.storeInto(cur, frame, value);
|
||||
return value;
|
||||
}
|
||||
|
||||
/// A Store *target* that may be NullName (no store).
|
||||
fn storeTarget(self: *Interp, cur: *Cursor, frame: *Frame, value: Object) Error!void {
|
||||
if (cur.peek() == 0x00) {
|
||||
cur.i += 1; // NullName
|
||||
return;
|
||||
}
|
||||
try self.storeInto(cur, frame, value);
|
||||
}
|
||||
|
||||
fn storeInto(self: *Interp, cur: *Cursor, frame: *Frame, value: Object) Error!void {
|
||||
const lead = cur.peek() orelse return error.Truncated;
|
||||
if (isNameStart(lead)) {
|
||||
const np = try cur.nameString();
|
||||
const node = self.ns.resolve(frame.scope, np.rooted, np.parents, np.slice()) orelse return;
|
||||
if (self.fields.get(node)) |bf| {
|
||||
try self.writeBufField(bf, try value.asInt());
|
||||
} else if (node.kind == .field) {
|
||||
try self.writeField(node, try value.asInt());
|
||||
} else {
|
||||
try self.dyn.put(self.arena, node, value);
|
||||
}
|
||||
return;
|
||||
}
|
||||
_ = try cur.byte();
|
||||
switch (lead) {
|
||||
0x00 => {}, // NullName
|
||||
op.local0_op...op.local7_op => frame.locals[lead - op.local0_op] = value,
|
||||
op.arg0_op...op.arg6_op => frame.args[lead - op.arg0_op] = value,
|
||||
op.index_op => {
|
||||
const src = try self.term(cur, frame);
|
||||
const idx: usize = @intCast(try self.evalInt(cur, frame));
|
||||
switch (src) {
|
||||
.buffer => |b| if (idx < b.len) {
|
||||
b[idx] = @truncate(try value.asInt());
|
||||
},
|
||||
.package => |p| if (idx < p.len) {
|
||||
p[idx] = value;
|
||||
},
|
||||
else => {},
|
||||
}
|
||||
},
|
||||
else => return error.Unsupported,
|
||||
}
|
||||
}
|
||||
|
||||
// --- CreateField (buffer patching) --------------------------------------
|
||||
|
||||
fn createField(self: *Interp, cur: *Cursor, frame: *Frame, bit_width: u32) Error!Object {
|
||||
const src = try self.term(cur, frame); // source buffer (as a reference or value)
|
||||
const bit_index = try self.evalInt(cur, frame);
|
||||
const np = try cur.nameString();
|
||||
const node = self.ns.resolve(frame.scope, np.rooted, np.parents, np.slice()) orelse return .uninitialized;
|
||||
|
||||
// Bind the new name to the source buffer's node so stores land in it.
|
||||
const buf_node: *Node = switch (src) {
|
||||
.reference => |n| n,
|
||||
else => return .uninitialized,
|
||||
};
|
||||
// Materialise the buffer into `dyn` so patches persist and are returned.
|
||||
if (self.dyn.get(buf_node) == null) {
|
||||
const val = try self.invoke(buf_node, &.{});
|
||||
try self.dyn.put(self.arena, buf_node, val);
|
||||
}
|
||||
const byte_off: usize = @intCast(bit_index / 8);
|
||||
try self.fields.put(self.arena, node, .{ .buf = buf_node, .byte_off = byte_off, .bit_width = bit_width });
|
||||
return .uninitialized;
|
||||
}
|
||||
|
||||
fn writeBufField(self: *Interp, bf: BufField, value: u64) Error!void {
|
||||
const obj = self.dyn.get(bf.buf) orelse return;
|
||||
const buf = switch (obj) {
|
||||
.buffer => |b| b,
|
||||
else => return,
|
||||
};
|
||||
const nbytes = (bf.bit_width + 7) / 8;
|
||||
var k: usize = 0;
|
||||
while (k < nbytes and bf.byte_off + k < buf.len) : (k += 1) {
|
||||
buf[bf.byte_off + k] = @truncate(value >> @intCast(k * 8));
|
||||
}
|
||||
}
|
||||
|
||||
// --- OperationRegion field access ---------------------------------------
|
||||
|
||||
fn readField(self: *Interp, field: *Node) Error!u64 {
|
||||
const region = field.region orelse return error.Unsupported;
|
||||
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
||||
const base = try self.regionBase(region);
|
||||
const start_byte = base + field.bit_offset / 8;
|
||||
const shift: u7 = @intCast(field.bit_offset % 8);
|
||||
const total = @as(usize, shift) + field.bit_width;
|
||||
const nbytes = (total + 7) / 8;
|
||||
var raw: u128 = 0;
|
||||
var k: usize = 0;
|
||||
while (k < nbytes) : (k += 1) {
|
||||
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
||||
}
|
||||
const masked = (raw >> shift) & bitMask(field.bit_width);
|
||||
return @truncate(masked);
|
||||
}
|
||||
|
||||
fn writeField(self: *Interp, field: *Node, value: u64) Error!void {
|
||||
const region = field.region orelse return error.Unsupported;
|
||||
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
||||
const base = try self.regionBase(region);
|
||||
const start_byte = base + field.bit_offset / 8;
|
||||
const shift: u7 = @intCast(field.bit_offset % 8);
|
||||
const total = @as(usize, shift) + field.bit_width;
|
||||
const nbytes = (total + 7) / 8;
|
||||
// Read-modify-write byte by byte.
|
||||
var raw: u128 = 0;
|
||||
var k: usize = 0;
|
||||
while (k < nbytes) : (k += 1) {
|
||||
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
||||
}
|
||||
const mask = bitMask(field.bit_width) << shift;
|
||||
raw = (raw & ~mask) | ((@as(u128, value) << shift) & mask);
|
||||
k = 0;
|
||||
while (k < nbytes) : (k += 1) {
|
||||
try self.writeRegionByte(region.region_space, start_byte + k, @truncate(raw >> @intCast(k * 8)));
|
||||
}
|
||||
}
|
||||
|
||||
fn regionBase(self: *Interp, region: *Node) Error!u64 {
|
||||
var cur = Cursor{ .b = region.region_offset_aml };
|
||||
var frame = Frame{ .scope = region.parent orelse self.ns.root };
|
||||
return (try self.term(&cur, &frame)).asInt();
|
||||
}
|
||||
|
||||
fn readRegionByte(self: *Interp, space: u8, addr: u64) Error!u8 {
|
||||
switch (space) {
|
||||
0 => { // SystemMemory
|
||||
self.hal.mapMmio(addr & ~@as(u64, 0xFFF), addr & ~@as(u64, 0xFFF), true);
|
||||
const p: *align(1) const volatile u8 = @ptrFromInt(addr);
|
||||
return p.*;
|
||||
},
|
||||
1 => return @truncate(self.hal.pioRead(1, @intCast(addr & 0xFFFF))), // SystemIO
|
||||
else => return error.Unsupported,
|
||||
}
|
||||
}
|
||||
|
||||
fn writeRegionByte(self: *Interp, space: u8, addr: u64, value: u8) Error!void {
|
||||
switch (space) {
|
||||
0 => {
|
||||
self.hal.mapMmio(addr & ~@as(u64, 0xFFF), addr & ~@as(u64, 0xFFF), true);
|
||||
const p: *align(1) volatile u8 = @ptrFromInt(addr);
|
||||
p.* = value;
|
||||
},
|
||||
1 => self.hal.pioWrite(1, @intCast(addr & 0xFFFF), value),
|
||||
else => return error.Unsupported,
|
||||
}
|
||||
}
|
||||
|
||||
// --- extended opcodes ---------------------------------------------------
|
||||
|
||||
fn ext(self: *Interp, cur: *Cursor, frame: *Frame) Error!Object {
|
||||
const e = try cur.byte();
|
||||
switch (e) {
|
||||
op.ext.debug => return .uninitialized,
|
||||
op.ext.revision => return .{ .integer = 2 },
|
||||
op.ext.timer => return .{ .integer = 0 },
|
||||
// Mutex/Event ops are no-ops in this single-threaded evaluator.
|
||||
op.ext.acquire => {
|
||||
_ = try self.term(cur, frame); // mutex SuperName
|
||||
_ = try cur.take(2); // timeout
|
||||
return .{ .integer = 0 }; // acquired
|
||||
},
|
||||
op.ext.release, op.ext.reset, op.ext.signal => {
|
||||
_ = try self.term(cur, frame);
|
||||
return .uninitialized;
|
||||
},
|
||||
op.ext.wait => {
|
||||
_ = try self.term(cur, frame);
|
||||
_ = try self.term(cur, frame);
|
||||
return .{ .integer = 0 };
|
||||
},
|
||||
op.ext.sleep, op.ext.stall => {
|
||||
_ = try self.term(cur, frame);
|
||||
return .uninitialized;
|
||||
},
|
||||
else => return error.Unsupported,
|
||||
}
|
||||
}
|
||||
|
||||
fn evalInt(self: *Interp, cur: *Cursor, frame: *Frame) Error!u64 {
|
||||
return (try self.term(cur, frame)).asInt();
|
||||
}
|
||||
};
|
||||
|
||||
fn bitMask(width: u32) u128 {
|
||||
if (width >= 128) return ~@as(u128, 0);
|
||||
return (@as(u128, 1) << @intCast(width)) - 1;
|
||||
}
|
||||
|
||||
fn isNameStart(b: u8) bool {
|
||||
return (b >= op.name_char_start and b <= op.name_char_end) or
|
||||
b == op.name_char_underscore or
|
||||
b == op.root_char or
|
||||
b == op.parent_prefix_char or
|
||||
b == op.dual_name_prefix or
|
||||
b == op.multi_name_prefix;
|
||||
}
|
||||
@@ -1,137 +0,0 @@
|
||||
//! AML opcode constants — the full ACPI Machine Language opcode table.
|
||||
//!
|
||||
//! Single-byte opcodes are plain values. Extended opcodes are a two-byte sequence
|
||||
//! `ext_prefix` (0x5B) followed by a byte listed under `ext`. A few comparison
|
||||
//! opcodes are `lnot_op` (0x92) followed by a second byte (see `lnot`).
|
||||
|
||||
// --- name / path characters -------------------------------------------------
|
||||
pub const zero_op = 0x00;
|
||||
pub const one_op = 0x01;
|
||||
pub const alias_op = 0x06;
|
||||
pub const name_op = 0x08;
|
||||
pub const byte_prefix = 0x0A;
|
||||
pub const word_prefix = 0x0B;
|
||||
pub const dword_prefix = 0x0C;
|
||||
pub const string_prefix = 0x0D;
|
||||
pub const qword_prefix = 0x0E;
|
||||
pub const scope_op = 0x10;
|
||||
pub const buffer_op = 0x11;
|
||||
pub const package_op = 0x12;
|
||||
pub const var_package_op = 0x13;
|
||||
pub const method_op = 0x14;
|
||||
pub const external_op = 0x15;
|
||||
|
||||
pub const dual_name_prefix = 0x2E;
|
||||
pub const multi_name_prefix = 0x2F;
|
||||
pub const ext_op_prefix = 0x5B;
|
||||
pub const root_char = 0x5C;
|
||||
pub const parent_prefix_char = 0x5E;
|
||||
pub const name_char_underscore = 0x5F;
|
||||
|
||||
pub const digit_char_start = 0x30;
|
||||
pub const digit_char_end = 0x39;
|
||||
pub const name_char_start = 0x41; // 'A'
|
||||
pub const name_char_end = 0x5A; // 'Z'
|
||||
|
||||
// --- locals / args ----------------------------------------------------------
|
||||
pub const local0_op = 0x60;
|
||||
pub const local7_op = 0x67;
|
||||
pub const arg0_op = 0x68;
|
||||
pub const arg6_op = 0x6E;
|
||||
|
||||
// --- store / references / arithmetic ---------------------------------------
|
||||
pub const store_op = 0x70;
|
||||
pub const ref_of_op = 0x71;
|
||||
pub const add_op = 0x72;
|
||||
pub const concat_op = 0x73;
|
||||
pub const subtract_op = 0x74;
|
||||
pub const increment_op = 0x75;
|
||||
pub const decrement_op = 0x76;
|
||||
pub const multiply_op = 0x77;
|
||||
pub const divide_op = 0x78;
|
||||
pub const shift_left_op = 0x79;
|
||||
pub const shift_right_op = 0x7A;
|
||||
pub const and_op = 0x7B;
|
||||
pub const nand_op = 0x7C;
|
||||
pub const or_op = 0x7D;
|
||||
pub const nor_op = 0x7E;
|
||||
pub const xor_op = 0x7F;
|
||||
pub const not_op = 0x80;
|
||||
pub const find_set_left_bit_op = 0x81;
|
||||
pub const find_set_right_bit_op = 0x82;
|
||||
pub const deref_of_op = 0x83;
|
||||
pub const concat_res_op = 0x84;
|
||||
pub const mod_op = 0x85;
|
||||
pub const notify_op = 0x86;
|
||||
pub const size_of_op = 0x87;
|
||||
pub const index_op = 0x88;
|
||||
pub const match_op = 0x89;
|
||||
pub const create_dword_field_op = 0x8A;
|
||||
pub const create_word_field_op = 0x8B;
|
||||
pub const create_byte_field_op = 0x8C;
|
||||
pub const create_bit_field_op = 0x8D;
|
||||
pub const object_type_op = 0x8E;
|
||||
pub const create_qword_field_op = 0x8F;
|
||||
|
||||
pub const land_op = 0x90;
|
||||
pub const lor_op = 0x91;
|
||||
pub const lnot_op = 0x92; // may be followed by a second byte (see `lnot`)
|
||||
pub const lequal_op = 0x93;
|
||||
pub const lgreater_op = 0x94;
|
||||
pub const lless_op = 0x95;
|
||||
pub const to_buffer_op = 0x96;
|
||||
pub const to_decimal_string_op = 0x97;
|
||||
pub const to_hex_string_op = 0x98;
|
||||
pub const to_integer_op = 0x99;
|
||||
pub const to_string_op = 0x9C;
|
||||
pub const copy_object_op = 0x9D;
|
||||
pub const mid_op = 0x9E;
|
||||
pub const continue_op = 0x9F;
|
||||
pub const if_op = 0xA0;
|
||||
pub const else_op = 0xA1;
|
||||
pub const while_op = 0xA2;
|
||||
pub const noop_op = 0xA3;
|
||||
pub const return_op = 0xA4;
|
||||
pub const break_op = 0xA5;
|
||||
pub const break_point_op = 0xCC;
|
||||
pub const ones_op = 0xFF;
|
||||
|
||||
/// Second bytes of the `lnot_op` (0x92) compound comparison opcodes.
|
||||
pub const lnot = struct {
|
||||
pub const not_equal = 0x93; // LNotEqualOp: 0x92 0x93
|
||||
pub const less_equal = 0x94; // LLessEqualOp: 0x92 0x94
|
||||
pub const greater_equal = 0x95; // LGreaterEqualOp: 0x92 0x95
|
||||
};
|
||||
|
||||
/// Second bytes of extended opcodes (prefixed by `ext_op_prefix`, 0x5B).
|
||||
pub const ext = struct {
|
||||
pub const mutex = 0x01;
|
||||
pub const event = 0x02;
|
||||
pub const cond_ref_of = 0x12;
|
||||
pub const create_field = 0x13;
|
||||
pub const load_table = 0x1F;
|
||||
pub const load = 0x20;
|
||||
pub const stall = 0x21;
|
||||
pub const sleep = 0x22;
|
||||
pub const acquire = 0x23;
|
||||
pub const signal = 0x24;
|
||||
pub const wait = 0x25;
|
||||
pub const reset = 0x26;
|
||||
pub const release = 0x27;
|
||||
pub const from_bcd = 0x28;
|
||||
pub const to_bcd = 0x29;
|
||||
pub const unload = 0x2A;
|
||||
pub const revision = 0x30;
|
||||
pub const debug = 0x31;
|
||||
pub const fatal = 0x32;
|
||||
pub const timer = 0x33;
|
||||
pub const op_region = 0x80;
|
||||
pub const field = 0x81;
|
||||
pub const device = 0x82;
|
||||
pub const processor = 0x83;
|
||||
pub const power_res = 0x84;
|
||||
pub const thermal_zone = 0x85;
|
||||
pub const index_field = 0x86;
|
||||
pub const bank_field = 0x87;
|
||||
pub const data_region = 0x88;
|
||||
};
|
||||
@@ -1,517 +0,0 @@
|
||||
//! Recursive-descent AML parser. Walks the entire byte stream — including method
|
||||
//! bodies — building the ACPI namespace as it goes. It does not *evaluate*
|
||||
//! anything (no OperationRegion reads, no arithmetic); it parses structure so the
|
||||
//! cursor stays aligned and every named object is recorded.
|
||||
//!
|
||||
//! The one genuine ambiguity in AML is method invocation: a bare NameString in an
|
||||
//! operand position is a call whose argument count is only known from the method's
|
||||
//! (earlier) declaration. Because we build the namespace in the same in-order pass,
|
||||
//! `resolve` finds that declaration and tells us how many operands to consume.
|
||||
//!
|
||||
//! Safety net: every object delimited by a PkgLength (Scope/Device/Method/If/While/
|
||||
//! Field/Buffer/Package/…) is parsed within its known extent, and the cursor is
|
||||
//! snapped to that extent afterwards. So a mis-resolved invocation can only desync
|
||||
//! *within* one such object; the enclosing walk realigns at the boundary.
|
||||
|
||||
const std = @import("std");
|
||||
const op = @import("opcodes.zig");
|
||||
const ns = @import("namespace.zig");
|
||||
const Namespace = ns.Namespace;
|
||||
const Node = ns.Node;
|
||||
|
||||
pub const Error = error{ Truncated, Malformed } || std.mem.Allocator.Error;
|
||||
|
||||
const max_segs = 64;
|
||||
|
||||
/// A parsed NameString: an optional root anchor or some parent hops, then a list
|
||||
/// of 4-byte segments.
|
||||
const NamePath = struct {
|
||||
rooted: bool = false,
|
||||
parents: u8 = 0,
|
||||
segs: [max_segs][4]u8 = undefined,
|
||||
count: usize = 0,
|
||||
|
||||
fn slice(self: *const NamePath) []const [4]u8 {
|
||||
return self.segs[0..self.count];
|
||||
}
|
||||
};
|
||||
|
||||
pub const Parser = struct {
|
||||
aml: []const u8,
|
||||
pos: usize = 0,
|
||||
namespace: *Namespace,
|
||||
|
||||
pub fn init(aml: []const u8, namespace: *Namespace) Parser {
|
||||
return .{ .aml = aml, .namespace = namespace };
|
||||
}
|
||||
|
||||
/// Parse the whole block as a TermList under the namespace root. Returns the
|
||||
/// number of bytes consumed — equal to `aml.len` for a clean full traversal.
|
||||
pub fn parseAll(self: *Parser) usize {
|
||||
self.termList(self.aml.len, self.namespace.root);
|
||||
return self.pos;
|
||||
}
|
||||
|
||||
// --- cursor primitives --------------------------------------------------
|
||||
|
||||
fn eof(self: *Parser) bool {
|
||||
return self.pos >= self.aml.len;
|
||||
}
|
||||
|
||||
fn peek(self: *Parser) ?u8 {
|
||||
return if (self.eof()) null else self.aml[self.pos];
|
||||
}
|
||||
|
||||
fn readByte(self: *Parser) Error!u8 {
|
||||
if (self.eof()) return error.Truncated;
|
||||
const b = self.aml[self.pos];
|
||||
self.pos += 1;
|
||||
return b;
|
||||
}
|
||||
|
||||
fn skip(self: *Parser, n: usize) Error!void {
|
||||
if (self.pos + n > self.aml.len) return error.Truncated;
|
||||
self.pos += n;
|
||||
}
|
||||
|
||||
fn skipCString(self: *Parser) Error!void {
|
||||
while (true) {
|
||||
const b = try self.readByte();
|
||||
if (b == 0) return;
|
||||
}
|
||||
}
|
||||
|
||||
/// AML PkgLength: the lead byte's top two bits give how many extra bytes
|
||||
/// follow; the value counts from the start of the PkgLength field.
|
||||
fn readPkgLength(self: *Parser) Error!usize {
|
||||
const lead = try self.readByte();
|
||||
const follow: usize = lead >> 6;
|
||||
if (follow == 0) return lead & 0x3F;
|
||||
var value: usize = lead & 0x0F;
|
||||
var i: usize = 0;
|
||||
while (i < follow) : (i += 1) {
|
||||
const b = try self.readByte();
|
||||
value |= @as(usize, b) << @intCast(4 + i * 8);
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
fn readNameSeg(self: *Parser) Error![4]u8 {
|
||||
if (self.pos + 4 > self.aml.len) return error.Truncated;
|
||||
const seg = self.aml[self.pos..][0..4].*;
|
||||
self.pos += 4;
|
||||
return seg;
|
||||
}
|
||||
|
||||
fn readNameString(self: *Parser) Error!NamePath {
|
||||
var np = NamePath{};
|
||||
// A NameString is either root-anchored or parent-relative, not both.
|
||||
if (self.peek() == op.root_char) {
|
||||
np.rooted = true;
|
||||
self.pos += 1;
|
||||
} else {
|
||||
while (self.peek() == op.parent_prefix_char) : (self.pos += 1) np.parents += 1;
|
||||
}
|
||||
|
||||
const lead = self.peek() orelse return np;
|
||||
switch (lead) {
|
||||
0x00 => self.pos += 1, // NullName
|
||||
op.dual_name_prefix => {
|
||||
self.pos += 1;
|
||||
try self.appendSeg(&np);
|
||||
try self.appendSeg(&np);
|
||||
},
|
||||
op.multi_name_prefix => {
|
||||
self.pos += 1;
|
||||
const cnt = try self.readByte();
|
||||
var i: usize = 0;
|
||||
while (i < cnt) : (i += 1) try self.appendSeg(&np);
|
||||
},
|
||||
else => {
|
||||
if (isNameStart(lead)) try self.appendSeg(&np);
|
||||
},
|
||||
}
|
||||
return np;
|
||||
}
|
||||
|
||||
fn appendSeg(self: *Parser, np: *NamePath) Error!void {
|
||||
const seg = try self.readNameSeg();
|
||||
if (np.count < max_segs) {
|
||||
np.segs[np.count] = seg;
|
||||
np.count += 1;
|
||||
}
|
||||
}
|
||||
|
||||
// --- term list / object -------------------------------------------------
|
||||
|
||||
/// Parse objects until `end`, then snap to `end`. Any parse error resyncs to
|
||||
/// the boundary rather than propagating — containment for the rare desync.
|
||||
fn termList(self: *Parser, end: usize, scope: *Node) void {
|
||||
while (self.pos < end) {
|
||||
self.object(scope) catch break;
|
||||
}
|
||||
self.pos = end;
|
||||
}
|
||||
|
||||
/// Parse exactly one object/term at the cursor. Used for both TermObjs and
|
||||
/// operands (TermArg / SuperName / Target all reduce to "one object" for the
|
||||
/// purpose of advancing the cursor).
|
||||
fn object(self: *Parser, scope: *Node) Error!void {
|
||||
const lead = self.peek() orelse return error.Truncated;
|
||||
if (isNameStart(lead)) return self.nameInvocation(scope);
|
||||
|
||||
_ = try self.readByte();
|
||||
switch (lead) {
|
||||
// constants and no-operand statements
|
||||
op.zero_op, op.one_op, op.ones_op => {},
|
||||
op.noop_op, op.continue_op, op.break_op, op.break_point_op => {},
|
||||
op.local0_op...op.local7_op => {},
|
||||
op.arg0_op...op.arg6_op => {},
|
||||
|
||||
// literal data
|
||||
op.byte_prefix => try self.skip(1),
|
||||
op.word_prefix => try self.skip(2),
|
||||
op.dword_prefix => try self.skip(4),
|
||||
op.qword_prefix => try self.skip(8),
|
||||
op.string_prefix => try self.skipCString(),
|
||||
|
||||
// data containers (contents skipped via their PkgLength)
|
||||
op.buffer_op, op.package_op, op.var_package_op => try self.skipPkg(),
|
||||
|
||||
// namespace modifiers / named objects
|
||||
op.name_op => try self.opName(scope),
|
||||
op.alias_op => try self.opAlias(scope),
|
||||
op.scope_op => try self.opScopeLike(scope, .scope),
|
||||
op.method_op => try self.opMethod(scope),
|
||||
op.external_op => try self.opExternal(scope),
|
||||
op.ext_op_prefix => try self.opExt(scope),
|
||||
|
||||
// control flow
|
||||
op.if_op => try self.opIf(scope),
|
||||
op.else_op => try self.opElse(scope),
|
||||
op.while_op => try self.opWhile(scope),
|
||||
op.return_op => try self.object(scope),
|
||||
op.notify_op => try self.args(scope, 2),
|
||||
|
||||
// stores / references / unary+target
|
||||
op.store_op => try self.args(scope, 2),
|
||||
op.ref_of_op, op.deref_of_op, op.size_of_op, op.object_type_op => try self.args(scope, 1),
|
||||
op.increment_op, op.decrement_op => try self.args(scope, 1),
|
||||
op.not_op, op.find_set_left_bit_op, op.find_set_right_bit_op => try self.args(scope, 2),
|
||||
op.to_buffer_op, op.to_decimal_string_op, op.to_hex_string_op, op.to_integer_op => try self.args(scope, 2),
|
||||
op.copy_object_op => try self.args(scope, 2),
|
||||
|
||||
// binary + target
|
||||
op.add_op, op.subtract_op, op.multiply_op, op.mod_op => try self.args(scope, 3),
|
||||
op.and_op, op.nand_op, op.or_op, op.nor_op, op.xor_op => try self.args(scope, 3),
|
||||
op.shift_left_op, op.shift_right_op, op.concat_op, op.concat_res_op, op.index_op => try self.args(scope, 3),
|
||||
op.divide_op => try self.args(scope, 4),
|
||||
op.to_string_op => try self.args(scope, 3),
|
||||
op.mid_op => try self.args(scope, 4),
|
||||
|
||||
// logical
|
||||
op.land_op, op.lor_op => try self.args(scope, 2),
|
||||
op.lequal_op, op.lgreater_op, op.lless_op => try self.args(scope, 2),
|
||||
op.lnot_op => try self.opLnot(scope),
|
||||
|
||||
op.match_op => try self.opMatch(scope),
|
||||
|
||||
// CreateXField: <source> <index> NameString
|
||||
op.create_dword_field_op,
|
||||
op.create_word_field_op,
|
||||
op.create_byte_field_op,
|
||||
op.create_bit_field_op,
|
||||
op.create_qword_field_op,
|
||||
=> try self.opCreateField(scope, 2),
|
||||
|
||||
else => return error.Malformed,
|
||||
}
|
||||
}
|
||||
|
||||
/// Parse `n` operands.
|
||||
fn args(self: *Parser, scope: *Node, n: usize) Error!void {
|
||||
var i: usize = 0;
|
||||
while (i < n) : (i += 1) try self.object(scope);
|
||||
}
|
||||
|
||||
/// A NameString in operand/statement position: a method invocation (consuming
|
||||
/// the callee's declared argument count) or a plain name reference.
|
||||
fn nameInvocation(self: *Parser, scope: *Node) Error!void {
|
||||
const np = try self.readNameString();
|
||||
if (self.namespace.resolve(scope, np.rooted, np.parents, np.slice())) |node| {
|
||||
if ((node.kind == .method or node.kind == .external) and node.arg_count > 0) {
|
||||
try self.args(scope, node.arg_count);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Skip a PkgLength-delimited body wholesale (Buffer / Package / VarPackage):
|
||||
/// the contents are pure data, never namespace declarations.
|
||||
fn skipPkg(self: *Parser) Error!void {
|
||||
const start = self.pos;
|
||||
const len = try self.readPkgLength();
|
||||
const end = start + len;
|
||||
if (end > self.aml.len) return error.Truncated;
|
||||
self.pos = end;
|
||||
}
|
||||
|
||||
// --- namespace objects --------------------------------------------------
|
||||
|
||||
fn opName(self: *Parser, scope: *Node) Error!void {
|
||||
const np = try self.readNameString();
|
||||
const val_start = self.pos;
|
||||
try self.object(scope); // the DataRefObject value
|
||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .name);
|
||||
node.value = self.aml[val_start..self.pos];
|
||||
}
|
||||
|
||||
fn opAlias(self: *Parser, scope: *Node) Error!void {
|
||||
_ = try self.readNameString(); // source
|
||||
const np = try self.readNameString(); // the alias name
|
||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .alias);
|
||||
}
|
||||
|
||||
fn opMethod(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.pos;
|
||||
const end = start + try self.readPkgLength();
|
||||
const np = try self.readNameString();
|
||||
const flags = try self.readByte();
|
||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .method);
|
||||
node.arg_count = flags & 0x7;
|
||||
// Capture the body for on-demand evaluation and skip it — objects declared
|
||||
// inside a method are created at *runtime*, not at load, so they must not
|
||||
// become permanent namespace nodes.
|
||||
node.value = self.aml[self.pos..@min(end, self.aml.len)];
|
||||
self.pos = end;
|
||||
}
|
||||
|
||||
fn opExternal(self: *Parser, scope: *Node) Error!void {
|
||||
const np = try self.readNameString();
|
||||
_ = try self.readByte(); // object type
|
||||
const arg_count = try self.readByte();
|
||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .external);
|
||||
node.arg_count = arg_count;
|
||||
}
|
||||
|
||||
/// Scope / Device / ThermalZone: PkgLength, NameString, then a nested TermList.
|
||||
fn opScopeLike(self: *Parser, scope: *Node, kind: ns.NodeKind) Error!void {
|
||||
const start = self.pos;
|
||||
const end = start + try self.readPkgLength();
|
||||
const np = try self.readNameString();
|
||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), kind);
|
||||
self.termList(end, node);
|
||||
}
|
||||
|
||||
fn opProcessor(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.pos;
|
||||
const end = start + try self.readPkgLength();
|
||||
const np = try self.readNameString();
|
||||
try self.skip(6); // ProcID(byte) + PblkAddr(dword) + PblkLen(byte)
|
||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .processor);
|
||||
self.termList(end, node);
|
||||
}
|
||||
|
||||
fn opPowerRes(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.pos;
|
||||
const end = start + try self.readPkgLength();
|
||||
const np = try self.readNameString();
|
||||
try self.skip(3); // SystemLevel(byte) + ResourceOrder(word)
|
||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .power_res);
|
||||
self.termList(end, node);
|
||||
}
|
||||
|
||||
/// OperationRegion: NameString, RegionSpace(byte), Offset(TermArg), Len(TermArg).
|
||||
/// The offset/length expressions are kept as AML for lazy evaluation.
|
||||
fn opRegion(self: *Parser, scope: *Node) Error!void {
|
||||
const np = try self.readNameString();
|
||||
const space = try self.readByte();
|
||||
const off_start = self.pos;
|
||||
try self.object(scope);
|
||||
const off_end = self.pos;
|
||||
try self.object(scope);
|
||||
const len_end = self.pos;
|
||||
const node = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .region);
|
||||
node.region_space = space;
|
||||
node.region_offset_aml = self.aml[off_start..off_end];
|
||||
node.region_len_aml = self.aml[off_end..len_end];
|
||||
}
|
||||
|
||||
fn opDataRegion(self: *Parser, scope: *Node) Error!void {
|
||||
const np = try self.readNameString();
|
||||
try self.args(scope, 3); // signature, oem id, oem table id (TermArgs)
|
||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .region);
|
||||
}
|
||||
|
||||
fn opMutex(self: *Parser, scope: *Node) Error!void {
|
||||
const np = try self.readNameString();
|
||||
try self.skip(1); // sync flags
|
||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .mutex);
|
||||
}
|
||||
|
||||
fn opEvent(self: *Parser, scope: *Node) Error!void {
|
||||
const np = try self.readNameString();
|
||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .event);
|
||||
}
|
||||
|
||||
/// CreateXField: `count` TermArgs then the new field's NameString.
|
||||
fn opCreateField(self: *Parser, scope: *Node, count: usize) Error!void {
|
||||
try self.args(scope, count);
|
||||
const np = try self.readNameString();
|
||||
_ = try self.namespace.place(scope, np.rooted, np.parents, np.slice(), .name);
|
||||
}
|
||||
|
||||
/// Field / IndexField / BankField: a region/bank reference, flags, then a
|
||||
/// FieldList whose NamedFields become nodes in the current scope. For a plain
|
||||
/// Field, the first NameString is the backing region — captured so field units
|
||||
/// carry a region + bit position the evaluator can read/write.
|
||||
fn opField(self: *Parser, scope: *Node, name_strings: u8, bank: bool) Error!void {
|
||||
const start = self.pos;
|
||||
const end = start + try self.readPkgLength();
|
||||
var region: ?*Node = null;
|
||||
var i: u8 = 0;
|
||||
while (i < name_strings) : (i += 1) {
|
||||
const np = try self.readNameString();
|
||||
// Only a plain Field's single NameString denotes an OperationRegion.
|
||||
if (name_strings == 1) region = self.namespace.resolve(scope, np.rooted, np.parents, np.slice());
|
||||
}
|
||||
if (bank) try self.object(scope); // bank value TermArg
|
||||
const flags = try self.readByte();
|
||||
self.fieldList(end, scope, region, flags & 0x0F);
|
||||
}
|
||||
|
||||
fn fieldList(self: *Parser, end: usize, scope: *Node, region: ?*Node, initial_access: u8) void {
|
||||
var bit_offset: u32 = 0;
|
||||
var access = initial_access;
|
||||
while (self.pos < end) {
|
||||
const lead = self.peek() orelse break;
|
||||
switch (lead) {
|
||||
0x00 => { // ReservedField: advances the bit position
|
||||
self.pos += 1;
|
||||
const width = self.readPkgLength() catch break;
|
||||
bit_offset += @intCast(width);
|
||||
},
|
||||
0x01 => { // AccessField: AccessType (low nibble) + AccessAttrib
|
||||
self.pos += 1;
|
||||
const at = self.readByte() catch break;
|
||||
self.skip(1) catch break;
|
||||
access = at & 0x0F;
|
||||
},
|
||||
0x02 => { // ConnectField: NameString | BufferData
|
||||
self.pos += 1;
|
||||
self.object(scope) catch break;
|
||||
},
|
||||
0x03 => { // ExtendedAccessField: type + attrib + length
|
||||
self.pos += 1;
|
||||
self.skip(3) catch break;
|
||||
},
|
||||
else => { // NamedField: NameSeg + PkgLength (bit width)
|
||||
const seg = self.readNameSeg() catch break;
|
||||
const width = self.readPkgLength() catch break;
|
||||
const unit = self.namespace.newFieldUnit(scope, seg) catch break;
|
||||
unit.region = region;
|
||||
unit.bit_offset = bit_offset;
|
||||
unit.bit_width = @intCast(width);
|
||||
unit.access_type = access;
|
||||
bit_offset += @intCast(width);
|
||||
},
|
||||
}
|
||||
}
|
||||
self.pos = end;
|
||||
}
|
||||
|
||||
// --- control flow -------------------------------------------------------
|
||||
|
||||
fn opIf(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.pos;
|
||||
const end = start + try self.readPkgLength();
|
||||
try self.object(scope); // predicate
|
||||
self.termList(end, scope);
|
||||
if (self.peek() == op.else_op) {
|
||||
self.pos += 1;
|
||||
try self.opElse(scope);
|
||||
}
|
||||
}
|
||||
|
||||
fn opElse(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.pos;
|
||||
const end = start + try self.readPkgLength();
|
||||
self.termList(end, scope);
|
||||
}
|
||||
|
||||
fn opWhile(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.pos;
|
||||
const end = start + try self.readPkgLength();
|
||||
try self.object(scope); // predicate
|
||||
self.termList(end, scope);
|
||||
}
|
||||
|
||||
fn opLnot(self: *Parser, scope: *Node) Error!void {
|
||||
// 0x92 followed by 0x93/94/95 is a compound comparison (two operands);
|
||||
// otherwise it is a plain LNot of one operand.
|
||||
const b = self.peek() orelse return error.Truncated;
|
||||
switch (b) {
|
||||
op.lnot.not_equal, op.lnot.less_equal, op.lnot.greater_equal => {
|
||||
self.pos += 1;
|
||||
try self.args(scope, 2);
|
||||
},
|
||||
else => try self.object(scope),
|
||||
}
|
||||
}
|
||||
|
||||
fn opMatch(self: *Parser, scope: *Node) Error!void {
|
||||
try self.object(scope); // search package
|
||||
try self.skip(1); // match opcode 1
|
||||
try self.object(scope); // operand 1
|
||||
try self.skip(1); // match opcode 2
|
||||
try self.object(scope); // operand 2
|
||||
try self.object(scope); // start index
|
||||
}
|
||||
|
||||
// --- extended opcodes (0x5B xx) -----------------------------------------
|
||||
|
||||
fn opExt(self: *Parser, scope: *Node) Error!void {
|
||||
const e = try self.readByte();
|
||||
switch (e) {
|
||||
op.ext.mutex => try self.opMutex(scope),
|
||||
op.ext.event => try self.opEvent(scope),
|
||||
op.ext.op_region => try self.opRegion(scope),
|
||||
op.ext.data_region => try self.opDataRegion(scope),
|
||||
op.ext.field => try self.opField(scope, 1, false),
|
||||
op.ext.index_field => try self.opField(scope, 2, false),
|
||||
op.ext.bank_field => try self.opField(scope, 2, true),
|
||||
op.ext.device => try self.opScopeLike(scope, .device),
|
||||
op.ext.thermal_zone => try self.opScopeLike(scope, .thermal_zone),
|
||||
op.ext.processor => try self.opProcessor(scope),
|
||||
op.ext.power_res => try self.opPowerRes(scope),
|
||||
|
||||
op.ext.cond_ref_of => try self.args(scope, 2), // SuperName, Target
|
||||
op.ext.create_field => try self.opCreateField(scope, 3),
|
||||
op.ext.load_table => try self.args(scope, 6),
|
||||
op.ext.load => try self.args(scope, 2), // NameString, Target
|
||||
op.ext.stall, op.ext.sleep => try self.args(scope, 1),
|
||||
op.ext.acquire => {
|
||||
try self.object(scope); // mutex SuperName
|
||||
try self.skip(2); // timeout WordData
|
||||
},
|
||||
op.ext.signal, op.ext.reset, op.ext.release, op.ext.unload => try self.args(scope, 1),
|
||||
op.ext.wait => try self.args(scope, 2),
|
||||
op.ext.from_bcd, op.ext.to_bcd => try self.args(scope, 2),
|
||||
op.ext.fatal => {
|
||||
try self.skip(5); // Type(byte) + Code(dword)
|
||||
try self.object(scope); // Arg TermArg
|
||||
},
|
||||
op.ext.revision, op.ext.debug, op.ext.timer => {},
|
||||
|
||||
else => return error.Malformed,
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
fn isNameStart(b: u8) bool {
|
||||
return (b >= op.name_char_start and b <= op.name_char_end) or
|
||||
b == op.name_char_underscore or
|
||||
b == op.root_char or
|
||||
b == op.parent_prefix_char or
|
||||
b == op.dual_name_prefix or
|
||||
b == op.multi_name_prefix;
|
||||
}
|
||||
@@ -1,76 +0,0 @@
|
||||
//! The firmware-agnostic discovery facade.
|
||||
//!
|
||||
//! The kernel calls `platform.discover()` and gets back a generic `DeviceTree`
|
||||
//! without ever naming ACPI or device-tree — the same way it imports `arch`
|
||||
//! without naming x86_64. Which backend runs is decided *at runtime* from what
|
||||
//! the bootloader handed us (an ACPI RSDP today, a device-tree blob later),
|
||||
//! because a single image — a future ARM kernel especially — may boot under
|
||||
//! either firmware. That's a deliberate divergence from `arch`, which is a
|
||||
//! compile-time choice.
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const device = @import("device.zig");
|
||||
const acpi = @import("acpi.zig");
|
||||
const power = @import("power.zig");
|
||||
const devicetree = @import("devicetree.zig");
|
||||
|
||||
pub const DeviceTree = device.DeviceTree;
|
||||
pub const Device = device.Device;
|
||||
pub const DeviceClass = device.DeviceClass;
|
||||
pub const Hal = device.Hal;
|
||||
pub const PowerInfo = acpi.PowerInfo;
|
||||
pub const AmlStats = acpi.AmlStats;
|
||||
pub const PlatformInfo = acpi.PlatformInfo;
|
||||
pub const RegAccess = acpi.RegAccess;
|
||||
pub const IsoEntry = acpi.IsoEntry;
|
||||
|
||||
/// The register map + sleep types discovery extracted, for logging/diagnostics.
|
||||
pub fn powerInfo() PowerInfo {
|
||||
return acpi.power_info;
|
||||
}
|
||||
|
||||
/// The scalar firmware facts the arch layer needs to avoid legacy assumptions
|
||||
/// (8259 presence, LAPIC base, PM timer, SPCR UART, IRQ overrides).
|
||||
pub fn platformInfo() PlatformInfo {
|
||||
return acpi.platform_info;
|
||||
}
|
||||
|
||||
/// AML parse integrity/diagnostics (namespace node count, bytes consumed).
|
||||
pub fn amlStats() AmlStats {
|
||||
return acpi.aml_stats;
|
||||
}
|
||||
|
||||
/// Enumerate hardware into a fresh device tree. `hal` supplies the hardware
|
||||
/// primitives the backend needs (MMIO mapping for PCIe config space, port I/O for
|
||||
/// ACPI registers); pass the arch implementation. Errors leave nothing to clean up
|
||||
/// beyond the tree's own allocations.
|
||||
pub fn discover(
|
||||
boot_info: *const danos.BootInfo,
|
||||
allocator: std.mem.Allocator,
|
||||
hal: Hal,
|
||||
) !DeviceTree {
|
||||
var dt = try DeviceTree.init(allocator);
|
||||
|
||||
if (boot_info.acpi_rsdp != 0) {
|
||||
try acpi.discover(boot_info.acpi_rsdp, &dt, hal);
|
||||
} else {
|
||||
// No ACPI RSDP. A device-tree boot would parse its blob here; today that
|
||||
// path is a stub, so this reports the machine described itself no way we
|
||||
// understand yet.
|
||||
try devicetree.discover(&dt);
|
||||
}
|
||||
|
||||
return dt;
|
||||
}
|
||||
|
||||
/// Restart the machine. Never returns on success; returns only if no reset method
|
||||
/// worked (extremely unlikely). Backend-agnostic entry the kernel calls.
|
||||
pub fn reboot(hal: Hal) void {
|
||||
power.reboot(hal);
|
||||
}
|
||||
|
||||
/// Power the machine off (ACPI S5). Never returns on success.
|
||||
pub fn shutdown(hal: Hal) void {
|
||||
power.shutdown(hal);
|
||||
}
|
||||
@@ -1,265 +0,0 @@
|
||||
//! x86_64 CPU operations. This is the "arch" module: the generic kernel imports
|
||||
//! it as `@import("arch")` and never names x86_64 directly, so a second
|
||||
//! architecture is added by pointing that module at a different directory in
|
||||
//! build.zig — no change to the generic code. Keep everything CPU-specific here
|
||||
//! (halt, the descriptor tables, later paging), and nothing generic.
|
||||
|
||||
const danos = @import("danos");
|
||||
const gdt = @import("gdt.zig");
|
||||
const tss = @import("tss.zig");
|
||||
const idt = @import("idt.zig");
|
||||
const paging = @import("paging.zig");
|
||||
const serial = @import("serial.zig");
|
||||
const apic = @import("apic.zig");
|
||||
const ioapic = @import("ioapic.zig");
|
||||
const io = @import("io.zig");
|
||||
|
||||
/// The saved register/trap frame passed to a fault handler.
|
||||
pub const CpuState = idt.CpuState;
|
||||
|
||||
/// Bring up the serial port (the kernel's machine-readable log). No dependencies,
|
||||
/// so it can be the very first thing called.
|
||||
pub fn serialInit() void {
|
||||
serial.init();
|
||||
}
|
||||
|
||||
/// Write bytes to the serial port.
|
||||
pub fn serialWrite(bytes: []const u8) void {
|
||||
serial.write(bytes);
|
||||
}
|
||||
|
||||
/// Set up the CPU's descriptor tables: our own GDT, the TSS (with an interrupt
|
||||
/// stack for double faults), then the IDT with exception handlers. After this a
|
||||
/// CPU fault is reported instead of triple-faulting. Install the fault handler
|
||||
/// (setFaultHandler) first so early faults are caught.
|
||||
pub fn init() void {
|
||||
gdt.init();
|
||||
tss.init();
|
||||
idt.init();
|
||||
}
|
||||
|
||||
/// Build the kernel's own page tables (with real permissions) and switch onto
|
||||
/// them. Needs the frame allocator and the boot info (for the memory map and the
|
||||
/// kernel's segment layout). Call once the frame allocator is up.
|
||||
pub fn enablePaging(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void {
|
||||
paging.init(allocFrame, boot_info);
|
||||
}
|
||||
|
||||
/// Map a page into the kernel address space (non-executable). For the heap, etc.
|
||||
pub fn mapPage(virt: u64, phys: u64, writable: bool) void {
|
||||
paging.map(virt, phys, writable);
|
||||
}
|
||||
|
||||
/// Remove a kernel mapping.
|
||||
pub fn unmapPage(virt: u64) void {
|
||||
paging.unmap(virt);
|
||||
}
|
||||
|
||||
/// CR3 holds the physical address of the active top-level page table.
|
||||
pub fn readCr3() u64 {
|
||||
return asm volatile ("mov %%cr3, %[out]"
|
||||
: [out] "=r" (-> u64),
|
||||
);
|
||||
}
|
||||
|
||||
/// Kernel tick rate: 1000 Hz (1 ms), the scheduler's time quantum.
|
||||
pub const timer_hz = 1000;
|
||||
|
||||
/// The ACPI PM timer, as a calibration reference (re-exported for the config).
|
||||
pub const PmTimer = apic.PmTimer;
|
||||
/// A MADT interrupt-source override (re-exported for the config).
|
||||
pub const IsoEntry = ioapic.IsoEntry;
|
||||
|
||||
/// Discovered platform facts the arch layer needs so it makes no legacy
|
||||
/// assumptions — sourced from the device tree + ACPI, passed in by the kernel.
|
||||
pub const PlatformConfig = struct {
|
||||
/// Whether the legacy 8259 PIC is present (skip programming it if not).
|
||||
pic_present: bool = true,
|
||||
/// HPET MMIO base (0 = none) — a calibration reference for the timer.
|
||||
hpet_base: u64 = 0,
|
||||
/// The ACPI PM timer, another calibration reference.
|
||||
pm_timer: ?PmTimer = null,
|
||||
/// I/O APIC MMIO base + its first global system interrupt (0 = none).
|
||||
ioapic_base: u64 = 0,
|
||||
ioapic_gsi_base: u32 = 0,
|
||||
/// MADT ISA-IRQ overrides, for I/O APIC routing.
|
||||
overrides: []const IsoEntry = &.{},
|
||||
};
|
||||
|
||||
/// Apply the discovered platform config. Must run before `startTimer` (the timer
|
||||
/// calibration reads `hpet_base`/`pm_timer`) and before any interrupt routing.
|
||||
/// Maps + masks the I/O APIC immediately.
|
||||
pub fn configurePlatform(cfg: PlatformConfig) void {
|
||||
apic.configure(cfg.pic_present, cfg.hpet_base, cfg.pm_timer);
|
||||
ioapic.configure(cfg.ioapic_base, cfg.ioapic_gsi_base, cfg.overrides);
|
||||
ioapic.init();
|
||||
}
|
||||
|
||||
/// Point the serial console at the UART ACPI's SPCR table named (MMIO or I/O port).
|
||||
pub fn serialReconfigure(is_mmio: bool, addr: u64) void {
|
||||
serial.reconfigure(is_mmio, addr);
|
||||
}
|
||||
|
||||
/// The reference clock the timer was calibrated against ("cpuid"/"hpet"/…).
|
||||
pub fn timerCalibrationSource() []const u8 {
|
||||
return apic.calibrationSource();
|
||||
}
|
||||
|
||||
/// I/O APIC diagnostics (for boot logging / verification).
|
||||
pub fn ioapicEntryCount() u32 {
|
||||
return ioapic.entryCount();
|
||||
}
|
||||
pub fn ioapicEntryLow(n: u32) u32 {
|
||||
return ioapic.entryLow(n);
|
||||
}
|
||||
|
||||
/// Enable the Local APIC, calibrate its timer against the best available reference
|
||||
/// (see apic.calibrate — no longer the PIT by default), and start it firing at
|
||||
/// `timer_hz` — the kernel's real-time heartbeat. Interrupts still have to be
|
||||
/// unmasked with enableInterrupts() to be delivered. Run `configurePlatform` first.
|
||||
pub fn startTimer() void {
|
||||
apic.init();
|
||||
apic.calibrate();
|
||||
idt.setHandler(apic.timer_vector, apic.timerTick);
|
||||
apic.initTimer(timer_hz);
|
||||
}
|
||||
|
||||
/// Number of timer ticks since startTimer().
|
||||
pub fn ticks() u64 {
|
||||
return apic.ticks();
|
||||
}
|
||||
|
||||
// Monotonic high-resolution clock (from the TSC), one function per resolution.
|
||||
pub fn nanos() u64 {
|
||||
return apic.nanos();
|
||||
}
|
||||
pub fn micros() u64 {
|
||||
return apic.micros();
|
||||
}
|
||||
pub fn millis() u64 {
|
||||
return apic.millis();
|
||||
}
|
||||
|
||||
/// Measured LAPIC timer / TSC frequencies in Hz (from calibration).
|
||||
pub fn lapicHz() u64 {
|
||||
return apic.lapicHz();
|
||||
}
|
||||
pub fn tscHz() u64 {
|
||||
return apic.tscHz();
|
||||
}
|
||||
|
||||
/// Unmask maskable interrupts (`sti`) so device interrupts get delivered.
|
||||
pub fn enableInterrupts() void {
|
||||
asm volatile ("sti");
|
||||
}
|
||||
|
||||
/// Mask maskable interrupts (`cli`).
|
||||
pub fn disableInterrupts() void {
|
||||
asm volatile ("cli");
|
||||
}
|
||||
|
||||
/// Disable interrupts and return the previous flags, so a nested critical section
|
||||
/// can restore the caller's state rather than blindly re-enabling. Pairs with
|
||||
/// restoreInterrupts.
|
||||
pub fn saveInterrupts() u64 {
|
||||
var flags: u64 = undefined;
|
||||
asm volatile (
|
||||
\\pushfq
|
||||
\\pop %[f]
|
||||
\\cli
|
||||
: [f] "=r" (flags),
|
||||
:
|
||||
: .{ .memory = true }
|
||||
);
|
||||
return flags;
|
||||
}
|
||||
|
||||
/// Re-enable interrupts only if they were enabled when `flags` was captured.
|
||||
pub fn restoreInterrupts(flags: u64) void {
|
||||
if (flags & 0x200 != 0) asm volatile ("sti" ::: .{ .memory = true }); // bit 9 = IF
|
||||
}
|
||||
|
||||
/// Register a callback the timer interrupt invokes each tick (e.g. the scheduler).
|
||||
pub fn setTickHook(hook: *const fn () void) void {
|
||||
apic.setTickHook(hook);
|
||||
}
|
||||
|
||||
// --- context switching (for the scheduler) -------------------------------
|
||||
|
||||
/// Save the current task's registers/stack and resume `new_rsp`; the old stack
|
||||
/// pointer is written to `old_rsp`. Defined in isr.s.
|
||||
extern fn switch_context(old_rsp: *usize, new_rsp: usize) callconv(.c) void;
|
||||
|
||||
pub fn switchContext(old_rsp: *usize, new_rsp: usize) void {
|
||||
switch_context(old_rsp, new_rsp);
|
||||
}
|
||||
|
||||
/// Build the initial stack for a new task so that switching to it lands in
|
||||
/// `task_trampoline`, which then calls `entry`. Returns the saved stack pointer.
|
||||
/// The layout must match switch_context's push order (callee-saved, then the
|
||||
/// return address on top); `entry` is smuggled in via the r15 slot.
|
||||
pub fn initTaskStack(stack_top: usize, entry: usize) usize {
|
||||
const trampoline = @extern(*const anyopaque, .{ .name = "task_trampoline" });
|
||||
var sp = stack_top;
|
||||
const push = struct {
|
||||
fn f(p: *usize, value: usize) void {
|
||||
p.* -= @sizeOf(usize);
|
||||
@as(*usize, @ptrFromInt(p.*)).* = value;
|
||||
}
|
||||
}.f;
|
||||
push(&sp, @intFromPtr(trampoline)); // return address for switch_context's `ret`
|
||||
push(&sp, 0); // rbx
|
||||
push(&sp, 0); // rbp
|
||||
push(&sp, 0); // r12
|
||||
push(&sp, 0); // r13
|
||||
push(&sp, 0); // r14
|
||||
push(&sp, entry); // r15 -> task entry, read by task_trampoline
|
||||
return sp;
|
||||
}
|
||||
|
||||
/// Route CPU exceptions to `handler`, which receives the trap frame and does not
|
||||
/// return. Until set, faults just halt the core.
|
||||
pub fn setFaultHandler(handler: *const fn (*const CpuState) noreturn) void {
|
||||
idt.on_fault = handler;
|
||||
}
|
||||
|
||||
/// A human-readable name for a CPU exception vector.
|
||||
pub fn vectorName(vector: u64) []const u8 {
|
||||
return idt.vectorName(vector);
|
||||
}
|
||||
|
||||
/// Read `width` bytes (1/2/4) from an I/O port. The generic device layer drives
|
||||
/// ACPI registers through this rather than naming x86 port instructions; on an
|
||||
/// MMIO-only architecture this would be implemented differently.
|
||||
pub fn pioRead(width: u8, port: u16) u32 {
|
||||
return switch (width) {
|
||||
1 => io.inb(port),
|
||||
2 => io.inw(port),
|
||||
4 => io.inl(port),
|
||||
else => 0,
|
||||
};
|
||||
}
|
||||
|
||||
/// Write `width` bytes (1/2/4) to an I/O port.
|
||||
pub fn pioWrite(width: u8, port: u16, value: u32) void {
|
||||
switch (width) {
|
||||
1 => io.outb(port, @truncate(value)),
|
||||
2 => io.outw(port, @truncate(value)),
|
||||
4 => io.outl(port, value),
|
||||
else => {},
|
||||
}
|
||||
}
|
||||
|
||||
/// CR2 holds the faulting linear address after a page fault (#PF, vector 14).
|
||||
pub fn readCr2() u64 {
|
||||
return asm volatile ("mov %%cr2, %[out]"
|
||||
: [out] "=r" (-> u64),
|
||||
);
|
||||
}
|
||||
|
||||
/// Park the core forever. `hlt` drops it into a low-power idle until the next
|
||||
/// interrupt; the loop re-halts on every wake so the stop is permanent. See
|
||||
/// docs/halting.md for the full reasoning.
|
||||
pub fn halt() noreturn {
|
||||
while (true) asm volatile ("hlt");
|
||||
}
|
||||
@@ -1,54 +0,0 @@
|
||||
//! Global Descriptor Table. In long mode segmentation is mostly vestigial, but
|
||||
//! the CPU still needs valid code/data segment descriptors, and the IDT's gates
|
||||
//! reference a code selector — so we install our own flat GDT with known
|
||||
//! selectors (0x08 kernel code, 0x10 kernel data) rather than trusting whatever
|
||||
//! the firmware left in place.
|
||||
|
||||
/// Selectors into the table below (index * 8).
|
||||
pub const kernel_code = 0x08;
|
||||
pub const kernel_data = 0x10;
|
||||
pub const tss_selector = 0x18;
|
||||
|
||||
/// Flat 64-bit descriptors. Base/limit are ignored in long mode; what matters is
|
||||
/// the access byte and, for code, the long-mode (L) flag.
|
||||
/// code: present, ring 0, executable, readable, L=1 -> 0x00AF9A00_0000FFFF
|
||||
/// data: present, ring 0, writable -> 0x00CF9200_0000FFFF
|
||||
/// The last two slots hold one 16-byte TSS descriptor, filled in by setTss.
|
||||
var table = [_]u64{
|
||||
0, // null descriptor (required)
|
||||
0x00AF9A000000FFFF, // kernel code (0x08)
|
||||
0x00CF92000000FFFF, // kernel data (0x10)
|
||||
0, // TSS descriptor low (0x18)
|
||||
0, // TSS descriptor high
|
||||
};
|
||||
|
||||
/// Fill the 64-bit TSS system descriptor (two GDT slots) so the task register can
|
||||
/// point at our TSS. Type 0x89 = present, ring 0, available 64-bit TSS.
|
||||
pub fn setTss(base: u64, limit: u64) void {
|
||||
table[3] = (limit & 0xFFFF) |
|
||||
((base & 0xFFFF) << 16) |
|
||||
(((base >> 16) & 0xFF) << 32) |
|
||||
(@as(u64, 0x89) << 40) |
|
||||
(((limit >> 16) & 0xF) << 48) |
|
||||
(((base >> 24) & 0xFF) << 56);
|
||||
table[4] = (base >> 32) & 0xFFFFFFFF;
|
||||
}
|
||||
|
||||
/// The operand `lgdt` wants: table byte-length minus one, then its address.
|
||||
const Descriptor = packed struct {
|
||||
limit: u16,
|
||||
base: u64,
|
||||
};
|
||||
|
||||
/// Loads the GDT and reloads the segment registers (including CS). Defined in
|
||||
/// isr.s — it uses the selectors 0x08 (code) and 0x10 (data) that match `table`.
|
||||
extern fn gdt_flush(descriptor: *const Descriptor) callconv(.c) void;
|
||||
|
||||
/// Install our GDT and switch onto its segments.
|
||||
pub fn init() void {
|
||||
const descriptor = Descriptor{
|
||||
.limit = @sizeOf(@TypeOf(table)) - 1,
|
||||
.base = @intFromPtr(&table),
|
||||
};
|
||||
gdt_flush(&descriptor);
|
||||
}
|
||||
@@ -1,97 +0,0 @@
|
||||
//! I/O APIC — routes external device interrupts (a device's line) to a LAPIC
|
||||
//! vector on a chosen CPU. Its address and the ISA-IRQ-to-GSI remappings come from
|
||||
//! ACPI's MADT (via discovery), never assumed.
|
||||
//!
|
||||
//! Status: groundwork. The only interrupt danos handles today is the LAPIC's own
|
||||
//! timer, which needs no I/O APIC — so nothing calls `routeIrq` yet. What runs now
|
||||
//! is `init`, which maps the I/O APIC and **masks every input**, the correct
|
||||
//! quiescent state on a legacy-free machine. `routeIrq` is ready for the first real
|
||||
//! device driver (a keyboard, say).
|
||||
|
||||
const paging = @import("paging.zig");
|
||||
|
||||
/// A MADT Interrupt Source Override: an ISA IRQ that appears at a different global
|
||||
/// system interrupt, with its own polarity/trigger (MPS INTI `flags`).
|
||||
pub const IsoEntry = struct { source: u8, gsi: u32, flags: u16 };
|
||||
|
||||
var base: u64 = 0; // 0 = no I/O APIC discovered
|
||||
var gsi_base: u32 = 0;
|
||||
var max_entries: u32 = 0;
|
||||
var overrides: [16]IsoEntry = undefined;
|
||||
var override_count: usize = 0;
|
||||
|
||||
// The I/O APIC exposes an index register (IOREGSEL) and a data window (IOWIN).
|
||||
const reg_ioregsel = 0x00;
|
||||
const reg_iowin = 0x10;
|
||||
const reg_version = 0x01;
|
||||
const redir_base = 0x10; // redirection table: two 32-bit regs per entry
|
||||
const redir_mask = 1 << 16; // mask bit in the low dword
|
||||
|
||||
/// Supply the discovered I/O APIC location + the MADT IRQ overrides. Call before `init`.
|
||||
pub fn configure(ioapic_base: u64, ioapic_gsi_base: u32, isos: []const IsoEntry) void {
|
||||
base = ioapic_base;
|
||||
gsi_base = ioapic_gsi_base;
|
||||
override_count = @min(isos.len, overrides.len);
|
||||
for (isos[0..override_count], 0..) |iso, i| overrides[i] = iso;
|
||||
}
|
||||
|
||||
fn regRead(index: u32) u32 {
|
||||
@as(*volatile u32, @ptrFromInt(base + reg_ioregsel)).* = index;
|
||||
return @as(*volatile u32, @ptrFromInt(base + reg_iowin)).*;
|
||||
}
|
||||
fn regWrite(index: u32, value: u32) void {
|
||||
@as(*volatile u32, @ptrFromInt(base + reg_ioregsel)).* = index;
|
||||
@as(*volatile u32, @ptrFromInt(base + reg_iowin)).* = value;
|
||||
}
|
||||
|
||||
fn writeEntry(n: u32, low: u32, high: u32) void {
|
||||
regWrite(redir_base + 2 * n, low);
|
||||
regWrite(redir_base + 2 * n + 1, high);
|
||||
}
|
||||
|
||||
/// Map the I/O APIC and mask every redirection entry — the safe quiescent state.
|
||||
pub fn init() void {
|
||||
if (base == 0) return;
|
||||
paging.map(base & ~@as(u64, 0xFFF), base & ~@as(u64, 0xFFF), true);
|
||||
max_entries = ((regRead(reg_version) >> 16) & 0xFF) + 1;
|
||||
var n: u32 = 0;
|
||||
while (n < max_entries) : (n += 1) writeEntry(n, redir_mask, 0);
|
||||
}
|
||||
|
||||
/// Route ISA `irq` to `vector` on the LAPIC `apic_id`, honouring a MADT override
|
||||
/// for its GSI/polarity/trigger, and unmask it. No caller yet — groundwork for the
|
||||
/// first device driver.
|
||||
pub fn routeIrq(irq: u8, vector: u8, apic_id: u8) void {
|
||||
if (base == 0) return;
|
||||
|
||||
var gsi: u32 = irq;
|
||||
var flags: u16 = 0;
|
||||
for (overrides[0..override_count]) |o| {
|
||||
if (o.source == irq) {
|
||||
gsi = o.gsi;
|
||||
flags = o.flags;
|
||||
}
|
||||
}
|
||||
if (gsi < gsi_base) return;
|
||||
const n = gsi - gsi_base;
|
||||
if (n >= max_entries) return;
|
||||
|
||||
// Low dword: vector + delivery mode fixed(0) + physical dest(0), unmasked.
|
||||
// MPS INTI flags: bits [1:0] polarity (3 = active low), [3:2] trigger (3 = level).
|
||||
var low: u32 = vector;
|
||||
if (flags & 0x3 == 3) low |= (1 << 13);
|
||||
if ((flags >> 2) & 0x3 == 3) low |= (1 << 15);
|
||||
const high: u32 = @as(u32, apic_id) << 24; // destination APIC ID
|
||||
writeEntry(n, low, high);
|
||||
}
|
||||
|
||||
/// Number of redirection entries the I/O APIC advertises (0 until `init`).
|
||||
pub fn entryCount() u32 {
|
||||
return max_entries;
|
||||
}
|
||||
|
||||
/// The low dword of redirection entry `n` — for diagnostics/read-back.
|
||||
pub fn entryLow(n: u32) u32 {
|
||||
if (base == 0) return 0;
|
||||
return regRead(redir_base + 2 * n);
|
||||
}
|
||||
@@ -1,180 +0,0 @@
|
||||
# x86_64 low-level entry code: the CPU-exception stubs, plus the GDT/IDT load
|
||||
# helpers. Kept in a dedicated assembly file rather than inline asm because these
|
||||
# need real labels and cross-symbol jumps/calls (isr_common, exceptionHandler),
|
||||
# and because `lgdt`/`lidt` memory operands aren't expressible in Zig inline asm.
|
||||
#
|
||||
# Each exception vector normalises the stack to a uniform trap frame — a dummy
|
||||
# error code where the CPU pushes none, then the vector number — and jumps to the
|
||||
# shared tail, which saves the general registers and calls the Zig handler with a
|
||||
# pointer to the frame (matching src/arch/x86_64/idt.zig's CpuState).
|
||||
|
||||
.text
|
||||
|
||||
# gdt_flush(rdi = *GDT descriptor): load the GDT, reload the data segment
|
||||
# registers to the data selector, and reload CS to the code selector. CS can't be
|
||||
# set with mov, so we far-return through the caller's own return address.
|
||||
.global gdt_flush
|
||||
gdt_flush:
|
||||
lgdt (%rdi)
|
||||
mov $0x10, %ax # kernel data selector
|
||||
mov %ax, %ds
|
||||
mov %ax, %es
|
||||
mov %ax, %ss
|
||||
mov %ax, %fs
|
||||
mov %ax, %gs
|
||||
pop %rax # caller's return address
|
||||
push $0x08 # kernel code selector (new CS)
|
||||
push %rax # return address (new RIP)
|
||||
lretq
|
||||
|
||||
# idt_flush(rdi = *IDT descriptor): load the IDT.
|
||||
.global idt_flush
|
||||
idt_flush:
|
||||
lidt (%rdi)
|
||||
ret
|
||||
|
||||
# load_tr(di = TSS selector): load the task register.
|
||||
.global load_tr
|
||||
load_tr:
|
||||
ltr %di
|
||||
ret
|
||||
|
||||
# switch_context(rdi = &old_task.rsp, rsi = new_task.rsp)
|
||||
# Cooperative context switch: save the callee-saved registers on the current
|
||||
# stack, stash the stack pointer in the old task, load the new task's stack
|
||||
# pointer, restore its callee-saved registers, and return into it. Caller-saved
|
||||
# registers are the compiler's responsibility (this looks like a normal call).
|
||||
.global switch_context
|
||||
switch_context:
|
||||
push %rbx
|
||||
push %rbp
|
||||
push %r12
|
||||
push %r13
|
||||
push %r14
|
||||
push %r15
|
||||
mov %rsp, (%rdi) # save old stack pointer into old_task.rsp
|
||||
mov %rsi, %rsp # switch to the new task's stack
|
||||
pop %r15
|
||||
pop %r14
|
||||
pop %r13
|
||||
pop %r12
|
||||
pop %rbp
|
||||
pop %rbx
|
||||
ret # return into the new task's saved instruction pointer
|
||||
|
||||
# task_trampoline: the first thing a freshly-spawned task runs. init_task_stack
|
||||
# leaves its entry function in r15. New tasks start with interrupts enabled.
|
||||
.global task_trampoline
|
||||
task_trampoline:
|
||||
sti
|
||||
call *%r15 # call the task entry (fn() void)
|
||||
1: hlt # if the entry returns, idle (still preemptible)
|
||||
jmp 1b
|
||||
|
||||
# Stub for a vector the CPU does NOT push an error code for: push a dummy 0.
|
||||
.macro STUB_NOERR vec
|
||||
.global isr\vec
|
||||
isr\vec:
|
||||
pushq $0
|
||||
pushq $\vec
|
||||
jmp isr_common
|
||||
.endm
|
||||
|
||||
# Stub for a vector the CPU DOES push an error code for: leave it in place.
|
||||
.macro STUB_ERR vec
|
||||
.global isr\vec
|
||||
isr\vec:
|
||||
pushq $\vec
|
||||
jmp isr_common
|
||||
.endm
|
||||
|
||||
STUB_NOERR 0
|
||||
STUB_NOERR 1
|
||||
STUB_NOERR 2
|
||||
STUB_NOERR 3
|
||||
STUB_NOERR 4
|
||||
STUB_NOERR 5
|
||||
STUB_NOERR 6
|
||||
STUB_NOERR 7
|
||||
STUB_ERR 8
|
||||
STUB_NOERR 9
|
||||
STUB_ERR 10
|
||||
STUB_ERR 11
|
||||
STUB_ERR 12
|
||||
STUB_ERR 13
|
||||
STUB_ERR 14
|
||||
STUB_NOERR 15
|
||||
STUB_NOERR 16
|
||||
STUB_ERR 17
|
||||
STUB_NOERR 18
|
||||
STUB_NOERR 19
|
||||
STUB_NOERR 20
|
||||
STUB_ERR 21
|
||||
STUB_NOERR 22
|
||||
STUB_NOERR 23
|
||||
STUB_NOERR 24
|
||||
STUB_NOERR 25
|
||||
STUB_NOERR 26
|
||||
STUB_NOERR 27
|
||||
STUB_NOERR 28
|
||||
STUB_NOERR 29
|
||||
STUB_NOERR 30
|
||||
STUB_NOERR 31
|
||||
|
||||
# Device-interrupt vectors (timer, spurious, room for more). None push an error
|
||||
# code, so they all use the dummy-zero form.
|
||||
STUB_NOERR 32
|
||||
STUB_NOERR 33
|
||||
STUB_NOERR 34
|
||||
STUB_NOERR 35
|
||||
STUB_NOERR 36
|
||||
STUB_NOERR 37
|
||||
STUB_NOERR 38
|
||||
STUB_NOERR 39
|
||||
STUB_NOERR 40
|
||||
STUB_NOERR 41
|
||||
STUB_NOERR 42
|
||||
STUB_NOERR 43
|
||||
STUB_NOERR 44
|
||||
STUB_NOERR 45
|
||||
STUB_NOERR 46
|
||||
STUB_NOERR 47
|
||||
|
||||
.extern interruptDispatch
|
||||
|
||||
# Shared tail. Register push order here defines the CpuState field order.
|
||||
isr_common:
|
||||
push %rax
|
||||
push %rbx
|
||||
push %rcx
|
||||
push %rdx
|
||||
push %rsi
|
||||
push %rdi
|
||||
push %rbp
|
||||
push %r8
|
||||
push %r9
|
||||
push %r10
|
||||
push %r11
|
||||
push %r12
|
||||
push %r13
|
||||
push %r14
|
||||
push %r15
|
||||
mov %rsp, %rdi # first argument: pointer to the trap frame
|
||||
call interruptDispatch
|
||||
pop %r15
|
||||
pop %r14
|
||||
pop %r13
|
||||
pop %r12
|
||||
pop %r11
|
||||
pop %r10
|
||||
pop %r9
|
||||
pop %r8
|
||||
pop %rbp
|
||||
pop %rdi
|
||||
pop %rsi
|
||||
pop %rdx
|
||||
pop %rcx
|
||||
pop %rbx
|
||||
pop %rax
|
||||
add $16, %rsp # drop the vector and error code
|
||||
iretq
|
||||
@@ -1,47 +0,0 @@
|
||||
/* Kernel link layout.
|
||||
*
|
||||
* The kernel is linked at a fixed low physical address (set by `image_base` in
|
||||
* build.zig). UEFI runs with memory identity-mapped, so the bootloader can load
|
||||
* each PT_LOAD segment to the physical address matching its virtual address and
|
||||
* jump straight to _start — no page tables to build yet. (Moving to a
|
||||
* higher-half virtual base is a later step, once the bootloader sets up paging.)
|
||||
*/
|
||||
|
||||
ENTRY(_start)
|
||||
|
||||
/* One loadable segment per permission set, so the loader can map .text as R+X,
|
||||
* .rodata as R, and .data/.bss as R+W. FLAGS bits: 1=X, 2=W, 4=R. */
|
||||
PHDRS {
|
||||
text PT_LOAD FLAGS(5); /* R + X */
|
||||
rodata PT_LOAD FLAGS(4); /* R */
|
||||
data PT_LOAD FLAGS(6); /* R + W */
|
||||
}
|
||||
|
||||
SECTIONS {
|
||||
.text ALIGN(4K) : {
|
||||
*(.text .text.*)
|
||||
} :text
|
||||
|
||||
.rodata ALIGN(4K) : {
|
||||
*(.rodata .rodata.*)
|
||||
} :rodata
|
||||
|
||||
.data ALIGN(4K) : {
|
||||
*(.data .data.*)
|
||||
} :data
|
||||
|
||||
/* .bss occupies memory but not file space. The loader zeroes it via the
|
||||
* gap between each PT_LOAD segment's file size and memory size, so no
|
||||
* boundary symbols are needed here. (Zig's self-hosted linker also does not
|
||||
* yet honour linker-script symbol assignments.) */
|
||||
.bss ALIGN(4K) : {
|
||||
*(.bss .bss.*)
|
||||
*(COMMON)
|
||||
} :data
|
||||
|
||||
/DISCARD/ : {
|
||||
*(.comment)
|
||||
*(.note .note.*)
|
||||
*(.eh_frame .eh_frame_hdr)
|
||||
}
|
||||
}
|
||||
@@ -1,153 +0,0 @@
|
||||
//! The kernel's page tables and virtual memory manager.
|
||||
//!
|
||||
//! Builds our own 4-level page tables and switches CR3 onto them, replacing the
|
||||
//! firmware's. Unlike the earlier bootstrap this maps with real permissions:
|
||||
//! RAM is identity-mapped read-write + no-execute, the kernel's own segments get
|
||||
//! their ELF permissions (code R+X, rodata R, data R+W+NX), and page 0 is left
|
||||
//! unmapped as a null guard. It also exposes map/unmap for on-demand mapping,
|
||||
//! which the kernel heap will build on.
|
||||
//!
|
||||
//! Everything is 4 KiB pages — precise and simple; the extra table memory is
|
||||
//! negligible against available RAM.
|
||||
|
||||
const danos = @import("danos");
|
||||
const io = @import("io.zig");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
|
||||
// Page-table entry bits.
|
||||
const present: u64 = 1 << 0;
|
||||
const writable: u64 = 1 << 1;
|
||||
const no_execute: u64 = 1 << 63;
|
||||
const addr_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||
|
||||
// ELF segment flags (p_flags).
|
||||
const pf_x: u32 = 1;
|
||||
const pf_w: u32 = 2;
|
||||
|
||||
// State kept after init so map()/unmap() can serve later callers (e.g. the heap).
|
||||
var kernel_pml4: u64 = 0;
|
||||
var alloc_frame: *const fn () ?u64 = undefined;
|
||||
|
||||
fn tableAt(phys: u64) *[512]u64 {
|
||||
return @ptrFromInt(phys);
|
||||
}
|
||||
|
||||
fn allocTable() u64 {
|
||||
const frame = alloc_frame() orelse @panic("paging: out of memory building page tables");
|
||||
@memset(tableAt(frame)[0..], 0);
|
||||
return frame;
|
||||
}
|
||||
|
||||
/// Return the table an entry points at, creating it if empty. Intermediate
|
||||
/// entries are writable and executable so the leaf's bits govern (a page is
|
||||
/// writable only if every level is; non-executable if any level is).
|
||||
fn descend(entry: *u64) u64 {
|
||||
if (entry.* & present != 0) return entry.* & addr_mask;
|
||||
const frame = allocTable();
|
||||
entry.* = frame | present | writable;
|
||||
return frame;
|
||||
}
|
||||
|
||||
/// Map one 4 KiB page `virt` -> `phys` with `flags` (present is added).
|
||||
fn mapPage(pml4: u64, virt: u64, phys: u64, flags: u64) void {
|
||||
const pml4e = &tableAt(pml4)[(virt >> 39) & 0x1FF];
|
||||
const pdpt = descend(pml4e);
|
||||
const pdpte = &tableAt(pdpt)[(virt >> 30) & 0x1FF];
|
||||
const pd = descend(pdpte);
|
||||
const pde = &tableAt(pd)[(virt >> 21) & 0x1FF];
|
||||
const pt = descend(pde);
|
||||
tableAt(pt)[(virt >> 12) & 0x1FF] = (phys & addr_mask) | flags | present;
|
||||
}
|
||||
|
||||
/// Identity-map [base, base+len) with `flags`, rounded out to whole pages.
|
||||
fn mapRangeIdentity(pml4: u64, base: u64, len: u64, flags: u64) void {
|
||||
var addr = base & ~@as(u64, page_size - 1);
|
||||
const end = base + len;
|
||||
while (addr < end) : (addr += page_size) {
|
||||
if (addr == 0) continue; // leave page 0 unmapped: the null guard
|
||||
mapPage(pml4, addr, addr, flags);
|
||||
}
|
||||
}
|
||||
|
||||
fn regions(mm: danos.MemoryMap) []const danos.MemoryRegion {
|
||||
return @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
||||
}
|
||||
|
||||
/// Enable the NX bit in the page-table format (EFER.NXE). Must happen before we
|
||||
/// load a CR3 whose entries set the NX bit, or those bits are reserved and fault.
|
||||
fn enableNx() void {
|
||||
const efer_msr = 0xC0000080;
|
||||
io.wrmsr(efer_msr, io.rdmsr(efer_msr) | (1 << 11));
|
||||
}
|
||||
|
||||
/// Build the address space and switch onto it.
|
||||
pub fn init(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void {
|
||||
alloc_frame = allocFrame;
|
||||
enableNx();
|
||||
const pml4 = allocTable();
|
||||
|
||||
// 1. All RAM identity-mapped RW + NX. Non-RAM (MMIO) is skipped and stays
|
||||
// unmapped unless mapped explicitly below.
|
||||
for (regions(boot_info.memory_map)) |r| {
|
||||
if (r.kind == .mmio) continue;
|
||||
mapRangeIdentity(pml4, r.base, r.pages * page_size, present | writable | no_execute);
|
||||
}
|
||||
|
||||
// 2. The framebuffer and the Local APIC (device memory we need), RW + NX.
|
||||
const fb = boot_info.framebuffer;
|
||||
mapRangeIdentity(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute);
|
||||
mapPage(pml4, 0xFEE00000, 0xFEE00000, present | writable | no_execute);
|
||||
|
||||
// 3. Overlay the kernel's own segments with their real ELF permissions,
|
||||
// replacing the blanket RW+NX from step 1: code becomes R+X, rodata R,
|
||||
// data R+W+NX. This is the W^X guarantee.
|
||||
for (boot_info.kernel_segments[0..boot_info.kernel_segment_count]) |seg| {
|
||||
var flags: u64 = present;
|
||||
if (seg.flags & pf_w != 0) flags |= writable;
|
||||
if (seg.flags & pf_x == 0) flags |= no_execute;
|
||||
var addr = seg.virt;
|
||||
const end = seg.virt + seg.pages * page_size;
|
||||
while (addr < end) : (addr += page_size) mapPage(pml4, addr, addr, flags);
|
||||
}
|
||||
|
||||
kernel_pml4 = pml4;
|
||||
asm volatile ("mov %[pml4], %%cr3"
|
||||
:
|
||||
: [pml4] "r" (pml4),
|
||||
: .{ .memory = true }
|
||||
);
|
||||
}
|
||||
|
||||
/// Map a page into the kernel address space on demand (for the heap, etc.).
|
||||
/// `writable_page` controls W; pages are always mapped non-executable.
|
||||
pub fn map(virt: u64, phys: u64, writable_page: bool) void {
|
||||
var flags: u64 = present | no_execute;
|
||||
if (writable_page) flags |= writable;
|
||||
mapPage(kernel_pml4, virt, phys, flags);
|
||||
invalidate(virt);
|
||||
}
|
||||
|
||||
/// Remove a mapping and flush it from the TLB.
|
||||
pub fn unmap(virt: u64) void {
|
||||
const pml4e = tableAt(kernel_pml4)[(virt >> 39) & 0x1FF];
|
||||
if (pml4e & present == 0) return;
|
||||
const pdpte = tableAt(pml4e & addr_mask)[(virt >> 30) & 0x1FF];
|
||||
if (pdpte & present == 0) return;
|
||||
const pde = tableAt(pdpte & addr_mask)[(virt >> 21) & 0x1FF];
|
||||
if (pde & present == 0) return;
|
||||
tableAt(pde & addr_mask)[(virt >> 12) & 0x1FF] = 0;
|
||||
invalidate(virt);
|
||||
}
|
||||
|
||||
fn invalidate(virt: u64) void {
|
||||
// invlpg needs its operand via a register-indirect memory reference that Zig
|
||||
// inline asm won't form directly, so stage the address in a register first.
|
||||
asm volatile (
|
||||
\\mov %[v], %%rax
|
||||
\\invlpg (%%rax)
|
||||
:
|
||||
: [v] "r" (virt),
|
||||
: .{ .rax = true, .memory = true }
|
||||
);
|
||||
}
|
||||
@@ -1,48 +0,0 @@
|
||||
//! Task State Segment and its interrupt stack. In long mode the TSS's main job
|
||||
//! is the Interrupt Stack Table: an IDT gate can name an IST entry, and the CPU
|
||||
//! switches to that stack when the exception fires — no matter how broken the
|
||||
//! interrupted stack was. We use IST1 for the double-fault handler, so a fault
|
||||
//! that happens *because* the current stack is unusable still lands on solid
|
||||
//! ground instead of triple-faulting.
|
||||
|
||||
const gdt = @import("gdt.zig");
|
||||
|
||||
/// x86_64 TSS. `packed` because several 64-bit fields sit at 4-byte-unaligned
|
||||
/// offsets (rsp0 at byte 4), which a normal struct would pad away.
|
||||
const Tss = packed struct {
|
||||
reserved0: u32 = 0,
|
||||
rsp0: u64 = 0,
|
||||
rsp1: u64 = 0,
|
||||
rsp2: u64 = 0,
|
||||
reserved1: u64 = 0,
|
||||
ist1: u64 = 0,
|
||||
ist2: u64 = 0,
|
||||
ist3: u64 = 0,
|
||||
ist4: u64 = 0,
|
||||
ist5: u64 = 0,
|
||||
ist6: u64 = 0,
|
||||
ist7: u64 = 0,
|
||||
reserved2: u64 = 0,
|
||||
reserved3: u16 = 0,
|
||||
iomap_base: u16 = 0,
|
||||
};
|
||||
|
||||
/// The IST slot (1-based, as the IDT gate encodes it) used for critical faults.
|
||||
pub const double_fault_ist = 1;
|
||||
|
||||
var tss: Tss align(16) = .{};
|
||||
|
||||
/// Dedicated stack for IST1. Static so it needs no allocator and is always valid.
|
||||
var ist1_stack: [16 * 1024]u8 align(16) = undefined;
|
||||
|
||||
/// Loads the task register with the TSS selector. Defined in isr.s.
|
||||
extern fn load_tr(selector: u16) callconv(.c) void;
|
||||
|
||||
/// Point IST1 at its stack, publish the TSS through the GDT, and load it into the
|
||||
/// task register. Requires the GDT to already be loaded (gdt.init first).
|
||||
pub fn init() void {
|
||||
tss.ist1 = @intFromPtr(&ist1_stack) + ist1_stack.len; // stacks grow down
|
||||
tss.iomap_base = @sizeOf(Tss); // == limit: no I/O permission bitmap
|
||||
gdt.setTss(@intFromPtr(&tss), @sizeOf(Tss) - 1);
|
||||
load_tr(gdt.tss_selector);
|
||||
}
|
||||
@@ -1,248 +0,0 @@
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const arch = @import("arch");
|
||||
const console = @import("console.zig");
|
||||
const pmm = @import("pmm.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const scheduler = @import("scheduler.zig");
|
||||
const platform = @import("platform");
|
||||
const tests = @import("tests.zig");
|
||||
const build_options = @import("build_options");
|
||||
const BootInfo = danos.BootInfo;
|
||||
|
||||
/// The calling convention used to enter the kernel. Pinned to SysV explicitly:
|
||||
/// the bootloader is built for the UEFI target, whose C convention is Microsoft
|
||||
/// x64 (first argument in RCX), while the kernel is SysV (first argument in
|
||||
/// RDI). Both sides reference this so the `boot_info` pointer lands in the
|
||||
/// register the other expects. `danos.kernel_abi` re-exports it to the loader.
|
||||
pub const kernel_abi = danos.kernel_abi;
|
||||
|
||||
/// The system console, valid once `kmain` has initialised it. Global so the
|
||||
/// panic handler can reach it too.
|
||||
var con: console.Console = undefined;
|
||||
var con_ready = false;
|
||||
|
||||
/// Kernel entry point. The bootloader jumps here after `ExitBootServices` with a
|
||||
/// pointer to the handoff data. There is no runtime, no stack unwinding, and no
|
||||
/// caller to return to, so this never returns.
|
||||
export fn _start(boot_info: *const BootInfo) callconv(kernel_abi) noreturn {
|
||||
kmain(boot_info);
|
||||
}
|
||||
|
||||
fn kmain(boot_info: *const BootInfo) noreturn {
|
||||
arch.serialInit(); // machine-readable log; console mirrors to it
|
||||
|
||||
const fb = boot_info.framebuffer;
|
||||
const serial0 = console.SerialConsole;
|
||||
con = console.Console.init(fb);
|
||||
con.clear();
|
||||
con_ready = true;
|
||||
|
||||
// Catch CPU exceptions before doing anything that might fault: install our
|
||||
// reporter, then bring up the GDT + IDT.
|
||||
arch.setFaultHandler(onException);
|
||||
arch.init();
|
||||
|
||||
con.write("danos: initalizing kernel...");
|
||||
|
||||
serial0.debugWrite("danos: framebuffer console online\n");
|
||||
serial0.debugWrite("danos: cpu tables online (GDT, IDT, TSS)\n");
|
||||
serial0.debugPrint(" resolution : {d}x{d}\n", .{ fb.width, fb.height });
|
||||
serial0.debugPrint(" pitch : {d} bytes\n", .{fb.pitch});
|
||||
serial0.debugPrint(" format : {s}\n", .{@tagName(fb.format)});
|
||||
serial0.debugPrint(" framebuffer: 0x{x:0>16}\n", .{fb.base});
|
||||
serial0.debugPrint (" footprint : {d} MiB\n", .{(fb.pitch * fb.height) / (1024 * 1024)});
|
||||
|
||||
// Summarise the physical memory the loader handed us. The array is danos's
|
||||
// own MemoryRegion, so this is a plain slice — no firmware layout in sight.
|
||||
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(boot_info.memory_map.regions))[0..boot_info.memory_map.len];
|
||||
var usable_pages: u64 = 0;
|
||||
var reserved_pages: u64 = 0; // reserved RAM only — MMIO is device space, not RAM
|
||||
for (regions) |r| {
|
||||
switch (r.kind) {
|
||||
.usable => usable_pages += r.pages,
|
||||
.reserved, .acpi_tables, .acpi_nvs => reserved_pages += r.pages,
|
||||
.mmio => {},
|
||||
}
|
||||
}
|
||||
const total_pages = usable_pages + reserved_pages;
|
||||
const total_bytes = total_pages * danos.page_size;
|
||||
const gib = 1 << 30;
|
||||
|
||||
serial0.debugWrite("\ndanos: physical memory\n");
|
||||
serial0.debugPrint(" total RAM : {d}.{d:0>2} GiB ({d} MiB) - RAM the firmware reported\n", .{ total_bytes / gib, (total_bytes % gib) * 100 / gib, mib(total_pages) });
|
||||
serial0.debugPrint(" usable : {d} MiB - free RAM (incl. reclaimed boot-services memory)\n", .{mib(usable_pages)});
|
||||
serial0.debugPrint(" reserved : {d} MiB - kernel image, boot stack, ACPI, runtime services\n", .{mib(reserved_pages)});
|
||||
serial0.debugPrint(" regions : {d} - entries in the firmware memory map\n", .{regions.len});
|
||||
|
||||
// Bring up the physical frame allocator over that map, and prove it works:
|
||||
// allocate three frames, then hand them back.
|
||||
pmm.init(boot_info.memory_map);
|
||||
const s1 = pmm.stats();
|
||||
serial0.debugPrint("\ndanos: frame allocator online\n", .{});
|
||||
serial0.debugPrint(" free frames: {d} ({d} MiB)\n", .{ s1.free_frames, mib(s1.free_frames) });
|
||||
const f0 = pmm.alloc();
|
||||
const f1 = pmm.alloc();
|
||||
const f2 = pmm.alloc();
|
||||
serial0.debugPrint(" alloc x3 : 0x{x} 0x{x} 0x{x}\n", .{ f0 orelse 0, f1 orelse 0, f2 orelse 0 });
|
||||
if (f0) |p| pmm.free(p);
|
||||
if (f1) |p| pmm.free(p);
|
||||
if (f2) |p| pmm.free(p);
|
||||
serial0.debugPrint(" after free : {d} frames free\n", .{pmm.stats().free_frames});
|
||||
|
||||
// Switch off the firmware's page tables onto our own (with real permissions).
|
||||
arch.enablePaging(pmm.alloc, boot_info);
|
||||
serial0.debugPrint("\ndanos: paging enabled\n", .{});
|
||||
serial0.debugPrint(" page tables: CR3 = 0x{x:0>16}\n", .{arch.readCr3()});
|
||||
serial0.debugPrint(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_info.kernel_segment_count});
|
||||
|
||||
// Bring up the kernel heap (dynamic allocation), built on the VMM.
|
||||
heap.init();
|
||||
serial0.debugWrite("\ndanos: kernel heap online\n");
|
||||
// Measure the amount of resources the kernel is actually using
|
||||
const s2 = pmm.stats();
|
||||
serial0.debugPrint(" Kernel footprint: {d} KiB\n", .{kib(s1.free_frames - s2.free_frames)});
|
||||
|
||||
// Enumerate hardware from the firmware tables (ACPI here) into a generic
|
||||
// device tree, then list it. Discovery walks ACPI memory directly (identity-
|
||||
// mapped) and maps PCIe config space on demand via the VMM. A failure here is
|
||||
// not fatal yet — log it and carry on.
|
||||
const hal = platform.Hal{
|
||||
.mapMmio = arch.mapPage,
|
||||
.pioRead = arch.pioRead,
|
||||
.pioWrite = arch.pioWrite,
|
||||
};
|
||||
if (platform.discover(boot_info, heap.allocator(), hal)) |devtree| {
|
||||
var dt = devtree;
|
||||
serial0.debugWrite("\ndanos: device discovery online\n");
|
||||
dt.dump(console.SerialConsole.debugWrite);
|
||||
|
||||
// Power register map extracted from the FADT + AML, for confidence it parsed.
|
||||
const pw = platform.powerInfo();
|
||||
serial0.debugWrite("danos: power\n");
|
||||
serial0.debugPrint(" pm1a_cnt : {s} 0x{x} (width {d})\n", .{ if (pw.pm1a_cnt.mmio) "mmio" else "io", pw.pm1a_cnt.address, pw.pm1a_cnt.width });
|
||||
if (pw.s5) |s| {
|
||||
serial0.debugPrint(" S5 slp_typ : a={d} b={d}\n", .{ s.slp_typ_a, s.slp_typ_b });
|
||||
} else {
|
||||
serial0.debugWrite(" S5 slp_typ : (not found)\n");
|
||||
}
|
||||
serial0.debugPrint(" reset : supported={} {s} 0x{x} val 0x{x}\n", .{ pw.reset_supported, if (pw.reset.mmio) "mmio" else "io", pw.reset.address, pw.reset_value });
|
||||
|
||||
// AML namespace parse integrity: consumed should equal total.
|
||||
const am = platform.amlStats();
|
||||
serial0.debugPrint(" aml : {d} namespace nodes, parsed {d}/{d} bytes\n", .{ am.nodes, am.consumed, am.total });
|
||||
|
||||
// Feed the arch layer the discovered addresses/facts so it makes no legacy
|
||||
// assumptions — the point of all this on UEFI Class 3 firmware. MMIO bases
|
||||
// (HPET, I/O APIC) come from the device tree; scalar facts from ACPI.
|
||||
const pinfo = platform.platformInfo();
|
||||
const hpet_base: u64 = if (dt.firstOfClass(.timer)) |t|
|
||||
(if (t.firstResource(.memory)) |r| r.start else 0)
|
||||
else
|
||||
0;
|
||||
var ioapic_base: u64 = 0;
|
||||
var ioapic_gsi: u32 = 0;
|
||||
if (dt.firstOfClass(.interrupt_controller)) |ic| {
|
||||
if (ic.firstResource(.memory)) |r| ioapic_base = r.start;
|
||||
if (ic.firstResource(.irq)) |r| ioapic_gsi = @intCast(r.start);
|
||||
}
|
||||
var isos: [16]arch.IsoEntry = undefined;
|
||||
const iso_n = @min(pinfo.override_count, isos.len);
|
||||
for (0..iso_n) |i| isos[i] = .{
|
||||
.source = pinfo.overrides[i].source,
|
||||
.gsi = pinfo.overrides[i].gsi,
|
||||
.flags = pinfo.overrides[i].flags,
|
||||
};
|
||||
const pm_timer: ?arch.PmTimer = if (pinfo.pm_timer.present())
|
||||
.{ .mmio = pinfo.pm_timer.mmio, .address = pinfo.pm_timer.address, .is_32bit = pinfo.pm_timer_32bit }
|
||||
else
|
||||
null;
|
||||
arch.configurePlatform(.{
|
||||
.pic_present = pinfo.pic_present,
|
||||
.hpet_base = hpet_base,
|
||||
.pm_timer = pm_timer,
|
||||
.ioapic_base = ioapic_base,
|
||||
.ioapic_gsi_base = ioapic_gsi,
|
||||
.overrides = isos[0..iso_n],
|
||||
});
|
||||
if (pinfo.spcr_uart) |u| arch.serialReconfigure(u.mmio, u.address);
|
||||
|
||||
serial0.debugWrite("danos: platform\n");
|
||||
serial0.debugPrint(" 8259 PIC : {s}\n", .{if (pinfo.pic_present) "present" else "absent"});
|
||||
serial0.debugPrint(" lapic base : 0x{x}\n", .{pinfo.lapic_base});
|
||||
serial0.debugPrint(" hpet base : 0x{x}\n", .{hpet_base});
|
||||
serial0.debugPrint(" pm timer : {s} 0x{x} ({s})\n", .{ if (pinfo.pm_timer.mmio) "mmio" else "io", pinfo.pm_timer.address, if (pinfo.pm_timer_32bit) "32-bit" else "24-bit" });
|
||||
if (pinfo.spcr_uart) |u| {
|
||||
serial0.debugPrint(" console UART: {s} 0x{x} (SPCR type {d})\n", .{ if (u.mmio) "mmio" else "io", u.address, pinfo.spcr_kind });
|
||||
} else {
|
||||
serial0.debugWrite(" console UART: none in SPCR -> legacy COM1\n");
|
||||
}
|
||||
serial0.debugPrint(" ioapic : base 0x{x}, {d} inputs (masked); entry0 low 0x{x}\n", .{ ioapic_base, arch.ioapicEntryCount(), arch.ioapicEntryLow(0) });
|
||||
} else |err| {
|
||||
serial0.debugPrint("\ndanos: device discovery failed: {s}\n", .{@errorName(err)});
|
||||
}
|
||||
|
||||
// Register the current context as the first task before enabling preemption.
|
||||
scheduler.init(4);
|
||||
serial0.debugWrite("\ndanos: scheduler online\n");
|
||||
|
||||
// Start the timer and unmask interrupts — the kernel now has a heartbeat, and
|
||||
// the timer preempts among tasks.
|
||||
arch.startTimer();
|
||||
arch.enableInterrupts();
|
||||
serial0.debugPrint("danos: timer online ({d} Hz tick; LAPIC {d} MHz, TSC {d} MHz; calibrated via {s})\n", .{ arch.timer_hz, arch.lapicHz() / 1_000_000, arch.tscHz() / 1_000_000, arch.timerCalibrationSource() });
|
||||
|
||||
// In a test build (`zig build -Dtest-case=<name>`), run that case and stop.
|
||||
// Normal builds fall through to the idle halt.
|
||||
if (build_options.test_case) |case| {
|
||||
tests.run(case, boot_info);
|
||||
arch.halt();
|
||||
}
|
||||
|
||||
con.write("kernel initialised.\n");
|
||||
|
||||
// TODO: init process
|
||||
|
||||
con.write("\nnothing left to do; halting CPU.\n");
|
||||
|
||||
arch.halt();
|
||||
}
|
||||
|
||||
/// Frames (4 KiB pages) to whole MiB.
|
||||
fn mib(pages: u64) u64 {
|
||||
return pages * danos.page_size / (1024 * 1024);
|
||||
}
|
||||
|
||||
fn kib(frames: u64) u64 {
|
||||
return frames * danos.page_size / (1024);
|
||||
}
|
||||
|
||||
/// Report a CPU exception in red and halt. There's no fault recovery yet, so any
|
||||
/// exception is terminal — but now it debugPrints what and where instead of silently
|
||||
/// resetting the machine.
|
||||
fn onException(state: *const arch.CpuState) noreturn {
|
||||
if (con_ready) {
|
||||
con.fg = 0x00ff_5555;
|
||||
con.print("\nCPU EXCEPTION: {s} (vector {d})\n", .{ arch.vectorName(state.vector), state.vector });
|
||||
con.print(" error code : 0x{x}\n", .{state.error_code});
|
||||
con.print(" RIP : 0x{x:0>16}\n", .{state.rip});
|
||||
con.print(" RSP : 0x{x:0>16}\n", .{state.rsp});
|
||||
if (state.vector == 14) con.print(" CR2 (addr) : 0x{x:0>16}\n", .{arch.readCr2()});
|
||||
}
|
||||
arch.halt();
|
||||
}
|
||||
|
||||
/// Freestanding has no OS to receive a panic. debugPrint it to the console (if it is
|
||||
/// up yet) in red, then halt.
|
||||
pub const panic = std.debug.FullPanic(struct {
|
||||
fn panic(msg: []const u8, first_trace_addr: ?usize) noreturn {
|
||||
_ = first_trace_addr;
|
||||
if (con_ready) {
|
||||
con.fg = 0x00ff_5555;
|
||||
con.write("\nKERNEL PANIC: ");
|
||||
con.write(msg);
|
||||
con.write("\n");
|
||||
}
|
||||
arch.halt();
|
||||
}
|
||||
}.panic);
|
||||
@@ -1,251 +0,0 @@
|
||||
//! The scheduler: fixed-priority preemptive multitasking.
|
||||
//!
|
||||
//! Tasks are kernel threads (ring 0, each with its own stack). The **highest-
|
||||
//! priority ready task always runs**; within a priority level, tasks round-robin.
|
||||
//! Selection is O(1) — a bitmap of non-empty priority levels plus a FIFO queue per
|
||||
//! level — which keeps scheduling deterministic, as a real-time kernel needs (see
|
||||
//! docs/vision.md).
|
||||
//!
|
||||
//! Switching happens both cooperatively (`yield`) and preemptively (the timer
|
||||
//! calls `tick`). See docs/scheduling.md for the interrupt-flag discipline that
|
||||
//! makes those two paths coexist.
|
||||
|
||||
const std = @import("std");
|
||||
const arch = @import("arch");
|
||||
const heap = @import("heap.zig");
|
||||
|
||||
/// Priority level: 0 (lowest) .. 7 (highest). 8 levels total.
|
||||
pub const Priority = u3;
|
||||
const num_priorities = 8;
|
||||
|
||||
const stack_size = 16 * 1024; // each task's kernel stack is 16 KiB
|
||||
const max_tasks = 16; // the maximum number of tasks alive at once is 16 in a static sized pool
|
||||
|
||||
const State = enum { free, ready, running, blocked };
|
||||
|
||||
const Task = struct {
|
||||
id: u32 = 0,
|
||||
state: State = .free,
|
||||
priority: Priority = 0,
|
||||
rsp: usize = 0, // saved stack pointer, valid while not running
|
||||
stack: []u8 = &.{},
|
||||
wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping
|
||||
next: ?*Task = null, // ready-queue link
|
||||
};
|
||||
|
||||
var tasks = [_]Task{.{}} ** max_tasks;
|
||||
var current: *Task = undefined;
|
||||
var next_id: u32 = 1;
|
||||
|
||||
// Per-priority FIFO ready queues, and a bitmap of which levels are non-empty.
|
||||
var ready_head: [num_priorities]?*Task = .{null} ** num_priorities;
|
||||
var ready_tail: [num_priorities]?*Task = .{null} ** num_priorities;
|
||||
var ready_bitmap: u8 = 0;
|
||||
|
||||
var preemption_enabled = true;
|
||||
|
||||
/// Register the currently-running kernel context as the first task, spawn the
|
||||
/// idle task, and hook the timer for preemption.
|
||||
pub fn init(boot_priority: Priority) void {
|
||||
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
|
||||
current = &tasks[0];
|
||||
spawn(idle, 0); // lowest priority, always runnable — runs when nothing else is
|
||||
arch.setTickHook(tick);
|
||||
}
|
||||
|
||||
/// The idle task: run when every other task is blocked or sleeping. `hlt` waits
|
||||
/// for the next interrupt at near-zero power (see docs/halting.md).
|
||||
fn idle() void {
|
||||
while (true) asm volatile ("hlt");
|
||||
}
|
||||
|
||||
fn enqueue(t: *Task) void {
|
||||
t.next = null;
|
||||
const p: usize = t.priority;
|
||||
if (ready_tail[p]) |tail| tail.next = t else ready_head[p] = t;
|
||||
ready_tail[p] = t;
|
||||
ready_bitmap |= levelBit(t.priority);
|
||||
}
|
||||
|
||||
fn dequeueHighest() ?*Task {
|
||||
if (ready_bitmap == 0) return null;
|
||||
const level: Priority = @intCast(num_priorities - 1 - @clz(ready_bitmap));
|
||||
const t = ready_head[level].?;
|
||||
ready_head[level] = t.next;
|
||||
if (ready_head[level] == null) {
|
||||
ready_tail[level] = null;
|
||||
ready_bitmap &= ~levelBit(level);
|
||||
}
|
||||
t.next = null;
|
||||
return t;
|
||||
}
|
||||
|
||||
fn levelBit(p: Priority) u8 {
|
||||
return @as(u8, 1) << p;
|
||||
}
|
||||
|
||||
/// Create a task that runs `entry` at `priority`. It becomes ready immediately.
|
||||
pub fn spawn(entry: *const fn () void, priority: Priority) void {
|
||||
const t = freeSlot() orelse @panic("sched: task table full");
|
||||
const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack");
|
||||
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack };
|
||||
next_id += 1;
|
||||
const top = @intFromPtr(stack.ptr) + stack.len;
|
||||
t.rsp = arch.initTaskStack(top, @intFromPtr(entry));
|
||||
enqueue(t);
|
||||
}
|
||||
|
||||
fn freeSlot() ?*Task {
|
||||
for (&tasks) |*t| {
|
||||
if (t.state == .free) return t;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Pick the highest-priority ready task and switch to it. Interrupts must be
|
||||
/// disabled by the caller.
|
||||
fn schedule() void {
|
||||
const prev = current;
|
||||
if (prev.state == .running) {
|
||||
prev.state = .ready;
|
||||
enqueue(prev); // back of its level's queue (round-robin)
|
||||
}
|
||||
const next = dequeueHighest() orelse {
|
||||
prev.state = .running; // nothing else ready — keep running
|
||||
return;
|
||||
};
|
||||
next.state = .running;
|
||||
current = next;
|
||||
if (next != prev) arch.switchContext(&prev.rsp, next.rsp);
|
||||
}
|
||||
|
||||
/// Voluntarily give up the CPU to the next ready task.
|
||||
pub fn yield() void {
|
||||
const flags = arch.saveInterrupts();
|
||||
schedule();
|
||||
arch.restoreInterrupts(flags);
|
||||
}
|
||||
|
||||
/// Block the current task for `ms` milliseconds, then let it become runnable
|
||||
/// again. The idle task (or other work) runs in the meantime.
|
||||
pub fn sleep(ms: u64) void {
|
||||
const flags = arch.saveInterrupts();
|
||||
current.wake_at = arch.millis() + ms;
|
||||
current.state = .blocked;
|
||||
schedule(); // current is blocked, so schedule() won't re-enqueue it
|
||||
arch.restoreInterrupts(flags);
|
||||
}
|
||||
|
||||
// --- event-based blocking -------------------------------------------------
|
||||
//
|
||||
// A WaitQueue is a set of tasks blocked waiting for something (a resource, a
|
||||
// message). Tasks link into it through the same `next` field the ready queues
|
||||
// use — a task is in exactly one queue at a time. These are the primitive locks,
|
||||
// semaphores and IPC channels are built on.
|
||||
|
||||
pub const WaitQueue = struct {
|
||||
head: ?*Task = null,
|
||||
};
|
||||
|
||||
/// Block the current task on `wq` and switch away. Precondition: interrupts are
|
||||
/// disabled (the caller holds them, so a condition can be checked and the block
|
||||
/// committed atomically). On return — when woken — interrupts are still disabled.
|
||||
pub fn waitLocked(wq: *WaitQueue) void {
|
||||
current.state = .blocked;
|
||||
current.next = wq.head;
|
||||
wq.head = current;
|
||||
schedule();
|
||||
}
|
||||
|
||||
/// Move the highest-priority waiter on `wq` (if any) to the ready queue.
|
||||
/// Precondition: interrupts disabled. Does not preempt — the caller decides.
|
||||
pub fn wakeLocked(wq: *WaitQueue) void {
|
||||
// Find the highest-priority waiter (bounded scan) and unlink it.
|
||||
var best_prev: ?*Task = null;
|
||||
var best: ?*Task = null;
|
||||
var prev: ?*Task = null;
|
||||
var cur = wq.head;
|
||||
while (cur) |t| : ({
|
||||
prev = t;
|
||||
cur = t.next;
|
||||
}) {
|
||||
if (best == null or t.priority > best.?.priority) {
|
||||
best = t;
|
||||
best_prev = prev;
|
||||
}
|
||||
}
|
||||
const t = best orelse return;
|
||||
if (best_prev) |p| p.next = t.next else wq.head = t.next;
|
||||
t.state = .ready;
|
||||
enqueue(t);
|
||||
}
|
||||
|
||||
/// Block on `wq` (a self-contained critical section).
|
||||
pub fn wait(wq: *WaitQueue) void {
|
||||
const flags = arch.saveInterrupts();
|
||||
waitLocked(wq);
|
||||
arch.restoreInterrupts(flags);
|
||||
}
|
||||
|
||||
/// Wake the highest-priority waiter on `wq`, preempting if it outranks us.
|
||||
pub fn wake(wq: *WaitQueue) void {
|
||||
const flags = arch.saveInterrupts();
|
||||
wakeLocked(wq);
|
||||
// If a higher-priority task is now ready, run it immediately.
|
||||
if (highestReadyPriority()) |p| {
|
||||
if (p > current.priority) schedule();
|
||||
}
|
||||
arch.restoreInterrupts(flags);
|
||||
}
|
||||
|
||||
fn highestReadyPriority() ?Priority {
|
||||
if (ready_bitmap == 0) return null;
|
||||
return @intCast(num_priorities - 1 - @clz(ready_bitmap));
|
||||
}
|
||||
|
||||
/// Wake any sleeping task whose deadline has passed. Bounded by the task count,
|
||||
/// so it stays deterministic. Called from the timer tick (interrupts disabled).
|
||||
fn wakeExpired() void {
|
||||
const now = arch.millis();
|
||||
for (&tasks) |*t| {
|
||||
if (t.state == .blocked and t.wake_at != 0 and now >= t.wake_at) {
|
||||
t.wake_at = 0;
|
||||
t.state = .ready;
|
||||
enqueue(t);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Called from the timer interrupt (interrupts already disabled): wake due
|
||||
/// sleepers, then preempt.
|
||||
pub fn tick() void {
|
||||
wakeExpired();
|
||||
if (preemption_enabled) schedule();
|
||||
}
|
||||
|
||||
/// Enable or disable timer-driven preemption (cooperative-only when off).
|
||||
pub fn setPreemption(enabled: bool) void {
|
||||
preemption_enabled = enabled;
|
||||
}
|
||||
|
||||
/// End the current task and switch away for good; never returns. The task's stack
|
||||
/// is leaked for now (no reaper yet).
|
||||
pub fn exit() noreturn {
|
||||
arch.disableInterrupts();
|
||||
current.state = .free;
|
||||
const next = dequeueHighest() orelse @panic("sched: no task left to run");
|
||||
next.state = .running;
|
||||
current = next;
|
||||
var discard: usize = 0;
|
||||
arch.switchContext(&discard, next.rsp);
|
||||
unreachable;
|
||||
}
|
||||
|
||||
pub fn currentId() u32 {
|
||||
return current.id;
|
||||
}
|
||||
|
||||
/// Change the running task's priority (takes effect next time it's enqueued).
|
||||
pub fn setPriority(p: Priority) void {
|
||||
current.priority = p;
|
||||
}
|
||||
@@ -1,491 +0,0 @@
|
||||
//! In-kernel test cases, run at the end of bring-up when the kernel is built with
|
||||
//! `-Dtest-case=<name>`. Each case writes structured markers to the serial port
|
||||
//! that the QEMU harness (test/qemu_test.py) asserts on:
|
||||
//!
|
||||
//! [PASS]/[FAIL] <check> per assertion
|
||||
//! DANOS-TEST-RESULT: PASS|FAIL overall, for non-faulting cases
|
||||
//!
|
||||
//! Faulting cases (fault-ud, fault-pf, fault-df) deliberately don't return a
|
||||
//! result line — they trigger a CPU exception, and the harness asserts on the
|
||||
//! exception report the handler prints (which also reaches serial).
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const arch = @import("arch");
|
||||
const platform = @import("platform");
|
||||
const pmm = @import("pmm.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const sched = @import("scheduler.zig");
|
||||
const ipc = @import("ipc.zig");
|
||||
|
||||
/// Formatted write straight to serial, independent of the framebuffer console.
|
||||
fn log(comptime fmt: []const u8, args: anytype) void {
|
||||
var buf: [128]u8 = undefined;
|
||||
arch.serialWrite(std.fmt.bufPrint(&buf, fmt, args) catch return);
|
||||
}
|
||||
|
||||
var passed: u32 = 0;
|
||||
var failed: u32 = 0;
|
||||
|
||||
fn check(name: []const u8, ok: bool) void {
|
||||
if (ok) {
|
||||
passed += 1;
|
||||
log("[PASS] {s}\n", .{name});
|
||||
} else {
|
||||
failed += 1;
|
||||
log("[FAIL] {s}\n", .{name});
|
||||
}
|
||||
}
|
||||
|
||||
/// Emit the overall result line the harness matches, then the done sentinel.
|
||||
fn result() void {
|
||||
log("DANOS-TEST-RESULT: {s} ({d} passed, {d} failed)\n", .{
|
||||
if (failed == 0) "PASS" else "FAIL",
|
||||
passed,
|
||||
failed,
|
||||
});
|
||||
log("DANOS-TEST-DONE\n", .{});
|
||||
}
|
||||
|
||||
pub fn run(case: []const u8, boot_info: *const BootInfo) void {
|
||||
if (eql(case, "smoke")) {
|
||||
smoke(boot_info);
|
||||
} else if (eql(case, "timer")) {
|
||||
timer();
|
||||
} else if (eql(case, "clock")) {
|
||||
clock();
|
||||
} else if (eql(case, "vmm")) {
|
||||
vmm();
|
||||
} else if (eql(case, "heap")) {
|
||||
heapTest();
|
||||
} else if (eql(case, "sched")) {
|
||||
schedTest();
|
||||
} else if (eql(case, "priority")) {
|
||||
priorityTest();
|
||||
} else if (eql(case, "sleep")) {
|
||||
sleepTest();
|
||||
} else if (eql(case, "event")) {
|
||||
eventTest();
|
||||
} else if (eql(case, "ipc")) {
|
||||
ipcTest();
|
||||
} else if (eql(case, "fault-ud")) {
|
||||
faultInvalidOpcode();
|
||||
} else if (eql(case, "fault-pf")) {
|
||||
faultPageFault();
|
||||
} else if (eql(case, "fault-df")) {
|
||||
faultDoubleFault();
|
||||
} else if (eql(case, "fault-nx")) {
|
||||
faultNoExecute();
|
||||
} else if (eql(case, "fault-null")) {
|
||||
faultNull();
|
||||
} else if (eql(case, "poweroff")) {
|
||||
powerTest(.off);
|
||||
} else if (eql(case, "reboot")) {
|
||||
powerTest(.reboot);
|
||||
} else {
|
||||
log("DANOS-TEST-RESULT: FAIL (unknown case '{s}')\n", .{case});
|
||||
}
|
||||
}
|
||||
|
||||
fn platformHal() platform.Hal {
|
||||
return .{
|
||||
.mapMmio = arch.mapPage,
|
||||
.pioRead = arch.pioRead,
|
||||
.pioWrite = arch.pioWrite,
|
||||
};
|
||||
}
|
||||
|
||||
/// Drive an ACPI power transition. On success the machine powers off or resets,
|
||||
/// so QEMU exits — the harness observes the process exit. If control returns, the
|
||||
/// transition failed and we emit a FAIL result.
|
||||
fn powerTest(comptime action: enum { off, reboot }) void {
|
||||
const name = if (action == .off) "poweroff" else "reboot";
|
||||
log("DANOS-TEST-BEGIN: {s}\n", .{name});
|
||||
const hal = platformHal();
|
||||
log("DANOS-POWER: attempting {s}\n", .{name});
|
||||
switch (action) {
|
||||
.off => platform.shutdown(hal),
|
||||
.reboot => platform.reboot(hal),
|
||||
}
|
||||
check("power transition took effect", false);
|
||||
result();
|
||||
}
|
||||
|
||||
const BootInfo = danos.BootInfo;
|
||||
|
||||
fn eql(a: []const u8, b: []const u8) bool {
|
||||
return std.mem.eql(u8, a, b);
|
||||
}
|
||||
|
||||
/// Non-destructive checks of the memory map and frame allocator.
|
||||
fn smoke(boot_info: *const BootInfo) void {
|
||||
log("DANOS-TEST-BEGIN: smoke\n", .{});
|
||||
|
||||
// The memory map has some usable RAM.
|
||||
const mm = boot_info.memory_map;
|
||||
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
||||
var usable: u64 = 0;
|
||||
for (regions) |r| {
|
||||
if (r.kind == .usable) usable += r.pages;
|
||||
}
|
||||
check("memory map reports usable RAM", usable > 0);
|
||||
|
||||
// The frame allocator hands out distinct, page-aligned frames.
|
||||
const a = pmm.alloc();
|
||||
const b = pmm.alloc();
|
||||
check("alloc returns a frame", a != null);
|
||||
check("alloc returns distinct frames", a != null and b != null and a.? != b.?);
|
||||
check("frames are page-aligned", (a orelse 1) % danos.page_size == 0);
|
||||
|
||||
// Freeing restores the count.
|
||||
const before = pmm.stats().free_frames;
|
||||
if (a) |p| pmm.free(p);
|
||||
if (b) |p| pmm.free(p);
|
||||
check("free returns frames to the pool", pmm.stats().free_frames == before + 2);
|
||||
|
||||
// Paging is active on our own tables (CR3 is non-zero and page-aligned).
|
||||
const cr3 = arch.readCr3();
|
||||
check("paging active (CR3 set)", cr3 != 0 and cr3 % danos.page_size == 0);
|
||||
|
||||
result();
|
||||
}
|
||||
|
||||
/// Verify device interrupts fire and return: the timer tick counter must advance
|
||||
/// on its own. Interrupts are already enabled by kmain before tests run.
|
||||
fn timer() void {
|
||||
log("DANOS-TEST-BEGIN: timer\n", .{});
|
||||
const start = arch.ticks();
|
||||
// Busy-wait for the counter to advance. arch.ticks() is a volatile load, so
|
||||
// the compiler re-reads it each iteration and sees the interrupt's update.
|
||||
// The cap is only a safety net; the harness timeout is the real backstop.
|
||||
var spins: u64 = 0;
|
||||
while (arch.ticks() == start and spins < 5_000_000_000) spins +%= 1;
|
||||
check("timer interrupts advance the tick count", arch.ticks() > start);
|
||||
|
||||
result();
|
||||
}
|
||||
|
||||
/// Verify the on-demand VMM: map a fresh frame at an unused virtual address, and
|
||||
/// check it's writable and reads back.
|
||||
fn vmm() void {
|
||||
log("DANOS-TEST-BEGIN: vmm\n", .{});
|
||||
const frame = pmm.alloc();
|
||||
check("frame available to map", frame != null);
|
||||
if (frame) |phys| {
|
||||
var virt: u64 = 0x0000_4000_0000_0000; // canonical, well clear of everything mapped
|
||||
arch.mapPage(virt, phys, true);
|
||||
const p: *volatile u64 = @ptrFromInt(virt);
|
||||
p.* = 0xdead_c0de_cafe_babe;
|
||||
check("mapped page is writable and reads back", p.* == 0xdead_c0de_cafe_babe);
|
||||
arch.unmapPage(virt);
|
||||
pmm.free(phys);
|
||||
virt += 0;
|
||||
}
|
||||
result();
|
||||
}
|
||||
|
||||
/// Exercise the kernel heap: basic alloc/write/free, reuse, growth beyond the
|
||||
/// initial region, and a std container backed by it.
|
||||
fn heapTest() void {
|
||||
log("DANOS-TEST-BEGIN: heap\n", .{});
|
||||
const a = heap.allocator();
|
||||
|
||||
// Allocate, write a pattern, read it back, free.
|
||||
const buf = a.alloc(u8, 4096) catch null;
|
||||
check("alloc 4096 bytes", buf != null);
|
||||
if (buf) |b| {
|
||||
@memset(b, 0xAB);
|
||||
check("heap memory is writable and reads back", b[0] == 0xAB and b[4095] == 0xAB);
|
||||
a.free(b);
|
||||
}
|
||||
|
||||
// Freeing then re-allocating the same size should reuse the block.
|
||||
const p1 = a.alloc(u64, 8) catch null;
|
||||
const addr1 = if (p1) |p| @intFromPtr(p.ptr) else 0;
|
||||
if (p1) |p| a.free(p);
|
||||
const p2 = a.alloc(u64, 8) catch null;
|
||||
const addr2 = if (p2) |p| @intFromPtr(p.ptr) else 0;
|
||||
check("freed block is reused", addr1 != 0 and addr1 == addr2);
|
||||
if (p2) |p| a.free(p);
|
||||
|
||||
// Force growth past the initial page and check every block is usable.
|
||||
var blocks: [64]?[]u8 = .{null} ** 64;
|
||||
var ok = true;
|
||||
for (&blocks, 0..) |*slot, i| {
|
||||
const b = a.alloc(u8, 4096) catch null;
|
||||
slot.* = b;
|
||||
if (b) |bb| @memset(bb, @intCast(i & 0xff)) else {
|
||||
ok = false;
|
||||
}
|
||||
}
|
||||
for (blocks, 0..) |slot, i| {
|
||||
if (slot) |bb| {
|
||||
if (bb[0] != @as(u8, @intCast(i & 0xff)) or bb[4095] != @as(u8, @intCast(i & 0xff))) ok = false;
|
||||
}
|
||||
}
|
||||
check("many allocations (heap growth) stay valid", ok);
|
||||
for (blocks) |slot| {
|
||||
if (slot) |bb| a.free(bb);
|
||||
}
|
||||
|
||||
// A std container backed by the kernel heap.
|
||||
var list: std.ArrayList(u32) = .empty;
|
||||
var sum: u64 = 0;
|
||||
var expected: u64 = 0;
|
||||
var i: u32 = 0;
|
||||
var list_ok = true;
|
||||
while (i < 1000) : (i += 1) {
|
||||
list.append(a, i) catch {
|
||||
list_ok = false;
|
||||
};
|
||||
expected += i;
|
||||
}
|
||||
for (list.items) |v| sum += v;
|
||||
list.deinit(a);
|
||||
check("std.ArrayList on the kernel heap", list_ok and sum == expected);
|
||||
|
||||
result();
|
||||
}
|
||||
|
||||
/// Verify the calibrated clocks: sane measured frequencies, monotonic uptime that
|
||||
/// advances with real ticks, and — the point of the TSC clock — nanosecond
|
||||
/// resolution far finer than the 1 ms tick, with the unit functions consistent.
|
||||
fn clock() void {
|
||||
log("DANOS-TEST-BEGIN: clock\n", .{});
|
||||
|
||||
const lapic = arch.lapicHz();
|
||||
check("LAPIC frequency measured", lapic > 1_000_000 and lapic < 100_000_000_000);
|
||||
const tsc = arch.tscHz();
|
||||
check("TSC frequency measured", tsc > 100_000_000 and tsc < 100_000_000_000);
|
||||
|
||||
// Uptime advances over ~5 real ticks (1000 Hz => 1 tick == 1 ms).
|
||||
const start_ticks = arch.ticks();
|
||||
const start_ms = arch.millis();
|
||||
var spins: u64 = 0;
|
||||
while (arch.ticks() < start_ticks + 5 and spins < 5_000_000_000) spins +%= 1;
|
||||
const elapsed_ms = arch.millis() - start_ms;
|
||||
check("uptime advances with ticks", elapsed_ms >= 5 and elapsed_ms < 100);
|
||||
|
||||
// Sub-millisecond resolution: spin until nanos() first advances, then confirm
|
||||
// that first step happened within a millisecond — so nanos() resolves finer
|
||||
// than the 1 ms tick (a tick clock's smallest step *is* 1 ms). Spinning to the
|
||||
// first change is robust to QEMU's coarse TSC update granularity.
|
||||
const n1 = arch.nanos();
|
||||
var s2: u64 = 0;
|
||||
while (arch.nanos() == n1 and s2 < 10_000_000) s2 +%= 1;
|
||||
const n2 = arch.nanos();
|
||||
check("nanos() has sub-millisecond resolution", n2 > n1 and (n2 - n1) < 1_000_000);
|
||||
|
||||
// The unit functions agree (within rounding).
|
||||
const ns = arch.nanos();
|
||||
check("nanos/micros/millis are consistent", diffWithin(arch.micros(), ns / 1000, 1000) and diffWithin(arch.millis(), ns / 1_000_000, 2));
|
||||
|
||||
result();
|
||||
}
|
||||
|
||||
fn diffWithin(a: u64, b: u64, tol: u64) bool {
|
||||
return if (a > b) a - b <= tol else b - a <= tol;
|
||||
}
|
||||
|
||||
// --- scheduler tests ------------------------------------------------------
|
||||
|
||||
var counters = [_]u64{0} ** 3;
|
||||
|
||||
fn spin0() void {
|
||||
const p: *volatile u64 = &counters[0];
|
||||
while (true) p.* = p.* +% 1;
|
||||
}
|
||||
fn spin1() void {
|
||||
const p: *volatile u64 = &counters[1];
|
||||
while (true) p.* = p.* +% 1;
|
||||
}
|
||||
fn spin2() void {
|
||||
const p: *volatile u64 = &counters[2];
|
||||
while (true) p.* = p.* +% 1;
|
||||
}
|
||||
|
||||
/// Preemption: spawn three tasks that busy-loop *without* yielding. If they all
|
||||
/// make progress, the timer must be preempting between them (and the context
|
||||
/// switch works) — because nothing yields voluntarily.
|
||||
fn schedTest() void {
|
||||
log("DANOS-TEST-BEGIN: sched\n", .{});
|
||||
counters = .{ 0, 0, 0 };
|
||||
sched.spawn(spin0, 4);
|
||||
sched.spawn(spin1, 4);
|
||||
sched.spawn(spin2, 4);
|
||||
|
||||
const c0: *volatile u64 = &counters[0];
|
||||
const c1: *volatile u64 = &counters[1];
|
||||
const c2: *volatile u64 = &counters[2];
|
||||
var spins: u64 = 0;
|
||||
while ((c0.* == 0 or c1.* == 0 or c2.* == 0) and spins < 5_000_000_000) spins +%= 1;
|
||||
|
||||
check("all three non-yielding tasks made progress (preemption)", c0.* > 0 and c1.* > 0 and c2.* > 0);
|
||||
result();
|
||||
}
|
||||
|
||||
var run_order = [_]u8{0} ** 4;
|
||||
var run_n: usize = 0;
|
||||
|
||||
fn recordExit(priority: u8) void {
|
||||
run_order[run_n] = priority;
|
||||
run_n += 1;
|
||||
sched.exit();
|
||||
}
|
||||
fn taskHigh() void {
|
||||
recordExit(6);
|
||||
}
|
||||
fn taskMid() void {
|
||||
recordExit(4);
|
||||
}
|
||||
fn taskLow() void {
|
||||
recordExit(2);
|
||||
}
|
||||
|
||||
/// Fixed priority: with preemption off (deterministic), spawn tasks at three
|
||||
/// priorities and let them run cooperatively. They must run highest-first.
|
||||
fn priorityTest() void {
|
||||
log("DANOS-TEST-BEGIN: priority\n", .{});
|
||||
sched.setPreemption(false);
|
||||
sched.setPriority(1); // above the idle task (0), below the workers — runs last
|
||||
run_n = 0;
|
||||
|
||||
sched.spawn(taskLow, 2);
|
||||
sched.spawn(taskMid, 4);
|
||||
sched.spawn(taskHigh, 6);
|
||||
|
||||
while (run_n < 3) sched.yield(); // regain control only once the workers are done
|
||||
|
||||
check("tasks ran highest-priority first", run_order[0] == 6 and run_order[1] == 4 and run_order[2] == 2);
|
||||
|
||||
sched.setPriority(4);
|
||||
sched.setPreemption(true);
|
||||
result();
|
||||
}
|
||||
|
||||
var event_wq: sched.WaitQueue = .{};
|
||||
var event_stage: u32 = 0;
|
||||
|
||||
fn eventWaiter() void {
|
||||
event_stage = 1; // reached the wait
|
||||
sched.wait(&event_wq); // block until woken
|
||||
event_stage = 3; // woken and resumed
|
||||
sched.exit();
|
||||
}
|
||||
|
||||
/// Event-based blocking: a task blocks on a wait queue and is woken. The waiter is
|
||||
/// higher priority, so waking it preempts us and it runs to completion at once.
|
||||
fn eventTest() void {
|
||||
log("DANOS-TEST-BEGIN: event\n", .{});
|
||||
event_stage = 0;
|
||||
sched.spawn(eventWaiter, 6); // higher priority than this task (4)
|
||||
|
||||
var spins: u64 = 0;
|
||||
while (event_stage != 1 and spins < 1_000_000_000) : (spins += 1) sched.yield();
|
||||
check("waiter reached the wait and blocked", event_stage == 1);
|
||||
|
||||
sched.wake(&event_wq);
|
||||
check("wake resumed the blocked waiter (preempting)", event_stage == 3);
|
||||
result();
|
||||
}
|
||||
|
||||
var channel: ipc.Channel(u64, 4) = .{};
|
||||
var recv_sum: u64 = 0;
|
||||
var recv_count: u64 = 0;
|
||||
|
||||
fn producer() void {
|
||||
var i: u64 = 1;
|
||||
while (i <= 100) : (i += 1) channel.send(i);
|
||||
sched.exit();
|
||||
}
|
||||
fn consumer() void {
|
||||
var n: u64 = 0;
|
||||
while (n < 100) : (n += 1) {
|
||||
recv_sum += channel.recv();
|
||||
recv_count += 1;
|
||||
}
|
||||
sched.exit();
|
||||
}
|
||||
|
||||
/// IPC: a producer and consumer pass 100 messages through a 4-slot channel. The
|
||||
/// small buffer forces the channel full and empty repeatedly, exercising both the
|
||||
/// blocking-send and blocking-recv paths. The messages must arrive intact.
|
||||
fn ipcTest() void {
|
||||
log("DANOS-TEST-BEGIN: ipc\n", .{});
|
||||
channel = .{};
|
||||
recv_sum = 0;
|
||||
recv_count = 0;
|
||||
sched.spawn(consumer, 5); // above this task (4) so they run and we observe after
|
||||
sched.spawn(producer, 5);
|
||||
|
||||
var spins: u64 = 0;
|
||||
while (recv_count < 100 and spins < 2_000_000_000) : (spins += 1) sched.yield();
|
||||
|
||||
check("all 100 messages received", recv_count == 100);
|
||||
check("messages arrived intact (sum 1..100 == 5050)", recv_sum == 5050);
|
||||
result();
|
||||
}
|
||||
|
||||
/// Blocking: sleep(50) should block this task for about 50 ms (measured on the
|
||||
/// calibrated clock) — not busy-wait — while the idle task runs.
|
||||
fn sleepTest() void {
|
||||
log("DANOS-TEST-BEGIN: sleep\n", .{});
|
||||
const t0 = arch.millis();
|
||||
sched.sleep(50);
|
||||
const elapsed = arch.millis() - t0;
|
||||
check("sleep(50) blocked for ~50 ms", elapsed >= 50 and elapsed <= 70);
|
||||
result();
|
||||
}
|
||||
|
||||
fn faultInvalidOpcode() void {
|
||||
log("DANOS-TEST-BEGIN: fault-ud\n", .{});
|
||||
asm volatile ("ud2");
|
||||
}
|
||||
|
||||
/// Verify NX: fetching an instruction from a data page (mapped no-execute) faults.
|
||||
fn faultNoExecute() void {
|
||||
log("DANOS-TEST-BEGIN: fault-nx\n", .{});
|
||||
var scratch: u64 = 0xC3; // a lone `ret` — harmless if NX somehow let it run
|
||||
const f: *const fn () void = @ptrFromInt(@intFromPtr(&scratch));
|
||||
f(); // instruction fetch from an NX page -> #PF before it executes
|
||||
log("DANOS-TEST-RESULT: FAIL (NX not enforced)\n", .{});
|
||||
}
|
||||
|
||||
/// Verify the null guard: dereferencing address 0 (page 0 left unmapped) faults.
|
||||
fn faultNull() void {
|
||||
log("DANOS-TEST-BEGIN: fault-null\n", .{});
|
||||
// Launder the address through empty asm so the compiler no longer knows it's
|
||||
// 0 (otherwise it folds a null-pointer safety panic instead of doing the real
|
||||
// access). `allowzero` skips the same null check on the cast. The write then
|
||||
// hits the unmapped page 0 and takes a real hardware #PF.
|
||||
var addr: u64 = 0;
|
||||
addr = asm ("" : [ret] "=r" (-> u64) : [in] "0" (addr));
|
||||
const p: *allowzero volatile u64 = @ptrFromInt(addr);
|
||||
p.* = 1;
|
||||
}
|
||||
|
||||
fn faultPageFault() void {
|
||||
log("DANOS-TEST-BEGIN: fault-pf\n", .{});
|
||||
// Runtime address so the backend emits a register store (not a `mov moffs`,
|
||||
// which the self-hosted x86_64 backend can't encode).
|
||||
var addr: u64 = 0xdeadbeef000; // well above all mapped RAM
|
||||
const p: *volatile u64 = @ptrFromInt(addr);
|
||||
p.* = 1;
|
||||
addr += 0;
|
||||
}
|
||||
|
||||
fn faultDoubleFault() void {
|
||||
log("DANOS-TEST-BEGIN: fault-df\n", .{});
|
||||
arch.disableInterrupts(); // so only the ud2 delivery (not a timer tick) triggers the #DF
|
||||
// Point RSP at unmapped memory, then fault: the CPU can't push the fault
|
||||
// frame, which escalates to #DF — survivable only because #DF runs on IST1.
|
||||
var bad_sp: u64 = 0x5000000000;
|
||||
asm volatile (
|
||||
\\mov %[sp], %%rsp
|
||||
\\ud2
|
||||
:
|
||||
: [sp] "r" (bad_sp),
|
||||
: .{ .memory = true }
|
||||
);
|
||||
bad_sp += 0;
|
||||
}
|
||||
-103
@@ -1,103 +0,0 @@
|
||||
//! Shared definitions that form the contract between a bootloader
|
||||
//! (src/boot/, e.g. efi.zig built as BOOTX64.efi) and the kernel (src/kernel/main.zig).
|
||||
//!
|
||||
//! Both binaries import this as the "danos" module, so the handoff layout is
|
||||
//! defined in exactly one place.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// Calling convention for the bootloader→kernel jump. Pinned to SysV so it does
|
||||
/// not depend on each binary's target default: the UEFI bootloader's C
|
||||
/// convention is Microsoft x64 (first arg in RCX), the freestanding kernel's is
|
||||
/// SysV (first arg in RDI). Both reference this to agree on where `*BootInfo`
|
||||
/// is passed.
|
||||
pub const kernel_abi: std.builtin.CallingConvention = .{ .x86_64_sysv = .{} };
|
||||
|
||||
/// Pixel byte order of the linear framebuffer the firmware handed us.
|
||||
pub const PixelFormat = enum(u32) {
|
||||
/// Byte 0 = Red, 1 = Green, 2 = Blue, 3 = reserved.
|
||||
rgbx,
|
||||
/// Byte 0 = Blue, 1 = Green, 2 = Red, 3 = reserved.
|
||||
bgrx,
|
||||
};
|
||||
|
||||
/// A linear framebuffer: `width`x`height` pixels, each a 32-bit value, with
|
||||
/// `pitch` bytes between the start of one row and the next (which may be larger
|
||||
/// than `width * 4` due to hardware padding).
|
||||
pub const Framebuffer = extern struct {
|
||||
base: usize, // the memory address where pixel data starts
|
||||
width: u32, // visible pixels per row (e.g. 1920)
|
||||
height: u32, // visible rows (e.g. 1080)
|
||||
pitch: u32, // bytes from the start of one row to the start of the next
|
||||
format: PixelFormat,
|
||||
};
|
||||
|
||||
/// Page size the memory map is measured in. 4 KiB on every architecture danos
|
||||
/// targets so far.
|
||||
pub const page_size = 4096;
|
||||
|
||||
/// danos's own classification of a span of physical memory — deliberately not
|
||||
/// UEFI's vocabulary. Each boot path (UEFI now, device tree later) translates its
|
||||
/// native memory description into these kinds, so the kernel never learns what
|
||||
/// booted it. [[arch]] keeps the same discipline for CPU code.
|
||||
pub const MemoryKind = enum(u32) {
|
||||
/// Free RAM the kernel may allocate. Each boot path folds its own transient
|
||||
/// memory into this once it's genuinely free (e.g. the UEFI loader classifies
|
||||
/// boot-services memory as usable after ExitBootServices), so the kernel never
|
||||
/// has to know about boot-protocol-specific "reclaimable" states.
|
||||
usable,
|
||||
/// Firmware, MMIO, the kernel image, our own boot buffers, the boot stack —
|
||||
/// never hand out.
|
||||
reserved,
|
||||
/// ACPI tables: parse, then reclaim.
|
||||
acpi_tables,
|
||||
/// ACPI non-volatile storage: preserve across sleep, do not allocate.
|
||||
acpi_nvs,
|
||||
/// Not backed by RAM: memory-mapped device registers or a reserved
|
||||
/// address-space window (e.g. PCIe config space). Kept distinct from
|
||||
/// `reserved` so RAM accounting doesn't count device address space.
|
||||
mmio,
|
||||
};
|
||||
|
||||
/// One contiguous span of physical memory. Because danos defines this layout
|
||||
/// itself (unlike the UEFI descriptor it's built from), `@sizeOf` is
|
||||
/// authoritative — the kernel walks a plain `[]MemoryRegion`, with none of the
|
||||
/// firmware's variable descriptor-stride to worry about.
|
||||
pub const MemoryRegion = extern struct {
|
||||
base: u64, // physical start address
|
||||
pages: u64, // length in `page_size` units
|
||||
kind: MemoryKind,
|
||||
_pad: u32 = 0,
|
||||
};
|
||||
|
||||
/// The physical memory layout handed to the kernel: a pointer to an array of
|
||||
/// `len` `MemoryRegion`s, in a buffer that outlives the loader.
|
||||
pub const MemoryMap = extern struct {
|
||||
regions: usize, // address of a `[len]MemoryRegion`
|
||||
len: usize,
|
||||
};
|
||||
|
||||
/// One PT_LOAD segment of the kernel image, so the kernel can re-map itself with
|
||||
/// correct permissions (code R+X, rodata R, data R+W+NX). `flags` are raw ELF
|
||||
/// segment flags: PF_X=1, PF_W=2, PF_R=4.
|
||||
pub const KernelSegment = extern struct {
|
||||
virt: u64,
|
||||
pages: u64,
|
||||
flags: u32,
|
||||
_pad: u32 = 0,
|
||||
};
|
||||
|
||||
/// Handoff structure the bootloader fills in and passes to the kernel's
|
||||
/// `_start` in RDI (the first argument under the SysV AMD64 C ABI).
|
||||
pub const BootInfo = extern struct {
|
||||
framebuffer: Framebuffer,
|
||||
memory_map: MemoryMap,
|
||||
/// The kernel's own PT_LOAD segments (it has three: text, rodata, data).
|
||||
kernel_segments: [8]KernelSegment,
|
||||
kernel_segment_count: u32,
|
||||
/// Physical address of the ACPI RSDP the firmware exposed, or 0 if none. The
|
||||
/// kernel's device layer parses the ACPI tables from here to discover hardware.
|
||||
/// A device-tree boot path leaves this 0 and (later) fills a `device_tree_blob`
|
||||
/// field instead, so the kernel discovers devices without knowing what booted it.
|
||||
acpi_rsdp: u64 = 0,
|
||||
};
|
||||
+191
@@ -0,0 +1,191 @@
|
||||
//! The **private kernel ↔ runtime** ABI: the raw system_call contract — the call
|
||||
//! numbers, `mmap` protection flags, the page size those calls work in, and the IPC
|
||||
//! name-registry ids and notification bit. Shared by the kernel dispatcher
|
||||
//! (system/kernel/process.zig) and the user-space runtime library (library/runtime/),
|
||||
//! so the two can never drift.
|
||||
//!
|
||||
//! **Application code does not speak this.** danos programs call the `runtime` library —
|
||||
//! the stable, danos-native ABI — and the runtime is the one thing that issues the
|
||||
//! actual system calls (POSIX code layers over the runtime, never on this directly). It
|
||||
//! is the same split as libSystem on macOS or win32 over the NT syscalls: the numbers
|
||||
//! here are an implementation detail the runtime hides and may renumber, not a public
|
||||
//! interface. See docs/coding-standards.md and library/runtime/.
|
||||
//!
|
||||
//! This is the *core* contract; the device half — `DeviceDescriptor` and friends, which
|
||||
//! also cross this boundary — lives with the device sub-project as [[device-abi]]
|
||||
//! (system/devices/device-abi.zig). The loader↔kernel handoff is [[boot-handoff]].
|
||||
|
||||
/// Page size every `mmap`/`munmap` grant and the boot memory map are measured in.
|
||||
/// 4 KiB on every architecture danos targets so far. Part of the ABI because the
|
||||
/// runtime aligns to it (grants are page-granular) and the kernel guarantees it.
|
||||
pub const page_size = 4096;
|
||||
|
||||
/// The kernel system_call numbers — the single source of truth shared by the kernel
|
||||
/// dispatcher (system/kernel/process.zig) and the user runtime library, so the two
|
||||
/// can never drift. The set is deliberately microkernel-minimal: file/device I/O
|
||||
/// is not here — it lives in user-space servers reached through the IPC calls.
|
||||
/// The table grows one milestone at a time; see docs/syscall.md.
|
||||
pub const SystemCall = enum(u64) {
|
||||
exit = 0, // exit(code): end the calling process
|
||||
yield = 1, // yield(): give up the rest of this quantum
|
||||
debug_write = 2, // debug_write(ptr, len): raw bytes to the kernel log (bring-up only)
|
||||
sleep = 3, // sleep(ms): block the caller for ms milliseconds
|
||||
mmap = 4, // mmap(len, prot) -> base: grant zeroed, page-aligned user pages
|
||||
munmap = 5, // munmap(base, len): release pages from a prior mmap
|
||||
create_ipc_endpoint = 6, // create_ipc_endpoint() -> handle: a new IPC endpoint
|
||||
ipc_register = 7, // ipc_register(service_id, handle): publish an endpoint by well-known id
|
||||
ipc_lookup = 8, // ipc_lookup(service_id) -> handle: find a published endpoint
|
||||
ipc_call = 9, // ipc_call(h, message, len, reply, cap) -> reply_len: send + block for reply
|
||||
ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, receive, cap) -> receive_len (+badge in rdx)
|
||||
device_enumerate = 11, // device_enumerate(buffer, maximum) -> count: snapshot the device table
|
||||
device_claim = 12, // device_claim(id) -> ok: take exclusive ownership of a device
|
||||
mmio_map = 13, // mmio_map(id, resource_index) -> vaddr: map a claimed device's MMIO into this AS
|
||||
irq_bind = 14, // irq_bind(id, resource_index, endpoint): deliver a device IRQ as an IPC notification
|
||||
irq_ack = 15, // irq_ack(id, resource_index): re-arm a bound IRQ after servicing it
|
||||
device_register = 16, // device_register(parent_id, descriptor) -> id: publish a child of a device you claimed
|
||||
system_spawn = 17, // system_spawn(name_ptr, name_len, arguments_ptr, arguments_len, exit_endpoint) -> child process id: start a named initial-ramdisk binary as a new ring-3 process
|
||||
dma_alloc = 18, // dma_alloc(len, flags) -> vaddr (rax), paddr (rdx): contiguous, pinned, uncacheable DMA memory
|
||||
dma_free = 19, // dma_free(vaddr, len) -> 0: release a prior dma_alloc
|
||||
msi_bind = 20, // msi_bind(device_id, endpoint) -> address (rax), data (rdx): a per-device MSI vector for a claimed device
|
||||
io_read = 21, // io_read(device_id, resource_index, offset, width) -> value: read a port in a claimed device's io_port resource
|
||||
io_write = 22, // io_write(device_id, resource_index, offset, width, value) -> 0: write a port in a claimed device's io_port resource
|
||||
clock = 23, // clock() -> nanoseconds since boot: a monotonic time source (for timeouts/delays)
|
||||
process_enumerate = 24, // process_enumerate(buffer, maximum) -> total: snapshot the task table
|
||||
process_kill = 25, // process_kill(id) -> 0/-errno: end a process this process spawned
|
||||
ipc_send = 26, // ipc_send(handle, message_ptr, message_len) -> 0/-errno: post a payload to an endpoint's async queue without blocking
|
||||
process_exit_reason = 27, // process_exit_reason(id) -> ExitReason/-errno: how a dead child ended (its supervisor only)
|
||||
process_subscribe = 28, // process_subscribe(endpoint) -> 0/-errno: subscribe to published exit events — every death posts a notification
|
||||
signal_bind = 29, // signal_bind(endpoint) -> 0/-errno: nominate the endpoint this process's signals arrive on
|
||||
process_signal = 30, // process_signal(id, signal) -> 0/-errno: post a signal to a child (or to yourself)
|
||||
timer_bind = 31, // timer_bind(endpoint, ms) -> 0/-errno: one-shot timer — posts a notification when ms elapse
|
||||
_,
|
||||
};
|
||||
|
||||
/// How a process ended — recorded by the kernel at death, queried by the
|
||||
/// supervisor with `process_exit_reason`, and the input to its restart decision
|
||||
/// (docs/process-lifecycle.md): a clean exit meant to stop, a fault wants a
|
||||
/// restart with backoff, killed means the supervisor did it itself. The faults
|
||||
/// mirror the CPU exceptions a ring-3 process can die of; they are exit reasons,
|
||||
/// never delivered to the faulting process (recovery is restart, not a handler).
|
||||
pub const ExitReason = enum(u8) {
|
||||
exited = 0, // returned from main / called exit
|
||||
aborted = 1, // deliberate self-termination (reserved: no abort path yet)
|
||||
segmentation_fault = 2, // page fault
|
||||
illegal_instruction = 3, // invalid opcode
|
||||
arithmetic_fault = 4, // divide error, x87 or SIMD fault
|
||||
protection_fault = 5, // general protection fault
|
||||
fault = 6, // any other CPU exception
|
||||
killed = 7, // process_kill
|
||||
};
|
||||
|
||||
/// The x86 MSI message address base (`0xFEE0_0000`): a device raises an MSI by writing
|
||||
/// `data` to this address, which the Local APIC turns into an interrupt at the vector
|
||||
/// in `data`. The kernel returns the concrete (address, data) from `msi_bind`; this is
|
||||
/// the fixed prefix, exposed so a driver's config-space programming reads clearly.
|
||||
pub const msi_address_base: u64 = 0xFEE0_0000;
|
||||
|
||||
/// `dma_alloc` flags. `coherent` (uncacheable) is the portable default; the others are
|
||||
/// opt-in for specific hardware. `write_combining` needs PAT programming (not yet — it
|
||||
/// currently falls back to coherent); see docs/driver-model.md (M14).
|
||||
pub const dma_coherent: u64 = 1; // strong-uncacheable — the default, the only portable one
|
||||
pub const dma_write_combining: u64 = 2; // write-combining (framebuffers); needs PAT
|
||||
pub const dma_below_4g: u64 = 4; // physical address must fit 32 bits (legacy DMA engines)
|
||||
|
||||
/// Set in the badge returned by `ipc_reply_wait` when what arrived is an
|
||||
/// **asynchronous notification** (a device interrupt bound with `irq_bind`, or a
|
||||
/// child-exit notice — see `notify_exit_bit`) rather than a message from a client.
|
||||
/// There is no payload and no reply owed; the low bits carry the source. Shared so
|
||||
/// the kernel's ISR and the driver's event loop can't disagree about which bit
|
||||
/// means "the hardware spoke".
|
||||
pub const notify_badge_bit: u64 = 1 << 63;
|
||||
|
||||
/// Set (alongside `notify_badge_bit`) in the badge of a **child-exit notification**:
|
||||
/// posted to the endpoint a supervisor passed to `system_spawn` when that child ends
|
||||
/// — by clean exit, by a fault, or by `process_kill`. The low bits carry the child's
|
||||
/// process id, so one endpoint can supervise many children (and even share with IRQ
|
||||
/// notifications, which never set this bit). The microkernel's SIGCHLD.
|
||||
pub const notify_exit_bit: u64 = 1 << 62;
|
||||
|
||||
/// Set (alongside `notify_badge_bit`) in the badge of a **buffered message** — a payload
|
||||
/// posted to an endpoint's async queue by `ipc_send`, delivered through `ipc_reply_wait`
|
||||
/// like a notification (no reply owed) but carrying bytes in the receive buffer, not just
|
||||
/// a badge. This is what distinguishes a payload-bearing async message from a bare IRQ /
|
||||
/// child-exit notification (which sets neither this nor `notify_exit_bit`). The low bits
|
||||
/// carry the sender's task id. The async counterpart of the synchronous `ipc_call`, for
|
||||
/// broadcasts where a rendezvous is the wrong shape (the input service is the first user).
|
||||
pub const notify_message_bit: u64 = 1 << 61;
|
||||
|
||||
/// Set (alongside `notify_badge_bit`) in the badge of a **signal notification** —
|
||||
/// the process-lifecycle vocabulary of docs/process-lifecycle.md, delivered to the
|
||||
/// endpoint the process nominated with `signal_bind`. The low bits carry the
|
||||
/// coalesced pending mask (bit positions = `Signal` values): signals are
|
||||
/// statements, not questions, and two pending terminates are one terminate.
|
||||
pub const notify_signal_bit: u64 = 1 << 60;
|
||||
|
||||
/// Set (alongside `notify_badge_bit`) in the badge of a **timer notification** —
|
||||
/// a one-shot `timer_bind` deadline landing. No payload bits: what to do when the
|
||||
/// deadline fires is whatever the receiver armed it for (a stop-sequence
|
||||
/// escalation, a restart backoff, an alarm).
|
||||
pub const notify_timer_bit: u64 = 1 << 59;
|
||||
|
||||
/// The signal vocabulary (docs/process-lifecycle.md): POSIX's concepts, danos's
|
||||
/// names, message delivery. The value is the bit position in the pending mask — a
|
||||
/// private kernel/runtime detail, free to change while they ship together. Kill
|
||||
/// is not here (it is `process_kill`, unhandleable by definition); faults are not
|
||||
/// here (they are `ExitReason`s — recovery is restart, not a handler); liveness is
|
||||
/// not here (a question, asked as the zero-length ping call, not a statement).
|
||||
pub const Signal = enum(u5) {
|
||||
terminate = 0, // finish up and exit (the polite half of the stop sequence)
|
||||
reload = 1, // re-read configuration / re-scan
|
||||
interrupt = 2, // interactive interrupt (no sender until a console exists)
|
||||
quit = 3, // as interrupt, by convention more final
|
||||
alarm = 4, // a timer the process armed for itself (unbuilt: no consumer yet)
|
||||
user_1 = 5, // service-defined
|
||||
user_2 = 6, // service-defined
|
||||
};
|
||||
|
||||
/// Capacity of `ProcessDescriptor.name` — matches the longest name `system_spawn`
|
||||
/// accepts, so a process's recorded name (its argv[0]) is never truncated.
|
||||
pub const maximum_process_name = 64;
|
||||
|
||||
/// What a process is doing right now, as reported by `process_enumerate`. Crosses
|
||||
/// the system_call boundary as `ProcessDescriptor.state`.
|
||||
pub const ProcessState = enum(u32) {
|
||||
ready = 0, // runnable, waiting for a core
|
||||
running = 1, // executing on a core right now
|
||||
blocked = 2, // waiting (sleeping, or blocked in IPC)
|
||||
};
|
||||
|
||||
/// One `process_enumerate` entry — the kernel's view of a live task, kernel tasks
|
||||
/// included (they carry an empty name and id 0 is the boot task). Fixed layout
|
||||
/// (extern) because it crosses the kernel↔user boundary by memory copy, like
|
||||
/// `DeviceDescriptor` in the device ABI.
|
||||
pub const ProcessDescriptor = extern struct {
|
||||
id: u32, // kernel-assigned process id; never reused (monotonic)
|
||||
supervisor: u32, // id of the process that spawned it (0 = the kernel)
|
||||
state: u32, // a ProcessState value
|
||||
priority: u32,
|
||||
name_length: u32,
|
||||
name: [maximum_process_name]u8, // argv[0] at spawn; empty for kernel tasks
|
||||
};
|
||||
|
||||
/// Well-known IPC service ids for the bootstrap name registry (create_ipc_endpoint +
|
||||
/// ipc_register/ipc_lookup). Small integers, so no string interning is needed
|
||||
/// during bring-up. The VFS server registers under `vfs`; clients look it up.
|
||||
pub const ServiceId = enum(u32) {
|
||||
vfs = 1,
|
||||
input = 2,
|
||||
ps2_bus = 3, // the 8042 owner; child device drivers attach here for raw bytes
|
||||
device_manager = 4, // the tree, the matcher, the supervisor (docs/device-manager.md)
|
||||
_,
|
||||
};
|
||||
|
||||
/// Protection flags for `mmap` (matching the usual C bit values).
|
||||
pub const prot_read: u64 = 1;
|
||||
pub const prot_write: u64 = 2;
|
||||
pub const prot_exec: u64 = 4;
|
||||
|
||||
/// `send_cap` / `received_cap` sentinel meaning "no capability" on the `ipc_call` /
|
||||
/// `ipc_reply_wait` cap-passing path (M13). `~0`, like `no_parent` — a real handle is
|
||||
/// a small index, so it can never collide.
|
||||
pub const no_cap: u64 = ~@as(u64, 0);
|
||||
@@ -0,0 +1,156 @@
|
||||
//! The **loader ↔ kernel** contract: everything a bootloader (boot/, e.g. efi.zig
|
||||
//! built as BOOTX64.efi) and the kernel (system/kernel/kernel.zig) must agree on to
|
||||
//! hand control over — the handoff structures the loader fills in, plus the kernel's
|
||||
//! virtual-memory layout and the physical↔virtual addressing both sides use.
|
||||
//!
|
||||
//! Both binaries import this as the `boot-handoff` module, so the layout is defined
|
||||
//! in exactly one place. **User space never sees this** — the kernel↔user contract is
|
||||
//! [[abi]] (system/abi.zig); device types are [[device-abi]] (system/devices/device-abi.zig).
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// Calling convention for the bootloader→kernel jump. Pinned to SystemV so it does
|
||||
/// not depend on each binary's target default: the UEFI bootloader's C
|
||||
/// convention is Microsoft x64 (first arg in RCX), the freestanding kernel's is
|
||||
/// SystemV (first arg in RDI). Both reference this to agree on where `*BootInformation`
|
||||
/// is passed.
|
||||
pub const kernel_abi: std.builtin.CallingConvention = .{ .x86_64_sysv = .{} };
|
||||
|
||||
/// Pixel byte order of the linear framebuffer the firmware handed us.
|
||||
pub const PixelFormat = enum(u32) {
|
||||
/// Byte 0 = Red, 1 = Green, 2 = Blue, 3 = reserved.
|
||||
rgbx,
|
||||
/// Byte 0 = Blue, 1 = Green, 2 = Red, 3 = reserved.
|
||||
bgrx,
|
||||
};
|
||||
|
||||
/// A linear framebuffer: `width`x`height` pixels, each a 32-bit value, with
|
||||
/// `pitch` bytes between the start of one row and the next (which may be larger
|
||||
/// than `width * 4` due to hardware padding).
|
||||
///
|
||||
/// A `base` of 0 means **no framebuffer** — the firmware exposed no Graphics
|
||||
/// Output Protocol (a headless server, say). The kernel must treat on-screen
|
||||
/// output as optional and never assume a framebuffer exists.
|
||||
pub const Framebuffer = extern struct {
|
||||
base: usize, // the memory address where pixel data starts (0 = none)
|
||||
width: u32, // visible pixels per row (e.g. 1920)
|
||||
height: u32, // visible rows (e.g. 1080)
|
||||
pitch: u32, // bytes from the start of one row to the start of the next
|
||||
format: PixelFormat,
|
||||
|
||||
/// Whether a usable framebuffer was handed over.
|
||||
pub fn present(self: Framebuffer) bool {
|
||||
return self.base != 0 and self.width != 0 and self.height != 0;
|
||||
}
|
||||
};
|
||||
|
||||
/// The kernel's virtual-memory layout (higher-half). The kernel is linked at
|
||||
/// `kernel_virt_base` but loaded at a low physical address; all of RAM (and the
|
||||
/// device MMIO windows) is also mapped at `physmap_base + physical`, so the kernel
|
||||
/// can reach any physical address by adding a constant. The low half is left
|
||||
/// entirely to user space.
|
||||
///
|
||||
/// user image + stack : 0x0000_7000_0000_0000 (PML4[224], low half)
|
||||
/// kernel heap : 0xFFFF_8000_0000_0000 (PML4[256])
|
||||
/// physmap : 0xFFFF_8800_0000_0000 (PML4[272]) + physical
|
||||
/// kernel image : 0xFFFF_FFFF_8000_0000 (PML4[511])
|
||||
pub const physmap_base: u64 = 0xFFFF_8800_0000_0000;
|
||||
pub const kernel_virt_base: u64 = 0xFFFF_FFFF_8000_0000;
|
||||
|
||||
/// Physical address -> its virtual address in the physmap. The single way the
|
||||
/// kernel dereferences a physical address once paging is up.
|
||||
///
|
||||
/// **Hazard:** valid only once the (bootstrap or final) page tables are live.
|
||||
/// The bootloader may use the *constant* `physmap_base` to build those tables,
|
||||
/// but must not call this to dereference memory before its own CR3 is loaded —
|
||||
/// it runs under the firmware's identity map, where these addresses are unmapped.
|
||||
pub inline fn physicalToVirtual(physical: u64) u64 {
|
||||
return physical + physmap_base;
|
||||
}
|
||||
|
||||
/// Physmap virtual address -> physical. Inverse of `physicalToVirtual`; for producing
|
||||
/// the physical address of something the kernel holds a physmap pointer to
|
||||
/// (e.g. a page-table frame for CR3, a post-mortem breadcrumb's RAM location).
|
||||
pub inline fn virtualToPhysical(virtual: u64) u64 {
|
||||
return virtual - physmap_base;
|
||||
}
|
||||
|
||||
/// danos's own classification of a span of physical memory — deliberately not
|
||||
/// UEFI's vocabulary. Each boot path (UEFI now, device tree later) translates its
|
||||
/// native memory description into these kinds, so the kernel never learns what
|
||||
/// booted it. [[architecture]] keeps the same discipline for CPU code.
|
||||
pub const MemoryKind = enum(u32) {
|
||||
/// Free RAM the kernel may allocate. Each boot path folds its own transient
|
||||
/// memory into this once it's genuinely free (e.g. the UEFI loader classifies
|
||||
/// boot-services memory as usable after ExitBootServices), so the kernel never
|
||||
/// has to know about boot-protocol-specific "reclaimable" states.
|
||||
usable,
|
||||
/// Firmware, MMIO, the kernel image, our own boot buffers, the boot stack —
|
||||
/// never hand out.
|
||||
reserved,
|
||||
/// ACPI tables: parse, then reclaim.
|
||||
acpi_tables,
|
||||
/// ACPI non-volatile storage: preserve across sleep, do not allocate.
|
||||
acpi_nvs,
|
||||
/// Not backed by RAM: memory-mapped device registers or a reserved
|
||||
/// address-space window (e.g. PCIe configuration space). Kept distinct from
|
||||
/// `reserved` so RAM accounting doesn't count device address space.
|
||||
mmio,
|
||||
};
|
||||
|
||||
/// One contiguous span of physical memory. Because danos defines this layout
|
||||
/// itself (unlike the UEFI descriptor it's built from), `@sizeOf` is
|
||||
/// authoritative — the kernel walks a plain `[]MemoryRegion`, with none of the
|
||||
/// firmware's variable descriptor-stride to worry about.
|
||||
pub const MemoryRegion = extern struct {
|
||||
base: u64, // physical start address
|
||||
pages: u64, // length in 4 KiB pages (the [[abi]] `page_size` unit)
|
||||
kind: MemoryKind,
|
||||
_pad: u32 = 0,
|
||||
};
|
||||
|
||||
/// The physical memory layout handed to the kernel: a pointer to an array of
|
||||
/// `len` `MemoryRegion`s, in a buffer that outlives the loader.
|
||||
pub const MemoryMap = extern struct {
|
||||
regions: usize, // address of a `[len]MemoryRegion`
|
||||
len: usize,
|
||||
};
|
||||
|
||||
/// One PT_LOAD segment of the kernel image, so the kernel can re-map itself with
|
||||
/// correct permissions (code R+X, rodata R, data R+W+NX). `flags` are raw ELF
|
||||
/// segment flags: PF_X=1, PF_W=2, PF_R=4. `virtual` is the higher-half link address;
|
||||
/// `physical` is where the loader actually placed the segment (they differ once the
|
||||
/// kernel links high — the loader records the real load address here).
|
||||
pub const KernelSegment = extern struct {
|
||||
virtual: u64,
|
||||
physical: u64,
|
||||
pages: u64,
|
||||
flags: u32,
|
||||
_pad: u32 = 0,
|
||||
};
|
||||
|
||||
/// Handoff structure the bootloader fills in and passes to the kernel's
|
||||
/// `_start` in RDI (the first argument under the SystemV AMD64 C ABI).
|
||||
pub const BootInformation = extern struct {
|
||||
framebuffer: Framebuffer,
|
||||
memory_map: MemoryMap,
|
||||
/// The kernel's own PT_LOAD segments (it has three: text, rodata, data).
|
||||
kernel_segments: [8]KernelSegment,
|
||||
kernel_segment_count: u32,
|
||||
/// Physical address of the ACPI RSDP the firmware exposed, or 0 if none. The
|
||||
/// kernel's device layer parses the ACPI tables from here to discover hardware.
|
||||
/// A device-tree boot path leaves this 0 and (later) fills a `device_tree_blob`
|
||||
/// field instead, so the kernel discovers devices without knowing what booted it.
|
||||
acpi_rsdp: u64 = 0,
|
||||
/// The raw `/system/services/init` ELF image, read off the boot volume by the loader
|
||||
/// into memory that survives the handoff (classified reserved, so the kernel
|
||||
/// identity-maps it and never allocates over it). 0/0 = no init found — the
|
||||
/// kernel boots without user space. Grows into a full initial_ramdisk handoff later.
|
||||
init_base: u64 = 0,
|
||||
init_len: u64 = 0,
|
||||
/// The initial_ramdisk image (a bundle of extra user binaries — the VFS server and
|
||||
/// device drivers), read off the boot volume into memory that survives the
|
||||
/// handoff, same as `init` above. 0/0 = no initial_ramdisk. See system/initial-ramdisk.zig.
|
||||
initial_ramdisk_base: u64 = 0,
|
||||
initial_ramdisk_len: u64 = 0,
|
||||
};
|
||||
@@ -0,0 +1,138 @@
|
||||
//! ACPI / PnP hardware-ID (`_HID`) names: the flat analog of pci-class.zig for
|
||||
//! `acpi_device` nodes. Unlike PCI, ACPI has no class/subclass/prog-IF taxonomy — a
|
||||
//! device's identity *is* its `_HID` string (`PNP0303` simply means "PS/2 keyboard"),
|
||||
//! so this is a plain id <-> name registry rather than a hierarchical decoder.
|
||||
//! The well-known PnP/ACPI IDs; vendor-specific ids (e.g. `QEMU0002`, `INTC1234`) have
|
||||
//! no standard name and decode to nothing. Pure reference data, so it is shared by
|
||||
//! kernel discovery (the device-tree dump) and any user-space driver or tool.
|
||||
//!
|
||||
//! Code that means a specific device names the `HardwareId` variant instead of its
|
||||
//! `_HID` string — `HardwareId.ps2_keyboard.hid()` reads without a registry lookup,
|
||||
//! where a bare `"PNP0303"` does not.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// The common standard PnP/ACPI hardware IDs, as named values. Prefix ranges hint at
|
||||
/// the grouping (PNP03xx keyboards, PNP0Fxx pointing devices, PNP0Cxx ACPI
|
||||
/// power/thermal, PNP0Axx buses), but there is no formal hierarchy — hence a flat
|
||||
/// enum over a flat registry.
|
||||
pub const HardwareId = enum {
|
||||
programmable_interrupt_controller,
|
||||
system_timer,
|
||||
high_precision_event_timer,
|
||||
dma_controller,
|
||||
ps2_keyboard,
|
||||
parallel_port,
|
||||
ecp_parallel_port,
|
||||
serial_port,
|
||||
floppy_disk_controller,
|
||||
system_speaker,
|
||||
pci_bus,
|
||||
generic_container,
|
||||
/// The second id the ACPI spec assigns the same "Generic Container Device" name.
|
||||
generic_container_extended,
|
||||
pci_express_root_bridge,
|
||||
real_time_clock,
|
||||
system_board,
|
||||
motherboard_reserved_resources,
|
||||
math_coprocessor,
|
||||
acpi_system_board,
|
||||
embedded_controller,
|
||||
control_method_battery,
|
||||
fan,
|
||||
power_button,
|
||||
lid,
|
||||
sleep_button,
|
||||
pci_interrupt_link,
|
||||
microsoft_ps2_mouse,
|
||||
ps2_mouse,
|
||||
ac_adapter,
|
||||
processor_device,
|
||||
processor_aggregator,
|
||||
processor_container,
|
||||
|
||||
const Entry = struct { hid: []const u8, name: []const u8 };
|
||||
|
||||
/// The registry row for this id: its `_HID` string and human-readable name.
|
||||
fn entry(self: HardwareId) Entry {
|
||||
return switch (self) {
|
||||
.programmable_interrupt_controller => .{ .hid = "PNP0000", .name = "Programmable Interrupt Controller (PIC)" },
|
||||
.system_timer => .{ .hid = "PNP0100", .name = "System Timer (PIT)" },
|
||||
.high_precision_event_timer => .{ .hid = "PNP0103", .name = "High Precision Event Timer (HPET)" },
|
||||
.dma_controller => .{ .hid = "PNP0200", .name = "DMA Controller" },
|
||||
.ps2_keyboard => .{ .hid = "PNP0303", .name = "PS/2 Keyboard" },
|
||||
.parallel_port => .{ .hid = "PNP0400", .name = "Standard LPT Parallel Port" },
|
||||
.ecp_parallel_port => .{ .hid = "PNP0401", .name = "ECP Parallel Port" },
|
||||
.serial_port => .{ .hid = "PNP0501", .name = "16550A-compatible Serial Port" },
|
||||
.floppy_disk_controller => .{ .hid = "PNP0700", .name = "PC Floppy Disk Controller" },
|
||||
.system_speaker => .{ .hid = "PNP0800", .name = "System Speaker" },
|
||||
.pci_bus => .{ .hid = "PNP0A03", .name = "PCI Bus" },
|
||||
.generic_container => .{ .hid = "PNP0A05", .name = "Generic Container Device" },
|
||||
.generic_container_extended => .{ .hid = "PNP0A06", .name = "Generic Container Device" },
|
||||
.pci_express_root_bridge => .{ .hid = "PNP0A08", .name = "PCI Express Root Bridge" },
|
||||
.real_time_clock => .{ .hid = "PNP0B00", .name = "Real-Time Clock (RTC)" },
|
||||
.system_board => .{ .hid = "PNP0C01", .name = "System Board" },
|
||||
.motherboard_reserved_resources => .{ .hid = "PNP0C02", .name = "Motherboard Reserved Resources" },
|
||||
.math_coprocessor => .{ .hid = "PNP0C04", .name = "Math Coprocessor" },
|
||||
.acpi_system_board => .{ .hid = "PNP0C08", .name = "ACPI System Board" },
|
||||
.embedded_controller => .{ .hid = "PNP0C09", .name = "ACPI Embedded Controller" },
|
||||
.control_method_battery => .{ .hid = "PNP0C0A", .name = "ACPI Control Method Battery" },
|
||||
.fan => .{ .hid = "PNP0C0B", .name = "ACPI Fan" },
|
||||
.power_button => .{ .hid = "PNP0C0C", .name = "ACPI Power Button" },
|
||||
.lid => .{ .hid = "PNP0C0D", .name = "ACPI Lid" },
|
||||
.sleep_button => .{ .hid = "PNP0C0E", .name = "ACPI Sleep Button" },
|
||||
.pci_interrupt_link => .{ .hid = "PNP0C0F", .name = "PCI Interrupt Link Device" },
|
||||
.microsoft_ps2_mouse => .{ .hid = "PNP0F03", .name = "Microsoft PS/2 Mouse" },
|
||||
.ps2_mouse => .{ .hid = "PNP0F13", .name = "PS/2 Mouse" },
|
||||
.ac_adapter => .{ .hid = "ACPI0003", .name = "AC Adapter" },
|
||||
.processor_device => .{ .hid = "ACPI0007", .name = "Processor Device" },
|
||||
.processor_aggregator => .{ .hid = "ACPI000C", .name = "Processor Aggregator" },
|
||||
.processor_container => .{ .hid = "ACPI0010", .name = "Processor Container" },
|
||||
};
|
||||
}
|
||||
|
||||
/// This id's `_HID` string (e.g. `.ps2_keyboard` -> "PNP0303").
|
||||
pub fn hid(self: HardwareId) []const u8 {
|
||||
return self.entry().hid;
|
||||
}
|
||||
|
||||
/// This id's human-readable name (e.g. `.ps2_keyboard` -> "PS/2 Keyboard").
|
||||
pub fn description(self: HardwareId) []const u8 {
|
||||
return self.entry().name;
|
||||
}
|
||||
|
||||
/// The named value for a `_HID` string, or null if it is not a known standard
|
||||
/// id (vendor-specific ids are not in the registry).
|
||||
pub fn fromHid(hid_string: []const u8) ?HardwareId {
|
||||
for (std.enums.values(HardwareId)) |id| {
|
||||
if (std.mem.eql(u8, id.hid(), hid_string)) return id;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
};
|
||||
|
||||
/// The human-readable name for a `_HID` string, or "" if it is not a known standard
|
||||
/// id (vendor-specific ids have no registry name — callers just print the raw HID).
|
||||
pub fn description(hid: []const u8) []const u8 {
|
||||
return (HardwareId.fromHid(hid) orelse return "").description();
|
||||
}
|
||||
|
||||
test "decodes standard PnP/ACPI ids and leaves the rest alone" {
|
||||
const eq = std.testing.expectEqualStrings;
|
||||
try eq("PS/2 Keyboard", description("PNP0303"));
|
||||
try eq("PS/2 Mouse", description("PNP0F13"));
|
||||
try eq("PCI Express Root Bridge", description("PNP0A08"));
|
||||
try eq("Real-Time Clock (RTC)", description("PNP0B00"));
|
||||
try eq("", description("QEMU0002")); // vendor-specific: no standard name
|
||||
try eq("", description("")); // no HID at all
|
||||
}
|
||||
|
||||
test "named values round-trip through their _HID strings" {
|
||||
const testing = std.testing;
|
||||
try testing.expectEqualStrings("PNP0303", HardwareId.ps2_keyboard.hid());
|
||||
try testing.expectEqual(@as(?HardwareId, .ps2_mouse), HardwareId.fromHid("PNP0F13"));
|
||||
try testing.expectEqual(@as(?HardwareId, null), HardwareId.fromHid("QEMU0002"));
|
||||
for (std.enums.values(HardwareId)) |id| {
|
||||
try testing.expectEqual(@as(?HardwareId, id), HardwareId.fromHid(id.hid()));
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -3,26 +3,24 @@
|
||||
//!
|
||||
//! This module has two stages. `parser.zig` walks the entire byte stream and
|
||||
//! records every named object into a namespace tree (`namespace.zig`), capturing
|
||||
//! method bodies and field/region layout. `interp.zig` then *evaluates* control
|
||||
//! method bodies and field/region layout. `interpreter.zig` then *evaluates* control
|
||||
//! methods on demand — running operators, control flow, and OperationRegion field
|
||||
//! access — so callers can resolve device status (`_STA`), current resource
|
||||
//! settings (`_CRS`), sleep states (`_Sx`), and the like against the live namespace.
|
||||
|
||||
const std = @import("std");
|
||||
const op = @import("opcodes.zig");
|
||||
const opcode = @import("opcodes.zig");
|
||||
const parser = @import("parser.zig");
|
||||
const namespace = @import("namespace.zig");
|
||||
const interp = @import("interp.zig");
|
||||
|
||||
pub const Namespace = namespace.Namespace;
|
||||
pub const Node = namespace.Node;
|
||||
pub const NodeKind = namespace.NodeKind;
|
||||
pub const Namespace = @import("namespace.zig").Namespace;
|
||||
pub const Node = @import("namespace.zig").Node;
|
||||
pub const NodeKind = @import("namespace.zig").NodeKind;
|
||||
|
||||
/// The AML evaluator: interprets control methods (and reads Names/Fields) far
|
||||
/// enough for device discovery. See `interp.zig`.
|
||||
pub const Interp = interp.Interp;
|
||||
pub const Object = interp.Object;
|
||||
pub const EvalHal = interp.Hal;
|
||||
/// enough for device discovery. See `interpreter.zig`.
|
||||
pub const Interpreter = @import("interpreter.zig").Interpreter;
|
||||
pub const Object = @import("interpreter.zig").Object;
|
||||
pub const EvaluateHal = @import("interpreter.zig").Hal;
|
||||
|
||||
/// The SLP_TYP values written to PM1a/PM1b control to enter a sleep state.
|
||||
pub const SleepType = struct {
|
||||
@@ -41,22 +39,36 @@ pub const ParseResult = struct {
|
||||
/// Parse the given AML blocks (DSDT first, then SSDTs) into one namespace. Later
|
||||
/// blocks extend the namespace built by earlier ones, exactly as ACPI intends.
|
||||
pub fn parse(allocator: std.mem.Allocator, blocks: []const []const u8) !ParseResult {
|
||||
var ns = try Namespace.init(allocator);
|
||||
var namespace = try Namespace.init(allocator);
|
||||
var consumed: usize = 0;
|
||||
var total: usize = 0;
|
||||
for (blocks) |block| {
|
||||
var p = parser.Parser.init(block, &ns);
|
||||
var p = parser.Parser.init(block, &namespace);
|
||||
consumed += p.parseAll();
|
||||
total += block.len;
|
||||
}
|
||||
return .{ .namespace = ns, .consumed = consumed, .total = total };
|
||||
return .{ .namespace = namespace, .consumed = consumed, .total = total };
|
||||
}
|
||||
|
||||
/// Count the Device objects in a parsed namespace — what the acpi service
|
||||
/// (docs/m19-m20-plan.md M20) reports, and what the kernel's own parse counts
|
||||
/// so the two can be checked equal across the ring-3 move.
|
||||
pub fn deviceCount(namespace: *const Namespace) usize {
|
||||
return countKind(namespace.root, .device);
|
||||
}
|
||||
|
||||
fn countKind(node: *const Node, kind: NodeKind) usize {
|
||||
var n: usize = if (node.kind == kind) 1 else 0;
|
||||
var c = node.first_child;
|
||||
while (c) |child| : (c = child.next_sibling) n += countKind(child, kind);
|
||||
return n;
|
||||
}
|
||||
|
||||
/// Look up the `\_S{state}` sleep package in a parsed namespace and return its
|
||||
/// first two integer elements (SLP_TYP for PM1a / PM1b), or null if absent.
|
||||
pub fn sleepState(ns: *Namespace, state: u8) ?SleepType {
|
||||
const seg = [4]u8{ '_', 'S', '0' + state, '_' };
|
||||
const node = ns.resolve(ns.root, false, 0, &.{seg}) orelse return null;
|
||||
pub fn sleepState(namespace: *Namespace, state: u8) ?SleepType {
|
||||
const segment = [4]u8{ '_', 'S', '0' + state, '_' };
|
||||
const node = namespace.resolve(namespace.root, false, 0, &.{segment}) orelse return null;
|
||||
if (node.kind != .name) return null;
|
||||
return parseSleepPackage(node.value);
|
||||
}
|
||||
@@ -64,20 +76,20 @@ pub fn sleepState(ns: *Namespace, state: u8) ?SleepType {
|
||||
/// Decode a `Package(){ SLP_TYPa, SLP_TYPb, ... }` from the raw AML of a Name's
|
||||
/// value. Returns the first two elements as bytes (missing elements default to 0).
|
||||
fn parseSleepPackage(value: []const u8) ?SleepType {
|
||||
if (value.len == 0 or value[0] != op.package_op) return null;
|
||||
if (value.len == 0 or value[0] != opcode.package_opcode) return null;
|
||||
var p: usize = 1;
|
||||
p += pkgLengthSize(value, p) orelse return null;
|
||||
p += packageLengthSize(value, p) orelse return null;
|
||||
if (p >= value.len) return null;
|
||||
const num_elements = value[p];
|
||||
const number_elements = value[p];
|
||||
p += 1;
|
||||
|
||||
const a: u8 = if (num_elements >= 1) @truncate(readInteger(value, &p) orelse 0) else 0;
|
||||
const b: u8 = if (num_elements >= 2) @truncate(readInteger(value, &p) orelse 0) else 0;
|
||||
const a: u8 = if (number_elements >= 1) @truncate(readInteger(value, &p) orelse 0) else 0;
|
||||
const b: u8 = if (number_elements >= 2) @truncate(readInteger(value, &p) orelse 0) else 0;
|
||||
return .{ .slp_typ_a = a, .slp_typ_b = b };
|
||||
}
|
||||
|
||||
/// Bytes a PkgLength field occupies at `p` (we only need to step over it here).
|
||||
fn pkgLengthSize(bytes: []const u8, p: usize) ?usize {
|
||||
fn packageLengthSize(bytes: []const u8, p: usize) ?usize {
|
||||
if (p >= bytes.len) return null;
|
||||
const follow: usize = bytes[p] >> 6;
|
||||
if (p + 1 + follow > bytes.len) return null;
|
||||
@@ -87,16 +99,16 @@ fn pkgLengthSize(bytes: []const u8, p: usize) ?usize {
|
||||
/// Read one AML integer data object at `p`, advancing `p`.
|
||||
fn readInteger(bytes: []const u8, p: *usize) ?u64 {
|
||||
if (p.* >= bytes.len) return null;
|
||||
const opcode = bytes[p.*];
|
||||
const opcode_byte = bytes[p.*];
|
||||
p.* += 1;
|
||||
return switch (opcode) {
|
||||
op.zero_op => 0,
|
||||
op.one_op => 1,
|
||||
op.ones_op => 0xFF,
|
||||
op.byte_prefix => readLittle(bytes, p, 1),
|
||||
op.word_prefix => readLittle(bytes, p, 2),
|
||||
op.dword_prefix => readLittle(bytes, p, 4),
|
||||
op.qword_prefix => readLittle(bytes, p, 8),
|
||||
return switch (opcode_byte) {
|
||||
opcode.zero_opcode => 0,
|
||||
opcode.one_opcode => 1,
|
||||
opcode.ones_opcode => 0xFF,
|
||||
opcode.byte_prefix => readLittle(bytes, p, 1),
|
||||
opcode.word_prefix => readLittle(bytes, p, 2),
|
||||
opcode.dword_prefix => readLittle(bytes, p, 4),
|
||||
opcode.qword_prefix => readLittle(bytes, p, 8),
|
||||
else => null,
|
||||
};
|
||||
}
|
||||
@@ -125,20 +137,25 @@ test "parses a nested namespace and finds the sleep package" {
|
||||
const blob = [_]u8{
|
||||
// Name(_S5, Package(2){Byte 0x05, Byte 0x00})
|
||||
0x08, 0x5F, 0x53, 0x35, 0x5F, 0x12, 0x06, 0x02, 0x0A, 0x05, 0x0A, 0x00,
|
||||
// Scope(\_SB) pkglen=0x27
|
||||
// Scope(\_SB) packagelen=0x27
|
||||
0x10, 0x27, 0x5C, 0x5F, 0x53, 0x42, 0x5F,
|
||||
// Device(PCI0) pkglen=0x1F
|
||||
0x5B, 0x82, 0x1F, 0x50, 0x43, 0x49, 0x30,
|
||||
// Device(PCI0) packagelen=0x1F
|
||||
0x5B, 0x82, 0x1F, 0x50, 0x43,
|
||||
0x49, 0x30,
|
||||
// Name(_HID, 0x11)
|
||||
0x08, 0x5F, 0x48, 0x49, 0x44, 0x0A, 0x11,
|
||||
// Method(MTHD, flags=1) empty, pkglen=0x06
|
||||
0x14, 0x06, 0x4D, 0x54, 0x48, 0x44, 0x01,
|
||||
// Method(CALL, flags=0) { MTHD(Zero) }, pkglen=0x0B
|
||||
0x14, 0x0B, 0x43, 0x41, 0x4C, 0x4C, 0x00, 0x4D, 0x54, 0x48, 0x44, 0x00,
|
||||
// Method(MTHD, flags=1) empty, packagelen=0x06
|
||||
0x14, 0x06, 0x4D,
|
||||
0x54, 0x48, 0x44, 0x01,
|
||||
// Method(CALL, flags=0) { MTHD(Zero) }, packagelen=0x0B
|
||||
0x14, 0x0B, 0x43, 0x41, 0x4C, 0x4C, 0x00, 0x4D,
|
||||
0x54, 0x48, 0x44, 0x00,
|
||||
// OperationRegion(DBG0, SystemIO, Word 0x0402, Byte 1)
|
||||
0x5B, 0x80, 0x44, 0x42, 0x47, 0x30, 0x01, 0x0B, 0x02, 0x04, 0x0A, 0x01,
|
||||
// Field(DBG0, flags=1) { DBGB, 8 }, pkglen=0x0B
|
||||
0x5B, 0x81, 0x0B, 0x44, 0x42, 0x47, 0x30, 0x01, 0x44, 0x42, 0x47, 0x42, 0x08,
|
||||
0x5B, 0x80, 0x44, 0x42, 0x47, 0x30, 0x01, 0x0B,
|
||||
0x02, 0x04, 0x0A, 0x01,
|
||||
// Field(DBG0, flags=1) { DBGB, 8 }, packagelen=0x0B
|
||||
0x5B, 0x81, 0x0B, 0x44, 0x42, 0x47, 0x30, 0x01,
|
||||
0x44, 0x42, 0x47, 0x42, 0x08,
|
||||
};
|
||||
|
||||
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
||||
@@ -149,31 +166,33 @@ test "parses a nested namespace and finds the sleep package" {
|
||||
try std.testing.expectEqual(blob.len, result.consumed);
|
||||
try std.testing.expectEqual(blob.len, result.total);
|
||||
|
||||
const ns = &result.namespace;
|
||||
const namespace = &result.namespace;
|
||||
|
||||
// Expected top-level nodes.
|
||||
const sb = ns.resolve(ns.root, false, 0, &.{.{ '_', 'S', 'B', '_' }}) orelse return error.NoSB;
|
||||
const sb = namespace.resolve(namespace.root, false, 0, &.{.{ '_', 'S', 'B', '_' }}) orelse return error.NoSB;
|
||||
try std.testing.expectEqual(NodeKind.scope, sb.kind);
|
||||
const pci0 = ns.resolve(sb, false, 0, &.{.{ 'P', 'C', 'I', '0' }}) orelse return error.NoPCI0;
|
||||
const pci0 = namespace.resolve(sb, false, 0, &.{.{ 'P', 'C', 'I', '0' }}) orelse return error.NoPCI0;
|
||||
try std.testing.expectEqual(NodeKind.device, pci0.kind);
|
||||
_ = ns.resolve(pci0, false, 0, &.{.{ '_', 'H', 'I', 'D' }}) orelse return error.NoHID;
|
||||
_ = namespace.resolve(pci0, false, 0, &.{.{ '_', 'H', 'I', 'D' }}) orelse return error.NoHID;
|
||||
|
||||
// The 1-arg method's arg count was parsed from its flags byte.
|
||||
const mthd = ns.resolve(pci0, false, 0, &.{.{ 'M', 'T', 'H', 'D' }}) orelse return error.NoMTHD;
|
||||
const mthd = namespace.resolve(pci0, false, 0, &.{.{ 'M', 'T', 'H', 'D' }}) orelse return error.NoMTHD;
|
||||
try std.testing.expectEqual(NodeKind.method, mthd.kind);
|
||||
try std.testing.expectEqual(@as(u8, 1), mthd.arg_count);
|
||||
|
||||
// OperationRegion and the Field unit made it into the namespace.
|
||||
_ = ns.resolve(ns.root, false, 0, &.{.{ 'D', 'B', 'G', '0' }}) orelse return error.NoRegion;
|
||||
_ = ns.resolve(ns.root, false, 0, &.{.{ 'D', 'B', 'G', 'B' }}) orelse return error.NoField;
|
||||
_ = namespace.resolve(namespace.root, false, 0, &.{.{ 'D', 'B', 'G', '0' }}) orelse return error.NoRegion;
|
||||
_ = namespace.resolve(namespace.root, false, 0, &.{.{ 'D', 'B', 'G', 'B' }}) orelse return error.NoField;
|
||||
|
||||
// The sleep package decoded.
|
||||
const s5 = sleepState(ns, 5) orelse return error.NoS5;
|
||||
const s5 = sleepState(namespace, 5) orelse return error.NoS5;
|
||||
try std.testing.expectEqual(@as(u8, 5), s5.slp_typ_a);
|
||||
try std.testing.expectEqual(@as(u8, 0), s5.slp_typ_b);
|
||||
}
|
||||
|
||||
fn noMap(_: u64, _: u64, _: bool) void {}
|
||||
fn noMap(physical: u64, _: u64, _: bool) u64 {
|
||||
return physical;
|
||||
}
|
||||
fn noRead(_: u8, _: u16) u32 {
|
||||
return 0;
|
||||
}
|
||||
@@ -196,13 +215,13 @@ test "interpreter runs a method with args, arithmetic, and control flow" {
|
||||
var arena = std.heap.ArenaAllocator.init(std.testing.allocator);
|
||||
defer arena.deinit();
|
||||
var result = try parse(arena.allocator(), &.{&blob});
|
||||
const ns = &result.namespace;
|
||||
const tst = ns.resolve(ns.root, false, 0, &.{.{ 'T', 'S', 'T', '_' }}) orelse return error.NoMethod;
|
||||
const namespace = &result.namespace;
|
||||
const tst = namespace.resolve(namespace.root, false, 0, &.{.{ 'T', 'S', 'T', '_' }}) orelse return error.NoMethod;
|
||||
|
||||
var ev = Interp.init(ns, .{ .mapMmio = noMap, .pioRead = noRead, .pioWrite = noWrite }, arena.allocator());
|
||||
var interpreter = Interpreter.init(namespace, .{ .mapMmio = noMap, .pioRead = noRead, .pioWrite = noWrite }, arena.allocator());
|
||||
|
||||
const hi = try ev.evaluate(tst, &.{.{ .integer = 7 }}); // 7+5=12 > 10 -> 1
|
||||
try std.testing.expectEqual(@as(u64, 1), try hi.asInt());
|
||||
const lo = try ev.evaluate(tst, &.{.{ .integer = 2 }}); // 2+5=7 !> 10 -> 0
|
||||
try std.testing.expectEqual(@as(u64, 0), try lo.asInt());
|
||||
const hi = try interpreter.evaluate(tst, &.{.{ .integer = 7 }}); // 7+5=12 > 10 -> 1
|
||||
try std.testing.expectEqual(@as(u64, 1), try hi.asInteger());
|
||||
const lo = try interpreter.evaluate(tst, &.{.{ .integer = 2 }}); // 2+5=7 !> 10 -> 0
|
||||
try std.testing.expectEqual(@as(u64, 0), try lo.asInteger());
|
||||
}
|
||||
@@ -0,0 +1,736 @@
|
||||
//! A tree-walking AML interpreter — the evaluation stage on top of the parser's
|
||||
//! structural namespace. It executes control methods (their bodies captured by
|
||||
//! the parser) far enough to serve device discovery: device status (`_STA`, is a
|
||||
//! device present), current resource settings (`_CRS`), and the operators, control
|
||||
//! flow, locals/args, and
|
||||
//! OperationRegion field access those methods reach for.
|
||||
//!
|
||||
//! Scope: integers, buffers, strings, packages, and references; If/Else/While/
|
||||
//! Return; the arithmetic/logic operators; method invocation; Name/Local/Arg
|
||||
//! access; CreateField buffer patching (the common current-resource-settings
|
||||
//! (`_CRS`) idiom); and field
|
||||
//! reads/writes against SystemMemory and SystemIO regions. Opcodes outside this
|
||||
//! set return `error.Unsupported`, which callers treat as "couldn't evaluate" and
|
||||
//! fall back — never a hard failure.
|
||||
|
||||
const std = @import("std");
|
||||
const opcode = @import("opcodes.zig");
|
||||
const Node = @import("namespace.zig").Node;
|
||||
const Namespace = @import("namespace.zig").Namespace;
|
||||
|
||||
/// Injected hardware access for OperationRegion reads/writes (the architecture VMM + pio).
|
||||
pub const Hal = struct {
|
||||
mapMmio: *const fn (physical: u64, len: u64, writable: bool) u64,
|
||||
pioRead: *const fn (width: u8, port: u16) u32,
|
||||
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
||||
};
|
||||
|
||||
pub const Error = error{ Unsupported, Truncated, DivByZero } || std.mem.Allocator.Error;
|
||||
|
||||
/// A runtime AML value.
|
||||
pub const Object = union(enum) {
|
||||
uninitialized,
|
||||
integer: u64,
|
||||
buffer: []u8,
|
||||
string: []u8,
|
||||
package: []Object,
|
||||
reference: *Node,
|
||||
|
||||
pub fn asInteger(self: Object) Error!u64 {
|
||||
return switch (self) {
|
||||
.integer => |v| v,
|
||||
.buffer => |b| blk: {
|
||||
var v: u64 = 0;
|
||||
for (b, 0..) |byte, i| {
|
||||
if (i >= 8) break;
|
||||
v |= @as(u64, byte) << @intCast(i * 8);
|
||||
}
|
||||
break :blk v;
|
||||
},
|
||||
else => error.Unsupported,
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
const maximum_segments = 16;
|
||||
const NamePath = struct {
|
||||
rooted: bool = false,
|
||||
parents: u8 = 0,
|
||||
segments: [maximum_segments][4]u8 = undefined,
|
||||
count: usize = 0,
|
||||
fn slice(self: *const NamePath) []const [4]u8 {
|
||||
return self.segments[0..self.count];
|
||||
}
|
||||
};
|
||||
|
||||
const Cursor = struct {
|
||||
b: []const u8,
|
||||
i: usize = 0,
|
||||
|
||||
fn eof(self: *Cursor) bool {
|
||||
return self.i >= self.b.len;
|
||||
}
|
||||
fn peek(self: *Cursor) ?u8 {
|
||||
return if (self.eof()) null else self.b[self.i];
|
||||
}
|
||||
fn byte(self: *Cursor) Error!u8 {
|
||||
if (self.eof()) return error.Truncated;
|
||||
const v = self.b[self.i];
|
||||
self.i += 1;
|
||||
return v;
|
||||
}
|
||||
fn take(self: *Cursor, n: usize) Error![]const u8 {
|
||||
if (self.i + n > self.b.len) return error.Truncated;
|
||||
const s = self.b[self.i .. self.i + n];
|
||||
self.i += n;
|
||||
return s;
|
||||
}
|
||||
fn packageLength(self: *Cursor) Error!usize {
|
||||
const lead = try self.byte();
|
||||
const follow: usize = lead >> 6;
|
||||
if (follow == 0) return lead & 0x3F;
|
||||
var value: usize = lead & 0x0F;
|
||||
var k: usize = 0;
|
||||
while (k < follow) : (k += 1) value |= @as(usize, try self.byte()) << @intCast(4 + k * 8);
|
||||
return value;
|
||||
}
|
||||
fn nameString(self: *Cursor) Error!NamePath {
|
||||
var name_path = NamePath{};
|
||||
if (self.peek() == opcode.root_char) {
|
||||
name_path.rooted = true;
|
||||
self.i += 1;
|
||||
} else {
|
||||
while (self.peek() == opcode.parent_prefix_char) : (self.i += 1) name_path.parents += 1;
|
||||
}
|
||||
const lead = self.peek() orelse return name_path;
|
||||
switch (lead) {
|
||||
0x00 => self.i += 1,
|
||||
opcode.dual_name_prefix => {
|
||||
self.i += 1;
|
||||
try self.segment(&name_path);
|
||||
try self.segment(&name_path);
|
||||
},
|
||||
opcode.multi_name_prefix => {
|
||||
self.i += 1;
|
||||
const count = try self.byte();
|
||||
var k: usize = 0;
|
||||
while (k < count) : (k += 1) try self.segment(&name_path);
|
||||
},
|
||||
else => try self.segment(&name_path),
|
||||
}
|
||||
return name_path;
|
||||
}
|
||||
fn segment(self: *Cursor, name_path: *NamePath) Error!void {
|
||||
const s = try self.take(4);
|
||||
if (name_path.count < maximum_segments) {
|
||||
name_path.segments[name_path.count] = s[0..4].*;
|
||||
name_path.count += 1;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
const Frame = struct {
|
||||
args: [7]Object = .{.uninitialized} ** 7,
|
||||
locals: [8]Object = .{.uninitialized} ** 8,
|
||||
scope: *Node,
|
||||
ret: Object = .uninitialized,
|
||||
returned: bool = false,
|
||||
broke: bool = false,
|
||||
};
|
||||
|
||||
/// A CreateField binding: a name that indexes into a buffer object.
|
||||
const BufferField = struct { buffer: *Node, byte_off: usize, bit_width: u32 };
|
||||
|
||||
pub const Interpreter = struct {
|
||||
namespace: *Namespace,
|
||||
hal: Hal,
|
||||
arena: std.mem.Allocator,
|
||||
/// Runtime object overrides for Name nodes (Store targets, patched buffers).
|
||||
dynamic_overrides: std.AutoHashMapUnmanaged(*Node, Object) = .{},
|
||||
/// CreateField bindings active for the current evaluation.
|
||||
fields: std.AutoHashMapUnmanaged(*Node, BufferField) = .{},
|
||||
|
||||
pub fn init(namespace: *Namespace, hal: Hal, arena: std.mem.Allocator) Interpreter {
|
||||
return .{ .namespace = namespace, .hal = hal, .arena = arena };
|
||||
}
|
||||
|
||||
/// Evaluate a namespace object: invoke a Method, read a Name's value, or read a
|
||||
/// Field. Resets per-evaluation runtime state first.
|
||||
pub fn evaluate(self: *Interpreter, node: *Node, args: []const Object) Error!Object {
|
||||
self.dynamic_overrides.clearRetainingCapacity();
|
||||
self.fields.clearRetainingCapacity();
|
||||
return self.invoke(node, args);
|
||||
}
|
||||
|
||||
fn invoke(self: *Interpreter, node: *Node, args: []const Object) Error!Object {
|
||||
switch (node.kind) {
|
||||
.method => {
|
||||
var frame = Frame{ .scope = node };
|
||||
for (args, 0..) |a, i| {
|
||||
if (i < frame.args.len) frame.args[i] = a;
|
||||
}
|
||||
var current = Cursor{ .b = node.value };
|
||||
try self.executeList(¤t, &frame);
|
||||
return frame.ret;
|
||||
},
|
||||
.name => {
|
||||
if (self.dynamic_overrides.get(node)) |o| return o;
|
||||
var current = Cursor{ .b = node.value };
|
||||
var frame = Frame{ .scope = node.parent orelse self.namespace.root };
|
||||
return self.term(¤t, &frame);
|
||||
},
|
||||
.field => return .{ .integer = try self.readField(node) },
|
||||
else => return .{ .reference = node },
|
||||
}
|
||||
}
|
||||
|
||||
/// Execute a TermList until it ends or the frame returns/breaks.
|
||||
fn executeList(self: *Interpreter, current: *Cursor, frame: *Frame) Error!void {
|
||||
while (!current.eof() and !frame.returned and !frame.broke) {
|
||||
_ = try self.term(current, frame);
|
||||
}
|
||||
}
|
||||
|
||||
/// Evaluate/execute one term, returning its value (`.uninitialized` for pure
|
||||
/// statements).
|
||||
fn term(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const lead = current.peek() orelse return error.Truncated;
|
||||
if (isNameStart(lead)) return self.nameReference(current, frame);
|
||||
_ = try current.byte();
|
||||
|
||||
return switch (lead) {
|
||||
opcode.zero_opcode => Object{ .integer = 0 },
|
||||
opcode.one_opcode => Object{ .integer = 1 },
|
||||
opcode.ones_opcode => Object{ .integer = ~@as(u64, 0) },
|
||||
opcode.byte_prefix => Object{ .integer = try self.readConstant(current, 1) },
|
||||
opcode.word_prefix => Object{ .integer = try self.readConstant(current, 2) },
|
||||
opcode.dword_prefix => Object{ .integer = try self.readConstant(current, 4) },
|
||||
opcode.qword_prefix => Object{ .integer = try self.readConstant(current, 8) },
|
||||
opcode.string_prefix => try self.readString(current),
|
||||
opcode.buffer_opcode => try self.buffer(current, frame),
|
||||
opcode.package_opcode, opcode.var_package_opcode => try self.package(current, frame, lead == opcode.var_package_opcode),
|
||||
|
||||
opcode.local0_opcode...opcode.local7_opcode => frame.locals[lead - opcode.local0_opcode],
|
||||
opcode.arg0_opcode...opcode.arg6_opcode => frame.args[lead - opcode.arg0_opcode],
|
||||
|
||||
opcode.return_opcode => blk: {
|
||||
frame.ret = try self.term(current, frame);
|
||||
frame.returned = true;
|
||||
break :blk .uninitialized;
|
||||
},
|
||||
opcode.break_opcode => blk: {
|
||||
frame.broke = true;
|
||||
break :blk .uninitialized;
|
||||
},
|
||||
opcode.continue_opcode, opcode.noop_opcode => .uninitialized,
|
||||
|
||||
opcode.if_opcode => try self.ifElse(current, frame),
|
||||
opcode.while_opcode => try self.whileLoop(current, frame),
|
||||
opcode.store_opcode => try self.store(current, frame),
|
||||
opcode.increment_opcode => try self.incDec(current, frame, 1),
|
||||
opcode.decrement_opcode => try self.incDec(current, frame, -1),
|
||||
|
||||
opcode.add_opcode => try self.binary(current, frame, .add),
|
||||
opcode.subtract_opcode => try self.binary(current, frame, .sub),
|
||||
opcode.multiply_opcode => try self.binary(current, frame, .mul),
|
||||
opcode.mod_opcode => try self.binary(current, frame, .mod),
|
||||
opcode.and_opcode => try self.binary(current, frame, .band),
|
||||
opcode.or_opcode => try self.binary(current, frame, .bor),
|
||||
opcode.xor_opcode => try self.binary(current, frame, .bxor),
|
||||
opcode.nand_opcode => try self.binary(current, frame, .nand),
|
||||
opcode.nor_opcode => try self.binary(current, frame, .nor),
|
||||
opcode.shift_left_opcode => try self.binary(current, frame, .shl),
|
||||
opcode.shift_right_opcode => try self.binary(current, frame, .shr),
|
||||
opcode.divide_opcode => try self.divide(current, frame),
|
||||
|
||||
opcode.land_opcode => try self.logic2(current, frame, .land),
|
||||
opcode.lor_opcode => try self.logic2(current, frame, .lor),
|
||||
opcode.lequal_opcode => try self.logic2(current, frame, .eq),
|
||||
opcode.lgreater_opcode => try self.logic2(current, frame, .gt),
|
||||
opcode.lless_opcode => try self.logic2(current, frame, .lt),
|
||||
opcode.lnot_opcode => try self.lnot(current, frame),
|
||||
|
||||
opcode.not_opcode => blk: {
|
||||
const v = try self.evaluateInteger(current, frame);
|
||||
const r = ~v;
|
||||
try self.storeTarget(current, frame, .{ .integer = r });
|
||||
break :blk .{ .integer = r };
|
||||
},
|
||||
|
||||
opcode.size_of_opcode => try self.sizeOf(current, frame),
|
||||
opcode.index_opcode => try self.index(current, frame),
|
||||
opcode.dereference_of_opcode => try self.dereferenceOf(current, frame),
|
||||
opcode.to_integer_opcode => blk: {
|
||||
const v = try self.evaluateInteger(current, frame);
|
||||
try self.storeTarget(current, frame, .{ .integer = v });
|
||||
break :blk .{ .integer = v };
|
||||
},
|
||||
opcode.to_buffer_opcode => try self.passThroughUnary(current, frame),
|
||||
|
||||
opcode.extended_opcode_prefix => try self.ext(current, frame),
|
||||
|
||||
// CreateXField: source, index, name (bit widths differ by op)
|
||||
opcode.create_bit_field_opcode => try self.createField(current, frame, 1),
|
||||
opcode.create_byte_field_opcode => try self.createField(current, frame, 8),
|
||||
opcode.create_word_field_opcode => try self.createField(current, frame, 16),
|
||||
opcode.create_dword_field_opcode => try self.createField(current, frame, 32),
|
||||
opcode.create_qword_field_opcode => try self.createField(current, frame, 64),
|
||||
|
||||
else => error.Unsupported,
|
||||
};
|
||||
}
|
||||
|
||||
// --- name references ----------------------------------------------------
|
||||
|
||||
fn nameReference(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const name_path = try current.nameString();
|
||||
const node = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice()) orelse
|
||||
return .uninitialized; // unknown name -> treat as uninitialised
|
||||
switch (node.kind) {
|
||||
.method => {
|
||||
var argbuf: [7]Object = undefined;
|
||||
var i: usize = 0;
|
||||
while (i < node.arg_count and i < argbuf.len) : (i += 1) argbuf[i] = try self.term(current, frame);
|
||||
return self.invoke(node, argbuf[0..@min(node.arg_count, argbuf.len)]);
|
||||
},
|
||||
.field => return .{ .integer = try self.readField(node) },
|
||||
.name => return self.invoke(node, &.{}),
|
||||
else => return .{ .reference = node },
|
||||
}
|
||||
}
|
||||
|
||||
// --- data objects -------------------------------------------------------
|
||||
|
||||
fn readConstant(self: *Interpreter, current: *Cursor, n: usize) Error!u64 {
|
||||
_ = self;
|
||||
const bytes = try current.take(n);
|
||||
var v: u64 = 0;
|
||||
for (bytes, 0..) |b, i| v |= @as(u64, b) << @intCast(i * 8);
|
||||
return v;
|
||||
}
|
||||
|
||||
fn readString(self: *Interpreter, current: *Cursor) Error!Object {
|
||||
const start = current.i;
|
||||
while (current.peek()) |c| {
|
||||
current.i += 1;
|
||||
if (c == 0) break;
|
||||
}
|
||||
const raw = current.b[start .. current.i - 1];
|
||||
const s = try self.arena.dupe(u8, raw);
|
||||
return .{ .string = s };
|
||||
}
|
||||
|
||||
fn buffer(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const start = current.i;
|
||||
const len = try current.packageLength();
|
||||
const end = @min(start + len, current.b.len);
|
||||
const size = try self.evaluateInteger(current, frame);
|
||||
const data = current.b[@min(current.i, end)..end];
|
||||
const bytes = try self.arena.alloc(u8, @intCast(size));
|
||||
@memset(bytes, 0);
|
||||
@memcpy(bytes[0..@min(bytes.len, data.len)], data[0..@min(bytes.len, data.len)]);
|
||||
current.i = end;
|
||||
return .{ .buffer = bytes };
|
||||
}
|
||||
|
||||
fn package(self: *Interpreter, current: *Cursor, frame: *Frame, variable: bool) Error!Object {
|
||||
const start = current.i;
|
||||
const len = try current.packageLength();
|
||||
const end = @min(start + len, current.b.len);
|
||||
const count: usize = if (variable) @intCast(try self.evaluateInteger(current, frame)) else try current.byte();
|
||||
const elems = try self.arena.alloc(Object, count);
|
||||
var i: usize = 0;
|
||||
while (i < count and current.i < end) : (i += 1) elems[i] = try self.term(current, frame);
|
||||
while (i < count) : (i += 1) elems[i] = .uninitialized;
|
||||
current.i = end;
|
||||
return .{ .package = elems };
|
||||
}
|
||||
|
||||
// --- operators ----------------------------------------------------------
|
||||
|
||||
const BinaryOperation = enum { add, sub, mul, mod, band, bor, bxor, nand, nor, shl, shr };
|
||||
|
||||
fn binary(self: *Interpreter, current: *Cursor, frame: *Frame, kind: BinaryOperation) Error!Object {
|
||||
const a = try self.evaluateInteger(current, frame);
|
||||
const b = try self.evaluateInteger(current, frame);
|
||||
const r: u64 = switch (kind) {
|
||||
.add => a +% b,
|
||||
.sub => a -% b,
|
||||
.mul => a *% b,
|
||||
.mod => if (b == 0) return error.DivByZero else a % b,
|
||||
.band => a & b,
|
||||
.bor => a | b,
|
||||
.bxor => a ^ b,
|
||||
.nand => ~(a & b),
|
||||
.nor => ~(a | b),
|
||||
.shl => if (b >= 64) 0 else a << @intCast(b),
|
||||
.shr => if (b >= 64) 0 else a >> @intCast(b),
|
||||
};
|
||||
try self.storeTarget(current, frame, .{ .integer = r });
|
||||
return .{ .integer = r };
|
||||
}
|
||||
|
||||
fn divide(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const a = try self.evaluateInteger(current, frame);
|
||||
const b = try self.evaluateInteger(current, frame);
|
||||
if (b == 0) return error.DivByZero;
|
||||
try self.storeTarget(current, frame, .{ .integer = a % b }); // remainder target
|
||||
try self.storeTarget(current, frame, .{ .integer = a / b }); // quotient target
|
||||
return .{ .integer = a / b };
|
||||
}
|
||||
|
||||
const LogicOperation = enum { land, lor, eq, gt, lt };
|
||||
|
||||
fn logic2(self: *Interpreter, current: *Cursor, frame: *Frame, kind: LogicOperation) Error!Object {
|
||||
const a = try self.evaluateInteger(current, frame);
|
||||
const b = try self.evaluateInteger(current, frame);
|
||||
const r = switch (kind) {
|
||||
.land => a != 0 and b != 0,
|
||||
.lor => a != 0 or b != 0,
|
||||
.eq => a == b,
|
||||
.gt => a > b,
|
||||
.lt => a < b,
|
||||
};
|
||||
return .{ .integer = if (r) ~@as(u64, 0) else 0 };
|
||||
}
|
||||
|
||||
fn lnot(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
// 0x92 0x93/94/95 are the compound comparisons.
|
||||
const b = current.peek() orelse return error.Truncated;
|
||||
switch (b) {
|
||||
opcode.lnot.not_equal => {
|
||||
current.i += 1;
|
||||
const x = try self.evaluateInteger(current, frame);
|
||||
const y = try self.evaluateInteger(current, frame);
|
||||
return .{ .integer = if (x != y) ~@as(u64, 0) else 0 };
|
||||
},
|
||||
opcode.lnot.less_equal => {
|
||||
current.i += 1;
|
||||
const x = try self.evaluateInteger(current, frame);
|
||||
const y = try self.evaluateInteger(current, frame);
|
||||
return .{ .integer = if (x <= y) ~@as(u64, 0) else 0 };
|
||||
},
|
||||
opcode.lnot.greater_equal => {
|
||||
current.i += 1;
|
||||
const x = try self.evaluateInteger(current, frame);
|
||||
const y = try self.evaluateInteger(current, frame);
|
||||
return .{ .integer = if (x >= y) ~@as(u64, 0) else 0 };
|
||||
},
|
||||
else => {
|
||||
const x = try self.evaluateInteger(current, frame);
|
||||
return .{ .integer = if (x == 0) ~@as(u64, 0) else 0 };
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
fn incDec(self: *Interpreter, current: *Cursor, frame: *Frame, delta: i64) Error!Object {
|
||||
// Operand is a SuperName that is both read and written.
|
||||
const save = current.i;
|
||||
const current_value = try self.term(current, frame);
|
||||
const v = try current_value.asInteger();
|
||||
const r = if (delta > 0) v +% 1 else v -% 1;
|
||||
var tcur = Cursor{ .b = current.b, .i = save };
|
||||
try self.storeInto(&tcur, frame, .{ .integer = r });
|
||||
return .{ .integer = r };
|
||||
}
|
||||
|
||||
fn sizeOf(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const o = try self.term(current, frame);
|
||||
return .{ .integer = switch (o) {
|
||||
.buffer => |b| b.len,
|
||||
.string => |s| s.len,
|
||||
.package => |p| p.len,
|
||||
else => 0,
|
||||
} };
|
||||
}
|
||||
|
||||
fn passThroughUnary(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const o = try self.term(current, frame);
|
||||
try self.storeTarget(current, frame, o);
|
||||
return o;
|
||||
}
|
||||
|
||||
fn index(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const source = try self.term(current, frame);
|
||||
const element_index: usize = @intCast(try self.evaluateInteger(current, frame));
|
||||
// Optional target (a reference); we don't materialise references, so store
|
||||
// the indexed value if a target is present.
|
||||
const value: Object = switch (source) {
|
||||
.buffer => |b| .{ .integer = if (element_index < b.len) b[element_index] else 0 },
|
||||
.package => |p| if (element_index < p.len) p[element_index] else .uninitialized,
|
||||
.string => |s| .{ .integer = if (element_index < s.len) s[element_index] else 0 },
|
||||
else => .uninitialized,
|
||||
};
|
||||
try self.storeTarget(current, frame, value);
|
||||
return value;
|
||||
}
|
||||
|
||||
fn dereferenceOf(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const o = try self.term(current, frame);
|
||||
return switch (o) {
|
||||
.reference => |n| self.invoke(n, &.{}),
|
||||
else => o,
|
||||
};
|
||||
}
|
||||
|
||||
// --- control flow -------------------------------------------------------
|
||||
|
||||
fn ifElse(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const start = current.i;
|
||||
const end = @min(start + try current.packageLength(), current.b.len);
|
||||
const cond = try self.evaluateInteger(current, frame);
|
||||
if (cond != 0) {
|
||||
var body = Cursor{ .b = current.b[0..end], .i = current.i };
|
||||
try self.executeList(&body, frame);
|
||||
current.i = end;
|
||||
// Skip a trailing Else.
|
||||
if (current.peek() == opcode.else_opcode) {
|
||||
current.i += 1;
|
||||
const es = current.i;
|
||||
const ee = @min(es + try current.packageLength(), current.b.len);
|
||||
current.i = ee;
|
||||
}
|
||||
} else {
|
||||
current.i = end;
|
||||
if (current.peek() == opcode.else_opcode) {
|
||||
current.i += 1;
|
||||
const es = current.i;
|
||||
const ee = @min(es + try current.packageLength(), current.b.len);
|
||||
var body = Cursor{ .b = current.b[0..ee], .i = current.i };
|
||||
try self.executeList(&body, frame);
|
||||
current.i = ee;
|
||||
}
|
||||
}
|
||||
return .uninitialized;
|
||||
}
|
||||
|
||||
fn whileLoop(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const start = current.i;
|
||||
const end = @min(start + try current.packageLength(), current.b.len);
|
||||
const pred_at = current.i;
|
||||
var guard: usize = 0;
|
||||
while (guard < 100_000) : (guard += 1) {
|
||||
var pc = Cursor{ .b = current.b[0..end], .i = pred_at };
|
||||
const cond = try self.evaluateInteger(&pc, frame);
|
||||
if (cond == 0) break;
|
||||
var body = Cursor{ .b = current.b[0..end], .i = pc.i };
|
||||
try self.executeList(&body, frame);
|
||||
if (frame.returned) break;
|
||||
if (frame.broke) {
|
||||
frame.broke = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
current.i = end;
|
||||
return .uninitialized;
|
||||
}
|
||||
|
||||
// --- store --------------------------------------------------------------
|
||||
|
||||
fn store(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const value = try self.term(current, frame);
|
||||
try self.storeInto(current, frame, value);
|
||||
return value;
|
||||
}
|
||||
|
||||
/// A Store *target* that may be NullName (no store).
|
||||
fn storeTarget(self: *Interpreter, current: *Cursor, frame: *Frame, value: Object) Error!void {
|
||||
if (current.peek() == 0x00) {
|
||||
current.i += 1; // NullName
|
||||
return;
|
||||
}
|
||||
try self.storeInto(current, frame, value);
|
||||
}
|
||||
|
||||
fn storeInto(self: *Interpreter, current: *Cursor, frame: *Frame, value: Object) Error!void {
|
||||
const lead = current.peek() orelse return error.Truncated;
|
||||
if (isNameStart(lead)) {
|
||||
const name_path = try current.nameString();
|
||||
const node = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice()) orelse return;
|
||||
if (self.fields.get(node)) |buffer_field| {
|
||||
try self.writeBufferField(buffer_field, try value.asInteger());
|
||||
} else if (node.kind == .field) {
|
||||
try self.writeField(node, try value.asInteger());
|
||||
} else {
|
||||
try self.dynamic_overrides.put(self.arena, node, value);
|
||||
}
|
||||
return;
|
||||
}
|
||||
_ = try current.byte();
|
||||
switch (lead) {
|
||||
0x00 => {}, // NullName
|
||||
opcode.local0_opcode...opcode.local7_opcode => frame.locals[lead - opcode.local0_opcode] = value,
|
||||
opcode.arg0_opcode...opcode.arg6_opcode => frame.args[lead - opcode.arg0_opcode] = value,
|
||||
opcode.index_opcode => {
|
||||
const source = try self.term(current, frame);
|
||||
const element_index: usize = @intCast(try self.evaluateInteger(current, frame));
|
||||
switch (source) {
|
||||
.buffer => |b| if (element_index < b.len) {
|
||||
b[element_index] = @truncate(try value.asInteger());
|
||||
},
|
||||
.package => |p| if (element_index < p.len) {
|
||||
p[element_index] = value;
|
||||
},
|
||||
else => {},
|
||||
}
|
||||
},
|
||||
else => return error.Unsupported,
|
||||
}
|
||||
}
|
||||
|
||||
// --- CreateField (buffer patching) --------------------------------------
|
||||
|
||||
fn createField(self: *Interpreter, current: *Cursor, frame: *Frame, bit_width: u32) Error!Object {
|
||||
const source = try self.term(current, frame); // source buffer (as a reference or value)
|
||||
const bit_index = try self.evaluateInteger(current, frame);
|
||||
const name_path = try current.nameString();
|
||||
const node = self.namespace.resolve(frame.scope, name_path.rooted, name_path.parents, name_path.slice()) orelse return .uninitialized;
|
||||
|
||||
// Bind the new name to the source buffer's node so stores land in it.
|
||||
const buffer_node: *Node = switch (source) {
|
||||
.reference => |n| n,
|
||||
else => return .uninitialized,
|
||||
};
|
||||
// Materialise the buffer into `dynamic_overrides` so patches persist and are returned.
|
||||
if (self.dynamic_overrides.get(buffer_node) == null) {
|
||||
const value = try self.invoke(buffer_node, &.{});
|
||||
try self.dynamic_overrides.put(self.arena, buffer_node, value);
|
||||
}
|
||||
const byte_off: usize = @intCast(bit_index / 8);
|
||||
try self.fields.put(self.arena, node, .{ .buffer = buffer_node, .byte_off = byte_off, .bit_width = bit_width });
|
||||
return .uninitialized;
|
||||
}
|
||||
|
||||
fn writeBufferField(self: *Interpreter, buffer_field: BufferField, value: u64) Error!void {
|
||||
const obj = self.dynamic_overrides.get(buffer_field.buffer) orelse return;
|
||||
const bytes = switch (obj) {
|
||||
.buffer => |b| b,
|
||||
else => return,
|
||||
};
|
||||
const byte_count = (buffer_field.bit_width + 7) / 8;
|
||||
var k: usize = 0;
|
||||
while (k < byte_count and buffer_field.byte_off + k < bytes.len) : (k += 1) {
|
||||
bytes[buffer_field.byte_off + k] = @truncate(value >> @intCast(k * 8));
|
||||
}
|
||||
}
|
||||
|
||||
// --- OperationRegion field access ---------------------------------------
|
||||
|
||||
fn readField(self: *Interpreter, field: *Node) Error!u64 {
|
||||
const region = field.region orelse return error.Unsupported;
|
||||
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
||||
const base = try self.regionBase(region);
|
||||
const start_byte = base + field.bit_offset / 8;
|
||||
const shift: u7 = @intCast(field.bit_offset % 8);
|
||||
const total = @as(usize, shift) + field.bit_width;
|
||||
const byte_count = (total + 7) / 8;
|
||||
var raw: u128 = 0;
|
||||
var k: usize = 0;
|
||||
while (k < byte_count) : (k += 1) {
|
||||
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
||||
}
|
||||
const masked = (raw >> shift) & bitMask(field.bit_width);
|
||||
return @truncate(masked);
|
||||
}
|
||||
|
||||
fn writeField(self: *Interpreter, field: *Node, value: u64) Error!void {
|
||||
const region = field.region orelse return error.Unsupported;
|
||||
if (field.bit_width == 0 or field.bit_width > 64) return error.Unsupported;
|
||||
const base = try self.regionBase(region);
|
||||
const start_byte = base + field.bit_offset / 8;
|
||||
const shift: u7 = @intCast(field.bit_offset % 8);
|
||||
const total = @as(usize, shift) + field.bit_width;
|
||||
const byte_count = (total + 7) / 8;
|
||||
// Read-modify-write byte by byte.
|
||||
var raw: u128 = 0;
|
||||
var k: usize = 0;
|
||||
while (k < byte_count) : (k += 1) {
|
||||
raw |= @as(u128, try self.readRegionByte(region.region_space, start_byte + k)) << @intCast(k * 8);
|
||||
}
|
||||
const mask = bitMask(field.bit_width) << shift;
|
||||
raw = (raw & ~mask) | ((@as(u128, value) << shift) & mask);
|
||||
k = 0;
|
||||
while (k < byte_count) : (k += 1) {
|
||||
try self.writeRegionByte(region.region_space, start_byte + k, @truncate(raw >> @intCast(k * 8)));
|
||||
}
|
||||
}
|
||||
|
||||
fn regionBase(self: *Interpreter, region: *Node) Error!u64 {
|
||||
var current = Cursor{ .b = region.region_offset_aml };
|
||||
var frame = Frame{ .scope = region.parent orelse self.namespace.root };
|
||||
return (try self.term(¤t, &frame)).asInteger();
|
||||
}
|
||||
|
||||
fn readRegionByte(self: *Interpreter, space: u8, address: u64) Error!u8 {
|
||||
switch (space) {
|
||||
0 => { // SystemMemory
|
||||
const virtual = self.hal.mapMmio(address & ~@as(u64, 0xFFF), 0x1000, true);
|
||||
const p: *align(1) const volatile u8 = @ptrFromInt(virtual + (address & 0xFFF));
|
||||
return p.*;
|
||||
},
|
||||
1 => return @truncate(self.hal.pioRead(1, @intCast(address & 0xFFFF))), // SystemIO
|
||||
else => return error.Unsupported,
|
||||
}
|
||||
}
|
||||
|
||||
fn writeRegionByte(self: *Interpreter, space: u8, address: u64, value: u8) Error!void {
|
||||
switch (space) {
|
||||
0 => {
|
||||
const virtual = self.hal.mapMmio(address & ~@as(u64, 0xFFF), 0x1000, true);
|
||||
const p: *align(1) volatile u8 = @ptrFromInt(virtual + (address & 0xFFF));
|
||||
p.* = value;
|
||||
},
|
||||
1 => self.hal.pioWrite(1, @intCast(address & 0xFFFF), value),
|
||||
else => return error.Unsupported,
|
||||
}
|
||||
}
|
||||
|
||||
// --- extended opcodes ---------------------------------------------------
|
||||
|
||||
fn ext(self: *Interpreter, current: *Cursor, frame: *Frame) Error!Object {
|
||||
const e = try current.byte();
|
||||
switch (e) {
|
||||
opcode.extended.debug => return .uninitialized,
|
||||
opcode.extended.revision => return .{ .integer = 2 },
|
||||
opcode.extended.timer => return .{ .integer = 0 },
|
||||
// Mutex/Event ops are no-ops in this single-threaded evaluator.
|
||||
opcode.extended.acquire => {
|
||||
_ = try self.term(current, frame); // mutex SuperName
|
||||
_ = try current.take(2); // timeout
|
||||
return .{ .integer = 0 }; // acquired
|
||||
},
|
||||
opcode.extended.release, opcode.extended.reset, opcode.extended.signal => {
|
||||
_ = try self.term(current, frame);
|
||||
return .uninitialized;
|
||||
},
|
||||
opcode.extended.wait => {
|
||||
_ = try self.term(current, frame);
|
||||
_ = try self.term(current, frame);
|
||||
return .{ .integer = 0 };
|
||||
},
|
||||
opcode.extended.sleep, opcode.extended.stall => {
|
||||
_ = try self.term(current, frame);
|
||||
return .uninitialized;
|
||||
},
|
||||
else => return error.Unsupported,
|
||||
}
|
||||
}
|
||||
|
||||
fn evaluateInteger(self: *Interpreter, current: *Cursor, frame: *Frame) Error!u64 {
|
||||
return (try self.term(current, frame)).asInteger();
|
||||
}
|
||||
};
|
||||
|
||||
fn bitMask(width: u32) u128 {
|
||||
if (width >= 128) return ~@as(u128, 0);
|
||||
return (@as(u128, 1) << @intCast(width)) - 1;
|
||||
}
|
||||
|
||||
fn isNameStart(b: u8) bool {
|
||||
return (b >= opcode.name_char_start and b <= opcode.name_char_end) or
|
||||
b == opcode.name_char_underscore or
|
||||
b == opcode.root_char or
|
||||
b == opcode.parent_prefix_char or
|
||||
b == opcode.dual_name_prefix or
|
||||
b == opcode.multi_name_prefix;
|
||||
}
|
||||
@@ -18,7 +18,7 @@ pub const NodeKind = enum {
|
||||
mutex,
|
||||
event,
|
||||
processor,
|
||||
power_res,
|
||||
power_resource,
|
||||
thermal_zone,
|
||||
alias,
|
||||
external,
|
||||
@@ -28,12 +28,12 @@ pub const NodeKind = enum {
|
||||
pub const Node = struct {
|
||||
/// The 4-byte NameSeg identifying this node within its parent. The root uses
|
||||
/// all-zero.
|
||||
seg: [4]u8 = .{ 0, 0, 0, 0 },
|
||||
segment: [4]u8 = .{ 0, 0, 0, 0 },
|
||||
kind: NodeKind = .other,
|
||||
/// For Method / External: the declared argument count (0..7). Used to resolve
|
||||
/// how many TermArgs a method invocation consumes.
|
||||
arg_count: u8 = 0,
|
||||
/// For Name: the AML bytes of its DataRefObject (so a value like a sleep
|
||||
/// For Name: the AML bytes of its DataReferenceObject (so a value like a sleep
|
||||
/// state's (`_Sx`) Package can be parsed on demand). For Method: the AML bytes of the body,
|
||||
/// interpreted on demand by the evaluator. Empty otherwise.
|
||||
value: []const u8 = &.{},
|
||||
@@ -78,49 +78,49 @@ pub const Namespace = struct {
|
||||
return self.root.subtreeCount();
|
||||
}
|
||||
|
||||
fn findChild(parent: *Node, seg: [4]u8) ?*Node {
|
||||
fn findChild(parent: *Node, segment: [4]u8) ?*Node {
|
||||
var c = parent.first_child;
|
||||
while (c) |child| : (c = child.next_sibling) {
|
||||
if (std.mem.eql(u8, &child.seg, &seg)) return child;
|
||||
if (std.mem.eql(u8, &child.segment, &segment)) return child;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// The direct child of `node` named `seg`, or null. Unlike `resolve`, this does
|
||||
/// The direct child of `node` named `segment`, or null. Unlike `resolve`, this does
|
||||
/// not apply the search-rule walk-up — it looks only at immediate children (for
|
||||
/// reading a device's own hardware ID (`_HID`) / current resource settings (`_CRS`)).
|
||||
pub fn childOf(node: *Node, seg: [4]u8) ?*Node {
|
||||
return findChild(node, seg);
|
||||
pub fn childOf(node: *Node, segment: [4]u8) ?*Node {
|
||||
return findChild(node, segment);
|
||||
}
|
||||
|
||||
fn newChild(self: *Namespace, parent: *Node, seg: [4]u8, kind: NodeKind) !*Node {
|
||||
fn newChild(self: *Namespace, parent: *Node, segment: [4]u8, kind: NodeKind) !*Node {
|
||||
const n = try self.allocator.create(Node);
|
||||
n.* = .{ .seg = seg, .kind = kind, .parent = parent };
|
||||
n.* = .{ .segment = segment, .kind = kind, .parent = parent };
|
||||
// Append at the tail so a dump reads in declaration order.
|
||||
if (parent.first_child == null) {
|
||||
parent.first_child = n;
|
||||
} else {
|
||||
var cur = parent.first_child.?;
|
||||
while (cur.next_sibling) |sib| cur = sib;
|
||||
cur.next_sibling = n;
|
||||
var current = parent.first_child.?;
|
||||
while (current.next_sibling) |sib| current = sib;
|
||||
current.next_sibling = n;
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
/// Create a Field unit node directly under `scope` (field units live in the
|
||||
/// scope of the Field/IndexField/BankField, not under the region).
|
||||
pub fn newFieldUnit(self: *Namespace, scope: *Node, seg: [4]u8) !*Node {
|
||||
return self.findOrCreate(scope, seg, .field);
|
||||
pub fn newFieldUnit(self: *Namespace, scope: *Node, segment: [4]u8) !*Node {
|
||||
return self.findOrCreate(scope, segment, .field);
|
||||
}
|
||||
|
||||
fn findOrCreate(self: *Namespace, parent: *Node, seg: [4]u8, kind: NodeKind) !*Node {
|
||||
if (findChild(parent, seg)) |existing| {
|
||||
fn findOrCreate(self: *Namespace, parent: *Node, segment: [4]u8, kind: NodeKind) !*Node {
|
||||
if (findChild(parent, segment)) |existing| {
|
||||
// Reopening a scope (e.g. Scope(\_SB) after Device \_SB) keeps the more
|
||||
// specific kind rather than downgrading to a plain scope.
|
||||
if (existing.kind == .scope and kind != .scope) existing.kind = kind;
|
||||
return existing;
|
||||
}
|
||||
return self.newChild(parent, seg, kind);
|
||||
return self.newChild(parent, segment, kind);
|
||||
}
|
||||
|
||||
/// The node a definition's NameString names, creating any intermediate scopes.
|
||||
@@ -131,16 +131,16 @@ pub const Namespace = struct {
|
||||
current: *Node,
|
||||
rooted: bool,
|
||||
parents: u8,
|
||||
segs: []const [4]u8,
|
||||
segments: []const [4]u8,
|
||||
kind: NodeKind,
|
||||
) !*Node {
|
||||
var base = startNode(self, current, rooted, parents);
|
||||
if (segs.len == 0) return base;
|
||||
if (segments.len == 0) return base;
|
||||
var i: usize = 0;
|
||||
while (i + 1 < segs.len) : (i += 1) {
|
||||
base = try self.findOrCreate(base, segs[i], .scope);
|
||||
while (i + 1 < segments.len) : (i += 1) {
|
||||
base = try self.findOrCreate(base, segments[i], .scope);
|
||||
}
|
||||
return self.findOrCreate(base, segs[segs.len - 1], kind);
|
||||
return self.findOrCreate(base, segments[segments.len - 1], kind);
|
||||
}
|
||||
|
||||
/// Resolve a NameString *reference* to an existing node, or null. A single
|
||||
@@ -151,22 +151,22 @@ pub const Namespace = struct {
|
||||
current: *Node,
|
||||
rooted: bool,
|
||||
parents: u8,
|
||||
segs: []const [4]u8,
|
||||
segments: []const [4]u8,
|
||||
) ?*Node {
|
||||
if (segs.len == 0) return null;
|
||||
if (segments.len == 0) return null;
|
||||
|
||||
if (!rooted and parents == 0 and segs.len == 1) {
|
||||
if (!rooted and parents == 0 and segments.len == 1) {
|
||||
// Search rule: this scope, then each ancestor up to the root.
|
||||
var scope: ?*Node = current;
|
||||
while (scope) |s| : (scope = s.parent) {
|
||||
if (findChild(s, segs[0])) |n| return n;
|
||||
if (findChild(s, segments[0])) |n| return n;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
var base = startNode(self, current, rooted, parents);
|
||||
for (segs) |seg| {
|
||||
base = findChild(base, seg) orelse return null;
|
||||
for (segments) |segment| {
|
||||
base = findChild(base, segment) orelse return null;
|
||||
}
|
||||
return base;
|
||||
}
|
||||
@@ -0,0 +1,137 @@
|
||||
//! AML opcode constants — the full ACPI Machine Language opcode table.
|
||||
//!
|
||||
//! Single-byte opcodes are plain values. Extended opcodes are a two-byte sequence
|
||||
//! `ext_prefix` (0x5B) followed by a byte listed under `ext`. A few comparison
|
||||
//! opcodes are `lnot_opcode` (0x92) followed by a second byte (see `lnot`).
|
||||
|
||||
// --- name / path characters -------------------------------------------------
|
||||
pub const zero_opcode = 0x00;
|
||||
pub const one_opcode = 0x01;
|
||||
pub const alias_opcode = 0x06;
|
||||
pub const name_opcode = 0x08;
|
||||
pub const byte_prefix = 0x0A;
|
||||
pub const word_prefix = 0x0B;
|
||||
pub const dword_prefix = 0x0C;
|
||||
pub const string_prefix = 0x0D;
|
||||
pub const qword_prefix = 0x0E;
|
||||
pub const scope_opcode = 0x10;
|
||||
pub const buffer_opcode = 0x11;
|
||||
pub const package_opcode = 0x12;
|
||||
pub const var_package_opcode = 0x13;
|
||||
pub const method_opcode = 0x14;
|
||||
pub const external_opcode = 0x15;
|
||||
|
||||
pub const dual_name_prefix = 0x2E;
|
||||
pub const multi_name_prefix = 0x2F;
|
||||
pub const extended_opcode_prefix = 0x5B;
|
||||
pub const root_char = 0x5C;
|
||||
pub const parent_prefix_char = 0x5E;
|
||||
pub const name_char_underscore = 0x5F;
|
||||
|
||||
pub const digit_char_start = 0x30;
|
||||
pub const digit_char_end = 0x39;
|
||||
pub const name_char_start = 0x41; // 'A'
|
||||
pub const name_char_end = 0x5A; // 'Z'
|
||||
|
||||
// --- locals / args ----------------------------------------------------------
|
||||
pub const local0_opcode = 0x60;
|
||||
pub const local7_opcode = 0x67;
|
||||
pub const arg0_opcode = 0x68;
|
||||
pub const arg6_opcode = 0x6E;
|
||||
|
||||
// --- store / references / arithmetic ---------------------------------------
|
||||
pub const store_opcode = 0x70;
|
||||
pub const ref_of_opcode = 0x71;
|
||||
pub const add_opcode = 0x72;
|
||||
pub const concat_opcode = 0x73;
|
||||
pub const subtract_opcode = 0x74;
|
||||
pub const increment_opcode = 0x75;
|
||||
pub const decrement_opcode = 0x76;
|
||||
pub const multiply_opcode = 0x77;
|
||||
pub const divide_opcode = 0x78;
|
||||
pub const shift_left_opcode = 0x79;
|
||||
pub const shift_right_opcode = 0x7A;
|
||||
pub const and_opcode = 0x7B;
|
||||
pub const nand_opcode = 0x7C;
|
||||
pub const or_opcode = 0x7D;
|
||||
pub const nor_opcode = 0x7E;
|
||||
pub const xor_opcode = 0x7F;
|
||||
pub const not_opcode = 0x80;
|
||||
pub const find_set_left_bit_opcode = 0x81;
|
||||
pub const find_set_right_bit_opcode = 0x82;
|
||||
pub const dereference_of_opcode = 0x83;
|
||||
pub const concat_resource_opcode = 0x84;
|
||||
pub const mod_opcode = 0x85;
|
||||
pub const notify_opcode = 0x86;
|
||||
pub const size_of_opcode = 0x87;
|
||||
pub const index_opcode = 0x88;
|
||||
pub const match_opcode = 0x89;
|
||||
pub const create_dword_field_opcode = 0x8A;
|
||||
pub const create_word_field_opcode = 0x8B;
|
||||
pub const create_byte_field_opcode = 0x8C;
|
||||
pub const create_bit_field_opcode = 0x8D;
|
||||
pub const object_type_opcode = 0x8E;
|
||||
pub const create_qword_field_opcode = 0x8F;
|
||||
|
||||
pub const land_opcode = 0x90;
|
||||
pub const lor_opcode = 0x91;
|
||||
pub const lnot_opcode = 0x92; // may be followed by a second byte (see `lnot`)
|
||||
pub const lequal_opcode = 0x93;
|
||||
pub const lgreater_opcode = 0x94;
|
||||
pub const lless_opcode = 0x95;
|
||||
pub const to_buffer_opcode = 0x96;
|
||||
pub const to_decimal_string_opcode = 0x97;
|
||||
pub const to_hex_string_opcode = 0x98;
|
||||
pub const to_integer_opcode = 0x99;
|
||||
pub const to_string_opcode = 0x9C;
|
||||
pub const copy_object_opcode = 0x9D;
|
||||
pub const mid_opcode = 0x9E;
|
||||
pub const continue_opcode = 0x9F;
|
||||
pub const if_opcode = 0xA0;
|
||||
pub const else_opcode = 0xA1;
|
||||
pub const while_opcode = 0xA2;
|
||||
pub const noop_opcode = 0xA3;
|
||||
pub const return_opcode = 0xA4;
|
||||
pub const break_opcode = 0xA5;
|
||||
pub const break_point_opcode = 0xCC;
|
||||
pub const ones_opcode = 0xFF;
|
||||
|
||||
/// Second bytes of the `lnot_opcode` (0x92) compound comparison opcodes.
|
||||
pub const lnot = struct {
|
||||
pub const not_equal = 0x93; // LNotEqualOp: 0x92 0x93
|
||||
pub const less_equal = 0x94; // LLessEqualOp: 0x92 0x94
|
||||
pub const greater_equal = 0x95; // LGreaterEqualOp: 0x92 0x95
|
||||
};
|
||||
|
||||
/// Second bytes of extended opcodes (prefixed by `extended_opcode_prefix`, 0x5B).
|
||||
pub const extended = struct {
|
||||
pub const mutex = 0x01;
|
||||
pub const event = 0x02;
|
||||
pub const conditional_reference_of = 0x12;
|
||||
pub const create_field = 0x13;
|
||||
pub const load_table = 0x1F;
|
||||
pub const load = 0x20;
|
||||
pub const stall = 0x21;
|
||||
pub const sleep = 0x22;
|
||||
pub const acquire = 0x23;
|
||||
pub const signal = 0x24;
|
||||
pub const wait = 0x25;
|
||||
pub const reset = 0x26;
|
||||
pub const release = 0x27;
|
||||
pub const from_bcd = 0x28;
|
||||
pub const to_bcd = 0x29;
|
||||
pub const unload = 0x2A;
|
||||
pub const revision = 0x30;
|
||||
pub const debug = 0x31;
|
||||
pub const fatal = 0x32;
|
||||
pub const timer = 0x33;
|
||||
pub const operation_region = 0x80;
|
||||
pub const field = 0x81;
|
||||
pub const device = 0x82;
|
||||
pub const processor = 0x83;
|
||||
pub const power_resource = 0x84;
|
||||
pub const thermal_zone = 0x85;
|
||||
pub const index_field = 0x86;
|
||||
pub const bank_field = 0x87;
|
||||
pub const data_region = 0x88;
|
||||
};
|
||||
@@ -0,0 +1,517 @@
|
||||
//! Recursive-descent AML parser. Walks the entire byte stream — including method
|
||||
//! bodies — building the ACPI namespace as it goes. It does not *evaluate*
|
||||
//! anything (no OperationRegion reads, no arithmetic); it parses structure so the
|
||||
//! cursor stays aligned and every named object is recorded.
|
||||
//!
|
||||
//! The one genuine ambiguity in AML is method invocation: a bare NameString in an
|
||||
//! operand position is a call whose argument count is only known from the method's
|
||||
//! (earlier) declaration. Because we build the namespace in the same in-order pass,
|
||||
//! `resolve` finds that declaration and tells us how many operands to consume.
|
||||
//!
|
||||
//! Safety net: every object delimited by a PkgLength (Scope/Device/Method/If/While/
|
||||
//! Field/Buffer/Package/…) is parsed within its known extent, and the cursor is
|
||||
//! snapped to that extent afterwards. So a mis-resolved invocation can only desync
|
||||
//! *within* one such object; the enclosing walk realigns at the boundary.
|
||||
|
||||
const std = @import("std");
|
||||
const opcode = @import("opcodes.zig");
|
||||
const Namespace = @import("namespace.zig").Namespace;
|
||||
const Node = @import("namespace.zig").Node;
|
||||
const NodeKind = @import("namespace.zig").NodeKind;
|
||||
|
||||
pub const Error = error{ Truncated, Malformed } || std.mem.Allocator.Error;
|
||||
|
||||
const maximum_segments = 64;
|
||||
|
||||
/// A parsed NameString: an optional root anchor or some parent hops, then a list
|
||||
/// of 4-byte segments.
|
||||
const NamePath = struct {
|
||||
rooted: bool = false,
|
||||
parents: u8 = 0,
|
||||
segments: [maximum_segments][4]u8 = undefined,
|
||||
count: usize = 0,
|
||||
|
||||
fn slice(self: *const NamePath) []const [4]u8 {
|
||||
return self.segments[0..self.count];
|
||||
}
|
||||
};
|
||||
|
||||
pub const Parser = struct {
|
||||
aml: []const u8,
|
||||
position: usize = 0,
|
||||
namespace: *Namespace,
|
||||
|
||||
pub fn init(aml: []const u8, namespace: *Namespace) Parser {
|
||||
return .{ .aml = aml, .namespace = namespace };
|
||||
}
|
||||
|
||||
/// Parse the whole block as a TermList under the namespace root. Returns the
|
||||
/// number of bytes consumed — equal to `aml.len` for a clean full traversal.
|
||||
pub fn parseAll(self: *Parser) usize {
|
||||
self.termList(self.aml.len, self.namespace.root);
|
||||
return self.position;
|
||||
}
|
||||
|
||||
// --- cursor primitives --------------------------------------------------
|
||||
|
||||
fn eof(self: *Parser) bool {
|
||||
return self.position >= self.aml.len;
|
||||
}
|
||||
|
||||
fn peek(self: *Parser) ?u8 {
|
||||
return if (self.eof()) null else self.aml[self.position];
|
||||
}
|
||||
|
||||
fn readByte(self: *Parser) Error!u8 {
|
||||
if (self.eof()) return error.Truncated;
|
||||
const b = self.aml[self.position];
|
||||
self.position += 1;
|
||||
return b;
|
||||
}
|
||||
|
||||
fn skip(self: *Parser, n: usize) Error!void {
|
||||
if (self.position + n > self.aml.len) return error.Truncated;
|
||||
self.position += n;
|
||||
}
|
||||
|
||||
fn skipCString(self: *Parser) Error!void {
|
||||
while (true) {
|
||||
const b = try self.readByte();
|
||||
if (b == 0) return;
|
||||
}
|
||||
}
|
||||
|
||||
/// AML PkgLength: the lead byte's top two bits give how many extra bytes
|
||||
/// follow; the value counts from the start of the PkgLength field.
|
||||
fn readPackageLength(self: *Parser) Error!usize {
|
||||
const lead = try self.readByte();
|
||||
const follow: usize = lead >> 6;
|
||||
if (follow == 0) return lead & 0x3F;
|
||||
var value: usize = lead & 0x0F;
|
||||
var i: usize = 0;
|
||||
while (i < follow) : (i += 1) {
|
||||
const b = try self.readByte();
|
||||
value |= @as(usize, b) << @intCast(4 + i * 8);
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
fn readNameSegment(self: *Parser) Error![4]u8 {
|
||||
if (self.position + 4 > self.aml.len) return error.Truncated;
|
||||
const segment = self.aml[self.position..][0..4].*;
|
||||
self.position += 4;
|
||||
return segment;
|
||||
}
|
||||
|
||||
fn readNameString(self: *Parser) Error!NamePath {
|
||||
var name_path = NamePath{};
|
||||
// A NameString is either root-anchored or parent-relative, not both.
|
||||
if (self.peek() == opcode.root_char) {
|
||||
name_path.rooted = true;
|
||||
self.position += 1;
|
||||
} else {
|
||||
while (self.peek() == opcode.parent_prefix_char) : (self.position += 1) name_path.parents += 1;
|
||||
}
|
||||
|
||||
const lead = self.peek() orelse return name_path;
|
||||
switch (lead) {
|
||||
0x00 => self.position += 1, // NullName
|
||||
opcode.dual_name_prefix => {
|
||||
self.position += 1;
|
||||
try self.appendSegment(&name_path);
|
||||
try self.appendSegment(&name_path);
|
||||
},
|
||||
opcode.multi_name_prefix => {
|
||||
self.position += 1;
|
||||
const count = try self.readByte();
|
||||
var i: usize = 0;
|
||||
while (i < count) : (i += 1) try self.appendSegment(&name_path);
|
||||
},
|
||||
else => {
|
||||
if (isNameStart(lead)) try self.appendSegment(&name_path);
|
||||
},
|
||||
}
|
||||
return name_path;
|
||||
}
|
||||
|
||||
fn appendSegment(self: *Parser, name_path: *NamePath) Error!void {
|
||||
const segment = try self.readNameSegment();
|
||||
if (name_path.count < maximum_segments) {
|
||||
name_path.segments[name_path.count] = segment;
|
||||
name_path.count += 1;
|
||||
}
|
||||
}
|
||||
|
||||
// --- term list / object -------------------------------------------------
|
||||
|
||||
/// Parse objects until `end`, then snap to `end`. Any parse error resyncs to
|
||||
/// the boundary rather than propagating — containment for the rare desync.
|
||||
fn termList(self: *Parser, end: usize, scope: *Node) void {
|
||||
while (self.position < end) {
|
||||
self.object(scope) catch break;
|
||||
}
|
||||
self.position = end;
|
||||
}
|
||||
|
||||
/// Parse exactly one object/term at the cursor. Used for both TermObjs and
|
||||
/// operands (TermArg / SuperName / Target all reduce to "one object" for the
|
||||
/// purpose of advancing the cursor).
|
||||
fn object(self: *Parser, scope: *Node) Error!void {
|
||||
const lead = self.peek() orelse return error.Truncated;
|
||||
if (isNameStart(lead)) return self.nameInvocation(scope);
|
||||
|
||||
_ = try self.readByte();
|
||||
switch (lead) {
|
||||
// constants and no-operand statements
|
||||
opcode.zero_opcode, opcode.one_opcode, opcode.ones_opcode => {},
|
||||
opcode.noop_opcode, opcode.continue_opcode, opcode.break_opcode, opcode.break_point_opcode => {},
|
||||
opcode.local0_opcode...opcode.local7_opcode => {},
|
||||
opcode.arg0_opcode...opcode.arg6_opcode => {},
|
||||
|
||||
// literal data
|
||||
opcode.byte_prefix => try self.skip(1),
|
||||
opcode.word_prefix => try self.skip(2),
|
||||
opcode.dword_prefix => try self.skip(4),
|
||||
opcode.qword_prefix => try self.skip(8),
|
||||
opcode.string_prefix => try self.skipCString(),
|
||||
|
||||
// data containers (contents skipped via their PkgLength)
|
||||
opcode.buffer_opcode, opcode.package_opcode, opcode.var_package_opcode => try self.skipPackage(),
|
||||
|
||||
// namespace modifiers / named objects
|
||||
opcode.name_opcode => try self.parseName(scope),
|
||||
opcode.alias_opcode => try self.parseAlias(scope),
|
||||
opcode.scope_opcode => try self.parseScopeLike(scope, .scope),
|
||||
opcode.method_opcode => try self.parseMethod(scope),
|
||||
opcode.external_opcode => try self.parseExternal(scope),
|
||||
opcode.extended_opcode_prefix => try self.parseExtended(scope),
|
||||
|
||||
// control flow
|
||||
opcode.if_opcode => try self.parseIf(scope),
|
||||
opcode.else_opcode => try self.parseElse(scope),
|
||||
opcode.while_opcode => try self.parseWhile(scope),
|
||||
opcode.return_opcode => try self.object(scope),
|
||||
opcode.notify_opcode => try self.args(scope, 2),
|
||||
|
||||
// stores / references / unary+target
|
||||
opcode.store_opcode => try self.args(scope, 2),
|
||||
opcode.ref_of_opcode, opcode.dereference_of_opcode, opcode.size_of_opcode, opcode.object_type_opcode => try self.args(scope, 1),
|
||||
opcode.increment_opcode, opcode.decrement_opcode => try self.args(scope, 1),
|
||||
opcode.not_opcode, opcode.find_set_left_bit_opcode, opcode.find_set_right_bit_opcode => try self.args(scope, 2),
|
||||
opcode.to_buffer_opcode, opcode.to_decimal_string_opcode, opcode.to_hex_string_opcode, opcode.to_integer_opcode => try self.args(scope, 2),
|
||||
opcode.copy_object_opcode => try self.args(scope, 2),
|
||||
|
||||
// binary + target
|
||||
opcode.add_opcode, opcode.subtract_opcode, opcode.multiply_opcode, opcode.mod_opcode => try self.args(scope, 3),
|
||||
opcode.and_opcode, opcode.nand_opcode, opcode.or_opcode, opcode.nor_opcode, opcode.xor_opcode => try self.args(scope, 3),
|
||||
opcode.shift_left_opcode, opcode.shift_right_opcode, opcode.concat_opcode, opcode.concat_resource_opcode, opcode.index_opcode => try self.args(scope, 3),
|
||||
opcode.divide_opcode => try self.args(scope, 4),
|
||||
opcode.to_string_opcode => try self.args(scope, 3),
|
||||
opcode.mid_opcode => try self.args(scope, 4),
|
||||
|
||||
// logical
|
||||
opcode.land_opcode, opcode.lor_opcode => try self.args(scope, 2),
|
||||
opcode.lequal_opcode, opcode.lgreater_opcode, opcode.lless_opcode => try self.args(scope, 2),
|
||||
opcode.lnot_opcode => try self.parseLnot(scope),
|
||||
|
||||
opcode.match_opcode => try self.parseMatch(scope),
|
||||
|
||||
// CreateXField: <source> <index> NameString
|
||||
opcode.create_dword_field_opcode,
|
||||
opcode.create_word_field_opcode,
|
||||
opcode.create_byte_field_opcode,
|
||||
opcode.create_bit_field_opcode,
|
||||
opcode.create_qword_field_opcode,
|
||||
=> try self.parseCreateField(scope, 2),
|
||||
|
||||
else => return error.Malformed,
|
||||
}
|
||||
}
|
||||
|
||||
/// Parse `n` operands.
|
||||
fn args(self: *Parser, scope: *Node, n: usize) Error!void {
|
||||
var i: usize = 0;
|
||||
while (i < n) : (i += 1) try self.object(scope);
|
||||
}
|
||||
|
||||
/// A NameString in operand/statement position: a method invocation (consuming
|
||||
/// the callee's declared argument count) or a plain name reference.
|
||||
fn nameInvocation(self: *Parser, scope: *Node) Error!void {
|
||||
const name_path = try self.readNameString();
|
||||
if (self.namespace.resolve(scope, name_path.rooted, name_path.parents, name_path.slice())) |node| {
|
||||
if ((node.kind == .method or node.kind == .external) and node.arg_count > 0) {
|
||||
try self.args(scope, node.arg_count);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Skip a PkgLength-delimited body wholesale (Buffer / Package / VarPackage):
|
||||
/// the contents are pure data, never namespace declarations.
|
||||
fn skipPackage(self: *Parser) Error!void {
|
||||
const start = self.position;
|
||||
const len = try self.readPackageLength();
|
||||
const end = start + len;
|
||||
if (end > self.aml.len) return error.Truncated;
|
||||
self.position = end;
|
||||
}
|
||||
|
||||
// --- namespace objects --------------------------------------------------
|
||||
|
||||
fn parseName(self: *Parser, scope: *Node) Error!void {
|
||||
const name_path = try self.readNameString();
|
||||
const value_start = self.position;
|
||||
try self.object(scope); // the DataReferenceObject value
|
||||
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .name);
|
||||
node.value = self.aml[value_start..self.position];
|
||||
}
|
||||
|
||||
fn parseAlias(self: *Parser, scope: *Node) Error!void {
|
||||
_ = try self.readNameString(); // source
|
||||
const name_path = try self.readNameString(); // the alias name
|
||||
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .alias);
|
||||
}
|
||||
|
||||
fn parseMethod(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.position;
|
||||
const end = start + try self.readPackageLength();
|
||||
const name_path = try self.readNameString();
|
||||
const flags = try self.readByte();
|
||||
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .method);
|
||||
node.arg_count = flags & 0x7;
|
||||
// Capture the body for on-demand evaluation and skip it — objects declared
|
||||
// inside a method are created at *runtime*, not at load, so they must not
|
||||
// become permanent namespace nodes.
|
||||
node.value = self.aml[self.position..@min(end, self.aml.len)];
|
||||
self.position = end;
|
||||
}
|
||||
|
||||
fn parseExternal(self: *Parser, scope: *Node) Error!void {
|
||||
const name_path = try self.readNameString();
|
||||
_ = try self.readByte(); // object type
|
||||
const arg_count = try self.readByte();
|
||||
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .external);
|
||||
node.arg_count = arg_count;
|
||||
}
|
||||
|
||||
/// Scope / Device / ThermalZone: PkgLength, NameString, then a nested TermList.
|
||||
fn parseScopeLike(self: *Parser, scope: *Node, kind: NodeKind) Error!void {
|
||||
const start = self.position;
|
||||
const end = start + try self.readPackageLength();
|
||||
const name_path = try self.readNameString();
|
||||
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), kind);
|
||||
self.termList(end, node);
|
||||
}
|
||||
|
||||
fn parseProcessor(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.position;
|
||||
const end = start + try self.readPackageLength();
|
||||
const name_path = try self.readNameString();
|
||||
try self.skip(6); // ProcID(byte) + PblkAddress(dword) + PblkLen(byte)
|
||||
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .processor);
|
||||
self.termList(end, node);
|
||||
}
|
||||
|
||||
fn parsePowerResource(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.position;
|
||||
const end = start + try self.readPackageLength();
|
||||
const name_path = try self.readNameString();
|
||||
try self.skip(3); // SystemLevel(byte) + ResourceOrder(word)
|
||||
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .power_resource);
|
||||
self.termList(end, node);
|
||||
}
|
||||
|
||||
/// OperationRegion: NameString, RegionSpace(byte), Offset(TermArg), Len(TermArg).
|
||||
/// The offset/length expressions are kept as AML for lazy evaluation.
|
||||
fn parseRegion(self: *Parser, scope: *Node) Error!void {
|
||||
const name_path = try self.readNameString();
|
||||
const space = try self.readByte();
|
||||
const off_start = self.position;
|
||||
try self.object(scope);
|
||||
const off_end = self.position;
|
||||
try self.object(scope);
|
||||
const len_end = self.position;
|
||||
const node = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .region);
|
||||
node.region_space = space;
|
||||
node.region_offset_aml = self.aml[off_start..off_end];
|
||||
node.region_len_aml = self.aml[off_end..len_end];
|
||||
}
|
||||
|
||||
fn parseDataRegion(self: *Parser, scope: *Node) Error!void {
|
||||
const name_path = try self.readNameString();
|
||||
try self.args(scope, 3); // signature, oem id, oem table id (TermArgs)
|
||||
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .region);
|
||||
}
|
||||
|
||||
fn parseMutex(self: *Parser, scope: *Node) Error!void {
|
||||
const name_path = try self.readNameString();
|
||||
try self.skip(1); // sync flags
|
||||
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .mutex);
|
||||
}
|
||||
|
||||
fn parseEvent(self: *Parser, scope: *Node) Error!void {
|
||||
const name_path = try self.readNameString();
|
||||
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .event);
|
||||
}
|
||||
|
||||
/// CreateXField: `count` TermArgs then the new field's NameString.
|
||||
fn parseCreateField(self: *Parser, scope: *Node, count: usize) Error!void {
|
||||
try self.args(scope, count);
|
||||
const name_path = try self.readNameString();
|
||||
_ = try self.namespace.place(scope, name_path.rooted, name_path.parents, name_path.slice(), .name);
|
||||
}
|
||||
|
||||
/// Field / IndexField / BankField: a region/bank reference, flags, then a
|
||||
/// FieldList whose NamedFields become nodes in the current scope. For a plain
|
||||
/// Field, the first NameString is the backing region — captured so field units
|
||||
/// carry a region + bit position the evaluator can read/write.
|
||||
fn parseField(self: *Parser, scope: *Node, name_strings: u8, bank: bool) Error!void {
|
||||
const start = self.position;
|
||||
const end = start + try self.readPackageLength();
|
||||
var region: ?*Node = null;
|
||||
var i: u8 = 0;
|
||||
while (i < name_strings) : (i += 1) {
|
||||
const name_path = try self.readNameString();
|
||||
// Only a plain Field's single NameString denotes an OperationRegion.
|
||||
if (name_strings == 1) region = self.namespace.resolve(scope, name_path.rooted, name_path.parents, name_path.slice());
|
||||
}
|
||||
if (bank) try self.object(scope); // bank value TermArg
|
||||
const flags = try self.readByte();
|
||||
self.fieldList(end, scope, region, flags & 0x0F);
|
||||
}
|
||||
|
||||
fn fieldList(self: *Parser, end: usize, scope: *Node, region: ?*Node, initial_access: u8) void {
|
||||
var bit_offset: u32 = 0;
|
||||
var access = initial_access;
|
||||
while (self.position < end) {
|
||||
const lead = self.peek() orelse break;
|
||||
switch (lead) {
|
||||
0x00 => { // ReservedField: advances the bit position
|
||||
self.position += 1;
|
||||
const width = self.readPackageLength() catch break;
|
||||
bit_offset += @intCast(width);
|
||||
},
|
||||
0x01 => { // AccessField: AccessType (low nibble) + AccessAttrib
|
||||
self.position += 1;
|
||||
const at = self.readByte() catch break;
|
||||
self.skip(1) catch break;
|
||||
access = at & 0x0F;
|
||||
},
|
||||
0x02 => { // ConnectField: NameString | BufferData
|
||||
self.position += 1;
|
||||
self.object(scope) catch break;
|
||||
},
|
||||
0x03 => { // ExtendedAccessField: type + attrib + length
|
||||
self.position += 1;
|
||||
self.skip(3) catch break;
|
||||
},
|
||||
else => { // NamedField: NameSegment + PkgLength (bit width)
|
||||
const segment = self.readNameSegment() catch break;
|
||||
const width = self.readPackageLength() catch break;
|
||||
const unit = self.namespace.newFieldUnit(scope, segment) catch break;
|
||||
unit.region = region;
|
||||
unit.bit_offset = bit_offset;
|
||||
unit.bit_width = @intCast(width);
|
||||
unit.access_type = access;
|
||||
bit_offset += @intCast(width);
|
||||
},
|
||||
}
|
||||
}
|
||||
self.position = end;
|
||||
}
|
||||
|
||||
// --- control flow -------------------------------------------------------
|
||||
|
||||
fn parseIf(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.position;
|
||||
const end = start + try self.readPackageLength();
|
||||
try self.object(scope); // predicate
|
||||
self.termList(end, scope);
|
||||
if (self.peek() == opcode.else_opcode) {
|
||||
self.position += 1;
|
||||
try self.parseElse(scope);
|
||||
}
|
||||
}
|
||||
|
||||
fn parseElse(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.position;
|
||||
const end = start + try self.readPackageLength();
|
||||
self.termList(end, scope);
|
||||
}
|
||||
|
||||
fn parseWhile(self: *Parser, scope: *Node) Error!void {
|
||||
const start = self.position;
|
||||
const end = start + try self.readPackageLength();
|
||||
try self.object(scope); // predicate
|
||||
self.termList(end, scope);
|
||||
}
|
||||
|
||||
fn parseLnot(self: *Parser, scope: *Node) Error!void {
|
||||
// 0x92 followed by 0x93/94/95 is a compound comparison (two operands);
|
||||
// otherwise it is a plain LNot of one operand.
|
||||
const b = self.peek() orelse return error.Truncated;
|
||||
switch (b) {
|
||||
opcode.lnot.not_equal, opcode.lnot.less_equal, opcode.lnot.greater_equal => {
|
||||
self.position += 1;
|
||||
try self.args(scope, 2);
|
||||
},
|
||||
else => try self.object(scope),
|
||||
}
|
||||
}
|
||||
|
||||
fn parseMatch(self: *Parser, scope: *Node) Error!void {
|
||||
try self.object(scope); // search package
|
||||
try self.skip(1); // match opcode 1
|
||||
try self.object(scope); // operand 1
|
||||
try self.skip(1); // match opcode 2
|
||||
try self.object(scope); // operand 2
|
||||
try self.object(scope); // start index
|
||||
}
|
||||
|
||||
// --- extended opcodes (0x5B xx) -----------------------------------------
|
||||
|
||||
fn parseExtended(self: *Parser, scope: *Node) Error!void {
|
||||
const e = try self.readByte();
|
||||
switch (e) {
|
||||
opcode.extended.mutex => try self.parseMutex(scope),
|
||||
opcode.extended.event => try self.parseEvent(scope),
|
||||
opcode.extended.operation_region => try self.parseRegion(scope),
|
||||
opcode.extended.data_region => try self.parseDataRegion(scope),
|
||||
opcode.extended.field => try self.parseField(scope, 1, false),
|
||||
opcode.extended.index_field => try self.parseField(scope, 2, false),
|
||||
opcode.extended.bank_field => try self.parseField(scope, 2, true),
|
||||
opcode.extended.device => try self.parseScopeLike(scope, .device),
|
||||
opcode.extended.thermal_zone => try self.parseScopeLike(scope, .thermal_zone),
|
||||
opcode.extended.processor => try self.parseProcessor(scope),
|
||||
opcode.extended.power_resource => try self.parsePowerResource(scope),
|
||||
|
||||
opcode.extended.conditional_reference_of => try self.args(scope, 2), // SuperName, Target
|
||||
opcode.extended.create_field => try self.parseCreateField(scope, 3),
|
||||
opcode.extended.load_table => try self.args(scope, 6),
|
||||
opcode.extended.load => try self.args(scope, 2), // NameString, Target
|
||||
opcode.extended.stall, opcode.extended.sleep => try self.args(scope, 1),
|
||||
opcode.extended.acquire => {
|
||||
try self.object(scope); // mutex SuperName
|
||||
try self.skip(2); // timeout WordData
|
||||
},
|
||||
opcode.extended.signal, opcode.extended.reset, opcode.extended.release, opcode.extended.unload => try self.args(scope, 1),
|
||||
opcode.extended.wait => try self.args(scope, 2),
|
||||
opcode.extended.from_bcd, opcode.extended.to_bcd => try self.args(scope, 2),
|
||||
opcode.extended.fatal => {
|
||||
try self.skip(5); // Type(byte) + Code(dword)
|
||||
try self.object(scope); // Arg TermArg
|
||||
},
|
||||
opcode.extended.revision, opcode.extended.debug, opcode.extended.timer => {},
|
||||
|
||||
else => return error.Malformed,
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
fn isNameStart(b: u8) bool {
|
||||
return (b >= opcode.name_char_start and b <= opcode.name_char_end) or
|
||||
b == opcode.name_char_underscore or
|
||||
b == opcode.root_char or
|
||||
b == opcode.parent_prefix_char or
|
||||
b == opcode.dual_name_prefix or
|
||||
b == opcode.multi_name_prefix;
|
||||
}
|
||||
@@ -0,0 +1,90 @@
|
||||
//! The **device ABI**: the flat, `extern` device types that cross the system_call
|
||||
//! boundary — what `device_enumerate` hands a user-space driver, what
|
||||
//! `device_register` takes back. This is the devices sub-project's *public
|
||||
//! interface*, exposed as its own `device-abi` module the same way the VFS server
|
||||
//! exposes `vfs-protocol` — so both the kernel and user space depend on the contract
|
||||
//! by name, and neither reaches into the other's files.
|
||||
//!
|
||||
//! It is also the **single source of truth** for `DeviceClass` and `ResourceKind`:
|
||||
//! the kernel's rich, pointer-based device tree (system/devices/device-model.zig,
|
||||
//! which user space must never import) re-exports these, so the enum that a driver
|
||||
//! matches on and the enum the kernel classifies with are the *same* type — no
|
||||
//! hand-kept "mirror in order" to drift. The core kernel↔user ABI is [[abi]]; the
|
||||
//! loader↔kernel handoff is [[boot-handoff]].
|
||||
|
||||
/// A coarse classification of a device, independent of the describing firmware.
|
||||
/// Kept small on purpose; refine as real drivers arrive. `enum(u32)` because the
|
||||
/// `@intFromEnum` value crosses the system_call boundary in `DeviceDescriptor.class`.
|
||||
pub const DeviceClass = enum(u32) {
|
||||
/// The synthetic root every discovered device hangs beneath.
|
||||
root,
|
||||
processor,
|
||||
interrupt_controller,
|
||||
timer,
|
||||
/// A PCI(e) host bridge — the root of a PCI segment (owns an ECAM window).
|
||||
pci_host_bridge,
|
||||
/// A single PCI function.
|
||||
pci_device,
|
||||
/// A device named in the ACPI namespace (from the DSDT/SSDT), carrying a
|
||||
/// hardware ID (`_HID`) and, where static, current resource settings (`_CRS`).
|
||||
acpi_device,
|
||||
/// The ACPI tables themselves, published as one node for the user-space acpi
|
||||
/// service (docs/m19-m20-plan.md M20): memory resources over the AML blobs,
|
||||
/// a broad io_port grant for OperationRegion access, and the SCI interrupt.
|
||||
/// The one node whose claimant is trusted to run firmware bytecode.
|
||||
acpi_tables,
|
||||
unknown,
|
||||
};
|
||||
|
||||
/// The kind of hardware resource a device occupies. `enum(u32)` for the same
|
||||
/// boundary-crossing reason as `DeviceClass` (see `ResourceDescriptor.kind`).
|
||||
pub const ResourceKind = enum(u32) {
|
||||
/// A memory-mapped I/O window: `start` is the physical base, `len` its size.
|
||||
memory,
|
||||
/// A legacy I/O-port range: `start` is the first port, `len` the count.
|
||||
io_port,
|
||||
/// An interrupt: `start` is the global system interrupt (GSI), `len` is 1.
|
||||
irq,
|
||||
/// A range of bus numbers owned by a bridge: `start`..`start+len`.
|
||||
bus_range,
|
||||
};
|
||||
|
||||
/// One device resource, as handed to a user-space driver (flat, extern).
|
||||
pub const ResourceDescriptor = extern struct {
|
||||
kind: u64, // a ResourceKind value
|
||||
start: u64,
|
||||
len: u64,
|
||||
};
|
||||
|
||||
pub const maximum_device_resources = 8;
|
||||
|
||||
/// `DeviceDescriptor.parent` for a device with no parent — a root of the device tree.
|
||||
pub const no_parent: u64 = ~@as(u64, 0);
|
||||
|
||||
/// `DeviceDescriptor.pci_class` for a device that is not a PCI function. (Zero would be
|
||||
/// ambiguous: 0x000000 is a real class code, "unclassified device".)
|
||||
pub const no_pci_class: u64 = ~@as(u64, 0);
|
||||
|
||||
/// A device, as snapshotted for user space by `device_enumerate`. A driver scans
|
||||
/// these to find the hardware it owns, claims it, and maps its MMIO.
|
||||
///
|
||||
/// `parent` makes the table a tree rather than a list, which is what a **bus driver**
|
||||
/// needs: it claims the bus, finds the devices below it, and publishes any it
|
||||
/// discovers itself with `device_register`. A registered child's resources must lie
|
||||
/// within its parent's (the kernel enforces this) — that containment is what makes
|
||||
/// delegation safe, since a device descriptor is otherwise a licence to map physical
|
||||
/// memory.
|
||||
pub const DeviceDescriptor = extern struct {
|
||||
id: u64,
|
||||
parent: u64, // a device id, or `no_parent`
|
||||
class: u64, // a DeviceClass value
|
||||
// The PCI class/subclass/prog-IF triple packed as 0xCCSSPP when this device is a PCI
|
||||
// function, or `no_pci_class` otherwise. This is how a manager tells *what* a
|
||||
// `pci_device` is (an xHCI controller, an AHCI controller) — decode the triple into
|
||||
// names with the pci-class module.
|
||||
pci_class: u64,
|
||||
hid_len: u64,
|
||||
resource_count: u64,
|
||||
hid: [8]u8,
|
||||
resources: [maximum_device_resources]ResourceDescriptor,
|
||||
};
|
||||
@@ -3,7 +3,7 @@
|
||||
//! Discovery backends (ACPI today, device-tree later) translate their native
|
||||
//! hardware description into this one shape, so the rest of the kernel walks a
|
||||
//! plain `Device` tree without knowing which firmware described the machine —
|
||||
//! the same discipline `root.zig`'s `MemoryKind` applies to memory and `arch`
|
||||
//! the same discipline `ps2-library.zig`'s `MemoryKind` applies to memory and `architecture`
|
||||
//! applies to the CPU.
|
||||
//!
|
||||
//! This is deliberately minimal: enough to *describe* what was discovered (a
|
||||
@@ -12,28 +12,27 @@
|
||||
//! on top of this — nothing here presumes them.
|
||||
|
||||
const std = @import("std");
|
||||
const device_abi = @import("device-abi");
|
||||
const pci_class = @import("pci-class");
|
||||
const acpi_ids = @import("acpi-ids");
|
||||
|
||||
/// The hardware primitives a discovery backend needs but can't express portably.
|
||||
/// The kernel injects an implementation (the arch VMM + port I/O), so the device
|
||||
/// layer touches hardware without importing `arch` — the same discipline that lets
|
||||
/// The kernel injects an implementation (the architecture VMM + port I/O), so the device
|
||||
/// layer touches hardware without importing `architecture` — the same discipline that lets
|
||||
/// it stay firmware-agnostic. `pioRead`/`pioWrite` take a width in bytes (1/2/4).
|
||||
pub const Hal = struct {
|
||||
mapMmio: *const fn (virt: u64, phys: u64, writable: bool) void,
|
||||
/// Map a physical MMIO range and return the virtual address to reach it at.
|
||||
/// The device layer dereferences the returned address and never learns how
|
||||
/// the kernel places it (identity, physmap, a window — the kernel's choice).
|
||||
mapMmio: *const fn (physical: u64, len: u64, writable: bool) u64,
|
||||
pioRead: *const fn (width: u8, port: u16) u32,
|
||||
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
||||
};
|
||||
|
||||
/// The kind of hardware resource a device occupies.
|
||||
pub const ResourceKind = enum {
|
||||
/// A memory-mapped I/O window: `start` is the physical base, `len` its size.
|
||||
memory,
|
||||
/// A legacy I/O-port range: `start` is the first port, `len` the count.
|
||||
io_port,
|
||||
/// An interrupt: `start` is the global system interrupt (GSI), `len` is 1.
|
||||
irq,
|
||||
/// A range of bus numbers owned by a bridge: `start`..`start+len`.
|
||||
bus_range,
|
||||
};
|
||||
/// The kind of hardware resource a device occupies. Canonically defined by the
|
||||
/// device ABI (system/devices/device-abi.zig) and re-exported here, so the kernel's
|
||||
/// internal tree and the descriptors it hands user space share one enum.
|
||||
pub const ResourceKind = device_abi.ResourceKind;
|
||||
|
||||
/// One hardware resource claimed by a device.
|
||||
pub const Resource = struct {
|
||||
@@ -43,22 +42,10 @@ pub const Resource = struct {
|
||||
};
|
||||
|
||||
/// A coarse classification of a device, independent of the describing firmware.
|
||||
/// Kept small on purpose; refine as real drivers arrive.
|
||||
pub const DeviceClass = enum {
|
||||
/// The synthetic root every discovered device hangs beneath.
|
||||
root,
|
||||
processor,
|
||||
interrupt_controller,
|
||||
timer,
|
||||
/// A PCI(e) host bridge — the root of a PCI segment (owns an ECAM window).
|
||||
pci_host_bridge,
|
||||
/// A single PCI function.
|
||||
pci_device,
|
||||
/// A device named in the ACPI namespace (from the DSDT/SSDT), carrying a
|
||||
/// hardware ID (`_HID`) and, where static, current resource settings (`_CRS`).
|
||||
acpi_device,
|
||||
unknown,
|
||||
};
|
||||
/// Canonically defined by the device ABI (system/devices/device-abi.zig) and
|
||||
/// re-exported here — the enum a driver matches on and the one the kernel classifies
|
||||
/// with are the same type. Kept small on purpose; refine as real drivers arrive.
|
||||
pub const DeviceClass = device_abi.DeviceClass;
|
||||
|
||||
/// Firmware-independent identity. Each backend fills only the fields it knows;
|
||||
/// the rest stay null. The generic layer never branches on *how* an id was
|
||||
@@ -71,29 +58,29 @@ pub const Ids = struct {
|
||||
pci_device: ?u16 = null,
|
||||
/// PCI class/subclass/prog-if packed as 0xCCSSPP.
|
||||
pci_class: ?u24 = null,
|
||||
/// PCI bus/device/function packed as (bus << 8) | (dev << 3) | func — the key
|
||||
/// PCI bus/device/function packed as (bus << 8) | (device << 3) | function — the key
|
||||
/// the ACPI address (`_ADR`) merge uses to match a namespace device to this node.
|
||||
pci_bdf: ?u16 = null,
|
||||
};
|
||||
|
||||
/// Upper bound on resources tracked per device (6 PCI BARs + a couple of IRQs is
|
||||
/// the busy case). Stored inline so a device is a single allocation.
|
||||
pub const max_resources = 8;
|
||||
pub const maximum_resources = 8;
|
||||
|
||||
/// One node in the device tree. Nodes are individually heap-allocated and linked
|
||||
/// intrusively (first-child / next-sibling), the classic device-tree layout —
|
||||
/// no per-node dynamic arrays to manage.
|
||||
pub const Device = struct {
|
||||
name_buf: [24]u8 = undefined,
|
||||
name_buffer: [24]u8 = undefined,
|
||||
name_len: u8 = 0,
|
||||
class: DeviceClass = .unknown,
|
||||
ids: Ids = .{},
|
||||
/// Human-readable hardware id (e.g. "PNP0A03"), when known. Backed inline like
|
||||
/// `name`; empty when unset. The generic layer stores/prints it without knowing
|
||||
/// how a backend encoded it.
|
||||
hid_buf: [8]u8 = undefined,
|
||||
hid_buffer: [8]u8 = undefined,
|
||||
hid_len: u8 = 0,
|
||||
resources: [max_resources]Resource = undefined,
|
||||
resources: [maximum_resources]Resource = undefined,
|
||||
resource_count: u8 = 0,
|
||||
|
||||
parent: ?*Device = null,
|
||||
@@ -103,30 +90,30 @@ pub const Device = struct {
|
||||
/// The device's short name (e.g. "cpu0", "pci0:00:1f.0"). Backed by an inline
|
||||
/// buffer, so it stays valid for the life of the node with no extra allocation.
|
||||
pub fn name(self: *const Device) []const u8 {
|
||||
return self.name_buf[0..self.name_len];
|
||||
return self.name_buffer[0..self.name_len];
|
||||
}
|
||||
|
||||
fn setName(self: *Device, s: []const u8) void {
|
||||
const n: u8 = @intCast(@min(s.len, self.name_buf.len));
|
||||
@memcpy(self.name_buf[0..n], s[0..n]);
|
||||
const n: u8 = @intCast(@min(s.len, self.name_buffer.len));
|
||||
@memcpy(self.name_buffer[0..n], s[0..n]);
|
||||
self.name_len = n;
|
||||
}
|
||||
|
||||
/// The device's hardware id string, or empty if none is set.
|
||||
pub fn hid(self: *const Device) []const u8 {
|
||||
return self.hid_buf[0..self.hid_len];
|
||||
return self.hid_buffer[0..self.hid_len];
|
||||
}
|
||||
|
||||
pub fn setHid(self: *Device, s: []const u8) void {
|
||||
const n: u8 = @intCast(@min(s.len, self.hid_buf.len));
|
||||
@memcpy(self.hid_buf[0..n], s[0..n]);
|
||||
const n: u8 = @intCast(@min(s.len, self.hid_buffer.len));
|
||||
@memcpy(self.hid_buffer[0..n], s[0..n]);
|
||||
self.hid_len = n;
|
||||
}
|
||||
|
||||
/// Record a resource. Silently drops beyond `max_resources` — discovery logs
|
||||
/// Record a resource. Silently drops beyond `maximum_resources` — discovery logs
|
||||
/// the truncation rather than failing the whole tree.
|
||||
pub fn addResource(self: *Device, kind: ResourceKind, start: u64, len: u64) bool {
|
||||
if (self.resource_count >= max_resources) return false;
|
||||
if (self.resource_count >= maximum_resources) return false;
|
||||
self.resources[self.resource_count] = .{ .kind = kind, .start = start, .len = len };
|
||||
self.resource_count += 1;
|
||||
return true;
|
||||
@@ -167,17 +154,17 @@ pub const DeviceTree = struct {
|
||||
self: *DeviceTree,
|
||||
parent: *Device,
|
||||
class: DeviceClass,
|
||||
dev_name: []const u8,
|
||||
device_name: []const u8,
|
||||
) !*Device {
|
||||
const d = try self.allocator.create(Device);
|
||||
d.* = .{ .class = class, .parent = parent };
|
||||
d.setName(dev_name);
|
||||
d.setName(device_name);
|
||||
if (parent.first_child == null) {
|
||||
parent.first_child = d;
|
||||
} else {
|
||||
var cur = parent.first_child.?;
|
||||
while (cur.next_sibling) |sib| cur = sib;
|
||||
cur.next_sibling = d;
|
||||
var current = parent.first_child.?;
|
||||
while (current.next_sibling) |sib| current = sib;
|
||||
current.next_sibling = d;
|
||||
}
|
||||
return d;
|
||||
}
|
||||
@@ -199,18 +186,37 @@ fn firstOfClassIn(node: *Device, class: DeviceClass) ?*Device {
|
||||
return null;
|
||||
}
|
||||
|
||||
fn dumpNode(dev: *const Device, depth: usize, emit: *const fn ([]const u8) void) void {
|
||||
fn dumpNode(device: *const Device, depth: usize, emit: *const fn ([]const u8) void) void {
|
||||
const indent = @min(depth * 2, 40);
|
||||
|
||||
var buf: [200]u8 = undefined;
|
||||
@memset(buf[0..indent], ' ');
|
||||
const body = if (dev.hid_len != 0)
|
||||
std.fmt.bufPrint(buf[indent..], "{s} [{s}] hid={s}\n", .{ dev.name(), @tagName(dev.class), dev.hid() }) catch return
|
||||
else
|
||||
std.fmt.bufPrint(buf[indent..], "{s} [{s}]\n", .{ dev.name(), @tagName(dev.class) }) catch return;
|
||||
emit(buf[0 .. indent + body.len]);
|
||||
var buffer: [200]u8 = undefined;
|
||||
@memset(buffer[0..indent], ' ');
|
||||
const body = if (device.hid_len != 0) blk: {
|
||||
// Decode the _HID to a human name when it's a known standard PnP/ACPI id.
|
||||
const desc = acpi_ids.description(device.hid());
|
||||
break :blk if (desc.len != 0)
|
||||
std.fmt.bufPrint(buffer[indent..], "{s} [{s}] hid={s} ({s})\n", .{ device.name(), @tagName(device.class), device.hid(), desc }) catch return
|
||||
else
|
||||
std.fmt.bufPrint(buffer[indent..], "{s} [{s}] hid={s}\n", .{ device.name(), @tagName(device.class), device.hid() }) catch return;
|
||||
} else std.fmt.bufPrint(buffer[indent..], "{s} [{s}]\n", .{ device.name(), @tagName(device.class) }) catch return;
|
||||
emit(buffer[0 .. indent + body.len]);
|
||||
|
||||
for (dev.resources[0..dev.resource_count]) |r| {
|
||||
// For a PCI function, decode its class code — the (class / subclass / prog-IF)
|
||||
// triple that says what it actually is, which the coarse `DeviceClass` can't.
|
||||
if (device.ids.pci_class) |packed_code| {
|
||||
const cc = pci_class.ClassCode.unpack(packed_code);
|
||||
var cbuf: [200]u8 = undefined;
|
||||
const pad = @min(indent + 2, 42);
|
||||
@memset(cbuf[0..pad], ' ');
|
||||
const pif = pci_class.progIfName(cc.base, cc.subclass, cc.prog_if);
|
||||
const cline = if (pif.len != 0)
|
||||
std.fmt.bufPrint(cbuf[pad..], "class 0x{x:0>2} ({s}) subclass 0x{x:0>2} ({s}) progif 0x{x:0>2} ({s})\n", .{ cc.base, pci_class.className(cc.base), cc.subclass, pci_class.subclassName(cc.base, cc.subclass), cc.prog_if, pif }) catch return
|
||||
else
|
||||
std.fmt.bufPrint(cbuf[pad..], "class 0x{x:0>2} ({s}) subclass 0x{x:0>2} ({s}) progif 0x{x:0>2}\n", .{ cc.base, pci_class.className(cc.base), cc.subclass, pci_class.subclassName(cc.base, cc.subclass), cc.prog_if }) catch return;
|
||||
emit(cbuf[0 .. pad + cline.len]);
|
||||
}
|
||||
|
||||
for (device.resources[0..device.resource_count]) |r| {
|
||||
var rbuf: [200]u8 = undefined;
|
||||
const pad = @min(indent + 2, 42);
|
||||
@memset(rbuf[0..pad], ' ');
|
||||
@@ -222,6 +228,6 @@ fn dumpNode(dev: *const Device, depth: usize, emit: *const fn ([]const u8) void)
|
||||
emit(rbuf[0 .. pad + rline.len]);
|
||||
}
|
||||
|
||||
var child = dev.first_child;
|
||||
var child = device.first_child;
|
||||
while (child) |c| : (child = c.next_sibling) dumpNode(c, depth + 1, emit);
|
||||
}
|
||||
@@ -7,10 +7,10 @@
|
||||
//! already routes to a backend rather than hard-coding ACPI — wiring the FDT
|
||||
//! parser in later is a local change here, not an architectural one.
|
||||
|
||||
const device = @import("device.zig");
|
||||
const device_model = @import("device-model.zig");
|
||||
|
||||
/// Populate `dt` from a device-tree blob. Not implemented yet.
|
||||
pub fn discover(dt: *device.DeviceTree) !void {
|
||||
_ = dt;
|
||||
/// Populate `device_tree` from a device-tree blob. Not implemented yet.
|
||||
pub fn discover(device_tree: *device_model.DeviceTree) !void {
|
||||
_ = device_tree;
|
||||
return error.Unsupported;
|
||||
}
|
||||
@@ -0,0 +1,261 @@
|
||||
//! PCI class-code decoding: turn the (class, subclass, prog-IF) triple a PCI function
|
||||
//! reports in its configuration header into human-readable names. Every PCI function
|
||||
//! carries a 24-bit class code — base class (config byte 0x0B), subclass (0x0A), and
|
||||
//! programming interface (0x09) — that says *what it is* far more precisely than
|
||||
//! danos's coarse `DeviceClass`: an ISA bridge, a SATA/AHCI controller, and an xHCI USB
|
||||
//! controller are all just `pci_device` by class, and only this triple tells them
|
||||
//! apart. Pure reference data (from the PCI spec; see https://wiki.osdev.org/PCI) — no
|
||||
//! hardware access — so it is shared by kernel discovery (the device-tree dump) and any
|
||||
//! user-space tool (a future lspci, driver matching).
|
||||
|
||||
/// The three bytes of a PCI class code, unpacked from the `0xCCSSPP` value discovery
|
||||
/// records in `Device.ids.pci_class` (CC = base class, SS = subclass, PP = prog-IF).
|
||||
pub const ClassCode = struct {
|
||||
base: u8, // class code (config offset 0x0B)
|
||||
subclass: u8, // subclass (0x0A)
|
||||
prog_if: u8, // programming interface (0x09)
|
||||
|
||||
pub fn unpack(packed_code: u24) ClassCode {
|
||||
return .{
|
||||
.base = @intCast((packed_code >> 16) & 0xFF),
|
||||
.subclass = @intCast((packed_code >> 8) & 0xFF),
|
||||
.prog_if = @intCast(packed_code & 0xFF),
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
/// Name of the base class (byte 0x0B), e.g. `0x06` -> "Bridge".
|
||||
pub fn className(base: u8) []const u8 {
|
||||
return switch (base) {
|
||||
0x00 => "Unclassified",
|
||||
0x01 => "Mass Storage Controller",
|
||||
0x02 => "Network Controller",
|
||||
0x03 => "Display Controller",
|
||||
0x04 => "Multimedia Controller",
|
||||
0x05 => "Memory Controller",
|
||||
0x06 => "Bridge",
|
||||
0x07 => "Simple Communication Controller",
|
||||
0x08 => "Base System Peripheral",
|
||||
0x09 => "Input Device Controller",
|
||||
0x0A => "Docking Station",
|
||||
0x0B => "Processor",
|
||||
0x0C => "Serial Bus Controller",
|
||||
0x0D => "Wireless Controller",
|
||||
0x0E => "Intelligent Controller",
|
||||
0x0F => "Satellite Communication Controller",
|
||||
0x10 => "Encryption Controller",
|
||||
0x11 => "Signal Processing Controller",
|
||||
0x12 => "Processing Accelerator",
|
||||
0x13 => "Non-Essential Instrumentation",
|
||||
0x40 => "Co-Processor",
|
||||
0xFF => "Unassigned Class (Vendor specific)",
|
||||
else => "Unknown",
|
||||
};
|
||||
}
|
||||
|
||||
/// Name of the subclass within its base class, e.g. `(0x06, 0x01)` -> "ISA Bridge".
|
||||
/// Subclass `0x80` is "Other" by PCI convention; anything unlisted is "Unknown".
|
||||
pub fn subclassName(base: u8, subclass: u8) []const u8 {
|
||||
return switch (base) {
|
||||
0x01 => switch (subclass) {
|
||||
0x00 => "SCSI Bus Controller",
|
||||
0x01 => "IDE Controller",
|
||||
0x02 => "Floppy Disk Controller",
|
||||
0x03 => "IPI Bus Controller",
|
||||
0x04 => "RAID Controller",
|
||||
0x05 => "ATA Controller",
|
||||
0x06 => "Serial ATA Controller",
|
||||
0x07 => "Serial Attached SCSI Controller",
|
||||
0x08 => "Non-Volatile Memory Controller",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x02 => switch (subclass) {
|
||||
0x00 => "Ethernet Controller",
|
||||
0x01 => "Token Ring Controller",
|
||||
0x02 => "FDDI Controller",
|
||||
0x03 => "ATM Controller",
|
||||
0x04 => "ISDN Controller",
|
||||
0x06 => "PICMG 2.14 Multi Computing Controller",
|
||||
0x07 => "Infiniband Controller",
|
||||
0x08 => "Fabric Controller",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x03 => switch (subclass) {
|
||||
0x00 => "VGA Compatible Controller",
|
||||
0x01 => "XGA Controller",
|
||||
0x02 => "3D Controller (Not VGA-Compatible)",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x04 => switch (subclass) {
|
||||
0x00 => "Multimedia Video Controller",
|
||||
0x01 => "Multimedia Audio Controller",
|
||||
0x02 => "Computer Telephony Device",
|
||||
0x03 => "Audio Device",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x05 => switch (subclass) {
|
||||
0x00 => "RAM Controller",
|
||||
0x01 => "Flash Controller",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x06 => switch (subclass) {
|
||||
0x00 => "Host Bridge",
|
||||
0x01 => "ISA Bridge",
|
||||
0x02 => "EISA Bridge",
|
||||
0x03 => "MCA Bridge",
|
||||
0x04 => "PCI-to-PCI Bridge",
|
||||
0x05 => "PCMCIA Bridge",
|
||||
0x06 => "NuBus Bridge",
|
||||
0x07 => "CardBus Bridge",
|
||||
0x08 => "RACEway Bridge",
|
||||
0x09 => "PCI-to-PCI Bridge (Semi-Transparent)",
|
||||
0x0A => "InfiniBand-to-PCI Host Bridge",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x07 => switch (subclass) {
|
||||
0x00 => "Serial Controller",
|
||||
0x01 => "Parallel Controller",
|
||||
0x02 => "Multiport Serial Controller",
|
||||
0x03 => "Modem",
|
||||
0x04 => "IEEE 488.1/2 (GPIB) Controller",
|
||||
0x05 => "Smart Card Controller",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x08 => switch (subclass) {
|
||||
0x00 => "PIC",
|
||||
0x01 => "DMA Controller",
|
||||
0x02 => "Timer",
|
||||
0x03 => "RTC Controller",
|
||||
0x04 => "PCI Hot-Plug Controller",
|
||||
0x05 => "SD Host Controller",
|
||||
0x06 => "IOMMU",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x09 => switch (subclass) {
|
||||
0x00 => "Keyboard Controller",
|
||||
0x01 => "Digitizer Pen",
|
||||
0x02 => "Mouse Controller",
|
||||
0x03 => "Scanner Controller",
|
||||
0x04 => "Gameport Controller",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x0C => switch (subclass) {
|
||||
0x00 => "FireWire (IEEE 1394) Controller",
|
||||
0x01 => "ACCESS Bus Controller",
|
||||
0x02 => "SSA",
|
||||
0x03 => "USB Controller",
|
||||
0x04 => "Fibre Channel",
|
||||
0x05 => "SMBus Controller",
|
||||
0x06 => "InfiniBand Controller",
|
||||
0x07 => "IPMI Interface",
|
||||
0x08 => "SERCOS Interface (IEC 61491)",
|
||||
0x09 => "CANbus Controller",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
0x0D => switch (subclass) {
|
||||
0x00 => "iRDA Compatible Controller",
|
||||
0x01 => "Consumer IR Controller",
|
||||
0x10 => "RF Controller",
|
||||
0x11 => "Bluetooth Controller",
|
||||
0x12 => "Broadband Controller",
|
||||
0x20 => "Ethernet Controller (802.1a)",
|
||||
0x21 => "Ethernet Controller (802.1b)",
|
||||
else => defaultSubclass(subclass),
|
||||
},
|
||||
else => defaultSubclass(subclass),
|
||||
};
|
||||
}
|
||||
|
||||
fn defaultSubclass(subclass: u8) []const u8 {
|
||||
return if (subclass == 0x80) "Other" else "Unknown";
|
||||
}
|
||||
|
||||
/// Name of the programming interface, for the subclasses that define standard ones
|
||||
/// (IDE modes, SATA/AHCI, NVMe, PCI-bridge decode, UART generation, USB host type).
|
||||
/// Returns "" when the prog-IF carries no standard meaning for this class/subclass —
|
||||
/// callers just print the hex byte in that case.
|
||||
pub fn progIfName(base: u8, subclass: u8, prog_if: u8) []const u8 {
|
||||
return switch (base) {
|
||||
0x01 => switch (subclass) {
|
||||
0x06 => switch (prog_if) { // Serial ATA
|
||||
0x00 => "Vendor Specific Interface",
|
||||
0x01 => "AHCI 1.0",
|
||||
0x02 => "Serial Storage Bus",
|
||||
else => "",
|
||||
},
|
||||
0x08 => switch (prog_if) { // Non-Volatile Memory
|
||||
0x01 => "NVMHCI",
|
||||
0x02 => "NVM Express",
|
||||
else => "",
|
||||
},
|
||||
else => "",
|
||||
},
|
||||
0x03 => switch (subclass) {
|
||||
0x00 => switch (prog_if) { // VGA Compatible
|
||||
0x00 => "VGA Controller",
|
||||
0x01 => "8514-Compatible Controller",
|
||||
else => "",
|
||||
},
|
||||
else => "",
|
||||
},
|
||||
0x06 => switch (subclass) {
|
||||
0x04 => switch (prog_if) { // PCI-to-PCI Bridge
|
||||
0x00 => "Normal Decode",
|
||||
0x01 => "Subtractive Decode",
|
||||
else => "",
|
||||
},
|
||||
else => "",
|
||||
},
|
||||
0x07 => switch (subclass) {
|
||||
0x00 => switch (prog_if) { // Serial Controller
|
||||
0x00 => "8250-Compatible (Generic XT)",
|
||||
0x01 => "16450-Compatible",
|
||||
0x02 => "16550-Compatible",
|
||||
0x03 => "16650-Compatible",
|
||||
0x04 => "16750-Compatible",
|
||||
0x05 => "16850-Compatible",
|
||||
0x06 => "16950-Compatible",
|
||||
else => "",
|
||||
},
|
||||
else => "",
|
||||
},
|
||||
0x0C => switch (subclass) {
|
||||
0x03 => switch (prog_if) { // USB Controller
|
||||
0x00 => "UHCI Controller",
|
||||
0x10 => "OHCI Controller",
|
||||
0x20 => "EHCI (USB2) Controller",
|
||||
0x30 => "XHCI (USB3) Controller",
|
||||
0x80 => "Unspecified",
|
||||
0xFE => "USB Device (not a host controller)",
|
||||
else => "",
|
||||
},
|
||||
else => "",
|
||||
},
|
||||
else => "",
|
||||
};
|
||||
}
|
||||
|
||||
test "decodes the common class codes" {
|
||||
const std = @import("std");
|
||||
const eq = std.testing.expectEqualStrings;
|
||||
|
||||
const isa = ClassCode.unpack(0x06_01_00);
|
||||
try std.testing.expectEqual(@as(u8, 0x06), isa.base);
|
||||
try std.testing.expectEqual(@as(u8, 0x01), isa.subclass);
|
||||
try eq("Bridge", className(isa.base));
|
||||
try eq("ISA Bridge", subclassName(isa.base, isa.subclass));
|
||||
|
||||
const ahci = ClassCode.unpack(0x01_06_01);
|
||||
try eq("Mass Storage Controller", className(ahci.base));
|
||||
try eq("Serial ATA Controller", subclassName(ahci.base, ahci.subclass));
|
||||
try eq("AHCI 1.0", progIfName(ahci.base, ahci.subclass, ahci.prog_if));
|
||||
|
||||
const xhci = ClassCode.unpack(0x0C_03_30);
|
||||
try eq("USB Controller", subclassName(xhci.base, xhci.subclass));
|
||||
try eq("XHCI (USB3) Controller", progIfName(xhci.base, xhci.subclass, xhci.prog_if));
|
||||
|
||||
// Unknowns and the "Other" convention.
|
||||
try eq("Other", subclassName(0x02, 0x80));
|
||||
try eq("Unknown", subclassName(0x06, 0x7E));
|
||||
try eq("", progIfName(0x06, 0x00, 0x00)); // host bridge: prog-IF has no standard name
|
||||
}
|
||||
@@ -0,0 +1,103 @@
|
||||
//! The firmware-agnostic discovery facade.
|
||||
//!
|
||||
//! The kernel calls `platform.discover()` and gets back a generic `DeviceTree`
|
||||
//! without ever naming ACPI or device-tree — the same way it imports `architecture`
|
||||
//! without naming x86_64. Which backend runs is decided *at runtime* from what
|
||||
//! the bootloader handed us (an ACPI RSDP today, a device-tree blob later),
|
||||
//! because a single image — a future ARM kernel especially — may boot under
|
||||
//! either firmware. That's a deliberate divergence from `architecture`, which is a
|
||||
//! compile-time choice.
|
||||
|
||||
const std = @import("std");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const device_model = @import("device-model.zig");
|
||||
const acpi = @import("acpi.zig");
|
||||
const power = @import("power.zig");
|
||||
const devicetree = @import("device-tree.zig");
|
||||
|
||||
pub const DeviceTree = device_model.DeviceTree;
|
||||
pub const Device = device_model.Device;
|
||||
pub const DeviceClass = device_model.DeviceClass;
|
||||
pub const Resource = device_model.Resource;
|
||||
pub const ResourceKind = device_model.ResourceKind;
|
||||
pub const Hal = device_model.Hal;
|
||||
pub const PowerInformation = acpi.PowerInformation;
|
||||
pub const AmlStats = acpi.AmlStats;
|
||||
pub const PlatformInformation = acpi.PlatformInformation;
|
||||
pub const RegisterAccess = acpi.RegisterAccess;
|
||||
pub const IsoEntry = acpi.IsoEntry;
|
||||
pub const Cpu = acpi.Cpu;
|
||||
|
||||
/// The register map + sleep types discovery extracted, for logging/diagnostics.
|
||||
pub fn powerInformation() PowerInformation {
|
||||
return acpi.power_information;
|
||||
}
|
||||
|
||||
/// The scalar firmware facts the architecture layer needs to avoid legacy assumptions
|
||||
/// (8259 presence, LAPIC base, PM timer, SPCR UART, IRQ overrides).
|
||||
pub fn platformInformation() PlatformInformation {
|
||||
return acpi.platform_information;
|
||||
}
|
||||
|
||||
/// AML parse integrity/diagnostics (namespace node count, bytes consumed).
|
||||
/// The number of Device objects in the kernel's own AML namespace, or 0 if the
|
||||
/// parse produced none — the `acpi-parse` test compares the ring-3 service's
|
||||
/// count against this.
|
||||
pub fn amlDeviceCount() usize {
|
||||
return acpi.amlDeviceCount();
|
||||
}
|
||||
|
||||
pub fn amlStats() AmlStats {
|
||||
return acpi.aml_stats;
|
||||
}
|
||||
|
||||
/// The usable logical processors discovered during enumeration — one entry per
|
||||
/// core danos may schedule on, each carrying the Local APIC ID an SMP wake targets.
|
||||
/// `len` is the hardware's degree of parallelism: how many tasks *could* run at the
|
||||
/// same instant once the application processors are started. Today only the
|
||||
/// bootstrap processor is actually running, so starting the rest is the pending SMP
|
||||
/// step (see docs/smp.md). Borrowed from static storage populated by `discover`.
|
||||
pub fn cpus() []const Cpu {
|
||||
return acpi.cpu_information.cpus[0..acpi.cpu_information.count];
|
||||
}
|
||||
|
||||
/// Non-zero only if enumeration found more processors than the static pool holds
|
||||
/// (the surplus were dropped from `cpus()`); surfaced so the cap is never silent.
|
||||
pub fn cpusDropped() usize {
|
||||
return acpi.cpu_information.dropped;
|
||||
}
|
||||
|
||||
/// Enumerate hardware into a fresh device tree. `hal` supplies the hardware
|
||||
/// primitives the backend needs (MMIO mapping for PCIe configuration space, port I/O for
|
||||
/// ACPI registers); pass the architecture implementation. Errors leave nothing to clean up
|
||||
/// beyond the tree's own allocations.
|
||||
pub fn discover(
|
||||
boot_information: *const boot_handoff.BootInformation,
|
||||
allocator: std.mem.Allocator,
|
||||
hal: Hal,
|
||||
) !DeviceTree {
|
||||
var device_tree = try DeviceTree.init(allocator);
|
||||
|
||||
if (boot_information.acpi_rsdp != 0) {
|
||||
const memory_regions = @as([*]const boot_handoff.MemoryRegion, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.memory_map.regions)))[0..boot_information.memory_map.len];
|
||||
try acpi.discover(boot_information.acpi_rsdp, memory_regions, &device_tree, hal);
|
||||
} else {
|
||||
// No ACPI RSDP. A device-tree boot would parse its blob here; today that
|
||||
// path is a stub, so this reports the machine described itself no way we
|
||||
// understand yet.
|
||||
try devicetree.discover(&device_tree);
|
||||
}
|
||||
|
||||
return device_tree;
|
||||
}
|
||||
|
||||
/// Restart the machine. Never returns on success; returns only if no reset method
|
||||
/// worked (extremely unlikely). Backend-agnostic entry the kernel calls.
|
||||
pub fn reboot(hal: Hal) void {
|
||||
power.reboot(hal);
|
||||
}
|
||||
|
||||
/// Power the machine off (ACPI S5). Never returns on success.
|
||||
pub fn shutdown(hal: Hal) void {
|
||||
power.shutdown(hal);
|
||||
}
|
||||
@@ -9,8 +9,8 @@
|
||||
//! re-initialisation, a milestone of its own.
|
||||
|
||||
const acpi = @import("acpi.zig");
|
||||
const device = @import("device.zig");
|
||||
const Hal = device.Hal;
|
||||
const device_model = @import("device-model.zig");
|
||||
const Hal = device_model.Hal;
|
||||
|
||||
const slp_en: u32 = 1 << 13; // SLP_EN: writing 1 triggers the sleep transition
|
||||
const sci_en: u32 = 1 << 0; // SCI_EN in PM1 control: set once ACPI mode is active
|
||||
@@ -19,27 +19,27 @@ const sci_en: u32 = 1 << 0; // SCI_EN in PM1 control: set once ACPI mode is acti
|
||||
/// register is live. A no-op when the firmware exposes no SMI command port (ACPI
|
||||
/// already enabled, as under QEMU/OVMF) — we still verify SCI_EN first.
|
||||
pub fn enable(hal: Hal) void {
|
||||
const pi = acpi.power_info;
|
||||
const pi = acpi.power_information;
|
||||
if (!pi.pm1a_cnt.present()) return;
|
||||
if (readReg(hal, pi.pm1a_cnt) & sci_en != 0) return; // already in ACPI mode
|
||||
if (readRegister(hal, pi.pm1a_cnt) & sci_en != 0) return; // already in ACPI mode
|
||||
if (pi.smi_cmd == 0 or pi.acpi_enable == 0) return; // no way to switch; assume fine
|
||||
|
||||
hal.pioWrite(1, pi.smi_cmd, pi.acpi_enable);
|
||||
var spins: usize = 0;
|
||||
while (readReg(hal, pi.pm1a_cnt) & sci_en == 0 and spins < 1_000_000) : (spins += 1) {}
|
||||
while (readRegister(hal, pi.pm1a_cnt) & sci_en == 0 and spins < 1_000_000) : (spins += 1) {}
|
||||
}
|
||||
|
||||
/// Restart the machine. Tries the ACPI reset register first, then the two legacy
|
||||
/// fallbacks. Returns only if every method failed (very unlikely).
|
||||
pub fn reboot(hal: Hal) void {
|
||||
const pi = acpi.power_info;
|
||||
const pi = acpi.power_information;
|
||||
|
||||
// 1. The FADT reset register, when the firmware advertises support.
|
||||
if (pi.reset_supported and pi.reset.present()) {
|
||||
writeReg(hal, pi.reset, pi.reset_value);
|
||||
writeRegister(hal, pi.reset, pi.reset_value);
|
||||
delay();
|
||||
}
|
||||
// 2. The PCI reset-control register at port 0xCF9 (RST_CPU | SYS_RST).
|
||||
// 2. The PCI reset-control register at port 0xCF9 (RST_CPU | SYSTEM_RST).
|
||||
hal.pioWrite(1, 0xCF9, 0x0E);
|
||||
hal.pioWrite(1, 0xCF9, 0x06);
|
||||
delay();
|
||||
@@ -52,14 +52,14 @@ pub fn reboot(hal: Hal) void {
|
||||
/// it wasn't found in the AML, there is nothing safe to do and this returns.
|
||||
pub fn shutdown(hal: Hal) void {
|
||||
enable(hal);
|
||||
const pi = acpi.power_info;
|
||||
const pi = acpi.power_information;
|
||||
const s5 = pi.s5 orelse return;
|
||||
|
||||
if (pi.pm1a_cnt.present()) {
|
||||
writeReg(hal, pi.pm1a_cnt, sleepValue(s5.slp_typ_a));
|
||||
writeRegister(hal, pi.pm1a_cnt, sleepValue(s5.slp_typ_a));
|
||||
}
|
||||
if (pi.pm1b_cnt.present()) {
|
||||
writeReg(hal, pi.pm1b_cnt, sleepValue(s5.slp_typ_b));
|
||||
writeRegister(hal, pi.pm1b_cnt, sleepValue(s5.slp_typ_b));
|
||||
}
|
||||
delay();
|
||||
}
|
||||
@@ -76,27 +76,25 @@ fn sleepValue(slp_typ: u8) u32 {
|
||||
return (@as(u32, slp_typ & 0x7) << 10) | slp_en;
|
||||
}
|
||||
|
||||
fn readReg(hal: Hal, reg: acpi.RegAccess) u32 {
|
||||
if (reg.mmio) {
|
||||
hal.mapMmio(reg.address, reg.address, true);
|
||||
const p: *align(1) volatile u32 = @ptrFromInt(reg.address);
|
||||
fn readRegister(hal: Hal, register: acpi.RegisterAccess) u32 {
|
||||
if (register.mmio) {
|
||||
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true));
|
||||
return p.*;
|
||||
}
|
||||
return hal.pioRead(reg.width, @intCast(reg.address));
|
||||
return hal.pioRead(register.width, @intCast(register.address));
|
||||
}
|
||||
|
||||
fn writeReg(hal: Hal, reg: acpi.RegAccess, value: u32) void {
|
||||
if (reg.mmio) {
|
||||
hal.mapMmio(reg.address, reg.address, true);
|
||||
const p: *align(1) volatile u32 = @ptrFromInt(reg.address);
|
||||
fn writeRegister(hal: Hal, register: acpi.RegisterAccess, value: u32) void {
|
||||
if (register.mmio) {
|
||||
const p: *align(1) volatile u32 = @ptrFromInt(hal.mapMmio(register.address, 4, true));
|
||||
p.* = value;
|
||||
} else {
|
||||
hal.pioWrite(reg.width, @intCast(reg.address), value);
|
||||
hal.pioWrite(register.width, @intCast(register.address), value);
|
||||
}
|
||||
}
|
||||
|
||||
/// A short busy-wait so a reset/power-off takes effect before we fall through to
|
||||
/// the next method. The empty asm is an arch-neutral barrier that keeps the loop
|
||||
/// the next method. The empty asm is an architecture-neutral barrier that keeps the loop
|
||||
/// from being optimised away.
|
||||
fn delay() void {
|
||||
var i: usize = 0;
|
||||
@@ -0,0 +1,857 @@
|
||||
//! USB device-framework wire ABI: the set-up packets, standard requests, and standard
|
||||
//! descriptors every USB device speaks over its default control pipe, as defined by chapter 9
|
||||
//! of the USB 2.0 specification (see https://wiki.osdev.org/Universal_Serial_Bus). Pure data
|
||||
//! definitions — no hardware access — shared by the host-controller bus drivers (which build
|
||||
//! the requests) and anything that parses what devices return (device naming, driver
|
||||
//! matching, configuration). The structs mirror the wire byte-for-byte: multi-byte fields are
|
||||
//! little-endian and align(1), so a descriptor can be bit-cast straight out of a transfer
|
||||
//! buffer at any offset, and bitmap bytes are packed structs so no caller ever needs a magic
|
||||
//! mask. Class, subclass, and protocol code tables live in usb-ids.zig.
|
||||
|
||||
const DeviceState = enum(u8) {
|
||||
// Immediately after the USB device is attached to the USB system, it is in this state.
|
||||
// The USB specifications do not define the state of a USB device that is detached from
|
||||
// a USB system.
|
||||
attached,
|
||||
// A device is in this state after it has both been attached to the bus, and the VBUS line is
|
||||
// applied to the device (the host controller drives the VBUS at +5V, however this is only
|
||||
// particularly important for hardware developers). In this state, the device must not respond
|
||||
// to any bus transactions. The USB specification recognizes three potential scenarios with
|
||||
// respect to how a device draws power:
|
||||
// - Self-Powered Devices draw power from an external power source (e.g, a USB printer plugs
|
||||
// into the wall as well as a USB port). Although the device may be considered
|
||||
// technically "powered" even before attachment to the USB, it is still only considered
|
||||
// powered after the VBUS line is applied to the device.
|
||||
// - Bus-Powered Devices draw power solely from the USB up to 100mA.
|
||||
// - Self- or Bus-Powered Devices may draw power from either the bus or an external power
|
||||
// source, depending on the configuration. These devices may change power source at any
|
||||
// time. If a device is currently self-powered and requires more than 100mA of power, but
|
||||
// switches to being bus-powered, then the device must return to the Address state.
|
||||
powered,
|
||||
// A device in the powered state enters the default state after receiving a bus reset. In this
|
||||
// state, the device is addressable at the default, reserved address of 0. At this point, the
|
||||
// device is operating at the correct speed. The host is expected to allow 10 milliseconds
|
||||
// before expecting the device to respond to data transfers after reset.
|
||||
default,
|
||||
// A device enters this state after the host assigns it an address via the default control pipe,
|
||||
// which is always accessible whether the device's address has been set or not.
|
||||
address,
|
||||
// A device is in this state after the host examines its possible configurations and selects
|
||||
// one. All endpoint's data toggle bits are initialized to zero when a device enters this state.
|
||||
configured,
|
||||
// When no traffic is observed on the bus for a period of 1 millisecond, a USB device enters
|
||||
// this state, characterized by its low power consumption. The device's address and
|
||||
// configuration settings are maintained while suspended. A device exits the suspended state as
|
||||
// soon as it begins seeing bus activity again. The host is expected to allow 10 milliseconds
|
||||
// before expecting the device to respond to data transfers after resume.
|
||||
suspended,
|
||||
};
|
||||
|
||||
const RequestCode = enum(u8) {
|
||||
get_status = 0,
|
||||
clear_feature = 1,
|
||||
set_feature = 3,
|
||||
set_address = 5,
|
||||
get_descriptor = 6,
|
||||
set_descriptor = 7,
|
||||
get_configuration = 8,
|
||||
set_configuration = 9,
|
||||
get_interface = 10,
|
||||
set_interface = 11,
|
||||
sync_frame = 12,
|
||||
};
|
||||
|
||||
// Direction of an endpoint, from the host's point of view
|
||||
const EndpointDirection = enum(u1) {
|
||||
out = 0,
|
||||
in = 1,
|
||||
};
|
||||
|
||||
// Identifier newtypes: distinct wire-sized types for values that identify something on the
|
||||
// device rather than count something. Each is a non-exhaustive enum whose values originate
|
||||
// in the descriptors below and flow, still typed, into the standard request constructors —
|
||||
// so an interface number can never be passed where a configuration value is expected.
|
||||
|
||||
// The bus address of a device, assigned by the host with SET_ADDRESS. Addresses are 7 bits
|
||||
// wide.
|
||||
const DeviceAddress = enum(u7) {
|
||||
// The default address every device answers at after a reset, until SET_ADDRESS
|
||||
// completes
|
||||
default = 0,
|
||||
_,
|
||||
};
|
||||
|
||||
// Identifies a configuration; from ConfigurationDescriptor.configuration_value.
|
||||
const ConfigurationValue = enum(u8) {
|
||||
// Not configured: returned by GET_CONFIGURATION while the device is in the address
|
||||
// state, and passed to SET_CONFIGURATION to return a configured device to the address
|
||||
// state
|
||||
none = 0,
|
||||
_,
|
||||
};
|
||||
|
||||
// Identifies an interface within a configuration; from
|
||||
// InterfaceDescriptor.interface_number.
|
||||
const InterfaceNumber = enum(u8) { _ };
|
||||
|
||||
// Selects between the alternate settings of one interface; from
|
||||
// InterfaceDescriptor.alternate_setting.
|
||||
const AlternateSetting = enum(u8) {
|
||||
// The default setting of an interface
|
||||
default = 0,
|
||||
_,
|
||||
};
|
||||
|
||||
// The number of an endpoint within a device, 4 bits wide. The direction bit carried
|
||||
// alongside it tells the two endpoints sharing a number apart.
|
||||
const EndpointNumber = enum(u4) {
|
||||
// Endpoint zero: the default control pipe every device provides
|
||||
default_control = 0,
|
||||
_,
|
||||
};
|
||||
|
||||
// Index of a STRING descriptor, stored in descriptors that reference a string and passed to
|
||||
// GET_DESCRIPTOR to read it.
|
||||
const StringIndex = enum(u8) {
|
||||
// The device has no string descriptor for this field
|
||||
none = 0,
|
||||
_,
|
||||
};
|
||||
|
||||
// Characteristics of a device request (the bmRequestType field of a set-up packet). Fields are
|
||||
// declared least-significant first: recipient occupies bits 4...0, kind bits 6...5, and
|
||||
// direction bit 7.
|
||||
const RequestType = packed struct(u8) {
|
||||
// The recipient of the request (values 4...31 are reserved)
|
||||
recipient: Recipient,
|
||||
// The type of the request
|
||||
kind: Kind,
|
||||
// Data transfer direction. The value of this bit is ignored when length is zero.
|
||||
direction: Direction,
|
||||
|
||||
const Recipient = enum(u5) {
|
||||
device = 0,
|
||||
interface = 1,
|
||||
endpoint = 2,
|
||||
other = 3,
|
||||
};
|
||||
|
||||
const Kind = enum(u2) {
|
||||
standard = 0,
|
||||
class = 1,
|
||||
vendor = 2,
|
||||
reserved = 3,
|
||||
};
|
||||
|
||||
const Direction = enum(u1) {
|
||||
host_to_device = 0,
|
||||
device_to_host = 1,
|
||||
};
|
||||
};
|
||||
|
||||
const Request = extern struct {
|
||||
// Characteristics of the request
|
||||
request_type: RequestType,
|
||||
// Specific request
|
||||
request_code: RequestCode,
|
||||
// Word-sized field that may (or may not) serve as a parameter to the request, depending
|
||||
// on the specific request. For GET_DESCRIPTOR and SET_DESCRIPTOR, bit-cast a
|
||||
// DescriptorValue into this field.
|
||||
value: u16 align(1),
|
||||
// Word-sized field that may (or may not) serve as a parameter to the request, depending
|
||||
// on the specific request. Typically this field holds an index or an offset value. When
|
||||
// request_type specifies an endpoint or an interface as the recipient, bit-cast an
|
||||
// EndpointIndex or an InterfaceIndex into this field.
|
||||
index: u16 align(1),
|
||||
// Number of bytes to transfer if there is a DATA stage.
|
||||
// - If this field is non-zero, and request_type indicates a transfer from
|
||||
// device-to-host, then the device must never return more than length bytes of data.
|
||||
// However, a device may return less.
|
||||
// - If this field is non-zero, and request_type indicates a transfer from
|
||||
// host-to-device, then the host must send exactly length bytes of data. If the host
|
||||
// sends more than length bytes, the behavior of the device is undefined.
|
||||
length: u16 align(1),
|
||||
|
||||
// The format of the index field when request_type specifies an endpoint as the
|
||||
// recipient. The host should always set the direction bit to zero (but the device
|
||||
// should accept either value) when the endpoint is part of a control pipe.
|
||||
const EndpointIndex = packed struct(u16) {
|
||||
// Endpoint number
|
||||
number: EndpointNumber,
|
||||
// Reserved (reset to zero)
|
||||
reserved: u3 = 0,
|
||||
// Selects the OUT or the IN endpoint with the specified endpoint number
|
||||
direction: EndpointDirection,
|
||||
// Reserved (reset to zero)
|
||||
reserved_high: u8 = 0,
|
||||
};
|
||||
|
||||
// The format of the index field when request_type specifies an interface as the
|
||||
// recipient.
|
||||
const InterfaceIndex = packed struct(u16) {
|
||||
// Interface number
|
||||
number: u8,
|
||||
// Reserved (reset to zero)
|
||||
reserved: u8 = 0,
|
||||
};
|
||||
|
||||
// The format of the value field of GET_DESCRIPTOR and SET_DESCRIPTOR requests: the
|
||||
// descriptor type in the high byte, and the descriptor index in the low byte. The index
|
||||
// is used to select a specific descriptor (only for CONFIGURATION and STRING
|
||||
// descriptors) when several descriptors of that type are implemented by a device.
|
||||
const DescriptorValue = packed struct(u16) {
|
||||
// Descriptor index
|
||||
index: u8 = 0,
|
||||
// Descriptor type
|
||||
kind: DescriptorType,
|
||||
};
|
||||
};
|
||||
|
||||
// Feature selectors, used as the value field of CLEAR_FEATURE and SET_FEATURE requests. The
|
||||
// comment on each value notes the recipient the selector applies to.
|
||||
const FeatureSelector = enum(u16) {
|
||||
// Halts an endpoint (recipient: endpoint)
|
||||
endpoint_halt = 0,
|
||||
// Enables or disables the device's remote wakeup capability (recipient: device)
|
||||
device_remote_wakeup = 1,
|
||||
// Puts a hi-speed device into a test mode, selected by a TestMode value in the high
|
||||
// byte of the index field (recipient: device)
|
||||
test_mode = 2,
|
||||
};
|
||||
|
||||
// Test mode selectors, passed in the high byte of the index field of a SET_FEATURE request
|
||||
// with the test_mode feature selector. Values 06h...3Fh are reserved for standard test
|
||||
// selectors and C0h...FFh for vendor-specific test modes; all other unlisted values are
|
||||
// reserved.
|
||||
const TestMode = enum(u8) {
|
||||
test_j = 0x01,
|
||||
test_k = 0x02,
|
||||
test_se0_nak = 0x03,
|
||||
test_packet = 0x04,
|
||||
test_force_enable = 0x05,
|
||||
_,
|
||||
};
|
||||
|
||||
// The two bytes returned by a GET_STATUS request directed at a device. Fields are declared
|
||||
// least-significant first.
|
||||
const DeviceStatus = packed struct(u16) {
|
||||
// Whether the device is currently self-powered (as opposed to bus-powered). This bit
|
||||
// cannot be changed with the SET_FEATURE or CLEAR_FEATURE requests.
|
||||
self_powered: bool,
|
||||
// Whether the device is currently enabled to request remote wakeup. Changed with the
|
||||
// SET_FEATURE and CLEAR_FEATURE requests using the device_remote_wakeup feature
|
||||
// selector.
|
||||
remote_wakeup: bool,
|
||||
// Reserved (reset to zero)
|
||||
reserved: u14,
|
||||
};
|
||||
|
||||
// The two bytes returned by a GET_STATUS request directed at an endpoint. (A GET_STATUS
|
||||
// request directed at an interface returns two bytes that are entirely reserved.)
|
||||
const EndpointStatus = packed struct(u16) {
|
||||
// Whether the endpoint is currently halted. Set with the SET_FEATURE request using the
|
||||
// endpoint_halt feature selector, and cleared with CLEAR_FEATURE.
|
||||
halted: bool,
|
||||
// Reserved (reset to zero)
|
||||
reserved: u15,
|
||||
};
|
||||
|
||||
// A target for the standard requests that may be directed at the device, an interface, or
|
||||
// an endpoint.
|
||||
const Target = union(enum) {
|
||||
device,
|
||||
interface: InterfaceNumber,
|
||||
endpoint: Request.EndpointIndex,
|
||||
|
||||
fn recipient(target: Target) RequestType.Recipient {
|
||||
return switch (target) {
|
||||
.device => .device,
|
||||
.interface => .interface,
|
||||
.endpoint => .endpoint,
|
||||
};
|
||||
}
|
||||
|
||||
fn index(target: Target) u16 {
|
||||
return switch (target) {
|
||||
.device => 0,
|
||||
.interface => |number| @intFromEnum(number),
|
||||
.endpoint => |endpoint| @bitCast(endpoint),
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
// Constructors for the standard device requests, one per RequestCode. Each returns a
|
||||
// ready-to-send set-up packet with the request_type, value, index, and length fields the
|
||||
// specification prescribes for that request.
|
||||
|
||||
// Reads the status of the given target: bit-cast the two bytes the device returns into a
|
||||
// DeviceStatus or an EndpointStatus. (The two bytes returned for an interface are entirely
|
||||
// reserved.)
|
||||
fn getStatus(target: Target) Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = target.recipient(),
|
||||
.kind = .standard,
|
||||
.direction = .device_to_host,
|
||||
},
|
||||
.request_code = .get_status,
|
||||
.value = 0,
|
||||
.index = target.index(),
|
||||
.length = 2,
|
||||
};
|
||||
}
|
||||
|
||||
// Clears or disables the given feature. A device cannot be taken out of a test mode with
|
||||
// this request; test_mode is only cleared by cycling power.
|
||||
fn clearFeature(feature: FeatureSelector, target: Target) Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = target.recipient(),
|
||||
.kind = .standard,
|
||||
.direction = .host_to_device,
|
||||
},
|
||||
.request_code = .clear_feature,
|
||||
.value = @intFromEnum(feature),
|
||||
.index = target.index(),
|
||||
.length = 0,
|
||||
};
|
||||
}
|
||||
|
||||
// Sets or enables the given feature. For the test_mode feature selector, use setTestMode
|
||||
// instead: the test selector rides in the high byte of the index field.
|
||||
fn setFeature(feature: FeatureSelector, target: Target) Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = target.recipient(),
|
||||
.kind = .standard,
|
||||
.direction = .host_to_device,
|
||||
},
|
||||
.request_code = .set_feature,
|
||||
.value = @intFromEnum(feature),
|
||||
.index = target.index(),
|
||||
.length = 0,
|
||||
};
|
||||
}
|
||||
|
||||
// Puts a hi-speed device into the given test mode: a SET_FEATURE request with the test_mode
|
||||
// feature selector and the test selector in the high byte of the index field.
|
||||
fn setTestMode(mode: TestMode) Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = .device,
|
||||
.kind = .standard,
|
||||
.direction = .host_to_device,
|
||||
},
|
||||
.request_code = .set_feature,
|
||||
.value = @intFromEnum(FeatureSelector.test_mode),
|
||||
.index = @as(u16, @intFromEnum(mode)) << 8,
|
||||
.length = 0,
|
||||
};
|
||||
}
|
||||
|
||||
// Assigns the device its bus address, moving it from the default state to the address
|
||||
// state. The device does not answer at the new address until the status stage of this
|
||||
// request completes.
|
||||
fn setAddress(address: DeviceAddress) Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = .device,
|
||||
.kind = .standard,
|
||||
.direction = .host_to_device,
|
||||
},
|
||||
.request_code = .set_address,
|
||||
.value = @intFromEnum(address),
|
||||
.index = 0,
|
||||
.length = 0,
|
||||
};
|
||||
}
|
||||
|
||||
// Reads a descriptor from the device.
|
||||
// - descriptor_index selects among descriptors of the same type, and is only used for
|
||||
// configuration and string descriptors.
|
||||
// - language_id selects the language of a string descriptor, and is zero otherwise.
|
||||
// - length is the number of bytes to read; a device never returns more than length bytes,
|
||||
// but may return less if the descriptor is shorter.
|
||||
fn getDescriptor(kind: DescriptorType, descriptor_index: u8, language_id: u16, length: u16) Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = .device,
|
||||
.kind = .standard,
|
||||
.direction = .device_to_host,
|
||||
},
|
||||
.request_code = .get_descriptor,
|
||||
.value = @bitCast(Request.DescriptorValue{ .index = descriptor_index, .kind = kind }),
|
||||
.index = language_id,
|
||||
.length = length,
|
||||
};
|
||||
}
|
||||
|
||||
// Updates an existing descriptor or adds a new one (optional; many devices do not support
|
||||
// this request). The parameters mirror getDescriptor; the descriptor itself is sent in the
|
||||
// DATA stage.
|
||||
fn setDescriptor(kind: DescriptorType, descriptor_index: u8, language_id: u16, length: u16) Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = .device,
|
||||
.kind = .standard,
|
||||
.direction = .host_to_device,
|
||||
},
|
||||
.request_code = .set_descriptor,
|
||||
.value = @bitCast(Request.DescriptorValue{ .index = descriptor_index, .kind = kind }),
|
||||
.index = language_id,
|
||||
.length = length,
|
||||
};
|
||||
}
|
||||
|
||||
// Reads the currently active configuration: @enumFromInt the byte the device returns into a
|
||||
// ConfigurationValue, which is none while the device is not configured.
|
||||
fn getConfiguration() Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = .device,
|
||||
.kind = .standard,
|
||||
.direction = .device_to_host,
|
||||
},
|
||||
.request_code = .get_configuration,
|
||||
.value = 0,
|
||||
.index = 0,
|
||||
.length = 1,
|
||||
};
|
||||
}
|
||||
|
||||
// Selects the configuration with the given configuration_value (from
|
||||
// ConfigurationDescriptor.configuration_value), moving the device from the address state to
|
||||
// the configured state. Selecting none returns the device to the address state.
|
||||
fn setConfiguration(configuration_value: ConfigurationValue) Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = .device,
|
||||
.kind = .standard,
|
||||
.direction = .host_to_device,
|
||||
},
|
||||
.request_code = .set_configuration,
|
||||
.value = @intFromEnum(configuration_value),
|
||||
.index = 0,
|
||||
.length = 0,
|
||||
};
|
||||
}
|
||||
|
||||
// Reads the alternate setting currently selected for the given interface: @enumFromInt the
|
||||
// byte the device returns into an AlternateSetting.
|
||||
fn getInterface(interface: InterfaceNumber) Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = .interface,
|
||||
.kind = .standard,
|
||||
.direction = .device_to_host,
|
||||
},
|
||||
.request_code = .get_interface,
|
||||
.value = 0,
|
||||
.index = @intFromEnum(interface),
|
||||
.length = 1,
|
||||
};
|
||||
}
|
||||
|
||||
// Selects an alternate setting (from InterfaceDescriptor.alternate_setting) for the given
|
||||
// interface.
|
||||
fn setInterface(interface: InterfaceNumber, alternate_setting: AlternateSetting) Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = .interface,
|
||||
.kind = .standard,
|
||||
.direction = .host_to_device,
|
||||
},
|
||||
.request_code = .set_interface,
|
||||
.value = @intFromEnum(alternate_setting),
|
||||
.index = @intFromEnum(interface),
|
||||
.length = 0,
|
||||
};
|
||||
}
|
||||
|
||||
// Reads the two-byte number of the frame in which the given isochronous endpoint's
|
||||
// repeating pattern of transfers begins.
|
||||
fn syncFrame(endpoint: Request.EndpointIndex) Request {
|
||||
return .{
|
||||
.request_type = .{
|
||||
.recipient = .endpoint,
|
||||
.kind = .standard,
|
||||
.direction = .device_to_host,
|
||||
},
|
||||
.request_code = .sync_frame,
|
||||
.value = 0,
|
||||
.index = @bitCast(endpoint),
|
||||
.length = 2,
|
||||
};
|
||||
}
|
||||
|
||||
const DescriptorType = enum(u8) {
|
||||
device = 1,
|
||||
configuration = 2,
|
||||
string = 3,
|
||||
interface = 4,
|
||||
endpoint = 5,
|
||||
device_qualifier = 6,
|
||||
other_speed_configuration = 7,
|
||||
interface_power = 8,
|
||||
_,
|
||||
};
|
||||
|
||||
const DeviceDescriptor = extern struct {
|
||||
// Size of this descriptor in bytes
|
||||
length: u8,
|
||||
// DEVICE Descriptor Type
|
||||
descriptor_type: DescriptorType,
|
||||
// USB Specification Release Number in Binary-Coded Decimal (i.e, 2.10 is expressed as 210h).
|
||||
// Identifies the release of the USB Specification with with the device and its
|
||||
// descriptors are compliant.
|
||||
bcd_usb: u16 align(1),
|
||||
// Class code (assigned by the USB-IF)
|
||||
// - This field is reset to zero if each interface within a configuration specifies its own
|
||||
// class information and the various interfaces operate independently.
|
||||
// - A value of FFh in this field indicates the device class is vendor-specific.
|
||||
device_class: u8,
|
||||
// Subclass Code (assigned by the USB-IF)
|
||||
// - The subclass code of a device is qualified by the class code of that device.
|
||||
// - If device_class is reset to zero, then this field must also be reset to zero.
|
||||
// - When device_class is not set to FFh, then all values for this field are reserved for
|
||||
// assignment by the USB-IF.
|
||||
device_subclass: u8,
|
||||
// Protocol code (assigned by the USB-IF)
|
||||
// - The protocol code of a device is qualified by both the class and subclass codes of
|
||||
// that device.
|
||||
// - A value of 00h in this field means that the device may specify class-specific
|
||||
// protocols on an interface basis, though this is not a requirement.
|
||||
// - If this field is set to FFh, then the device uses a vendor-specific protocol.
|
||||
device_protocol: u8,
|
||||
// Maximum packet size for endpoint zero (8, 16, 32, or 64 are the only valid options)
|
||||
max_packet_size_0: u8,
|
||||
// Vendor ID (assigned by the USB-IF)
|
||||
vendor_id: u16 align(1),
|
||||
// Product ID (assigned by the USB-IF)
|
||||
product_id: u16 align(1),
|
||||
// Device release number in binary-coded decimal
|
||||
bcd_device: u16 align(1),
|
||||
// Index of STRING descriptor describing manufacturer
|
||||
manufacturer_index: StringIndex,
|
||||
// Index of STRING descriptor describing product
|
||||
product_index: StringIndex,
|
||||
// Index of STRING descriptor describing the device's serial number
|
||||
serial_number_index: StringIndex,
|
||||
// Number of possible configurations
|
||||
configuration_count: u8,
|
||||
};
|
||||
|
||||
const DeviceQualifierDescriptor = extern struct {
|
||||
// Size of this descriptor in bytes
|
||||
length: u8,
|
||||
// DEVICE_QUALIFIER Descriptor Type
|
||||
descriptor_type: DescriptorType,
|
||||
// USB Specification Release Number in Binary-Coded Decimal (i.e, 2.00 is expressed as 200h).
|
||||
// Identifies the release of the USB Specification with with the device and its
|
||||
// descriptors are compliant. This field must be at least 0200h.
|
||||
bcd_usb: u16 align(1),
|
||||
// Class code (assigned by the USB-IF)
|
||||
device_class: u8,
|
||||
// Subclass Code (assigned by the USB-IF)
|
||||
device_subclass: u8,
|
||||
// Protocol code (assigned by the USB-IF)
|
||||
device_protocol: u8,
|
||||
// Maximum packet size for endpoint zero (8, 16, 32, or 64 are the only valid options)
|
||||
max_packet_size_0: u8,
|
||||
// Number of possible configurations
|
||||
configuration_count: u8,
|
||||
// Reserved for future uses, must be zero.
|
||||
reserved: u8,
|
||||
};
|
||||
|
||||
const ConfigurationDescriptor = extern struct {
|
||||
// Size of this descriptor in bytes
|
||||
length: u8,
|
||||
// CONFIGURATION Descriptor Type
|
||||
descriptor_type: DescriptorType,
|
||||
// The total combined length in bytes of all the descriptors returned with the request for
|
||||
// this CONFIGURATION descriptor (including CONFIGURATION, INTERFACE, ENDPOINT, class- and
|
||||
// vendor-specific descriptors).
|
||||
total_length: u16 align(1),
|
||||
// Number of interfaces supported by this configuration
|
||||
interface_count: u8,
|
||||
// Value which when used as an argument in the SET_CONFIGURATION request, causes the device
|
||||
// to assume the configuration described by this descriptor.
|
||||
configuration_value: ConfigurationValue,
|
||||
// Index of STRING descriptor describing this configuration.
|
||||
configuration_index: StringIndex,
|
||||
// Configuration Characteristics
|
||||
attributes: Attributes,
|
||||
// Maximum power consumption of this device from the bus when fully operational and using
|
||||
// this configuration. Expressed in units of 2mA (i.e., a value of 50 in this field
|
||||
// indicates 100mA).
|
||||
// - A device reports with the attributes field whether the configuration is bus- or
|
||||
// self-powered, but the device status (retrieved with a GET_STATUS request) reports
|
||||
// whether the device is currently self-powered.
|
||||
// - If a device is disconnected from an external power source, it may not draw more
|
||||
// power from the bus than specified in this field.
|
||||
max_power: u8,
|
||||
|
||||
// Configuration characteristics. Fields are declared least-significant first.
|
||||
const Attributes = packed struct(u8) {
|
||||
// Reserved, reset to zero (D4...0)
|
||||
reserved: u5,
|
||||
// Whether Remote Wakeup is supported by this configuration (D5)
|
||||
remote_wakeup: bool,
|
||||
// Self-Powered (D6)
|
||||
// - false: Device runs on power supplied by the bus
|
||||
// - true: Device provides a local power source; if max_power is non-zero, the
|
||||
// device also may use bus power.
|
||||
self_powered: bool,
|
||||
// Reserved, must be set to one for historical reasons (D7)
|
||||
reserved_one: u1,
|
||||
};
|
||||
};
|
||||
|
||||
// This descriptor describes the configuration of a high-speed device if it were operating at
|
||||
// its alternative speed. The structure of the OTHER_SPEED_CONFIGURATION is identical to that
|
||||
// of the CONFIGURATION descriptor; the only difference is that the descriptor_type field
|
||||
// reflects that the descriptor is an OTHER_SPEED_CONFIGURATION descriptor.
|
||||
const OtherSpeedConfigurationDescriptor = ConfigurationDescriptor;
|
||||
|
||||
const InterfaceDescriptor = extern struct {
|
||||
// Size of this descriptor in bytes
|
||||
length: u8,
|
||||
// INTERFACE Descriptor Type
|
||||
descriptor_type: DescriptorType,
|
||||
// Number of this interface. Zero-based value which identifies the index of this interface
|
||||
// in the array of interfaces supported within a configuration.
|
||||
interface_number: InterfaceNumber,
|
||||
// Value used to select the alternate settings described by this INTERFACE descriptor for
|
||||
// the interface with the interface_number in the previous field. This value is zero if
|
||||
// this descriptor describes the default settings for a particular interface.
|
||||
alternate_setting: AlternateSetting,
|
||||
// Number of endpoints used by this interface, not including endpoint zero.
|
||||
endpoint_count: u8,
|
||||
// Class code (assigned by the USB-IF)
|
||||
// - A value of zero here is reserved for future standardization.
|
||||
// - If this value is FFh, the interface class is vendor-specific.
|
||||
// - All other values are reserved for assignment by the USB-IF.
|
||||
interface_class: u8,
|
||||
// Subclass code (assigned by the USB-IF)
|
||||
// - The subclass code in this field is qualified by the value of the interface_class
|
||||
// field.
|
||||
// - If interface_class is reset to zero, then this field must also be reset to zero.
|
||||
// - If interface_class is not set to the value of FFh, then all values of this field are
|
||||
// reserved for assignment by the USB-IF.
|
||||
interface_subclass: u8,
|
||||
// Protocol code (assigned by the USB-IF)
|
||||
// - The protocol code in this field is qualified by the values of the interface_class
|
||||
// and interface_subclass fields.
|
||||
// - If an interface supports class-specific requests, then this field identifies the
|
||||
// protocols that the device uses as defined by the specifications of the device class.
|
||||
// - If this field is reset to zero, then the device does not use a class-specific
|
||||
// protocol on this interface.
|
||||
// - If this field is set to FFh, then the device uses a vendor-specific protocol on
|
||||
// this interface.
|
||||
interface_protocol: u8,
|
||||
// Index of STRING descriptor describing this interface
|
||||
interface_index: StringIndex,
|
||||
};
|
||||
|
||||
const EndpointDescriptor = extern struct {
|
||||
// Size of this descriptor in bytes
|
||||
length: u8,
|
||||
// ENDPOINT Descriptor Type
|
||||
descriptor_type: DescriptorType,
|
||||
// The address of the endpoint on the USB device described by this descriptor
|
||||
endpoint_address: Address,
|
||||
// The endpoint's attributes
|
||||
attributes: Attributes,
|
||||
// Maximum packet size that this endpoint is capable of sending or receiving. For
|
||||
// isochronous endpoints, this value is used to reserve bus time; the pipe, however, may
|
||||
// not always use all of the reserved bus time.
|
||||
max_packet_size: MaxPacketSize align(1),
|
||||
// Interval for polling a device during a data transfer, expressed in units of microframes
|
||||
// for high-speed devices, and frames for low- and full-speed devices. The exact meaning of
|
||||
// the value in this field depends on the endpoint type and the operating speed of the
|
||||
// device:
|
||||
// - Full- and High-speed isochronous endpoints, and high-speed interrupt endpoints:
|
||||
// This field must be in the range from 1 to 16, and is used to calculate the period
|
||||
// as 2^(interval - 1). That is, a value of 4 calculates to 2^(4 - 1) = 2^3 = 8.
|
||||
// - Full- and Low-speed interrupt endpoints: This field must be in the range from
|
||||
// 1 to 255.
|
||||
// - High-speed bulk and control OUT endpoints: This field must be in the range from
|
||||
// 0 to 255, and specifies the maximum NAK rate of the endpoint. A value of zero
|
||||
// indicates that the endpoint never NAKs; other values indicate at most 1 NAK each
|
||||
// interval number of microframes.
|
||||
interval: u8,
|
||||
|
||||
// The address of an endpoint. Fields are declared least-significant first.
|
||||
const Address = packed struct(u8) {
|
||||
// Endpoint Number (D3...0)
|
||||
number: EndpointNumber,
|
||||
// Reserved, reset to zero (D6...4)
|
||||
reserved: u3,
|
||||
// Direction, ignored for control endpoints (D7)
|
||||
direction: EndpointDirection,
|
||||
};
|
||||
|
||||
// An endpoint's attributes. Fields are declared least-significant first.
|
||||
const Attributes = packed struct(u8) {
|
||||
// Transfer Type (D1...0)
|
||||
transfer_type: TransferType,
|
||||
// Synchronization Type; isochronous endpoints only, reserved and reset to zero for
|
||||
// other endpoint types (D3...2)
|
||||
synchronization: Synchronization,
|
||||
// Usage Type; isochronous endpoints only, reserved and reset to zero for other
|
||||
// endpoints (D5...4)
|
||||
usage: Usage,
|
||||
// Reserved, reset to zero (D7...6)
|
||||
reserved: u2,
|
||||
};
|
||||
|
||||
const TransferType = enum(u2) {
|
||||
control = 0,
|
||||
isochronous = 1,
|
||||
bulk = 2,
|
||||
interrupt = 3,
|
||||
};
|
||||
|
||||
const Synchronization = enum(u2) {
|
||||
none = 0,
|
||||
asynchronous = 1,
|
||||
adaptive = 2,
|
||||
synchronous = 3,
|
||||
};
|
||||
|
||||
const Usage = enum(u2) {
|
||||
data = 0,
|
||||
feedback = 1,
|
||||
implicit_feedback_data = 2,
|
||||
_,
|
||||
};
|
||||
|
||||
// The maximum packet size of an endpoint. Fields are declared least-significant first.
|
||||
const MaxPacketSize = packed struct(u16) {
|
||||
// Maximum packet size in bytes (bits 10...0)
|
||||
size: u11,
|
||||
// Number of additional transaction opportunities per microframe, for high-speed
|
||||
// isochronous and interrupt endpoints; reserved and reset to zero for other
|
||||
// endpoints (bits 12...11)
|
||||
additional_transactions: AdditionalTransactions,
|
||||
// Reserved, must be reset to zero (bits 15...13)
|
||||
reserved: u3,
|
||||
};
|
||||
|
||||
const AdditionalTransactions = enum(u2) {
|
||||
// None (1 transaction per microframe)
|
||||
none = 0,
|
||||
// 1 additional (2 transactions per microframe)
|
||||
one = 1,
|
||||
// 2 additional (3 transactions per microframe)
|
||||
two = 2,
|
||||
_,
|
||||
};
|
||||
};
|
||||
|
||||
// A STRING descriptor at index zero returns the list of LANGID codes supported by the
|
||||
// device; all other indices return a Unicode string. Both forms start with this two-byte
|
||||
// header, followed by the variable-length payload:
|
||||
// - index 0: an array of two-byte LANGID codes (wLangID[0] through wLangID[x])
|
||||
// - other indices: a Unicode string of N bytes
|
||||
const StringDescriptor = extern struct {
|
||||
// Size of this descriptor in bytes
|
||||
length: u8,
|
||||
// STRING Descriptor Type
|
||||
descriptor_type: DescriptorType,
|
||||
};
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
test "wire sizes and offsets match the specification" {
|
||||
const expectEqual = std.testing.expectEqual;
|
||||
|
||||
try expectEqual(8, @sizeOf(Request));
|
||||
try expectEqual(18, @sizeOf(DeviceDescriptor));
|
||||
try expectEqual(10, @sizeOf(DeviceQualifierDescriptor));
|
||||
try expectEqual(9, @sizeOf(ConfigurationDescriptor));
|
||||
try expectEqual(9, @sizeOf(InterfaceDescriptor));
|
||||
try expectEqual(7, @sizeOf(EndpointDescriptor));
|
||||
try expectEqual(2, @sizeOf(StringDescriptor));
|
||||
|
||||
try expectEqual(2, @offsetOf(DeviceDescriptor, "bcd_usb"));
|
||||
try expectEqual(8, @offsetOf(DeviceDescriptor, "vendor_id"));
|
||||
try expectEqual(17, @offsetOf(DeviceDescriptor, "configuration_count"));
|
||||
try expectEqual(2, @offsetOf(ConfigurationDescriptor, "total_length"));
|
||||
try expectEqual(4, @offsetOf(EndpointDescriptor, "max_packet_size"));
|
||||
}
|
||||
|
||||
test "bitmap packings match the specification" {
|
||||
const expectEqual = std.testing.expectEqual;
|
||||
const expect = std.testing.expect;
|
||||
|
||||
// bmRequestType for GET_DESCRIPTOR: device-to-host | standard | device = 80h
|
||||
const request_type = RequestType{
|
||||
.recipient = .device,
|
||||
.kind = .standard,
|
||||
.direction = .device_to_host,
|
||||
};
|
||||
try expectEqual(0x80, @as(u8, @bitCast(request_type)));
|
||||
|
||||
// wValue for GET_DESCRIPTOR(CONFIGURATION, index 0) = 0200h
|
||||
const descriptor_value = Request.DescriptorValue{ .kind = .configuration };
|
||||
try expectEqual(0x0200, @as(u16, @bitCast(descriptor_value)));
|
||||
|
||||
// wIndex for the IN endpoint 1 = 0081h
|
||||
const endpoint_index = Request.EndpointIndex{ .number = @enumFromInt(1), .direction = .in };
|
||||
try expectEqual(0x0081, @as(u16, @bitCast(endpoint_index)));
|
||||
|
||||
// Endpoint address 81h = IN endpoint 1
|
||||
const address: EndpointDescriptor.Address = @bitCast(@as(u8, 0x81));
|
||||
try expectEqual(1, @intFromEnum(address.number));
|
||||
try expectEqual(.in, address.direction);
|
||||
|
||||
// Endpoint attributes 03h = interrupt transfer
|
||||
const attributes: EndpointDescriptor.Attributes = @bitCast(@as(u8, 0x03));
|
||||
try expectEqual(.interrupt, attributes.transfer_type);
|
||||
|
||||
// wMaxPacketSize 0008h = 8 bytes, no additional transactions
|
||||
const max_packet_size: EndpointDescriptor.MaxPacketSize = @bitCast(@as(u16, 0x0008));
|
||||
try expectEqual(8, max_packet_size.size);
|
||||
try expectEqual(.none, max_packet_size.additional_transactions);
|
||||
|
||||
// Configuration attributes C0h = self-powered, with the historical D7 bit set
|
||||
const configuration_attributes: ConfigurationDescriptor.Attributes = @bitCast(@as(u8, 0xC0));
|
||||
try expect(configuration_attributes.self_powered);
|
||||
try expect(!configuration_attributes.remote_wakeup);
|
||||
try expectEqual(1, configuration_attributes.reserved_one);
|
||||
|
||||
// GET_STATUS words: device 0001h = self-powered; endpoint 0001h = halted
|
||||
const device_status: DeviceStatus = @bitCast(@as(u16, 0x0001));
|
||||
try expect(device_status.self_powered and !device_status.remote_wakeup);
|
||||
const endpoint_status: EndpointStatus = @bitCast(@as(u16, 0x0001));
|
||||
try expect(endpoint_status.halted);
|
||||
|
||||
// DescriptorType is non-exhaustive: class-specific values (HID = 21h) pass through
|
||||
const hid_type: DescriptorType = @enumFromInt(0x21);
|
||||
try expectEqual(0x21, @intFromEnum(hid_type));
|
||||
try expect(hid_type != .device);
|
||||
}
|
||||
|
||||
fn expectRequestBytes(request: Request, expected: [8]u8) !void {
|
||||
try std.testing.expectEqualSlices(u8, &expected, std.mem.asBytes(&request));
|
||||
}
|
||||
|
||||
test "standard request constructors encode the specification's set-up packets" {
|
||||
try expectRequestBytes(getStatus(.device), .{ 0x80, 0, 0, 0, 0, 0, 2, 0 });
|
||||
try expectRequestBytes(getStatus(.{ .interface = @enumFromInt(3) }), .{ 0x81, 0, 0, 0, 3, 0, 2, 0 });
|
||||
try expectRequestBytes(getStatus(.{ .endpoint = .{ .number = @enumFromInt(2), .direction = .in } }), .{ 0x82, 0, 0, 0, 0x82, 0, 2, 0 });
|
||||
try expectRequestBytes(clearFeature(.endpoint_halt, .{ .endpoint = .{ .number = @enumFromInt(1), .direction = .out } }), .{ 0x02, 1, 0, 0, 0x01, 0, 0, 0 });
|
||||
try expectRequestBytes(setFeature(.device_remote_wakeup, .device), .{ 0x00, 3, 1, 0, 0, 0, 0, 0 });
|
||||
try expectRequestBytes(setTestMode(.test_packet), .{ 0x00, 3, 2, 0, 0, 0x04, 0, 0 });
|
||||
try expectRequestBytes(setAddress(@enumFromInt(5)), .{ 0x00, 5, 5, 0, 0, 0, 0, 0 });
|
||||
try expectRequestBytes(getDescriptor(.device, 0, 0, 18), .{ 0x80, 6, 0, 1, 0, 0, 18, 0 });
|
||||
try expectRequestBytes(getDescriptor(.string, 2, 0x0409, 255), .{ 0x80, 6, 2, 3, 0x09, 0x04, 255, 0 });
|
||||
try expectRequestBytes(setDescriptor(.string, 2, 0x0409, 16), .{ 0x00, 7, 2, 3, 0x09, 0x04, 16, 0 });
|
||||
try expectRequestBytes(getConfiguration(), .{ 0x80, 8, 0, 0, 0, 0, 1, 0 });
|
||||
try expectRequestBytes(setConfiguration(@enumFromInt(1)), .{ 0x00, 9, 1, 0, 0, 0, 0, 0 });
|
||||
try expectRequestBytes(getInterface(@enumFromInt(2)), .{ 0x81, 10, 0, 0, 2, 0, 1, 0 });
|
||||
try expectRequestBytes(setInterface(@enumFromInt(2), @enumFromInt(1)), .{ 0x01, 11, 1, 0, 2, 0, 0, 0 });
|
||||
try expectRequestBytes(syncFrame(.{ .number = @enumFromInt(3), .direction = .in }), .{ 0x82, 12, 0, 0, 0x83, 0, 2, 0 });
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user