moving kernel code to kernel/

This commit is contained in:
2026-07-05 10:19:11 +01:00
parent 7c3cffb337
commit 6f6ccc8bc9
37 changed files with 63 additions and 61 deletions
+7 -5
View File
@@ -2,10 +2,12 @@
Codename: Shodan
Version: 1
A small x86-64 operating system kernel, written from scratch in Zig. It boots via
UEFI, and so far has a framebuffer console, a physical frame allocator, its own
paging with W^X permissions, interrupt/exception handling, a LAPIC timer, and a
kernel heap. See [`docs/`](docs/README.md) for how each piece works.
A small operating system, written from scratch in Zig — a bootloader (`src/boot/`)
and a microkernel (`src/kernel/`), sharing a neutral handoff contract (`src/root.zig`).
It boots x86-64 via UEFI, and so far has a framebuffer console, a physical frame
allocator, its own paging with W^X permissions, interrupt/exception handling, a
LAPIC timer, a kernel heap, a fixed-priority preemptive scheduler, and in-kernel IPC
channels. See [`docs/`](docs/README.md) for how each piece works.
## Prerequisites
@@ -29,7 +31,7 @@ zig build
```
Produces the UEFI bootloader (`zig-out/bin/BOOTX64.efi`) and the kernel ELF
(`zig-out/bin/danos`).
(`zig-out/bin/kernel`).
## Run
+7 -7
View File
@@ -45,18 +45,18 @@ pub fn build(b: *std.Build) void {
// The generic kernel imports this as "arch" and never names x86_64, so a new
// architecture is a matter of pointing this module at a different directory.
const arch_mod = b.addModule("arch", .{
.root_source_file = b.path("src/arch/x86_64/cpu.zig"),
.root_source_file = b.path("src/kernel/arch/x86_64/cpu.zig"),
.imports = &.{
.{ .name = "danos", .module = mod }, // paging uses the shared BootInfo/memory-map types
},
});
// CPU-exception stubs — real assembly, since they need cross-symbol
// jumps/calls that Zig inline asm can't express (see the file's header).
arch_mod.addAssemblyFile(b.path("src/arch/x86_64/isr.s"));
arch_mod.addAssemblyFile(b.path("src/kernel/arch/x86_64/isr.s"));
// Compile-time config the kernel reads as `@import("build_options")`. The
// QEMU test harness sets -Dtest-case=<name> to run one self-test at boot.
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see src/tests.zig)");
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see src/kernel/tests.zig)");
const build_options = b.addOptions();
build_options.addOption(?[]const u8, "test_case", test_case);
const build_options_mod = build_options.createModule();
@@ -72,9 +72,9 @@ pub fn build(b: *std.Build) void {
});
const exe = b.addExecutable(.{
.name = "danos",
.name = "kernel",
.root_module = b.createModule(.{
.root_source_file = b.path("src/main.zig"),
.root_source_file = b.path("src/kernel/main.zig"),
.target = kernel_target,
.optimize = optimize,
.code_model = .small, // kernel is linked in the low 2 GiB (see image_base)
@@ -90,7 +90,7 @@ pub fn build(b: *std.Build) void {
},
}),
});
exe.setLinkerScript(b.path("src/arch/x86_64/linker.ld"));
exe.setLinkerScript(b.path("src/kernel/arch/x86_64/linker.ld"));
exe.entry = .{ .symbol_name = "_start" };
// Physical address the bootloader loads the kernel to (identity-mapped under
// UEFI). Overrides Zig's default image base so the linker script's layout is
@@ -153,7 +153,7 @@ pub fn build(b: *std.Build) void {
.dest_dir = .{ .override = .{ .custom = "esp/EFI/BOOT" } },
});
// The bootloader loads the kernel by name from the volume root, so drop the
// kernel ELF at esp/danos.
// kernel ELF at esp/kernel.
const kernel_install = b.addInstallArtifact(exe, .{
.dest_dir = .{ .override = .{ .custom = "esp" } },
});
+8 -8
View File
@@ -92,14 +92,14 @@ behind the [arch](arch.md) boundary, and when idle, or on a panic, it **halts**
| Area | Code |
|------|------|
| Boot methods (one per way of booting the kernel) | `src/boot/` — `efi.zig` (UEFI) → `BOOTX64.efi` |
| Kernel entry, panic, bring-up | `src/main.zig` |
| Kernel entry, panic, bring-up | `src/kernel/main.zig` |
| Shared loader↔kernel contract (`BootInfo`, `Framebuffer`, `MemoryMap`, ABI) | `src/root.zig` |
| Physical frame allocator | `src/pmm.zig` |
| Kernel heap (`std.mem.Allocator`) | `src/heap.zig` |
| Scheduler (fixed-priority preemptive; blocking, wait queues) | `src/sched.zig` |
| IPC channels (message passing) | `src/ipc.zig` |
| Framebuffer text console (mirrors to serial) | `src/console.zig` |
| In-kernel test cases | `src/tests.zig` |
| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/timer, serial, linker script) | `src/arch/x86_64/` |
| Physical frame allocator | `src/kernel/pmm.zig` |
| Kernel heap (`std.mem.Allocator`) | `src/kernel/heap.zig` |
| Scheduler (fixed-priority preemptive; blocking, wait queues) | `src/kernel/sched.zig` |
| IPC channels (message passing) | `src/kernel/ipc.zig` |
| Framebuffer text console (mirrors to serial) | `src/kernel/console.zig` |
| In-kernel test cases | `src/kernel/tests.zig` |
| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/timer, serial, linker script) | `src/kernel/arch/x86_64/` |
| Build + `run-x86-64` (QEMU/OVMF) | `build.zig` |
| QEMU integration test harness | `test/qemu_test.py` |
+11 -11
View File
@@ -13,7 +13,7 @@ runtime dispatch. `build.zig` exposes one architecture's code as a module called
```zig
const arch_mod = b.addModule("arch", .{
.root_source_file = b.path("src/arch/x86_64/cpu.zig"),
.root_source_file = b.path("src/kernel/arch/x86_64/cpu.zig"),
});
```
@@ -26,7 +26,7 @@ arch.halt(); // never says "x86_64"
```
Adding a second architecture is then a build-time choice: create
`src/arch/aarch64/`, and point the `arch` module at it when the target CPU is
`src/kernel/arch/aarch64/`, and point the `arch` module at it when the target CPU is
AArch64. `main.zig` and `console.zig` don't change. **That compiler-checked module
boundary _is_ the architecture interface** — when a new arch is missing a function
the generic kernel calls, the build fails and names exactly what's missing.
@@ -36,7 +36,7 @@ the generic kernel calls, the build fails and names exactly what's missing.
The split follows a simple test: does it name a CPU instruction, a hardware
register, or a memory-management structure? If so, it's arch-specific.
| Arch-specific — `src/arch/x86_64/` | Generic — kernel core |
| Arch-specific — `src/kernel/arch/x86_64/` | Generic — kernel core |
|---|---|
| `cpu.zig`: `halt()` (`hlt`), later GDT/IDT/paging | `console.zig` — pure pixel math, works anywhere |
| `linker.ld` — link layout, load address | `main.zig` — `kmain` orchestration, panic handler |
@@ -51,7 +51,7 @@ should end up on the generic side; the arch module stays small.
There are really two independent questions, and it's worth not conflating them:
- **CPU architecture** (x86_64 vs AArch64): instructions, MMU, interrupts →
`src/arch/<cpu>/`.
`src/kernel/arch/<cpu>/`.
- **Boot protocol** (UEFI vs Raspberry Pi firmware + device tree): handled
*separately*, because loaders are their own binaries. `src/boot/efi.zig` builds
`BOOTX64.efi`, a distinct executable from the kernel ELF. On a Pi there is no
@@ -61,24 +61,24 @@ There are really two independent questions, and it's worth not conflating them:
## Current x86_64 contents
- **`src/arch/x86_64/cpu.zig`** — the `arch` module root. Exposes `halt()` (see
- **`src/kernel/arch/x86_64/cpu.zig`** — the `arch` module root. Exposes `halt()` (see
[halting.md](halting.md)), `init()` (bring up the descriptor tables),
`enablePaging()`, `setFaultHandler`, `readCr2`/`readCr3`, and the `CpuState`
trap frame.
- **`src/arch/x86_64/gdt.zig`** / **`idt.zig`** / **`tss.zig`** — the GDT, IDT and
- **`src/kernel/arch/x86_64/gdt.zig`** / **`idt.zig`** / **`tss.zig`** — the GDT, IDT and
TSS plus CPU-exception handling (see [interrupts.md](interrupts.md)).
- **`src/arch/x86_64/paging.zig`** — the kernel's page tables (see
- **`src/kernel/arch/x86_64/paging.zig`** — the kernel's page tables (see
[paging.md](paging.md)).
- **`src/arch/x86_64/apic.zig`** — the Local APIC and its timer, the source of
- **`src/kernel/arch/x86_64/apic.zig`** — the Local APIC and its timer, the source of
device interrupts (see [device-interrupts.md](device-interrupts.md)).
- **`src/arch/x86_64/serial.zig`** / **`io.zig`** — the COM1 UART (the kernel's
- **`src/kernel/arch/x86_64/serial.zig`** / **`io.zig`** — the COM1 UART (the kernel's
machine-readable log channel, see [testing.md](testing.md)) and the shared
port-I/O + MSR primitives.
- **`src/arch/x86_64/isr.s`** — the exception stubs, the `lgdt`/`lidt`/`ltr` load
- **`src/kernel/arch/x86_64/isr.s`** — the exception stubs, the `lgdt`/`lidt`/`ltr` load
helpers, and the context switch (`switch_context` / `task_trampoline`, see
[scheduling.md](scheduling.md)) — real assembly, since Zig inline asm can't
express them.
- **`src/arch/x86_64/linker.ld`** — the kernel link layout (fixed low load
- **`src/kernel/arch/x86_64/linker.ld`** — the kernel link layout (fixed low load
address, one PT_LOAD per permission set).
The kernel entry point `_start` currently still lives in the generic `main.zig` as
+3 -3
View File
@@ -18,7 +18,7 @@ matters for understanding why. This page maps the landscape so the
new ISA.
They are as different from each other as either is from x86-64: separate registers,
page-table formats, and calling conventions. Each needs its own `src/arch/<name>/`.
page-table formats, and calling conventions. Each needs its own `src/kernel/arch/<name>/`.
## The Raspberry Pi models
@@ -57,9 +57,9 @@ the DTB/ACPI tells you what devices exist.
## What danos needs, layer by layer
- **One CPU arch module: `src/arch/aarch64/`** — covering the Zero 2 W and Pi 3-5,
- **One CPU arch module: `src/kernel/arch/aarch64/`** — covering the Zero 2 W and Pi 3-5,
providing the same `arch` interface as x86_64: `halt`, context switch,
interrupt/exception vectors, page tables, a UART, a timer. No `src/arch/arm/` is
interrupt/exception vectors, page tables, a UART, a timer. No `src/kernel/arch/arm/` is
planned (see the decision above), so there's a single ARM backend to write.
- **A device-tree boot path.** Since stock Pis boot via DTB, danos needs an entry
that parses the DTB's `/memory` and `/reserved-memory` into the neutral
+1 -1
View File
@@ -18,7 +18,7 @@ Interrupt delivery on modern x86 goes through the **APIC**, not the legacy 8259
PIC. There are two halves; we only need one so far:
- The **Local APIC** (per-CPU, memory-mapped at physical `0xFEE00000`) handles the
CPU's own timer and receives interrupts routed to it. `src/arch/x86_64/apic.zig`.
CPU's own timer and receives interrupts routed to it. `src/kernel/arch/x86_64/apic.zig`.
- The **IO-APIC** routes *external* device lines (keyboard, etc.) to LAPIC vectors.
Not needed for the timer — it'll arrive with the keyboard.
+2 -2
View File
@@ -25,7 +25,7 @@ esp/EFI/BOOT/BOOTX64.efi <- the "removable media" default for x86-64
That's exactly the layout `build.zig` assembles. It builds `src/boot/efi.zig` for the
`uefi` target, installs it to `esp/EFI/BOOT/BOOTX64.efi`, and drops the kernel ELF
at `esp/danos`. The `run-x86-64` step then points QEMU at OVMF (UEFI firmware for
at `esp/kernel`. The `run-x86-64` step then points QEMU at OVMF (UEFI firmware for
virtual machines) and presents that `esp/` directory to the guest as a FAT drive.
The firmware finds `BOOTX64.efi` and runs it — that's our `main()`.
@@ -170,7 +170,7 @@ power on
-> loadKernel (read danos ELF, load PT_LOAD segments to 0x100000)
-> exitBootServices (retry until the memory-map key holds)
-> jump to e_entry, boot_info pointer in RDI
-> kernel _start (src/main.zig: framebuffer console, then halt)
-> kernel _start (src/kernel/main.zig: framebuffer console, then halt)
```
Bottom line: **UEFI's job is to give us a CPU, memory, and a framebuffer, then
+2 -2
View File
@@ -3,7 +3,7 @@
Once the kernel knows what RAM exists ([memory-map.md](memory-map.md)), it needs a
way to *hand out* that RAM: give me a free page of physical memory, and later,
here's one back. That's the **physical frame allocator** (a "physical memory
manager", hence `src/pmm.zig`). It deals only in fixed 4 KiB **frames** — the
manager", hence `src/kernel/pmm.zig`). It deals only in fixed 4 KiB **frames** — the
natural unit because that's the granularity the CPU's paging hardware maps — and
it is the primitive everything above it stands on: page tables, the kernel heap,
per-process memory all ultimately ask the frame allocator for pages.
@@ -33,7 +33,7 @@ RAM is 32768 frames — a **4 KiB bitmap, a single frame**. Even 64 GiB needs on
## How it works
State lives in `src/pmm.zig`: the `bitmap` slice, `total_frames`, `used_frames`,
State lives in `src/kernel/pmm.zig`: the `bitmap` slice, `total_frames`, `used_frames`,
and a `next_hint` marking where the next allocation scan should start.
### init(map) — building it from the memory map
+1 -1
View File
@@ -9,7 +9,7 @@ write a 32-bit value to the right address, and a pixel changes color. That's
exactly what `Console.pixel` does:
```zig
self.rowPtr(y)[x] = color; // src/console.zig
self.rowPtr(y)[x] = color; // src/kernel/console.zig
```
Our `Framebuffer` struct (`src/root.zig`) is the four facts you need to
+2 -2
View File
@@ -16,7 +16,7 @@ safely, until the machine is reset or powered off.
## The core of it: `hlt`
Everything comes down to one x86 instruction. It's CPU-specific, so it lives in
the arch module, `src/arch/x86_64/cpu.zig` (see [arch.md](arch.md)), and the
the arch module, `src/kernel/arch/x86_64/cpu.zig` (see [arch.md](arch.md)), and the
generic kernel calls it as `arch.halt()`:
```zig
@@ -83,7 +83,7 @@ treats the call:
signature for a kernel entry point — the bootloader jumps in and nothing ever
jumps back out.
You can see the chain in `src/main.zig`: `_start` is `noreturn`, it calls
You can see the chain in `src/kernel/main.zig`: `_start` is `noreturn`, it calls
`kmain` which is `noreturn`, which ends by calling `arch.halt()` which is
`noreturn`. The "never returns" property is threaded all the way down.
+1 -1
View File
@@ -7,7 +7,7 @@ top of both to provide what the rest of the kernel actually wants: `alloc(n)` /
the thing that unlocks dynamic data structures — lists, hash maps, driver state,
eventually a process table.
It's generic kernel code (`src/heap.zig`): the allocator logic is
It's generic kernel code (`src/kernel/heap.zig`): the allocator logic is
architecture-neutral, using `arch.mapPage` and the frame allocator underneath.
## A growable free-list allocator
+5 -5
View File
@@ -9,7 +9,7 @@ reboot is miserable.
This is the machinery that catches those faults and prints what happened instead.
It's all x86_64-specific, so it lives behind the [arch](arch.md) boundary in
`src/arch/x86_64/`. Only the 32 CPU-defined exception vectors are wired up so far;
`src/kernel/arch/x86_64/`. Only the 32 CPU-defined exception vectors are wired up so far;
device interrupts (timer, keyboard, via the APIC) come later, on the same IDT.
## First the GDT
@@ -20,7 +20,7 @@ IDT gate names a code-segment *selector* that must resolve in the current GDT. T
firmware left a GDT in place, but we don't control it, so we install our own with
known selectors: `0x08` kernel code, `0x10` kernel data.
`src/arch/x86_64/gdt.zig` holds three flat descriptors — a required null entry,
`src/kernel/arch/x86_64/gdt.zig` holds three flat descriptors — a required null entry,
plus code and data — where the only bits that matter in long mode are the access
byte and the code segment's long-mode (`L`) flag. Loading it (`gdt_flush` in
`isr.s`) does two things: `lgdt`, then reload the segment registers. The data
@@ -33,7 +33,7 @@ into CS:RIP.
The **Interrupt Descriptor Table** maps each of 256 vectors to a handler. Each
entry is a 16-byte *gate* holding the handler's address (split across three
fields, a quirk of the format), the code selector (`0x08`), and flags: `0x8E`
means present, ring 0, 64-bit interrupt gate. `src/arch/x86_64/idt.zig` builds the
means present, ring 0, 64-bit interrupt gate. `src/kernel/arch/x86_64/idt.zig` builds the
table, points the first 32 vectors at their stubs, and loads it with `lidt`
(`idt_flush`).
@@ -49,7 +49,7 @@ hit a fault *while trying to deliver another fault* — very often because the
current stack pointer is bad, so pushing the exception frame itself faulted. If
the #DF handler then tried to push onto that same bad stack, it would fault a
third time and **triple-fault** — an instant reset. So the #DF gate is pointed at
**IST1**, a small dedicated stack (`src/arch/x86_64/tss.zig`) that's always valid.
**IST1**, a small dedicated stack (`src/kernel/arch/x86_64/tss.zig`) that's always valid.
Bringing it up: fill in the TSS's IST1 pointer, publish the TSS through a
descriptor in the GDT (`gdt.setTss`), and load it into the task register with
@@ -60,7 +60,7 @@ which is why the GDT grew from three entries to five.
On an exception the CPU pushes a small frame (SS, RSP, RFLAGS, CS, RIP) and, for
*some* vectors, an **error code**. That inconsistency is a nuisance, so each stub
in `src/arch/x86_64/isr.s` normalises it: vectors that don't get a hardware error
in `src/kernel/arch/x86_64/isr.s` normalises it: vectors that don't get a hardware error
code push a dummy `0`, then every stub pushes its **vector number** and jumps to a
shared tail, `isr_common`. The tail pushes all the general registers and calls the
Zig handler with a pointer to the whole thing.
+1 -1
View File
@@ -6,7 +6,7 @@ just call each other — a request becomes a **message**. In a microkernel, what
was a function call across a monolithic kernel is IPC, so it's a first-class
concern, not an afterthought.
This first form is a **bounded blocking channel** (`src/ipc.zig`): a fixed-size
This first form is a **bounded blocking channel** (`src/kernel/ipc.zig`): a fixed-size
ring buffer of messages with a producer/consumer rendezvous, built on the
scheduler's [wait queues](scheduling.md).
+1 -1
View File
@@ -7,7 +7,7 @@ which live in memory we'd like to reclaim and don't control), switches CR3 onto
them, and — crucially — maps with **real permissions**.
It's x86_64-specific (the 4-level table format is an Intel/AMD thing), so it lives
behind the [arch](arch.md) boundary in `src/arch/x86_64/paging.zig`.
behind the [arch](arch.md) boundary in `src/kernel/arch/x86_64/paging.zig`.
## The format
+2 -2
View File
@@ -6,8 +6,8 @@ ready task always runs, and tasks at the same priority take turns. That model is
chosen for [real-time](vision.md) — it's predictable (you can reason about which
task runs when) and its decisions are O(1), unlike a fair-share scheduler.
The scheduler proper (`src/sched.zig`) is generic; the context switch and new-task
stack setup are architecture-specific (`src/arch/x86_64/`, see [arch](arch.md)).
The scheduler proper (`src/kernel/sched.zig`) is generic; the context switch and new-task
stack setup are architecture-specific (`src/kernel/arch/x86_64/`, see [arch](arch.md)).
## Tasks
+4 -4
View File
@@ -17,7 +17,7 @@ There are two layers:
The framebuffer console draws pixels, which a test can't read without
screen-scraping. So the kernel also writes everything to a **serial port**
(`src/arch/x86_64/serial.zig`, a 16550 UART on COM1). `Console.write` mirrors every
(`src/kernel/arch/x86_64/serial.zig`, a 16550 UART on COM1). `Console.write` mirrors every
byte to it, so all kernel output — boot log, memory summary, exception reports —
appears on serial as plain text.
@@ -29,7 +29,7 @@ a new architecture's UART is what makes the same tests run there.
## In-kernel test cases
Building with `-Dtest-case=<name>` makes the kernel, after normal bring-up, run one
self-test from `src/tests.zig` instead of idling. Each case writes structured
self-test from `src/kernel/tests.zig` instead of idling. Each case writes structured
markers to serial:
```
@@ -108,7 +108,7 @@ firmware, boot method, serial device). The cases are architecture-neutral —
So bringing up a second architecture — an AArch64 Raspberry Pi is the motivating
one — means:
1. implement `src/arch/aarch64/` (CPU ops, its UART, exception vectors, page
1. implement `src/kernel/arch/aarch64/` (CPU ops, its UART, exception vectors, page
tables) behind the same `arch` interface,
2. add an `aarch64` entry to `ARCHES` with its `qemu-system-aarch64` invocation,
@@ -118,7 +118,7 @@ architectures".
## Writing a new case
1. Add a function to `src/tests.zig` and dispatch it in `run` on its name.
1. Add a function to `src/kernel/tests.zig` and dispatch it in `run` on its name.
2. Emit `[PASS]/[FAIL]` lines and a `DANOS-TEST-RESULT:` line (non-faulting cases),
or trigger the condition and rely on the handler's output (faulting cases).
3. Add an entry to `CASES` in `test/qemu_test.py` with the regex that proves it.
+1 -1
View File
@@ -9,7 +9,7 @@ const MemoryMapSlice = uefi.tables.MemoryMapSlice;
/// Name of the kernel ELF on the boot volume (installed to the ESP root by
/// build.zig). UEFI wants a UTF-16, null-terminated path.
const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("danos");
const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("kernel");
/// Physical page size, and the sentinel UEFI uses to seek to end-of-file.
const page_size = 4096;
View File
View File
+1 -1
View File
@@ -1,5 +1,5 @@
//! Shared definitions that form the contract between a bootloader
//! (src/boot/, e.g. efi.zig built as BOOTX64.efi) and the kernel (src/main.zig).
//! (src/boot/, e.g. efi.zig built as BOOTX64.efi) and the kernel (src/kernel/main.zig).
//!
//! Both binaries import this as the "danos" module, so the handoff layout is
//! defined in exactly one place.
+3 -3
View File
@@ -4,7 +4,7 @@
For each test case it builds the kernel with `-Dtest-case=<name>`, boots it
headless in QEMU with the serial port captured to a file, and asserts that the
expected marker appears in that output before a timeout. The kernel's serial log
is machine-readable (see src/tests.zig and src/arch/*/serial.zig), so no
is machine-readable (see src/kernel/tests.zig and src/kernel/arch/*/serial.zig), so no
screen-scraping is involved.
Structured per-architecture so a second CPU (e.g. an AArch64 Raspberry Pi) is a
@@ -56,7 +56,7 @@ ARCHES = {
"/usr/local/share/qemu/edk2-i386-vars.fd", # macOS Homebrew (Intel)
],
"efi_app": ("EFI/BOOT/BOOTX64.efi", "BOOTX64.efi"), # (dest in ESP, name in zig-out/bin)
"kernel": ("danos", "danos"),
"kernel": ("kernel", "kernel"),
# Built as a function so we can splice in per-run paths.
"qemu_args": lambda a, esp, vars_fd, serial: [
"-machine", "q35", "-m", "128M",
@@ -71,7 +71,7 @@ ARCHES = {
],
},
# To add an architecture, e.g. "aarch64": provide its qemu binary, firmware,
# boot method, and the AArch64 kernel/serial support in src/arch/aarch64/.
# boot method, and the AArch64 kernel/serial support in src/kernel/arch/aarch64/.
}
# --- Test cases ------------------------------------------------------------