From 0cc71ec8aae94f63ee497675e4e54a963e79be35 Mon Sep 17 00:00:00 2001 From: Daniel Samson Date: Fri, 3 Jul 2026 12:05:41 +0100 Subject: [PATCH] Built GDT + IDT + exception handlers --- build.zig | 3 + docs/README.md | 14 +++-- docs/arch.md | 9 ++- docs/interrupts.md | 103 +++++++++++++++++++++++++++++++++ src/arch/x86_64/cpu.zig | 34 ++++++++++- src/arch/x86_64/gdt.zig | 38 +++++++++++++ src/arch/x86_64/idt.zig | 122 +++++++++++++++++++++++++++++++++++++++ src/arch/x86_64/isr.s | 123 ++++++++++++++++++++++++++++++++++++++++ src/main.zig | 35 +++++++----- 9 files changed, 458 insertions(+), 23 deletions(-) create mode 100644 docs/interrupts.md create mode 100644 src/arch/x86_64/gdt.zig create mode 100644 src/arch/x86_64/idt.zig create mode 100644 src/arch/x86_64/isr.s diff --git a/build.zig b/build.zig index 4659110..e44f498 100644 --- a/build.zig +++ b/build.zig @@ -17,6 +17,9 @@ pub fn build(b: *std.Build) void { const arch_mod = b.addModule("arch", .{ .root_source_file = b.path("src/arch/x86_64/cpu.zig"), }); + // CPU-exception stubs — real assembly, since they need cross-symbol + // jumps/calls that Zig inline asm can't express (see the file's header). + arch_mod.addAssemblyFile(b.path("src/arch/x86_64/isr.s")); // --- Kernel: freestanding x86_64 ELF, jumped to by the bootloader --- // SSE2 is part of the x86_64 baseline and UEFI leaves it enabled at handoff, diff --git a/docs/README.md b/docs/README.md index 100ae4b..6f9ccb2 100644 --- a/docs/README.md +++ b/docs/README.md @@ -19,7 +19,10 @@ rather than restate it. Roughly in the order things happen at runtime: 5. **[frame-allocator.md](frame-allocator.md) — the physical frame allocator.** The bitmap allocator that hands out and reclaims 4 KiB physical frames from that map — the primitive page tables and the heap will be built on. -6. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and +6. **[interrupts.md](interrupts.md) — interrupts and exceptions.** The GDT and IDT, + the exception stubs, and the handler that reports a CPU fault in red instead of + letting it triple-fault into a silent reset. +7. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and how `while (true) hlt` parks the CPU safely once there's nothing left to do. Cutting across all of these: @@ -34,9 +37,10 @@ The boot flow ties them together: UEFI runs the loader ([efi.md](efi.md)), which queries the **GOP** to pick a graphics mode ([gop.md](gop.md)), hands the kernel a **framebuffer** to draw into ([framebuffer.md](framebuffer.md)) and a **memory map** of physical RAM ([memory-map.md](memory-map.md)); the kernel turns that map -into a **frame allocator** ([frame-allocator.md](frame-allocator.md)), runs — its -CPU-specific bits behind the [arch](arch.md) boundary — and when it has finished, -or panics, it **halts** ([halting.md](halting.md)). +into a **frame allocator** ([frame-allocator.md](frame-allocator.md)), installs +its **descriptor tables** so CPU faults are caught ([interrupts.md](interrupts.md)), +runs — its CPU-specific bits behind the [arch](arch.md) boundary — and when it has +finished, or panics, it **halts** ([halting.md](halting.md)). ## Source map @@ -47,5 +51,5 @@ or panics, it **halts** ([halting.md](halting.md)). | Shared loader↔kernel contract (`BootInfo`, `Framebuffer`, `MemoryMap`, ABI) | `src/root.zig` | | Physical frame allocator | `src/pmm.zig` | | Framebuffer text console | `src/console.zig` | -| Arch-specific kernel code (`halt`, linker script) | `src/arch/x86_64/` | +| Arch-specific kernel code (`halt`, GDT/IDT, exception stubs, linker script) | `src/arch/x86_64/` | | Build + `run-efi` (QEMU/OVMF) | `build.zig` | diff --git a/docs/arch.md b/docs/arch.md index 385ef33..4b05a91 100644 --- a/docs/arch.md +++ b/docs/arch.md @@ -62,8 +62,13 @@ There are really two independent questions, and it's worth not conflating them: ## Current x86_64 contents - **`src/arch/x86_64/cpu.zig`** — the `arch` module root. Exposes `halt()` (see - [halting.md](halting.md)); GDT, IDT and paging will join it here as the kernel - grows. + [halting.md](halting.md)), `init()` (bring up the descriptor tables), + `setFaultHandler`, `readCr2`, and the `CpuState` trap frame. Paging will join it + here as the kernel grows. +- **`src/arch/x86_64/gdt.zig`** / **`idt.zig`** — the GDT and IDT plus CPU-exception + handling (see [interrupts.md](interrupts.md)). +- **`src/arch/x86_64/isr.s`** — the exception stubs and the `lgdt`/`lidt` load + helpers, in real assembly because Zig inline asm can't express them. - **`src/arch/x86_64/linker.ld`** — the kernel link layout (fixed low load address, one PT_LOAD per permission set). diff --git a/docs/interrupts.md b/docs/interrupts.md new file mode 100644 index 0000000..1a555cf --- /dev/null +++ b/docs/interrupts.md @@ -0,0 +1,103 @@ +# Interrupts and exceptions + +When something goes wrong on the CPU — a bad pointer, a divide by zero, a +malformed page table — the processor raises an **exception**. If nothing is set +up to catch it, the fault escalates: the CPU tries to invoke a handler, finds +none, faults again trying to handle *that*, and on the third strike triple-faults, +which on real hardware and in QEMU means a silent reset. Debugging by spontaneous +reboot is miserable. + +This is the machinery that catches those faults and prints what happened instead. +It's all x86_64-specific, so it lives behind the [arch](arch.md) boundary in +`src/arch/x86_64/`. Only the 32 CPU-defined exception vectors are wired up so far; +device interrupts (timer, keyboard, via the APIC) come later, on the same IDT. + +## First the GDT + +In 64-bit long mode, segmentation is mostly switched off — but the CPU still +requires valid **segment descriptors** for code and data, and, crucially, every +IDT gate names a code-segment *selector* that must resolve in the current GDT. The +firmware left a GDT in place, but we don't control it, so we install our own with +known selectors: `0x08` kernel code, `0x10` kernel data. + +`src/arch/x86_64/gdt.zig` holds three flat descriptors — a required null entry, +plus code and data — where the only bits that matter in long mode are the access +byte and the code segment's long-mode (`L`) flag. Loading it (`gdt_flush` in +`isr.s`) does two things: `lgdt`, then reload the segment registers. The data +registers take a plain `mov`, but **CS can't** — so we reload it with a far +return, pushing the new selector and a return address and letting `lretq` pop them +into CS:RIP. + +## Then the IDT + +The **Interrupt Descriptor Table** maps each of 256 vectors to a handler. Each +entry is a 16-byte *gate* holding the handler's address (split across three +fields, a quirk of the format), the code selector (`0x08`), and flags: `0x8E` +means present, ring 0, 64-bit interrupt gate. `src/arch/x86_64/idt.zig` builds the +table, points the first 32 vectors at their stubs, and loads it with `lidt` +(`idt_flush`). + +## The stubs and the trap frame + +On an exception the CPU pushes a small frame (SS, RSP, RFLAGS, CS, RIP) and, for +*some* vectors, an **error code**. That inconsistency is a nuisance, so each stub +in `src/arch/x86_64/isr.s` normalises it: vectors that don't get a hardware error +code push a dummy `0`, then every stub pushes its **vector number** and jumps to a +shared tail, `isr_common`. The tail pushes all the general registers and calls the +Zig handler with a pointer to the whole thing. + +The result on the stack is a uniform **`CpuState`** — register block, then vector +and error code, then the CPU's frame. Its field order in `idt.zig` is exactly the +push order in `isr.s`; the two must stay in sync. + +### Why a separate `.s` file + +The stubs and table-loads are real assembly rather than Zig inline asm because +they need things inline asm on this toolchain can't express: cross-symbol +`jmp`/`call` (a stub jumping to `isr_common`, which calls the exported +`exceptionHandler`), and the `lgdt`/`lidt` memory operands (which LLVM rejects +inline). `build.zig` adds `isr.s` to the arch module. + +## Reporting a fault + +`isr_common` calls `exceptionHandler`, which forwards to a swappable `on_fault` +hook. The generic kernel installs a reporter (`onException` in `main.zig`) that +prints, in red, the exception name and vector, the error code, the faulting RIP +and RSP, and — for a page fault (#PF, vector 14) — the faulting address from +**CR2**. Then it halts. There's no fault *recovery* yet, so every exception is +terminal; the point is that it's now **visible** instead of a silent reset. + +The hook is set before `arch.init()` in `kmain`, so a fault during setup is still +caught. + +## Verifying it + +A temporary `ud2` (unconditional invalid-opcode instruction) in `kmain` produced, +in red: + +``` +CPU EXCEPTION: invalid opcode (vector 6) + error code : 0x0 + RIP : 0x000000000010dadd <- the ud2, in the kernel image at 0x100000+ + RSP : 0x0000000007e8aed0 +``` + +Vector 6 with no error code, a RIP inside the loaded kernel, and a sane RSP +together confirm the whole path: the GDT is active (we're still executing), the +IDT vectored to the right stub, the stub built a correct `CpuState`, and the Zig +handler read it and reported instead of triple-faulting. + +## What's next (not done here) + +- **A TSS with an IST** (interrupt stack table) so the double-fault handler runs + on a known-good stack — important because a double fault often means the current + stack is unusable, and without an IST the handler would itself fault. +- **Device interrupts**: program the local APIC and IO-APIC, wire a timer and the + keyboard onto vectors ≥ 32, and (unlike exceptions) actually *return* from them + with `iretq` — which `isr_common` already does. +- **SSE state**: the stubs save general registers but not the vector registers, so + recoverable interrupts that return to SSE-using code will need that added. Fine + for now, since exceptions here don't return. + +With faults now debuggable, the paging work that comes next — where a wrong +page-table entry means an instant #PF — is far less painful. diff --git a/src/arch/x86_64/cpu.zig b/src/arch/x86_64/cpu.zig index 375f656..cad8ebd 100644 --- a/src/arch/x86_64/cpu.zig +++ b/src/arch/x86_64/cpu.zig @@ -2,7 +2,39 @@ //! it as `@import("arch")` and never names x86_64 directly, so a second //! architecture is added by pointing that module at a different directory in //! build.zig — no change to the generic code. Keep everything CPU-specific here -//! (halt now; GDT, IDT and paging will join it), and nothing generic. +//! (halt, the descriptor tables, later paging), and nothing generic. + +const gdt = @import("gdt.zig"); +const idt = @import("idt.zig"); + +/// The saved register/trap frame passed to a fault handler. +pub const CpuState = idt.CpuState; + +/// Set up the CPU's descriptor tables: our own GDT, then the IDT with exception +/// handlers. After this a CPU fault is reported instead of triple-faulting. +/// Install the fault handler (setFaultHandler) first so early faults are caught. +pub fn init() void { + gdt.init(); + idt.init(); +} + +/// Route CPU exceptions to `handler`, which receives the trap frame and does not +/// return. Until set, faults just halt the core. +pub fn setFaultHandler(handler: *const fn (*const CpuState) noreturn) void { + idt.on_fault = handler; +} + +/// A human-readable name for a CPU exception vector. +pub fn vectorName(vector: u64) []const u8 { + return idt.vectorName(vector); +} + +/// CR2 holds the faulting linear address after a page fault (#PF, vector 14). +pub fn readCr2() u64 { + return asm volatile ("mov %%cr2, %[out]" + : [out] "=r" (-> u64), + ); +} /// Park the core forever. `hlt` drops it into a low-power idle until the next /// interrupt; the loop re-halts on every wake so the stop is permanent. See diff --git a/src/arch/x86_64/gdt.zig b/src/arch/x86_64/gdt.zig new file mode 100644 index 0000000..7da69eb --- /dev/null +++ b/src/arch/x86_64/gdt.zig @@ -0,0 +1,38 @@ +//! Global Descriptor Table. In long mode segmentation is mostly vestigial, but +//! the CPU still needs valid code/data segment descriptors, and the IDT's gates +//! reference a code selector — so we install our own flat GDT with known +//! selectors (0x08 kernel code, 0x10 kernel data) rather than trusting whatever +//! the firmware left in place. + +/// Selectors into the table below (index * 8). +pub const kernel_code = 0x08; +pub const kernel_data = 0x10; + +/// Flat 64-bit descriptors. Base/limit are ignored in long mode; what matters is +/// the access byte and, for code, the long-mode (L) flag. +/// code: present, ring 0, executable, readable, L=1 -> 0x00AF9A00_0000FFFF +/// data: present, ring 0, writable -> 0x00CF9200_0000FFFF +var table = [_]u64{ + 0, // null descriptor (required) + 0x00AF9A000000FFFF, // kernel code + 0x00CF92000000FFFF, // kernel data +}; + +/// The operand `lgdt` wants: table byte-length minus one, then its address. +const Descriptor = packed struct { + limit: u16, + base: u64, +}; + +/// Loads the GDT and reloads the segment registers (including CS). Defined in +/// isr.s — it uses the selectors 0x08 (code) and 0x10 (data) that match `table`. +extern fn gdt_flush(descriptor: *const Descriptor) callconv(.c) void; + +/// Install our GDT and switch onto its segments. +pub fn init() void { + const descriptor = Descriptor{ + .limit = @sizeOf(@TypeOf(table)) - 1, + .base = @intFromPtr(&table), + }; + gdt_flush(&descriptor); +} diff --git a/src/arch/x86_64/idt.zig b/src/arch/x86_64/idt.zig new file mode 100644 index 0000000..b2a4d1e --- /dev/null +++ b/src/arch/x86_64/idt.zig @@ -0,0 +1,122 @@ +//! Interrupt Descriptor Table and the CPU-exception handlers. Without this, any +//! fault (a stray pointer, a bad page-table entry) triple-faults and silently +//! resets the machine. With it, the CPU vectors into our stubs, which capture the +//! register state and hand it to a reporter that prints what went wrong. +//! +//! Only the 32 architecture-defined exception vectors are wired up here; device +//! interrupts (the APIC, timer, keyboard) come later. + +const gdt = @import("gdt.zig"); + +/// The register + trap frame the ISR stubs build on the stack, laid out so the +/// lowest address (where RSP points when we call the handler) is the first field. +/// See the push order in `isrCommon` below. +pub const CpuState = extern struct { + r15: u64, + r14: u64, + r13: u64, + r12: u64, + r11: u64, + r10: u64, + r9: u64, + r8: u64, + rbp: u64, + rdi: u64, + rsi: u64, + rdx: u64, + rcx: u64, + rbx: u64, + rax: u64, + vector: u64, // pushed by the per-vector stub + error_code: u64, // real one from the CPU, or 0 pushed by the stub + rip: u64, // from here down: pushed by the CPU on entry + cs: u64, + rflags: u64, + rsp: u64, + ss: u64, +}; + +/// Where a fault is reported. The kernel overrides this (see setFaultHandler) with +/// something that prints to the console; until then, just stop. +pub var on_fault: *const fn (*const CpuState) noreturn = defaultFault; + +fn defaultFault(_: *const CpuState) noreturn { + while (true) asm volatile ("hlt"); +} + +/// Names for the 32 defined exception vectors, for readable output. +const names = [_][]const u8{ + "divide error", "debug", + "NMI", "breakpoint", + "overflow", "bound range exceeded", + "invalid opcode", "device not available", + "double fault", "coprocessor segment overrun", + "invalid TSS", "segment not present", + "stack-segment fault", "general protection fault", + "page fault", "reserved (15)", + "x87 floating-point", "alignment check", + "machine check", "SIMD floating-point", + "virtualization", "control protection", + "reserved (22)", "reserved (23)", + "reserved (24)", "reserved (25)", + "reserved (26)", "reserved (27)", + "hypervisor injection", "VMM communication", + "security exception", "reserved (31)", +}; + +pub fn vectorName(vector: u64) []const u8 { + return if (vector < names.len) names[vector] else "unknown"; +} + +/// A 64-bit IDT gate descriptor (16 bytes). +const Gate = packed struct { + offset_low: u16, + selector: u16, + ist: u8, // interrupt-stack-table index; 0 = use the current stack + flags: u8, // present, DPL, gate type + offset_mid: u16, + offset_high: u32, + reserved: u32 = 0, +}; + +var idt = [_]Gate{std.mem.zeroes(Gate)} ** 256; + +const Descriptor = packed struct { + limit: u16, + base: u64, +}; + +/// Loads the IDT (`lidt`). Defined in isr.s. +extern fn idt_flush(descriptor: *const Descriptor) callconv(.c) void; + +fn setGate(vector: usize, handler: u64) void { + idt[vector] = .{ + .offset_low = @truncate(handler), + .selector = gdt.kernel_code, + .ist = 0, + .flags = 0x8E, // present, ring 0, 64-bit interrupt gate + .offset_mid = @truncate(handler >> 16), + .offset_high = @truncate(handler >> 32), + }; +} + +/// Point the first 32 vectors at the stubs defined in isr.s and load the IDT. +pub fn init() void { + inline for (0..32) |vector| { + const stub = @extern(*const anyopaque, .{ .name = std.fmt.comptimePrint("isr{d}", .{vector}) }); + setGate(vector, @intFromPtr(stub)); + } + const descriptor = Descriptor{ + .limit = @sizeOf(@TypeOf(idt)) - 1, + .base = @intFromPtr(&idt), + }; + idt_flush(&descriptor); +} + +/// Called by isr_common (isr.s) with a pointer to the trap frame. Exported so the +/// assembly stubs can `call` it by name. +export fn exceptionHandler(state: *const CpuState) callconv(.c) void { + on_fault(state); +} + +const std = @import("std"); diff --git a/src/arch/x86_64/isr.s b/src/arch/x86_64/isr.s new file mode 100644 index 0000000..fb3e990 --- /dev/null +++ b/src/arch/x86_64/isr.s @@ -0,0 +1,123 @@ +# x86_64 low-level entry code: the CPU-exception stubs, plus the GDT/IDT load +# helpers. Kept in a dedicated assembly file rather than inline asm because these +# need real labels and cross-symbol jumps/calls (isr_common, exceptionHandler), +# and because `lgdt`/`lidt` memory operands aren't expressible in Zig inline asm. +# +# Each exception vector normalises the stack to a uniform trap frame — a dummy +# error code where the CPU pushes none, then the vector number — and jumps to the +# shared tail, which saves the general registers and calls the Zig handler with a +# pointer to the frame (matching src/arch/x86_64/idt.zig's CpuState). + +.text + +# gdt_flush(rdi = *GDT descriptor): load the GDT, reload the data segment +# registers to the data selector, and reload CS to the code selector. CS can't be +# set with mov, so we far-return through the caller's own return address. +.global gdt_flush +gdt_flush: + lgdt (%rdi) + mov $0x10, %ax # kernel data selector + mov %ax, %ds + mov %ax, %es + mov %ax, %ss + mov %ax, %fs + mov %ax, %gs + pop %rax # caller's return address + push $0x08 # kernel code selector (new CS) + push %rax # return address (new RIP) + lretq + +# idt_flush(rdi = *IDT descriptor): load the IDT. +.global idt_flush +idt_flush: + lidt (%rdi) + ret + +# Stub for a vector the CPU does NOT push an error code for: push a dummy 0. +.macro STUB_NOERR vec +.global isr\vec +isr\vec: + pushq $0 + pushq $\vec + jmp isr_common +.endm + +# Stub for a vector the CPU DOES push an error code for: leave it in place. +.macro STUB_ERR vec +.global isr\vec +isr\vec: + pushq $\vec + jmp isr_common +.endm + +STUB_NOERR 0 +STUB_NOERR 1 +STUB_NOERR 2 +STUB_NOERR 3 +STUB_NOERR 4 +STUB_NOERR 5 +STUB_NOERR 6 +STUB_NOERR 7 +STUB_ERR 8 +STUB_NOERR 9 +STUB_ERR 10 +STUB_ERR 11 +STUB_ERR 12 +STUB_ERR 13 +STUB_ERR 14 +STUB_NOERR 15 +STUB_NOERR 16 +STUB_ERR 17 +STUB_NOERR 18 +STUB_NOERR 19 +STUB_NOERR 20 +STUB_ERR 21 +STUB_NOERR 22 +STUB_NOERR 23 +STUB_NOERR 24 +STUB_NOERR 25 +STUB_NOERR 26 +STUB_NOERR 27 +STUB_NOERR 28 +STUB_NOERR 29 +STUB_NOERR 30 +STUB_NOERR 31 + +.extern exceptionHandler + +# Shared tail. Register push order here defines the CpuState field order. +isr_common: + push %rax + push %rbx + push %rcx + push %rdx + push %rsi + push %rdi + push %rbp + push %r8 + push %r9 + push %r10 + push %r11 + push %r12 + push %r13 + push %r14 + push %r15 + mov %rsp, %rdi # first argument: pointer to the trap frame + call exceptionHandler + pop %r15 + pop %r14 + pop %r13 + pop %r12 + pop %r11 + pop %r10 + pop %r9 + pop %r8 + pop %rbp + pop %rdi + pop %rsi + pop %rdx + pop %rcx + pop %rbx + pop %rax + add $16, %rsp # drop the vector and error code + iretq diff --git a/src/main.zig b/src/main.zig index ef89db1..3ca79d7 100644 --- a/src/main.zig +++ b/src/main.zig @@ -30,6 +30,11 @@ fn kmain(boot_info: *const BootInfo) noreturn { con.clear(); con_ready = true; + // Catch CPU exceptions before doing anything that might fault: install our + // reporter, then bring up the GDT + IDT. + arch.setFaultHandler(onException); + arch.init(); + con.write("danos: framebuffer console online\n"); con.print(" resolution : {d}x{d}\n", .{ fb.width, fb.height }); con.print(" pitch : {d} bytes\n", .{fb.pitch}); @@ -76,21 +81,6 @@ fn kmain(boot_info: *const BootInfo) noreturn { if (f2) |p| pmm.free(p); con.print(" after free : {d} frames free\n", .{pmm.stats().free_frames}); - // Bring up the physical frame allocator over that map, and prove it works: - // allocate three frames, then hand them back. - pmm.init(boot_info.memory_map); - const s = pmm.stats(); - con.print("\ndanos: frame allocator online\n", .{}); - con.print(" free frames: {d} ({d} MiB)\n", .{ s.free_frames, s.free_frames * danos.page_size / (1024 * 1024) }); - const f0 = pmm.alloc(); - const f1 = pmm.alloc(); - const f2 = pmm.alloc(); - con.print(" alloc x3 : 0x{x} 0x{x} 0x{x}\n", .{ f0 orelse 0, f1 orelse 0, f2 orelse 0 }); - if (f0) |p| pmm.free(p); - if (f1) |p| pmm.free(p); - if (f2) |p| pmm.free(p); - con.print(" after free : {d} frames free\n", .{pmm.stats().free_frames}); - con.write("\nkernel initialised; nothing left to do, halting.\n"); arch.halt(); @@ -101,6 +91,21 @@ fn mib(pages: u64) u64 { return pages * danos.page_size / (1024 * 1024); } +/// Report a CPU exception in red and halt. There's no fault recovery yet, so any +/// exception is terminal — but now it prints what and where instead of silently +/// resetting the machine. +fn onException(state: *const arch.CpuState) noreturn { + if (con_ready) { + con.fg = 0x00ff_5555; + con.print("\nCPU EXCEPTION: {s} (vector {d})\n", .{ arch.vectorName(state.vector), state.vector }); + con.print(" error code : 0x{x}\n", .{state.error_code}); + con.print(" RIP : 0x{x:0>16}\n", .{state.rip}); + con.print(" RSP : 0x{x:0>16}\n", .{state.rsp}); + if (state.vector == 14) con.print(" CR2 (addr) : 0x{x:0>16}\n", .{arch.readCr2()}); + } + arch.halt(); +} + /// Freestanding has no OS to receive a panic. Print it to the console (if it is /// up yet) in red, then halt. pub const panic = std.debug.FullPanic(struct {