From 20b4661ff37df3c8e9047fbcdcfed038e0547add Mon Sep 17 00:00:00 2001 From: Daniel Samson Date: Fri, 3 Jul 2026 13:29:34 +0100 Subject: [PATCH] Hardened paging into a real VMM --- build.zig | 3 + docs/paging.md | 133 +++++++++++++++++------------- docs/testing.md | 4 + src/arch/x86_64/cpu.zig | 20 ++++- src/arch/x86_64/paging.zig | 165 ++++++++++++++++++++++++++++--------- src/efi.zig | 25 ++++-- src/main.zig | 5 +- src/root.zig | 13 +++ src/tests.zig | 73 +++++++++++++--- test/qemu_test.py | 7 ++ 10 files changed, 328 insertions(+), 120 deletions(-) diff --git a/build.zig b/build.zig index c7b5a9a..681b5ab 100644 --- a/build.zig +++ b/build.zig @@ -16,6 +16,9 @@ pub fn build(b: *std.Build) void { // architecture is a matter of pointing this module at a different directory. const arch_mod = b.addModule("arch", .{ .root_source_file = b.path("src/arch/x86_64/cpu.zig"), + .imports = &.{ + .{ .name = "danos", .module = mod }, // paging uses the shared BootInfo/memory-map types + }, }); // CPU-exception stubs — real assembly, since they need cross-symbol // jumps/calls that Zig inline asm can't express (see the file's header). diff --git a/docs/paging.md b/docs/paging.md index 14d5175..362efbe 100644 --- a/docs/paging.md +++ b/docs/paging.md @@ -1,78 +1,101 @@ -# Paging: the kernel's own page tables +# Paging: the kernel's page tables and VMM Every memory access the CPU makes goes through the **page tables**: hardware walks them to translate a virtual address into a physical one, and faults if there's no -valid mapping. Up to now danos ran on the *firmware's* page tables — which live in -memory we'd like to reclaim and which we don't control. This step builds the -kernel's own tables and switches onto them. +valid mapping. danos builds its own tables (rather than staying on the firmware's, +which live in memory we'd like to reclaim and don't control), switches CR3 onto +them, and — crucially — maps with **real permissions**. It's x86_64-specific (the 4-level table format is an Intel/AMD thing), so it lives behind the [arch](arch.md) boundary in `src/arch/x86_64/paging.zig`. -## What we map, and why identity +## The format x86_64 uses **4 levels**: PML4 → PDPT → PD → PT, each a 512-entry table, with 9 -bits of the virtual address indexing each level. A leaf can be a 4 KiB page (at -the PT level) or, with the "huge" bit, a 2 MiB page straight from the PD. +bits of the virtual address indexing each level and the low 12 bits the offset into +the final 4 KiB page. Each entry holds a physical address plus flag bits — +present, writable, and (bit 63) **no-execute**. danos maps everything with 4 KiB +pages: precise, and the extra table memory is negligible against available RAM. -For this first set of tables we **identity-map the low 4 GiB** — virtual address -equals physical address. Identity mapping is the pragmatic bootstrap: it keeps -everything that's already running valid across the CR3 switch without relocating a -single thing. The kernel image (at 1 MiB), the stack, the frame allocator's -bitmap, the framebuffer (at 2 GiB), and device MMIO all sit in the low 4 GiB, so -one flat identity range covers the lot. +## What gets mapped, and with what permissions -Using 2 MiB pages makes the whole map tiny: 4 GiB ÷ 2 MiB = 2048 leaves, which is -one PML4, one PDPT, and four page directories — six frames total, from the -[frame allocator](frame-allocator.md). +The address space is built in three passes (`init`): -## Building and switching +1. **All RAM, identity-mapped RW + NX.** Every non-MMIO region from the + [memory map](memory-map.md) is mapped virtual == physical, read-write and + *non-executable*. Identity mapping keeps everything already running valid across + the CR3 switch (the frame allocator addresses frames by physical address, page + tables are reached the same way, the stack stays put). +2. **The framebuffer and the Local APIC**, the device memory we actually touch, + also RW + NX. Everything else — unbacked address space, other MMIO — is simply + left unmapped, so a stray access faults instead of silently succeeding. +3. **The kernel's own segments, overlaid with their true ELF permissions.** This is + the interesting part. -`paging.init` takes a frame allocator and: +### W^X from the ELF program headers -1. Allocates and zeroes a PML4. **Zeroing matters**: the frame is recycled memory, - and any stale non-zero entry would map a bogus region. -2. Walks PML4 → PDPT → PD for each 2 MiB address, creating intermediate tables on - demand (`descend`) and writing the leaf with present + writable + huge. -3. Loads the PML4's physical address into **CR3**, which both switches address - spaces and flushes the TLB in one instruction. +Blanket RW+NX is fine for data but wrong for the kernel's own code, which must be +executable — and its code must *not* be writable (W^X: no page is both). We get the +right permissions per region straight from the kernel ELF: the **loader already +parses the program headers**, so `efi.zig` records each `PT_LOAD` segment's +address, size and R/W/X flags into `BootInfo`. Pass 3 re-maps those ranges with +flags derived from the ELF flags: -This all works because the firmware's identity map is still active *while we -build*, so a freshly allocated frame's physical address is usable directly as a -pointer. After the CR3 load we're on our own tables — and since those page-table -frames came from low RAM, they remain mapped (and thus editable) for later. +| segment | ELF flags | mapped as | +|---------|-----------|-----------| +| `.text` | R + X | present, **not** writable, **not** NX | +| `.rodata` | R | present, not writable, NX | +| `.data`/`.bss` | R + W | present, writable, NX | + +So code can execute but not be written, and data can be written but not executed. +(Intermediate table entries are left writable and executable so the *leaf's* bits +govern — a page is writable only if every level is, and non-executable if any level +is.) NX itself has to be switched on first via `EFER.NXE`, or the NX bit would be a +reserved bit and fault. + +### The null guard + +Page 0 is deliberately left unmapped. A null (or near-null) pointer dereference now +takes a page fault instead of quietly reading or writing real memory — turning a +whole class of silent bugs into an immediate, located crash. + +## Switching on, and the on-demand API + +Loading the PML4's physical address into **CR3** switches address spaces and +flushes the TLB in one step. This works because the firmware's identity map is +still active *while we build*, so freshly allocated table frames are reachable by +physical address; afterwards they're covered by pass 1. + +`init` keeps the PML4 and the frame allocator around and exposes `map(virt, phys, +writable)` / `unmap(virt)` (with `invlpg` TLB invalidation) — the primitive the +kernel heap will build on to map pages on demand. ## Verifying it -Two temporary tests confirmed both that we switched tables and that faults are -caught with the right detail: +Four tests (see [testing.md](testing.md)) pin down the guarantees: -- **Touch an address above 4 GiB** (`0xdeadbeef000`). Under the firmware's tables - this *didn't* fault (they map a huge range); under ours it does: +- **`vmm`** — map a fresh frame at an unused virtual address, write and read it + back. Proves `map` works end to end. +- **`fault-pf`** — an access far above all mapped RAM faults, with the address in + CR2. Proves we're on our own (deliberately sparse) map. +- **`fault-nx`** — calling into a data page (NX) faults on the instruction fetch. + Proves NX is enforced. +- **`fault-null`** — writing to address 0 faults. Proves the null guard. - ``` - CPU EXCEPTION: page fault (vector 14) - error code : 0x2 (write, page not present) - CR2 (addr) : 0x00000deadbeef000 (the faulting address) - ``` - - A #PF at exactly the unmapped address is proof we're on our own, deliberately - smaller, map — and that CR2 reporting works (see [interrupts.md](interrupts.md)). -- The console keeps working *after* the switch, confirming the framebuffer, kernel - code, and stack are all still mapped. - -> Toolchain note: a volatile store to a *compile-time-constant* address tripped a -> codegen bug in the Zig self-hosted x86_64 backend ("no encoding for mov moffs"). -> Computing the address in a runtime variable sidesteps it (register-relative -> store). Worth remembering when poking fixed MMIO addresses. +> Toolchain notes, both hit while writing the tests: a volatile access to a +> compile-time-*constant* address either trips the self-hosted backend's +> `mov moffs` gap or (for address 0) Zig's null-pointer safety check — so the +> null-guard test launders the address through empty asm and uses an `allowzero` +> pointer to force a real hardware access. And `invlpg`, like `lgdt`, needs its +> operand staged through a register in inline asm. ## What's next (not done here) -- **A higher-half kernel**: map the kernel at a high virtual base (e.g. - `0xffffffff80000000`) so user address space can own the low half later. -- **Real permissions**: today every page is writable and executable. Map code - read-execute, data read-write + no-execute (needs setting EFER.NXE and the NX - bit), and leave a guard page unmapped to catch null-ish dereferences. -- **4 KiB pages / a general `map(virt, phys, flags)`** for fine-grained mappings, - and unmapping with TLB invalidation (`invlpg`). -- **Per-address-space tables** once there are user processes. +- **A kernel heap** — the first real user of `map`, giving the kernel dynamic + allocation. This is the natural next milestone. +- **A higher-half kernel**: relink the kernel at a high virtual base so a future + user address space can own the low half. +- **Per-address-space tables** once there are user processes, and shared/copy-on- + write mappings. +- **Uncacheable MMIO**: the APIC/framebuffer pages are mapped writeback-cacheable; + real hardware wants MMIO marked uncacheable. diff --git a/docs/testing.md b/docs/testing.md index ff532b1..86d1d1e 100644 --- a/docs/testing.md +++ b/docs/testing.md @@ -46,9 +46,13 @@ Current cases: | Case | What it checks | How the harness confirms it | |------|----------------|-----------------------------| | `smoke` | memory map has usable RAM; frame alloc/free; paging active | `DANOS-TEST-RESULT: PASS` | +| `timer` | device interrupts fire and return (tick count advances) | `DANOS-TEST-RESULT: PASS` | +| `vmm` | on-demand `map` works: a mapped page is writable and reads back | `DANOS-TEST-RESULT: PASS` | | `fault-ud` | invalid-opcode exception is caught | serial shows `invalid opcode (vector 6)` | | `fault-pf` | page fault caught with CR2 | `page fault (vector 14)` | | `fault-df` | double fault caught on IST1 (not a triple-fault reset) | `double fault (vector 8)` | +| `fault-nx` | executing a data page (NX) faults | `page fault (vector 14)` | +| `fault-null` | dereferencing the unmapped page 0 faults | `page fault (vector 14)` | The faulting cases don't print a result line — they deliberately raise a CPU exception, and the harness asserts on the [exception report](interrupts.md) the diff --git a/src/arch/x86_64/cpu.zig b/src/arch/x86_64/cpu.zig index fd891e5..9a4b7df 100644 --- a/src/arch/x86_64/cpu.zig +++ b/src/arch/x86_64/cpu.zig @@ -4,6 +4,7 @@ //! build.zig — no change to the generic code. Keep everything CPU-specific here //! (halt, the descriptor tables, later paging), and nothing generic. +const danos = @import("danos"); const gdt = @import("gdt.zig"); const tss = @import("tss.zig"); const idt = @import("idt.zig"); @@ -35,10 +36,21 @@ pub fn init() void { idt.init(); } -/// Build the kernel's own page tables and switch onto them. Needs a physical -/// frame allocator; call once the frame allocator is up. -pub fn enablePaging(allocFrame: *const fn () ?u64) void { - paging.init(allocFrame); +/// Build the kernel's own page tables (with real permissions) and switch onto +/// them. Needs the frame allocator and the boot info (for the memory map and the +/// kernel's segment layout). Call once the frame allocator is up. +pub fn enablePaging(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void { + paging.init(allocFrame, boot_info); +} + +/// Map a page into the kernel address space (non-executable). For the heap, etc. +pub fn mapPage(virt: u64, phys: u64, writable: bool) void { + paging.map(virt, phys, writable); +} + +/// Remove a kernel mapping. +pub fn unmapPage(virt: u64) void { + paging.unmap(virt); } /// CR3 holds the physical address of the active top-level page table. diff --git a/src/arch/x86_64/paging.zig b/src/arch/x86_64/paging.zig index 3f9097d..ae450aa 100644 --- a/src/arch/x86_64/paging.zig +++ b/src/arch/x86_64/paging.zig @@ -1,70 +1,153 @@ -//! The kernel's own 4-level page tables. Until now we've been running on the -//! firmware's page tables, which live in memory we'd like to reclaim and which we -//! don't control. This builds our own set, identity-mapping the low 4 GiB, and -//! loads CR3 to switch onto them. +//! The kernel's page tables and virtual memory manager. //! -//! "Identity map" means virtual address == physical address, which keeps -//! everything already running — kernel image, stack, framebuffer, the frame -//! allocator's bitmap, MMIO — valid across the switch without having to relocate -//! anything. 4 GiB comfortably covers all of that (RAM low down, the framebuffer -//! at 2 GiB, device MMIO below 4 GiB). Higher-half mapping and per-region -//! permissions come later; this is the bootstrap. +//! Builds our own 4-level page tables and switches CR3 onto them, replacing the +//! firmware's. Unlike the earlier bootstrap this maps with real permissions: +//! RAM is identity-mapped read-write + no-execute, the kernel's own segments get +//! their ELF permissions (code R+X, rodata R, data R+W+NX), and page 0 is left +//! unmapped as a null guard. It also exposes map/unmap for on-demand mapping, +//! which the kernel heap will build on. //! -//! We use 2 MiB pages, so the whole map is cheap: a PML4, a PDPT, and four page -//! directories. +//! Everything is 4 KiB pages — precise and simple; the extra table memory is +//! negligible against available RAM. -const KiB = 1024; -const MiB = 1024 * KiB; -const GiB = 1024 * MiB; +const danos = @import("danos"); +const io = @import("io.zig"); +const page_size = danos.page_size; + +// Page-table entry bits. const present: u64 = 1 << 0; const writable: u64 = 1 << 1; -const huge: u64 = 1 << 7; // in a PD entry: this maps a 2 MiB page directly -const addr_mask: u64 = 0x000F_FFFF_FFFF_F000; // physical address bits of an entry +const no_execute: u64 = 1 << 63; +const addr_mask: u64 = 0x000F_FFFF_FFFF_F000; + +// ELF segment flags (p_flags). +const pf_x: u32 = 1; +const pf_w: u32 = 2; + +// State kept after init so map()/unmap() can serve later callers (e.g. the heap). +var kernel_pml4: u64 = 0; +var alloc_frame: *const fn () ?u64 = undefined; -/// A page table is 512 64-bit entries. While building the tables the firmware's -/// identity map is still active, so a physical frame address is usable directly. fn tableAt(phys: u64) *[512]u64 { return @ptrFromInt(phys); } -/// Allocate and zero a fresh page-table frame. Zeroing matters: the frame comes -/// from previously-used memory, and any stale non-zero entry would map a bogus -/// region. -fn allocTable(allocFrame: *const fn () ?u64) u64 { - const frame = allocFrame() orelse @panic("paging: out of memory building page tables"); +fn allocTable() u64 { + const frame = alloc_frame() orelse @panic("paging: out of memory building page tables"); @memset(tableAt(frame)[0..], 0); return frame; } -/// Return the table an entry points at, creating it if the entry is empty. -fn descend(entry: *u64, allocFrame: *const fn () ?u64) u64 { +/// Return the table an entry points at, creating it if empty. Intermediate +/// entries are writable and executable so the leaf's bits govern (a page is +/// writable only if every level is; non-executable if any level is). +fn descend(entry: *u64) u64 { if (entry.* & present != 0) return entry.* & addr_mask; - const frame = allocTable(allocFrame); + const frame = allocTable(); entry.* = frame | present | writable; return frame; } -/// Identity-map one 2 MiB page: walk PML4 -> PDPT -> PD and write the leaf. -fn mapHugePage(pml4: u64, addr: u64, allocFrame: *const fn () ?u64) void { - const pml4e = &tableAt(pml4)[(addr >> 39) & 0x1FF]; - const pdpt = descend(pml4e, allocFrame); - const pdpte = &tableAt(pdpt)[(addr >> 30) & 0x1FF]; - const pd = descend(pdpte, allocFrame); - tableAt(pd)[(addr >> 21) & 0x1FF] = addr | present | writable | huge; +/// Map one 4 KiB page `virt` -> `phys` with `flags` (present is added). +fn mapPage(pml4: u64, virt: u64, phys: u64, flags: u64) void { + const pml4e = &tableAt(pml4)[(virt >> 39) & 0x1FF]; + const pdpt = descend(pml4e); + const pdpte = &tableAt(pdpt)[(virt >> 30) & 0x1FF]; + const pd = descend(pdpte); + const pde = &tableAt(pd)[(virt >> 21) & 0x1FF]; + const pt = descend(pde); + tableAt(pt)[(virt >> 12) & 0x1FF] = (phys & addr_mask) | flags | present; } -/// Build the tables, identity-map the low 4 GiB, and switch CR3 onto them. -pub fn init(allocFrame: *const fn () ?u64) void { - const pml4 = allocTable(allocFrame); - var addr: u64 = 0; - while (addr < 4 * GiB) : (addr += 2 * MiB) { - mapHugePage(pml4, addr, allocFrame); +/// Identity-map [base, base+len) with `flags`, rounded out to whole pages. +fn mapRangeIdentity(pml4: u64, base: u64, len: u64, flags: u64) void { + var addr = base & ~@as(u64, page_size - 1); + const end = base + len; + while (addr < end) : (addr += page_size) { + if (addr == 0) continue; // leave page 0 unmapped: the null guard + mapPage(pml4, addr, addr, flags); } - // Loading CR3 switches address spaces and flushes the TLB in one step. +} + +fn regions(mm: danos.MemoryMap) []const danos.MemoryRegion { + return @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len]; +} + +/// Enable the NX bit in the page-table format (EFER.NXE). Must happen before we +/// load a CR3 whose entries set the NX bit, or those bits are reserved and fault. +fn enableNx() void { + const efer_msr = 0xC0000080; + io.wrmsr(efer_msr, io.rdmsr(efer_msr) | (1 << 11)); +} + +/// Build the address space and switch onto it. +pub fn init(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) void { + alloc_frame = allocFrame; + enableNx(); + const pml4 = allocTable(); + + // 1. All RAM identity-mapped RW + NX. Non-RAM (MMIO) is skipped and stays + // unmapped unless mapped explicitly below. + for (regions(boot_info.memory_map)) |r| { + if (r.kind == .mmio) continue; + mapRangeIdentity(pml4, r.base, r.pages * page_size, present | writable | no_execute); + } + + // 2. The framebuffer and the Local APIC (device memory we need), RW + NX. + const fb = boot_info.framebuffer; + mapRangeIdentity(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute); + mapPage(pml4, 0xFEE00000, 0xFEE00000, present | writable | no_execute); + + // 3. Overlay the kernel's own segments with their real ELF permissions, + // replacing the blanket RW+NX from step 1: code becomes R+X, rodata R, + // data R+W+NX. This is the W^X guarantee. + for (boot_info.kernel_segments[0..boot_info.kernel_segment_count]) |seg| { + var flags: u64 = present; + if (seg.flags & pf_w != 0) flags |= writable; + if (seg.flags & pf_x == 0) flags |= no_execute; + var addr = seg.virt; + const end = seg.virt + seg.pages * page_size; + while (addr < end) : (addr += page_size) mapPage(pml4, addr, addr, flags); + } + + kernel_pml4 = pml4; asm volatile ("mov %[pml4], %%cr3" : : [pml4] "r" (pml4), : .{ .memory = true } ); } + +/// Map a page into the kernel address space on demand (for the heap, etc.). +/// `writable_page` controls W; pages are always mapped non-executable. +pub fn map(virt: u64, phys: u64, writable_page: bool) void { + var flags: u64 = present | no_execute; + if (writable_page) flags |= writable; + mapPage(kernel_pml4, virt, phys, flags); + invalidate(virt); +} + +/// Remove a mapping and flush it from the TLB. +pub fn unmap(virt: u64) void { + const pml4e = tableAt(kernel_pml4)[(virt >> 39) & 0x1FF]; + if (pml4e & present == 0) return; + const pdpte = tableAt(pml4e & addr_mask)[(virt >> 30) & 0x1FF]; + if (pdpte & present == 0) return; + const pde = tableAt(pdpte & addr_mask)[(virt >> 21) & 0x1FF]; + if (pde & present == 0) return; + tableAt(pde & addr_mask)[(virt >> 12) & 0x1FF] = 0; + invalidate(virt); +} + +fn invalidate(virt: u64) void { + // invlpg needs its operand via a register-indirect memory reference that Zig + // inline asm won't form directly, so stage the address in a register first. + asm volatile ( + \\mov %[v], %%rax + \\invlpg (%%rax) + : + : [v] "r" (virt), + : .{ .rax = true, .memory = true } + ); +} diff --git a/src/efi.zig b/src/efi.zig index a11df67..e691f65 100644 --- a/src/efi.zig +++ b/src/efi.zig @@ -36,9 +36,11 @@ fn boot() !noreturn { var boot_info: BootInfo = .{ .framebuffer = try queryFramebuffer(bs), .memory_map = undefined, // filled by exitBootServices, just below + .kernel_segments = undefined, // filled by loadKernel + .kernel_segment_count = 0, }; - const entry = try loadKernel(bs); + const entry = try loadKernel(bs, &boot_info); log("danos: kernel loaded, exiting boot services\r\n"); boot_info.memory_map = try exitBootServices(bs); @@ -152,7 +154,7 @@ fn edidNative(edid: []const u8) ?Resolution { /// Open the kernel on the volume we booted from, read it into a pool buffer, /// load its segments, and return the physical entry-point address. -fn loadKernel(bs: *uefi.tables.BootServices) !usize { +fn loadKernel(bs: *uefi.tables.BootServices, boot_info: *BootInfo) !usize { const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse return error.NoLoadedImage; const device = loaded.device_handle orelse return error.NoBootDevice; @@ -181,11 +183,13 @@ fn loadKernel(bs: *uefi.tables.BootServices) !usize { read_total += n; } - return loadElf(bs, image); + return loadElf(bs, image, boot_info); } -/// Validate the ELF and copy every PT_LOAD segment to its physical address. -fn loadElf(bs: *uefi.tables.BootServices, image: []u8) !usize { +/// Validate the ELF, copy every PT_LOAD segment to its physical address, and +/// record each segment's layout so the kernel can re-map itself with the right +/// permissions. +fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_info: *BootInfo) !usize { if (image.len < @sizeOf(elf.Elf64_Ehdr)) return error.NotElf; const ehdr: *const elf.Elf64_Ehdr = @ptrCast(@alignCast(image.ptr)); @@ -214,6 +218,17 @@ fn loadElf(bs: *uefi.tables.BootServices, image: []u8) !usize { const off: usize = @intCast(phdr.p_offset); @memcpy(bytes[0..file_sz], image[off..][0..file_sz]); @memset(bytes[file_sz..mem_sz], 0); + + // Record it (identity-loaded: virtual == physical) for the kernel's VMM. + const n = boot_info.kernel_segment_count; + if (n < boot_info.kernel_segments.len) { + boot_info.kernel_segments[n] = .{ + .virt = phdr.p_vaddr, + .pages = pages, + .flags = phdr.p_flags, + }; + boot_info.kernel_segment_count = n + 1; + } } return @intCast(ehdr.e_entry); diff --git a/src/main.zig b/src/main.zig index 5dba7bc..f2b5c94 100644 --- a/src/main.zig +++ b/src/main.zig @@ -86,10 +86,11 @@ fn kmain(boot_info: *const BootInfo) noreturn { if (f2) |p| pmm.free(p); con.print(" after free : {d} frames free\n", .{pmm.stats().free_frames}); - // Switch off the firmware's page tables onto our own. - arch.enablePaging(pmm.alloc); + // Switch off the firmware's page tables onto our own (with real permissions). + arch.enablePaging(pmm.alloc, boot_info); con.print("\ndanos: paging enabled\n", .{}); con.print(" page tables: CR3 = 0x{x:0>16}\n", .{arch.readCr3()}); + con.print(" kernel segs: {d} (mapped with W^X permissions)\n", .{boot_info.kernel_segment_count}); // Start the timer and unmask interrupts — the kernel now has a heartbeat. arch.startTimer(); diff --git a/src/root.zig b/src/root.zig index 87be8d7..23203c0 100644 --- a/src/root.zig +++ b/src/root.zig @@ -76,9 +76,22 @@ pub const MemoryMap = extern struct { len: usize, }; +/// One PT_LOAD segment of the kernel image, so the kernel can re-map itself with +/// correct permissions (code R+X, rodata R, data R+W+NX). `flags` are raw ELF +/// segment flags: PF_X=1, PF_W=2, PF_R=4. +pub const KernelSegment = extern struct { + virt: u64, + pages: u64, + flags: u32, + _pad: u32 = 0, +}; + /// Handoff structure the bootloader fills in and passes to the kernel's /// `_start` in RDI (the first argument under the SysV AMD64 C ABI). pub const BootInfo = extern struct { framebuffer: Framebuffer, memory_map: MemoryMap, + /// The kernel's own PT_LOAD segments (it has three: text, rodata, data). + kernel_segments: [8]KernelSegment, + kernel_segment_count: u32, }; diff --git a/src/tests.zig b/src/tests.zig index 83fd77a..685855f 100644 --- a/src/tests.zig +++ b/src/tests.zig @@ -33,17 +33,33 @@ fn check(name: []const u8, ok: bool) void { } } +/// Emit the overall result line the harness matches, then the done sentinel. +fn result() void { + log("DANOS-TEST-RESULT: {s} ({d} passed, {d} failed)\n", .{ + if (failed == 0) "PASS" else "FAIL", + passed, + failed, + }); + log("DANOS-TEST-DONE\n", .{}); +} + pub fn run(case: []const u8, boot_info: *const BootInfo) void { if (eql(case, "smoke")) { smoke(boot_info); } else if (eql(case, "timer")) { timer(); + } else if (eql(case, "vmm")) { + vmm(); } else if (eql(case, "fault-ud")) { faultInvalidOpcode(); } else if (eql(case, "fault-pf")) { faultPageFault(); } else if (eql(case, "fault-df")) { faultDoubleFault(); + } else if (eql(case, "fault-nx")) { + faultNoExecute(); + } else if (eql(case, "fault-null")) { + faultNull(); } else { log("DANOS-TEST-RESULT: FAIL (unknown case '{s}')\n", .{case}); } @@ -85,12 +101,7 @@ fn smoke(boot_info: *const BootInfo) void { const cr3 = arch.readCr3(); check("paging active (CR3 set)", cr3 != 0 and cr3 % danos.page_size == 0); - log("DANOS-TEST-RESULT: {s} ({d} passed, {d} failed)\n", .{ - if (failed == 0) "PASS" else "FAIL", - passed, - failed, - }); - log("DANOS-TEST-DONE\n", .{}); + result(); } /// Verify device interrupts fire and return: the timer tick counter must advance @@ -105,12 +116,26 @@ fn timer() void { while (arch.ticks() == start and spins < 5_000_000_000) spins +%= 1; check("timer interrupts advance the tick count", arch.ticks() > start); - log("DANOS-TEST-RESULT: {s} ({d} passed, {d} failed)\n", .{ - if (failed == 0) "PASS" else "FAIL", - passed, - failed, - }); - log("DANOS-TEST-DONE\n", .{}); + result(); +} + +/// Verify the on-demand VMM: map a fresh frame at an unused virtual address, and +/// check it's writable and reads back. +fn vmm() void { + log("DANOS-TEST-BEGIN: vmm\n", .{}); + const frame = pmm.alloc(); + check("frame available to map", frame != null); + if (frame) |phys| { + var virt: u64 = 0x0000_4000_0000_0000; // canonical, well clear of everything mapped + arch.mapPage(virt, phys, true); + const p: *volatile u64 = @ptrFromInt(virt); + p.* = 0xdead_c0de_cafe_babe; + check("mapped page is writable and reads back", p.* == 0xdead_c0de_cafe_babe); + arch.unmapPage(virt); + pmm.free(phys); + virt += 0; + } + result(); } fn faultInvalidOpcode() void { @@ -118,11 +143,33 @@ fn faultInvalidOpcode() void { asm volatile ("ud2"); } +/// Verify NX: fetching an instruction from a data page (mapped no-execute) faults. +fn faultNoExecute() void { + log("DANOS-TEST-BEGIN: fault-nx\n", .{}); + var scratch: u64 = 0xC3; // a lone `ret` — harmless if NX somehow let it run + const f: *const fn () void = @ptrFromInt(@intFromPtr(&scratch)); + f(); // instruction fetch from an NX page -> #PF before it executes + log("DANOS-TEST-RESULT: FAIL (NX not enforced)\n", .{}); +} + +/// Verify the null guard: dereferencing address 0 (page 0 left unmapped) faults. +fn faultNull() void { + log("DANOS-TEST-BEGIN: fault-null\n", .{}); + // Launder the address through empty asm so the compiler no longer knows it's + // 0 (otherwise it folds a null-pointer safety panic instead of doing the real + // access). `allowzero` skips the same null check on the cast. The write then + // hits the unmapped page 0 and takes a real hardware #PF. + var addr: u64 = 0; + addr = asm ("" : [ret] "=r" (-> u64) : [in] "0" (addr)); + const p: *allowzero volatile u64 = @ptrFromInt(addr); + p.* = 1; +} + fn faultPageFault() void { log("DANOS-TEST-BEGIN: fault-pf\n", .{}); // Runtime address so the backend emits a register store (not a `mov moffs`, // which the self-hosted x86_64 backend can't encode). - var addr: u64 = 0xdeadbeef000; // above our identity-mapped 4 GiB + var addr: u64 = 0xdeadbeef000; // well above all mapped RAM const p: *volatile u64 = @ptrFromInt(addr); p.* = 1; addr += 0; diff --git a/test/qemu_test.py b/test/qemu_test.py index 1fec8cb..d7c54c7 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -66,9 +66,16 @@ CASES = [ {"name": "timer", "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, + {"name": "vmm", + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, {"name": "fault-ud", "expect": r"invalid opcode \(vector 6\)"}, {"name": "fault-pf", "expect": r"page fault \(vector 14\)"}, {"name": "fault-df", "expect": r"double fault \(vector 8\)"}, + {"name": "fault-nx", + "expect": r"page fault \(vector 14\)", + "fail": r"NX not enforced"}, + {"name": "fault-null", "expect": r"page fault \(vector 14\)"}, ] TIMEOUT = 30 # seconds per case