From bb7597ea0bc241f55b16f3a4188b20d55492a802 Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Wed, 8 Jul 2026 22:54:52 +0100 Subject: [PATCH] =?UTF-8?q?M2=20step=205:=20drop=20the=20low=20half=20?= =?UTF-8?q?=E2=80=94=20a=20true=20higher-half=20kernel?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit paging.init now maps only the physmap, the framebuffer/LAPIC windows, and the kernel's own segments; the entire low canonical half is left to user space. Every higher-half PML4 entry is pre-created so a per-process address space can share the kernel half by copying PML4[256..512), with an assert against late top-half entries and a 4 GiB guard on pre-switch table frames. The AP trampoline's low identity page is now created transiently by arm() and unmapped by disarm(); startAp asserts the page-table root is 32-bit addressable. Docs (paging.md) updated. Suite 27/27; 4-core normal boot reaches /sbin/init. Co-Authored-By: Claude Fable 5 --- docs/paging.md | 61 ++++++++++++++++++++++------ src/kernel/arch/x86_64/paging.zig | 66 ++++++++++++++++++++----------- src/kernel/arch/x86_64/smp.zig | 3 ++ src/kernel/tests.zig | 2 +- 4 files changed, 96 insertions(+), 36 deletions(-) diff --git a/docs/paging.md b/docs/paging.md index 9d8d9a3..a9806fb 100644 --- a/docs/paging.md +++ b/docs/paging.md @@ -17,20 +17,54 @@ the final 4 KiB page. Each entry holds a physical address plus flag bits — present, writable, and (bit 63) **no-execute**. danos maps everything with 4 KiB pages: precise, and the extra table memory is negligible against available RAM. +## Higher half: the address-space layout + +danos is a **higher-half kernel**. The kernel is linked to run at +`0xFFFF_FFFF_8000_0000` but loaded low (the linker script's `AT()` gives each +segment a physical load address at 1 MiB up; the bootloader maps the high link +address to the low load address in its bootstrap tables and jumps in). The entire +**low canonical half is reserved for user space**; the kernel lives in the top half +alongside a **physmap** — a straight window onto all of physical memory at +`physmap_base + phys`. Wherever the kernel needs to touch a physical address (a +page-table frame, an ACPI table, a device register), it adds that constant: +`danos.physToVirt(phys)`. The layout constants live in `src/root.zig`: + +| region | virtual base | PML4 slot | +|--------|--------------|-----------| +| user image + stack | `0x0000_7000_0000_0000` | 224 (low half) | +| kernel heap | `0xFFFF_8000_0000_0000` | 256 | +| physmap (all RAM + MMIO windows) | `0xFFFF_8800_0000_0000` + phys | 272 | +| kernel image | `0xFFFF_FFFF_8000_0000` | 511 | + +The bootloader builds temporary **bootstrap tables** (identity + a 4 GiB physmap + +the high kernel) so it can switch CR3 and jump to the high entry; the kernel then +builds its own precise tables below and abandons them. Because both use the same +`physmap_base`, any physmap pointer minted before the switch stays valid after it. + ## What gets mapped, and with what permissions -The address space is built in three passes (`init`): +The address space is built in four passes (`init`): -1. **All RAM, identity-mapped RW + NX.** Every non-MMIO region from the - [memory map](memory-map.md) is mapped virtual == physical, read-write and - *non-executable*. Identity mapping keeps everything already running valid across - the CR3 switch (the frame allocator addresses frames by physical address, page - tables are reached the same way, the stack stays put). -2. **The framebuffer and the Local APIC**, the device memory we actually touch, - also RW + NX. Everything else — unbacked address space, other MMIO — is simply - left unmapped, so a stray access faults instead of silently succeeding. -3. **The kernel's own segments, overlaid with their true ELF permissions.** This is +1. **All RAM in the physmap, RW + NX.** Every non-MMIO region from the + [memory map](memory-map.md) is mapped at `physToVirt(phys)`, read-write and + *non-executable*. There is **no low/identity mapping** — the low half is user + space. (Frames the kernel touches while still building these tables are reached + through the loader's bootstrap physmap, which covers the low 4 GiB; both the + frame allocator and the table builder scan low-address-up, so those frames stay + under that limit.) +2. **The framebuffer and the Local APIC**, the device memory the kernel touches + directly, as physmap windows (RW + NX). Other MMIO is mapped on demand by + `mapMmio`, also into the physmap; everything else is left unmapped, so a stray + access faults instead of silently succeeding. +3. **The kernel's own segments, overlaid with their true ELF permissions**, at + their high link addresses mapped to their low physical load addresses. This is the interesting part. +4. **Every higher-half PML4 entry pre-created** (an empty PDPT where none exists + yet). The kernel half is then a fixed set of top-level slots, so a per-process + address space can share it by copying `PML4[256..512)` once — growth beneath + those slots (heap, on-demand MMIO) propagates to every address space because + they share the PDPTs. `init` asserts no new higher-half PML4 entry appears + afterward. ### W^X from the ELF program headers @@ -55,9 +89,10 @@ reserved bit and fault. ### The null guard -Page 0 is deliberately left unmapped. A null (or near-null) pointer dereference now -takes a page fault instead of quietly reading or writing real memory — turning a -whole class of silent bugs into an immediate, located crash. +The whole low half is unmapped except for explicit user mappings, so page 0 (and +every near-null address) is unmapped by construction. A null (or near-null) pointer +dereference in the kernel takes a page fault instead of quietly reading or writing +real memory — turning a whole class of silent bugs into an immediate, located crash. ## Switching on, and the on-demand API diff --git a/src/kernel/arch/x86_64/paging.zig b/src/kernel/arch/x86_64/paging.zig index 6d4ff75..987ea06 100644 --- a/src/kernel/arch/x86_64/paging.zig +++ b/src/kernel/arch/x86_64/paging.zig @@ -30,6 +30,23 @@ const pf_w: u32 = 2; var kernel_pml4: u64 = 0; var alloc_frame: *const fn () ?u64 = undefined; +/// Set once the kernel is running on its own tables (past the CR3 load in +/// `init`). Before that, the kernel reaches page-table frames through the +/// *loader's* bootstrap physmap, which only covers the low 4 GiB — so every +/// frame allocated for a table during that window must be below 4 GiB. Both the +/// frame allocator and this code scan from low addresses up, so it holds +/// naturally; the assertion in `allocTable` makes a violation loud rather than +/// a silent fault. After the switch the kernel's own physmap covers all RAM. +var on_own_tables = false; + +/// Set at the end of `init`. Guards against a new *higher-half* PML4 entry being +/// created afterward: the kernel half is pre-populated at init and then shared +/// by copying PML4[256..512) into every process address space (M3), so a late +/// top-half entry would be invisible to already-created address spaces. +var init_done = false; + +const bootstrap_physmap_limit: u64 = 4 << 30; + /// Dereference a page-table frame by its physical address, via the physmap. /// This is the single hinge for the higher-half move: page tables hold physical /// frame addresses (pmm gives out physical frames, and CR3/PTEs must be @@ -42,6 +59,8 @@ fn tableAt(phys: u64) *[512]u64 { fn allocTable() u64 { const frame = alloc_frame() orelse @panic("paging: out of memory building page tables"); + if (!on_own_tables and frame >= bootstrap_physmap_limit) + @panic("paging: table frame above the 4 GiB bootstrap physmap"); @memset(tableAt(frame)[0..], 0); return frame; } @@ -59,6 +78,11 @@ fn descend(entry: *u64) u64 { /// Map one 4 KiB page `virt` -> `phys` with `flags` (present is added). fn mapPage(pml4: u64, virt: u64, phys: u64, flags: u64) void { const pml4e = &tableAt(pml4)[(virt >> 39) & 0x1FF]; + // The kernel half is fixed after init: every top-half PML4 entry is + // pre-created so address spaces can share it by copying these slots. A new + // one here would be invisible to address spaces already made. + if (init_done and (virt >> 63) == 1 and pml4e.* & present == 0) + @panic("paging: new higher-half PML4 entry after init"); const pdpt = descend(pml4e); const pdpte = &tableAt(pdpt)[(virt >> 30) & 0x1FF]; const pd = descend(pdpte); @@ -67,16 +91,6 @@ fn mapPage(pml4: u64, virt: u64, phys: u64, flags: u64) void { tableAt(pt)[(virt >> 12) & 0x1FF] = (phys & addr_mask) | flags | present; } -/// Identity-map [base, base+len) with `flags`, rounded out to whole pages. -fn mapRangeIdentity(pml4: u64, base: u64, len: u64, flags: u64) void { - var addr = base & ~@as(u64, page_size - 1); - const end = base + len; - while (addr < end) : (addr += page_size) { - if (addr == 0) continue; // leave page 0 unmapped: the null guard - mapPage(pml4, addr, addr, flags); - } -} - /// Map [phys_base, phys_base+len) into the physmap (at physToVirt(phys)) with /// `flags`, rounded out to whole pages. This is how the kernel keeps a permanent /// window onto physical memory once the low identity map goes away. @@ -105,27 +119,23 @@ pub fn init(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) voi enableNx(); const pml4 = allocTable(); - // 1. All RAM in the physmap (physToVirt(phys)) RW + NX, plus — during the - // higher-half transition — a low identity map so any not-yet-converted - // physical deref still resolves. Non-RAM (MMIO) is skipped here and - // mapped explicitly below. The identity half is removed in a later step. + // 1. All RAM in the physmap (physToVirt(phys)) RW + NX. No identity/low-half + // mapping: the low half belongs to user space. MMIO is skipped here and + // mapped on demand (mapMmio) or explicitly below. for (regions(boot_info.memory_map)) |r| { if (r.kind == .mmio) continue; mapRangePhysmap(pml4, r.base, r.pages * page_size, present | writable | no_execute); - mapRangeIdentity(pml4, r.base, r.pages * page_size, present | writable | no_execute); } - // 2. The framebuffer and the Local APIC (device memory we need), RW + NX — - // in the physmap and (transitionally) identity. + // 2. Physmap windows for the framebuffer and the Local APIC (device memory + // the kernel touches directly), RW + NX. const fb = boot_info.framebuffer; mapRangePhysmap(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute); - mapRangeIdentity(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute); mapPage(pml4, danos.physToVirt(0xFEE00000), 0xFEE00000, present | writable | no_execute); - mapPage(pml4, 0xFEE00000, 0xFEE00000, present | writable | no_execute); - // 3. Overlay the kernel's own segments with their real ELF permissions, - // replacing the blanket RW+NX from step 1: code becomes R+X, rodata R, - // data R+W+NX. This is the W^X guarantee. + // 3. The kernel's own segments at their higher-half link addresses, mapped + // to their low physical load addresses with real ELF permissions: code + // R+X, rodata R, data R+W+NX. This is the W^X guarantee. for (boot_info.kernel_segments[0..boot_info.kernel_segment_count]) |seg| { var flags: u64 = present; if (seg.flags & pf_w != 0) flags |= writable; @@ -136,12 +146,24 @@ pub fn init(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) voi } } + // 4. Pre-create every higher-half PML4 entry (an empty PDPT where none + // exists yet), so the whole kernel half is a fixed set of top-level + // slots. A process address space (M3) then shares the kernel half simply + // by copying PML4[256..512) — growth beneath these slots (heap, on-demand + // MMIO) propagates to every address space because they share the PDPTs. + for (256..512) |i| { + const e = &tableAt(pml4)[i]; + if (e.* & present == 0) e.* = allocTable() | present | writable; + } + kernel_pml4 = pml4; asm volatile ("mov %[pml4], %%cr3" : : [pml4] "r" (pml4), : .{ .memory = true } ); + on_own_tables = true; // now on the kernel's physmap (covers all RAM) + init_done = true; // the kernel half is fixed from here } /// Map a page into the kernel address space on demand (for the heap, etc.). diff --git a/src/kernel/arch/x86_64/smp.zig b/src/kernel/arch/x86_64/smp.zig index 2b92ba4..949dd8a 100644 --- a/src/kernel/arch/x86_64/smp.zig +++ b/src/kernel/arch/x86_64/smp.zig @@ -118,6 +118,9 @@ fn param(comptime name: []const u8) *align(1) volatile u64 { /// the running system). `cr3` is the kernel page tables the AP adopts. Precondition: /// `setTrampolinePage` has run. pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, index: usize, cr3: u64) bool { + // The trampoline loads CR3 with a 32-bit `movl` before it reaches long mode, + // so the page-table root must be addressable in 32 bits. + if (cr3 >= (1 << 32)) @panic("smp: kernel page tables above 4 GiB"); arm(); defer disarm(); diff --git a/src/kernel/tests.zig b/src/kernel/tests.zig index 60b4432..557f11f 100644 --- a/src/kernel/tests.zig +++ b/src/kernel/tests.zig @@ -144,7 +144,7 @@ fn smoke(boot_info: *const BootInfo) void { // The memory map has some usable RAM. const mm = boot_info.memory_map; - const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len]; + const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physToVirt(mm.regions)))[0..mm.len]; var usable: u64 = 0; for (regions) |r| { if (r.kind == .usable) usable += r.pages;