M2 step 4: relink the kernel into the higher half

The kernel now links at 0xFFFF_FFFF_8000_0000 (code_model .kernel) and
loads low via the linker script's AT() clauses (.text at 1 MiB). A new
asm _start installs a 64 KiB kernel-owned .bss stack — the loader stack
is a low address that goes away with the identity map — and calls the
Zig entry, which reaches boot_info through the physmap. The user-pf test
blob reads the LAPIC through the physmap window (still present|user, ec
0x5). The low identity map still coexists in the kernel's tables as the
safety net. Suite 27/27.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Daniel Samson
2026-07-08 22:45:56 +01:00
co-authored by Claude Fable 5
parent 724de7bbd0
commit f57a73e8a1
4 changed files with 58 additions and 29 deletions
+7 -7
View File
@@ -121,7 +121,7 @@ pub fn build(b: *std.Build) void {
.root_source_file = b.path("src/kernel/main.zig"),
.target = kernel_target,
.optimize = optimize,
.code_model = .small, // kernel is linked in the low 2 GiB (see image_base)
.code_model = .kernel, // kernel runs in the top 2 GiB (higher half)
.red_zone = false, // interrupts would corrupt the SysV red zone
.single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
@@ -139,14 +139,14 @@ pub fn build(b: *std.Build) void {
exe.setLinkerScript(b.path("src/kernel/arch/x86_64/linker.ld"));
exe.entry = .{ .symbol_name = "_start" };
// The self-hosted linker ignores parts of the linker script (PHDRS,
// /DISCARD/, section order); the higher-half layout depends on the script
// being authoritative, so pin the kernel to LLVM + LLD.
// /DISCARD/, AT(), section order); the higher-half layout depends on the
// script being authoritative, so pin the kernel to LLVM + LLD.
exe.use_llvm = true;
exe.use_lld = true;
// Physical address the bootloader loads the kernel to (identity-mapped under
// UEFI). Overrides Zig's default image base so the linker script's layout is
// honoured; adjust here if it collides with firmware-reserved memory.
exe.image_base = 0x100000; // 1 MiB
// Higher-half virtual base (matches KERNEL_VIRT_BASE in linker.ld); the
// linker's AT() clauses give each segment a low physical load address
// (.text at 1 MiB), which the loader allocates and copies into.
exe.image_base = 0xFFFFFFFF80100000;
b.installArtifact(exe);
+27 -5
View File
@@ -10,6 +10,28 @@
.text
# _start: the kernel entry. The loader jumps here (higher-half address) with
# boot_info in RDI, still on the loader's low stack. Switch to a kernel-owned
# stack in .bss (the loader stack is a low address that goes away once the low
# half is dropped), keeping RDI, then call the Zig entry. kmainEntry never
# returns; the hlt loop is a belt-and-braces backstop.
.global _start
_start:
leaq bootstrap_stack_top(%rip), %rsp
call kmainEntry
1: hlt
jmp 1b
# The kernel's initial stack (used until the scheduler hands each task its own).
# 64 KiB: kmain's discovery path includes the recursive AML interpreter, so it
# needs more than a token stack. Lives in .bss (zeroed, higher-half).
.section .bss
.balign 16
bootstrap_stack:
.skip 65536
bootstrap_stack_top:
.text
# gdt_flush(rdi = *GDT descriptor): load the GDT, reload the data segment
# registers to the data selector, and reload CS to the code selector. CS can't be
# set with mov, so we far-return through the caller's own return address.
@@ -152,14 +174,14 @@ user_prog_start:
user_prog_end:
# The isolation-proof program: read a kernel-only page from ring 3. The LAPIC
# page is unconditionally mapped supervisor (paging.zig), so this must take a
# #PF with error code 0x5 (present | user) before any access happens. The
# address goes through %ecx (zero-extending) — an absolute `mov 0xFEE00000,%rax`
# would sign-extend the disp32 to 0xFFFFFFFF_FEE00000 and miss the page.
# lives in the kernel's physmap (physmap_base + 0xFEE00000) as a supervisor
# page, so this must take a #PF with error code 0x5 (present | user) before any
# access happens. movabs loads the full 64-bit higher-half address (a disp32
# would sign-extend and miss).
.global user_pf_start
.global user_pf_end
user_pf_start:
mov $0xFEE00000, %ecx
movabs $0xFFFF8800FEE00000, %rcx
mov (%rcx), %rax
1: jmp 1b
user_pf_end:
+17 -12
View File
@@ -1,12 +1,18 @@
/* Kernel link layout.
/* Kernel link layout — higher half.
*
* The kernel is linked at a fixed low physical address (set by `image_base` in
* build.zig). UEFI runs with memory identity-mapped, so the bootloader can load
* each PT_LOAD segment to the physical address matching its virtual address and
* jump straight to _start — no page tables to build yet. (Moving to a
* higher-half virtual base is a later step, once the bootloader sets up paging.)
* The kernel is linked to *run* in the higher half (virtual base
* 0xFFFF_FFFF_8000_0000, matching danos.kernel_virt_base and build.zig's
* image_base) but is *loaded* low. Each section's load address (LMA) is its
* virtual address minus KERNEL_VIRT_BASE via AT(), so the ELF's p_paddr lands
* at a low physical address (.text at 1 MiB) that the loader can allocate and
* copy into. The loader maps p_vaddr (high) -> p_paddr (low) in its bootstrap
* tables and jumps to the high entry; the kernel then builds its own tables
* with the physmap and abandons the identity map. Requires LLD (build.zig pins
* it) — the self-hosted linker ignores PHDRS/AT()/section order.
*/
KERNEL_VIRT_BASE = 0xFFFFFFFF80000000;
ENTRY(_start)
/* One loadable segment per permission set, so the loader can map .text as R+X,
@@ -18,23 +24,22 @@ PHDRS {
}
SECTIONS {
.text ALIGN(4K) : {
.text ALIGN(4K) : AT(ADDR(.text) - KERNEL_VIRT_BASE) {
*(.text .text.*)
} :text
.rodata ALIGN(4K) : {
.rodata ALIGN(4K) : AT(ADDR(.rodata) - KERNEL_VIRT_BASE) {
*(.rodata .rodata.*)
} :rodata
.data ALIGN(4K) : {
.data ALIGN(4K) : AT(ADDR(.data) - KERNEL_VIRT_BASE) {
*(.data .data.*)
} :data
/* .bss occupies memory but not file space. The loader zeroes it via the
* gap between each PT_LOAD segment's file size and memory size, so no
* boundary symbols are needed here. (Zig's self-hosted linker also does not
* yet honour linker-script symbol assignments.) */
.bss ALIGN(4K) : {
* boundary symbols are needed here. */
.bss ALIGN(4K) : AT(ADDR(.bss) - KERNEL_VIRT_BASE) {
*(.bss .bss.*)
*(COMMON)
} :data
+7 -5
View File
@@ -40,11 +40,13 @@ var ap_trampoline_page: u64 = 0;
/// Kernel entry point. The bootloader jumps here after `ExitBootServices` with a
/// pointer to the handoff data. There is no runtime, no stack unwinding, and no
/// caller to return to, so this never returns.
export fn _start(boot_info: *const BootInfo) callconv(kernel_abi) noreturn {
// The loader passes a low physical pointer (its own stack). Reach it — and
// everything it points at — through the physmap, so it stays valid once the
// low identity map is gone. The physmap base is the same under the loader's
// bootstrap tables and the kernel's own.
/// The real entry (`_start`, in isr.s) installs a kernel-owned stack in .bss
/// then calls this with the loader's `boot_info` pointer in RDI. We can't keep
/// running on the loader's stack: it's a low physical address that the identity
/// map covers only transitionally, and vanishes once the kernel drops the low
/// half. `boot_info` (also low) is reached through the physmap — its base is the
/// same under the loader's bootstrap tables and the kernel's own.
export fn kmainEntry(boot_info: *const BootInfo) callconv(kernel_abi) noreturn {
kmain(@ptrFromInt(danos.physToVirt(@intFromPtr(boot_info))));
}