diff --git a/build.zig b/build.zig index a688e6c..3c5a9be 100644 --- a/build.zig +++ b/build.zig @@ -121,7 +121,7 @@ pub fn build(b: *std.Build) void { .root_source_file = b.path("src/kernel/main.zig"), .target = kernel_target, .optimize = optimize, - .code_model = .small, // kernel is linked in the low 2 GiB (see image_base) + .code_model = .kernel, // kernel runs in the top 2 GiB (higher half) .red_zone = false, // interrupts would corrupt the SysV red zone .single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores .sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide @@ -139,14 +139,14 @@ pub fn build(b: *std.Build) void { exe.setLinkerScript(b.path("src/kernel/arch/x86_64/linker.ld")); exe.entry = .{ .symbol_name = "_start" }; // The self-hosted linker ignores parts of the linker script (PHDRS, - // /DISCARD/, section order); the higher-half layout depends on the script - // being authoritative, so pin the kernel to LLVM + LLD. + // /DISCARD/, AT(), section order); the higher-half layout depends on the + // script being authoritative, so pin the kernel to LLVM + LLD. exe.use_llvm = true; exe.use_lld = true; - // Physical address the bootloader loads the kernel to (identity-mapped under - // UEFI). Overrides Zig's default image base so the linker script's layout is - // honoured; adjust here if it collides with firmware-reserved memory. - exe.image_base = 0x100000; // 1 MiB + // Higher-half virtual base (matches KERNEL_VIRT_BASE in linker.ld); the + // linker's AT() clauses give each segment a low physical load address + // (.text at 1 MiB), which the loader allocates and copies into. + exe.image_base = 0xFFFFFFFF80100000; b.installArtifact(exe); diff --git a/src/kernel/arch/x86_64/isr.s b/src/kernel/arch/x86_64/isr.s index 3d8c050..ab39ce8 100644 --- a/src/kernel/arch/x86_64/isr.s +++ b/src/kernel/arch/x86_64/isr.s @@ -10,6 +10,28 @@ .text +# _start: the kernel entry. The loader jumps here (higher-half address) with +# boot_info in RDI, still on the loader's low stack. Switch to a kernel-owned +# stack in .bss (the loader stack is a low address that goes away once the low +# half is dropped), keeping RDI, then call the Zig entry. kmainEntry never +# returns; the hlt loop is a belt-and-braces backstop. +.global _start +_start: + leaq bootstrap_stack_top(%rip), %rsp + call kmainEntry +1: hlt + jmp 1b + +# The kernel's initial stack (used until the scheduler hands each task its own). +# 64 KiB: kmain's discovery path includes the recursive AML interpreter, so it +# needs more than a token stack. Lives in .bss (zeroed, higher-half). +.section .bss +.balign 16 +bootstrap_stack: + .skip 65536 +bootstrap_stack_top: +.text + # gdt_flush(rdi = *GDT descriptor): load the GDT, reload the data segment # registers to the data selector, and reload CS to the code selector. CS can't be # set with mov, so we far-return through the caller's own return address. @@ -152,14 +174,14 @@ user_prog_start: user_prog_end: # The isolation-proof program: read a kernel-only page from ring 3. The LAPIC -# page is unconditionally mapped supervisor (paging.zig), so this must take a -# #PF with error code 0x5 (present | user) before any access happens. The -# address goes through %ecx (zero-extending) — an absolute `mov 0xFEE00000,%rax` -# would sign-extend the disp32 to 0xFFFFFFFF_FEE00000 and miss the page. +# lives in the kernel's physmap (physmap_base + 0xFEE00000) as a supervisor +# page, so this must take a #PF with error code 0x5 (present | user) before any +# access happens. movabs loads the full 64-bit higher-half address (a disp32 +# would sign-extend and miss). .global user_pf_start .global user_pf_end user_pf_start: - mov $0xFEE00000, %ecx + movabs $0xFFFF8800FEE00000, %rcx mov (%rcx), %rax 1: jmp 1b user_pf_end: diff --git a/src/kernel/arch/x86_64/linker.ld b/src/kernel/arch/x86_64/linker.ld index 50bf499..523bdd2 100644 --- a/src/kernel/arch/x86_64/linker.ld +++ b/src/kernel/arch/x86_64/linker.ld @@ -1,12 +1,18 @@ -/* Kernel link layout. +/* Kernel link layout — higher half. * - * The kernel is linked at a fixed low physical address (set by `image_base` in - * build.zig). UEFI runs with memory identity-mapped, so the bootloader can load - * each PT_LOAD segment to the physical address matching its virtual address and - * jump straight to _start — no page tables to build yet. (Moving to a - * higher-half virtual base is a later step, once the bootloader sets up paging.) + * The kernel is linked to *run* in the higher half (virtual base + * 0xFFFF_FFFF_8000_0000, matching danos.kernel_virt_base and build.zig's + * image_base) but is *loaded* low. Each section's load address (LMA) is its + * virtual address minus KERNEL_VIRT_BASE via AT(), so the ELF's p_paddr lands + * at a low physical address (.text at 1 MiB) that the loader can allocate and + * copy into. The loader maps p_vaddr (high) -> p_paddr (low) in its bootstrap + * tables and jumps to the high entry; the kernel then builds its own tables + * with the physmap and abandons the identity map. Requires LLD (build.zig pins + * it) — the self-hosted linker ignores PHDRS/AT()/section order. */ +KERNEL_VIRT_BASE = 0xFFFFFFFF80000000; + ENTRY(_start) /* One loadable segment per permission set, so the loader can map .text as R+X, @@ -18,23 +24,22 @@ PHDRS { } SECTIONS { - .text ALIGN(4K) : { + .text ALIGN(4K) : AT(ADDR(.text) - KERNEL_VIRT_BASE) { *(.text .text.*) } :text - .rodata ALIGN(4K) : { + .rodata ALIGN(4K) : AT(ADDR(.rodata) - KERNEL_VIRT_BASE) { *(.rodata .rodata.*) } :rodata - .data ALIGN(4K) : { + .data ALIGN(4K) : AT(ADDR(.data) - KERNEL_VIRT_BASE) { *(.data .data.*) } :data /* .bss occupies memory but not file space. The loader zeroes it via the * gap between each PT_LOAD segment's file size and memory size, so no - * boundary symbols are needed here. (Zig's self-hosted linker also does not - * yet honour linker-script symbol assignments.) */ - .bss ALIGN(4K) : { + * boundary symbols are needed here. */ + .bss ALIGN(4K) : AT(ADDR(.bss) - KERNEL_VIRT_BASE) { *(.bss .bss.*) *(COMMON) } :data diff --git a/src/kernel/main.zig b/src/kernel/main.zig index 5e13062..e93edd8 100644 --- a/src/kernel/main.zig +++ b/src/kernel/main.zig @@ -40,11 +40,13 @@ var ap_trampoline_page: u64 = 0; /// Kernel entry point. The bootloader jumps here after `ExitBootServices` with a /// pointer to the handoff data. There is no runtime, no stack unwinding, and no /// caller to return to, so this never returns. -export fn _start(boot_info: *const BootInfo) callconv(kernel_abi) noreturn { - // The loader passes a low physical pointer (its own stack). Reach it — and - // everything it points at — through the physmap, so it stays valid once the - // low identity map is gone. The physmap base is the same under the loader's - // bootstrap tables and the kernel's own. +/// The real entry (`_start`, in isr.s) installs a kernel-owned stack in .bss +/// then calls this with the loader's `boot_info` pointer in RDI. We can't keep +/// running on the loader's stack: it's a low physical address that the identity +/// map covers only transitionally, and vanishes once the kernel drops the low +/// half. `boot_info` (also low) is reached through the physmap — its base is the +/// same under the loader's bootstrap tables and the kernel's own. +export fn kmainEntry(boot_info: *const BootInfo) callconv(kernel_abi) noreturn { kmain(@ptrFromInt(danos.physToVirt(@intFromPtr(boot_info)))); }