M2 step 4: relink the kernel into the higher half
The kernel now links at 0xFFFF_FFFF_8000_0000 (code_model .kernel) and loads low via the linker script's AT() clauses (.text at 1 MiB). A new asm _start installs a 64 KiB kernel-owned .bss stack — the loader stack is a low address that goes away with the identity map — and calls the Zig entry, which reaches boot_info through the physmap. The user-pf test blob reads the LAPIC through the physmap window (still present|user, ec 0x5). The low identity map still coexists in the kernel's tables as the safety net. Suite 27/27. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Fable 5
parent
724de7bbd0
commit
f57a73e8a1
@@ -121,7 +121,7 @@ pub fn build(b: *std.Build) void {
|
|||||||
.root_source_file = b.path("src/kernel/main.zig"),
|
.root_source_file = b.path("src/kernel/main.zig"),
|
||||||
.target = kernel_target,
|
.target = kernel_target,
|
||||||
.optimize = optimize,
|
.optimize = optimize,
|
||||||
.code_model = .small, // kernel is linked in the low 2 GiB (see image_base)
|
.code_model = .kernel, // kernel runs in the top 2 GiB (higher half)
|
||||||
.red_zone = false, // interrupts would corrupt the SysV red zone
|
.red_zone = false, // interrupts would corrupt the SysV red zone
|
||||||
.single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores
|
.single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores
|
||||||
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
.sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide
|
||||||
@@ -139,14 +139,14 @@ pub fn build(b: *std.Build) void {
|
|||||||
exe.setLinkerScript(b.path("src/kernel/arch/x86_64/linker.ld"));
|
exe.setLinkerScript(b.path("src/kernel/arch/x86_64/linker.ld"));
|
||||||
exe.entry = .{ .symbol_name = "_start" };
|
exe.entry = .{ .symbol_name = "_start" };
|
||||||
// The self-hosted linker ignores parts of the linker script (PHDRS,
|
// The self-hosted linker ignores parts of the linker script (PHDRS,
|
||||||
// /DISCARD/, section order); the higher-half layout depends on the script
|
// /DISCARD/, AT(), section order); the higher-half layout depends on the
|
||||||
// being authoritative, so pin the kernel to LLVM + LLD.
|
// script being authoritative, so pin the kernel to LLVM + LLD.
|
||||||
exe.use_llvm = true;
|
exe.use_llvm = true;
|
||||||
exe.use_lld = true;
|
exe.use_lld = true;
|
||||||
// Physical address the bootloader loads the kernel to (identity-mapped under
|
// Higher-half virtual base (matches KERNEL_VIRT_BASE in linker.ld); the
|
||||||
// UEFI). Overrides Zig's default image base so the linker script's layout is
|
// linker's AT() clauses give each segment a low physical load address
|
||||||
// honoured; adjust here if it collides with firmware-reserved memory.
|
// (.text at 1 MiB), which the loader allocates and copies into.
|
||||||
exe.image_base = 0x100000; // 1 MiB
|
exe.image_base = 0xFFFFFFFF80100000;
|
||||||
|
|
||||||
b.installArtifact(exe);
|
b.installArtifact(exe);
|
||||||
|
|
||||||
|
|||||||
@@ -10,6 +10,28 @@
|
|||||||
|
|
||||||
.text
|
.text
|
||||||
|
|
||||||
|
# _start: the kernel entry. The loader jumps here (higher-half address) with
|
||||||
|
# boot_info in RDI, still on the loader's low stack. Switch to a kernel-owned
|
||||||
|
# stack in .bss (the loader stack is a low address that goes away once the low
|
||||||
|
# half is dropped), keeping RDI, then call the Zig entry. kmainEntry never
|
||||||
|
# returns; the hlt loop is a belt-and-braces backstop.
|
||||||
|
.global _start
|
||||||
|
_start:
|
||||||
|
leaq bootstrap_stack_top(%rip), %rsp
|
||||||
|
call kmainEntry
|
||||||
|
1: hlt
|
||||||
|
jmp 1b
|
||||||
|
|
||||||
|
# The kernel's initial stack (used until the scheduler hands each task its own).
|
||||||
|
# 64 KiB: kmain's discovery path includes the recursive AML interpreter, so it
|
||||||
|
# needs more than a token stack. Lives in .bss (zeroed, higher-half).
|
||||||
|
.section .bss
|
||||||
|
.balign 16
|
||||||
|
bootstrap_stack:
|
||||||
|
.skip 65536
|
||||||
|
bootstrap_stack_top:
|
||||||
|
.text
|
||||||
|
|
||||||
# gdt_flush(rdi = *GDT descriptor): load the GDT, reload the data segment
|
# gdt_flush(rdi = *GDT descriptor): load the GDT, reload the data segment
|
||||||
# registers to the data selector, and reload CS to the code selector. CS can't be
|
# registers to the data selector, and reload CS to the code selector. CS can't be
|
||||||
# set with mov, so we far-return through the caller's own return address.
|
# set with mov, so we far-return through the caller's own return address.
|
||||||
@@ -152,14 +174,14 @@ user_prog_start:
|
|||||||
user_prog_end:
|
user_prog_end:
|
||||||
|
|
||||||
# The isolation-proof program: read a kernel-only page from ring 3. The LAPIC
|
# The isolation-proof program: read a kernel-only page from ring 3. The LAPIC
|
||||||
# page is unconditionally mapped supervisor (paging.zig), so this must take a
|
# lives in the kernel's physmap (physmap_base + 0xFEE00000) as a supervisor
|
||||||
# #PF with error code 0x5 (present | user) before any access happens. The
|
# page, so this must take a #PF with error code 0x5 (present | user) before any
|
||||||
# address goes through %ecx (zero-extending) — an absolute `mov 0xFEE00000,%rax`
|
# access happens. movabs loads the full 64-bit higher-half address (a disp32
|
||||||
# would sign-extend the disp32 to 0xFFFFFFFF_FEE00000 and miss the page.
|
# would sign-extend and miss).
|
||||||
.global user_pf_start
|
.global user_pf_start
|
||||||
.global user_pf_end
|
.global user_pf_end
|
||||||
user_pf_start:
|
user_pf_start:
|
||||||
mov $0xFEE00000, %ecx
|
movabs $0xFFFF8800FEE00000, %rcx
|
||||||
mov (%rcx), %rax
|
mov (%rcx), %rax
|
||||||
1: jmp 1b
|
1: jmp 1b
|
||||||
user_pf_end:
|
user_pf_end:
|
||||||
|
|||||||
@@ -1,12 +1,18 @@
|
|||||||
/* Kernel link layout.
|
/* Kernel link layout — higher half.
|
||||||
*
|
*
|
||||||
* The kernel is linked at a fixed low physical address (set by `image_base` in
|
* The kernel is linked to *run* in the higher half (virtual base
|
||||||
* build.zig). UEFI runs with memory identity-mapped, so the bootloader can load
|
* 0xFFFF_FFFF_8000_0000, matching danos.kernel_virt_base and build.zig's
|
||||||
* each PT_LOAD segment to the physical address matching its virtual address and
|
* image_base) but is *loaded* low. Each section's load address (LMA) is its
|
||||||
* jump straight to _start — no page tables to build yet. (Moving to a
|
* virtual address minus KERNEL_VIRT_BASE via AT(), so the ELF's p_paddr lands
|
||||||
* higher-half virtual base is a later step, once the bootloader sets up paging.)
|
* at a low physical address (.text at 1 MiB) that the loader can allocate and
|
||||||
|
* copy into. The loader maps p_vaddr (high) -> p_paddr (low) in its bootstrap
|
||||||
|
* tables and jumps to the high entry; the kernel then builds its own tables
|
||||||
|
* with the physmap and abandons the identity map. Requires LLD (build.zig pins
|
||||||
|
* it) — the self-hosted linker ignores PHDRS/AT()/section order.
|
||||||
*/
|
*/
|
||||||
|
|
||||||
|
KERNEL_VIRT_BASE = 0xFFFFFFFF80000000;
|
||||||
|
|
||||||
ENTRY(_start)
|
ENTRY(_start)
|
||||||
|
|
||||||
/* One loadable segment per permission set, so the loader can map .text as R+X,
|
/* One loadable segment per permission set, so the loader can map .text as R+X,
|
||||||
@@ -18,23 +24,22 @@ PHDRS {
|
|||||||
}
|
}
|
||||||
|
|
||||||
SECTIONS {
|
SECTIONS {
|
||||||
.text ALIGN(4K) : {
|
.text ALIGN(4K) : AT(ADDR(.text) - KERNEL_VIRT_BASE) {
|
||||||
*(.text .text.*)
|
*(.text .text.*)
|
||||||
} :text
|
} :text
|
||||||
|
|
||||||
.rodata ALIGN(4K) : {
|
.rodata ALIGN(4K) : AT(ADDR(.rodata) - KERNEL_VIRT_BASE) {
|
||||||
*(.rodata .rodata.*)
|
*(.rodata .rodata.*)
|
||||||
} :rodata
|
} :rodata
|
||||||
|
|
||||||
.data ALIGN(4K) : {
|
.data ALIGN(4K) : AT(ADDR(.data) - KERNEL_VIRT_BASE) {
|
||||||
*(.data .data.*)
|
*(.data .data.*)
|
||||||
} :data
|
} :data
|
||||||
|
|
||||||
/* .bss occupies memory but not file space. The loader zeroes it via the
|
/* .bss occupies memory but not file space. The loader zeroes it via the
|
||||||
* gap between each PT_LOAD segment's file size and memory size, so no
|
* gap between each PT_LOAD segment's file size and memory size, so no
|
||||||
* boundary symbols are needed here. (Zig's self-hosted linker also does not
|
* boundary symbols are needed here. */
|
||||||
* yet honour linker-script symbol assignments.) */
|
.bss ALIGN(4K) : AT(ADDR(.bss) - KERNEL_VIRT_BASE) {
|
||||||
.bss ALIGN(4K) : {
|
|
||||||
*(.bss .bss.*)
|
*(.bss .bss.*)
|
||||||
*(COMMON)
|
*(COMMON)
|
||||||
} :data
|
} :data
|
||||||
|
|||||||
+7
-5
@@ -40,11 +40,13 @@ var ap_trampoline_page: u64 = 0;
|
|||||||
/// Kernel entry point. The bootloader jumps here after `ExitBootServices` with a
|
/// Kernel entry point. The bootloader jumps here after `ExitBootServices` with a
|
||||||
/// pointer to the handoff data. There is no runtime, no stack unwinding, and no
|
/// pointer to the handoff data. There is no runtime, no stack unwinding, and no
|
||||||
/// caller to return to, so this never returns.
|
/// caller to return to, so this never returns.
|
||||||
export fn _start(boot_info: *const BootInfo) callconv(kernel_abi) noreturn {
|
/// The real entry (`_start`, in isr.s) installs a kernel-owned stack in .bss
|
||||||
// The loader passes a low physical pointer (its own stack). Reach it — and
|
/// then calls this with the loader's `boot_info` pointer in RDI. We can't keep
|
||||||
// everything it points at — through the physmap, so it stays valid once the
|
/// running on the loader's stack: it's a low physical address that the identity
|
||||||
// low identity map is gone. The physmap base is the same under the loader's
|
/// map covers only transitionally, and vanishes once the kernel drops the low
|
||||||
// bootstrap tables and the kernel's own.
|
/// half. `boot_info` (also low) is reached through the physmap — its base is the
|
||||||
|
/// same under the loader's bootstrap tables and the kernel's own.
|
||||||
|
export fn kmainEntry(boot_info: *const BootInfo) callconv(kernel_abi) noreturn {
|
||||||
kmain(@ptrFromInt(danos.physToVirt(@intFromPtr(boot_info))));
|
kmain(@ptrFromInt(danos.physToVirt(@intFromPtr(boot_info))));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user