wake application processors to long mode

INIT-SIPI-SIPI plus a self-relocating real-mode trampoline.
This commit is contained in:
Daniel Samson
2026-07-08 12:35:30 +01:00
parent 36c29d2d6d
commit ed7f542006
12 changed files with 482 additions and 8 deletions
+101
View File
@@ -0,0 +1,101 @@
//! Application-processor (AP) bring-up: waking the cores the firmware left parked.
//!
//! The firmware starts only the bootstrap processor (BSP); the others sit idle until
//! the kernel wakes them with an INIT–SIPI–SIPI sequence (Intel SDM Vol.3, "MP
//! Initialization"). A woken core begins in 16-bit real mode at a low physical page,
//! runs the [trampoline](trampoline.s) up into 64-bit long mode, and lands in
//! `apEntry` here. This module copies the trampoline into place, patches its
//! per-AP parameters, drives the wake IPIs, and waits for each core to report in.
//!
//! Cores are brought up **one at a time**: a single trampoline page and parameter
//! block are reused, so the BSP patches, wakes, and waits for one AP before the
//! next. The mechanism-vs-policy split matches the rest of the kernel — the generic
//! scheduler decides *what* runs where; this just gets a core executing 64-bit code.
//!
//! This is step 3a: an AP climbs to long mode, publishes its per-CPU pointer, marks
//! itself alive, and parks. Entering the scheduler (its own TSS, LAPIC timer, and
//! the run loop) is the next step.
const io = @import("io.zig");
const apic = @import("apic.zig");
/// IA32_GS_BASE — the per-CPU data pointer (see cpu.zig; kept in sync here so the AP
/// path doesn't depend on cpu.zig and risk an import cycle).
const ia32_gs_base = 0xC000_0101;
/// Physical address of the trampoline page (page-aligned, below 1 MiB). Set by
/// `prepare`; the low 20 bits are always zero, so `phys >> 12` is the SIPI vector.
var tramp_phys: u64 = 0;
/// Set to 1 by a freshly-woken AP once it reaches `apEntry` and finishes its own
/// bring-up. The BSP clears it before each wake and polls it afterwards — a simple
/// one-at-a-time handshake (only one AP is being started at any moment).
var ap_alive: u32 = 0;
/// Copy the trampoline blob to its low page. Call once, after the page has been
/// allocated and made executable, before waking any AP.
pub fn prepare(phys: u64) void {
tramp_phys = phys;
const start = @extern([*]const u8, .{ .name = "ap_trampoline_start" });
const end = @extern([*]const u8, .{ .name = "ap_trampoline_end" });
const len = @intFromPtr(end) - @intFromPtr(start);
const dst: [*]u8 = @ptrFromInt(phys);
@memcpy(dst[0..len], start[0..len]);
}
/// Address of a patchable trampoline parameter, by symbol name: the copied blob's
/// base plus the field's offset within it (a same-section symbol difference). The
/// pointer is `align(1)` — the fields aren't 8-aligned within the blob, and x86
/// tolerates unaligned stores, so we don't force layout constraints on the asm.
fn param(comptime name: []const u8) *align(1) volatile u64 {
const start = @intFromPtr(@extern([*]const u8, .{ .name = "ap_trampoline_start" }));
const sym = @intFromPtr(@extern([*]const u8, .{ .name = name }));
return @ptrFromInt(tramp_phys + (sym - start));
}
/// Wake the core with Local APIC id `apic_id`, hand it `stack_top` and `percpu` (its
/// per-CPU pointer), and wait for it to come alive. Returns false if it doesn't
/// report in within the timeout (left parked, no harm to the running system).
/// `cr3` is the kernel page tables the AP adopts. Precondition: `prepare` has run.
pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, cr3: u64) bool {
param("ap_tramp_cr3").* = cr3;
param("ap_tramp_stack").* = stack_top;
param("ap_tramp_entry").* = @intFromPtr(&apEntry);
param("ap_tramp_percpu").* = percpu;
@atomicStore(u32, &ap_alive, 0, .seq_cst);
const vector: u8 = @intCast(tramp_phys >> 12);
apic.sendInit(apic_id);
delayMicros(10_000); // 10 ms INIT settle
apic.sendStartup(apic_id, vector);
delayMicros(200);
apic.sendStartup(apic_id, vector);
// Wait up to 100 ms for the AP to reach apEntry and set the flag.
const deadline = apic.millis() + 100;
while (apic.millis() < deadline) {
if (@atomicLoad(u32, &ap_alive, .acquire) != 0) return true;
asm volatile ("pause");
}
return false;
}
/// Busy-wait `us` microseconds against the calibrated TSC clock (the AP wake happens
/// after the timer is up, so the clock is available).
fn delayMicros(us: u64) void {
const start = apic.micros();
while (apic.micros() - start < us) asm volatile ("pause");
}
/// The 64-bit entry every AP lands on, called from the trampoline with its per-CPU
/// pointer in RDI. Adopts the shared descriptor tables, publishes its per-CPU
/// pointer, signals the BSP it's alive, and (for now) parks. Never returns.
fn apEntry(percpu: usize) callconv(.c) noreturn {
// Step 3a: minimal. The core keeps the trampoline's descriptor tables, publishes
// its per-CPU pointer, signals the BSP, and parks with interrupts off. Loading
// this core's own kernel GDT/IDT/TSS and entering the scheduler is step 3b.
io.wrmsr(ia32_gs_base, percpu); // publish per-CPU pointer (GS base)
@atomicStore(u32, &ap_alive, 1, .release); // "I'm up" — BSP is polling this
while (true) asm volatile ("hlt"); // parked (3b enters the scheduler here)
}