wake application processors to long mode

INIT-SIPI-SIPI plus a self-relocating real-mode trampoline.
This commit is contained in:
Daniel Samson
2026-07-08 12:35:30 +01:00
parent 36c29d2d6d
commit ed7f542006
12 changed files with 482 additions and 8 deletions
+37
View File
@@ -44,11 +44,15 @@ const spurious_vector = 47;
// LAPIC register offsets.
const reg_spurious = 0x0F0;
const reg_eoi = 0x0B0;
const reg_icr_low = 0x300; // interrupt command register, low dword (writing it sends)
const reg_icr_high = 0x310; // ICR high dword (destination APIC id in bits 24-31)
const reg_lvt_timer = 0x320;
const reg_timer_initial = 0x380;
const reg_timer_current = 0x390;
const reg_timer_divide = 0x3E0;
const icr_delivery_pending = 1 << 12; // ICR low bit 12: a previous IPI is still in flight
const lvt_masked = 1 << 16;
const lvt_periodic = 1 << 17;
const timer_divide_16 = 0x3;
@@ -120,6 +124,39 @@ pub fn init() void {
write(reg_spurious, 0x100 | spurious_vector); // bit 8 = software enable
}
/// Software-enable *this* core's Local APIC — the application-processor counterpart
/// of `init`, minus the one-time PIC remap (the BSP already masked it) and minus
/// calibration (the timer rate is a shared hardware constant, measured once). Each
/// core has its own LAPIC at the same MMIO address, so no per-core base is needed.
pub fn initSecondary() void {
const msr = io.rdmsr(ia32_apic_base_msr);
io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable
write(reg_spurious, 0x100 | spurious_vector); // software enable
}
// --- application-processor wakeup (INIT–SIPI–SIPI) --------------------------
/// Send an INIT IPI to the core with Local APIC id `apic_id` — the first step of
/// the wake sequence. Blocks until the LAPIC reports the IPI was delivered.
pub fn sendInit(apic_id: u32) void {
write(reg_icr_high, apic_id << 24);
write(reg_icr_low, 0x4500); // INIT, physical destination, assert, edge-triggered
waitIcrIdle();
}
/// Send a STARTUP IPI (SIPI) telling the target core to begin executing at physical
/// address `vector << 12` (in real mode). Per the Intel bring-up protocol this is
/// sent twice after the INIT; both calls block until delivery completes.
pub fn sendStartup(apic_id: u32, vector: u8) void {
write(reg_icr_high, apic_id << 24);
write(reg_icr_low, 0x4600 | @as(u32, vector)); // STARTUP with the page vector
waitIcrIdle();
}
fn waitIcrIdle() void {
while (read(reg_icr_low) & icr_delivery_pending != 0) {}
}
/// The calibration window: we time everything against a 10 ms reference interval.
const calib_ms = 10;
+22
View File
@@ -13,6 +13,7 @@ const serial = @import("serial.zig");
const apic = @import("apic.zig");
const ioapic = @import("ioapic.zig");
const io = @import("io.zig");
const smp = @import("smp.zig");
/// The saved register/trap frame passed to a fault handler.
pub const CpuState = idt.CpuState;
@@ -99,6 +100,27 @@ pub fn cpuLocal() usize {
return io.rdmsr(ia32_gs_base);
}
// --- SMP: application-processor bring-up ----------------------------------
/// Make a low RAM page executable (clear its NX bit) — the AP trampoline is fetched
/// from it under paging. Delegates to the VMM; see paging.setExecutable.
pub fn setPageExecutable(phys: u64) void {
paging.setExecutable(phys);
}
/// Copy the AP trampoline into its low page (allocated + made executable by the
/// caller). Run once before waking any application processor.
pub fn prepareSecondaries(tramp_phys: u64) void {
smp.prepare(tramp_phys);
}
/// Wake the core with Local APIC id `apic_id`, giving it `stack_top` and its per-CPU
/// pointer `percpu`; it adopts the current (kernel) page tables. Returns false if it
/// doesn't come online within the timeout. Blocks until the core reports in.
pub fn startSecondary(apic_id: u32, stack_top: usize, percpu: usize) bool {
return smp.startAp(apic_id, stack_top, percpu, readCr3());
}
/// Kernel tick rate: 1000 Hz (1 ms), the scheduler's time quantum.
pub const timer_hz = 1000;
+11 -2
View File
@@ -44,11 +44,20 @@ const Descriptor = packed struct {
/// isr.s — it uses the selectors 0x08 (code) and 0x10 (data) that match `table`.
extern fn gdt_flush(descriptor: *const Descriptor) callconv(.c) void;
/// Install our GDT and switch onto its segments.
pub fn init() void {
/// Load our GDT on the current core and switch onto its segments. The table is
/// shared across all cores (the descriptors are flat and read-only); each core just
/// needs to point its GDTR at it. Called by the BSP in `init` and by every AP during
/// bring-up. Note this reloads the segment registers, which zeroes the GS base — so
/// a core must publish its per-CPU pointer (setCpuLocal) *after* calling this.
pub fn loadOnThisCpu() void {
const descriptor = Descriptor{
.limit = @sizeOf(@TypeOf(table)) - 1,
.base = @intFromPtr(&table),
};
gdt_flush(&descriptor);
}
/// Install our GDT and switch onto its segments (bootstrap processor).
pub fn init() void {
loadOnThisCpu();
}
+7
View File
@@ -128,6 +128,13 @@ pub fn init() void {
// Run the double-fault handler (vector 8) on IST1: a #DF usually means the
// current stack is unusable, so it needs a guaranteed-good one. See tss.zig.
idt[8].ist = tss.double_fault_ist;
loadOnThisCpu();
}
/// Load the (shared, already-populated) IDT on the current core. The gate table is
/// read-only after `init`, so every core points its IDTR at the same one. Called by
/// the BSP via `init` and by each AP during bring-up.
pub fn loadOnThisCpu() void {
const descriptor = Descriptor{
.limit = @sizeOf(@TypeOf(idt)) - 1,
.base = @intFromPtr(&idt),
+10
View File
@@ -128,6 +128,16 @@ pub fn map(virt: u64, phys: u64, writable_page: bool) void {
invalidate(virt);
}
/// Make an already-identity-mapped RAM page **executable** (clear its NX bit),
/// leaving it present and writable. The blanket RAM mapping is NX for W^X, but the
/// application processors fetch the AP trampoline from a low RAM page under paging —
/// so that one page must be executable. A deliberate, temporary W^X exception for a
/// single bring-up page; the caller frees it once every AP is up.
pub fn setExecutable(phys: u64) void {
mapPage(kernel_pml4, phys, phys, present | writable); // note: no no_execute
invalidate(phys);
}
/// Remove a mapping and flush it from the TLB.
pub fn unmap(virt: u64) void {
const pml4e = tableAt(kernel_pml4)[(virt >> 39) & 0x1FF];
+101
View File
@@ -0,0 +1,101 @@
//! Application-processor (AP) bring-up: waking the cores the firmware left parked.
//!
//! The firmware starts only the bootstrap processor (BSP); the others sit idle until
//! the kernel wakes them with an INIT–SIPI–SIPI sequence (Intel SDM Vol.3, "MP
//! Initialization"). A woken core begins in 16-bit real mode at a low physical page,
//! runs the [trampoline](trampoline.s) up into 64-bit long mode, and lands in
//! `apEntry` here. This module copies the trampoline into place, patches its
//! per-AP parameters, drives the wake IPIs, and waits for each core to report in.
//!
//! Cores are brought up **one at a time**: a single trampoline page and parameter
//! block are reused, so the BSP patches, wakes, and waits for one AP before the
//! next. The mechanism-vs-policy split matches the rest of the kernel — the generic
//! scheduler decides *what* runs where; this just gets a core executing 64-bit code.
//!
//! This is step 3a: an AP climbs to long mode, publishes its per-CPU pointer, marks
//! itself alive, and parks. Entering the scheduler (its own TSS, LAPIC timer, and
//! the run loop) is the next step.
const io = @import("io.zig");
const apic = @import("apic.zig");
/// IA32_GS_BASE — the per-CPU data pointer (see cpu.zig; kept in sync here so the AP
/// path doesn't depend on cpu.zig and risk an import cycle).
const ia32_gs_base = 0xC000_0101;
/// Physical address of the trampoline page (page-aligned, below 1 MiB). Set by
/// `prepare`; the low 20 bits are always zero, so `phys >> 12` is the SIPI vector.
var tramp_phys: u64 = 0;
/// Set to 1 by a freshly-woken AP once it reaches `apEntry` and finishes its own
/// bring-up. The BSP clears it before each wake and polls it afterwards — a simple
/// one-at-a-time handshake (only one AP is being started at any moment).
var ap_alive: u32 = 0;
/// Copy the trampoline blob to its low page. Call once, after the page has been
/// allocated and made executable, before waking any AP.
pub fn prepare(phys: u64) void {
tramp_phys = phys;
const start = @extern([*]const u8, .{ .name = "ap_trampoline_start" });
const end = @extern([*]const u8, .{ .name = "ap_trampoline_end" });
const len = @intFromPtr(end) - @intFromPtr(start);
const dst: [*]u8 = @ptrFromInt(phys);
@memcpy(dst[0..len], start[0..len]);
}
/// Address of a patchable trampoline parameter, by symbol name: the copied blob's
/// base plus the field's offset within it (a same-section symbol difference). The
/// pointer is `align(1)` — the fields aren't 8-aligned within the blob, and x86
/// tolerates unaligned stores, so we don't force layout constraints on the asm.
fn param(comptime name: []const u8) *align(1) volatile u64 {
const start = @intFromPtr(@extern([*]const u8, .{ .name = "ap_trampoline_start" }));
const sym = @intFromPtr(@extern([*]const u8, .{ .name = name }));
return @ptrFromInt(tramp_phys + (sym - start));
}
/// Wake the core with Local APIC id `apic_id`, hand it `stack_top` and `percpu` (its
/// per-CPU pointer), and wait for it to come alive. Returns false if it doesn't
/// report in within the timeout (left parked, no harm to the running system).
/// `cr3` is the kernel page tables the AP adopts. Precondition: `prepare` has run.
pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, cr3: u64) bool {
param("ap_tramp_cr3").* = cr3;
param("ap_tramp_stack").* = stack_top;
param("ap_tramp_entry").* = @intFromPtr(&apEntry);
param("ap_tramp_percpu").* = percpu;
@atomicStore(u32, &ap_alive, 0, .seq_cst);
const vector: u8 = @intCast(tramp_phys >> 12);
apic.sendInit(apic_id);
delayMicros(10_000); // 10 ms INIT settle
apic.sendStartup(apic_id, vector);
delayMicros(200);
apic.sendStartup(apic_id, vector);
// Wait up to 100 ms for the AP to reach apEntry and set the flag.
const deadline = apic.millis() + 100;
while (apic.millis() < deadline) {
if (@atomicLoad(u32, &ap_alive, .acquire) != 0) return true;
asm volatile ("pause");
}
return false;
}
/// Busy-wait `us` microseconds against the calibrated TSC clock (the AP wake happens
/// after the timer is up, so the clock is available).
fn delayMicros(us: u64) void {
const start = apic.micros();
while (apic.micros() - start < us) asm volatile ("pause");
}
/// The 64-bit entry every AP lands on, called from the trampoline with its per-CPU
/// pointer in RDI. Adopts the shared descriptor tables, publishes its per-CPU
/// pointer, signals the BSP it's alive, and (for now) parks. Never returns.
fn apEntry(percpu: usize) callconv(.c) noreturn {
// Step 3a: minimal. The core keeps the trampoline's descriptor tables, publishes
// its per-CPU pointer, signals the BSP, and parks with interrupts off. Loading
// this core's own kernel GDT/IDT/TSS and entering the scheduler is step 3b.
io.wrmsr(ia32_gs_base, percpu); // publish per-CPU pointer (GS base)
@atomicStore(u32, &ap_alive, 1, .release); // "I'm up" — BSP is polling this
while (true) asm volatile ("hlt"); // parked (3b enters the scheduler here)
}
+140
View File
@@ -0,0 +1,140 @@
# AP trampoline: brings a waking application processor from the 16-bit real mode it
# starts in (after INIT-SIPI-SIPI) up through protected mode into 64-bit long mode,
# then jumps to the Zig AP entry (arch/x86_64/smp.zig:apEntry).
#
# A STARTUP IPI vectors a core to physical address `vector << 12` in real mode, so
# this blob is copied to a low (<1 MiB) page and started there; at entry CS = that
# page >> 4 and IP = 0. It is fully **position-independent**: it derives its own
# linear base (CS << 4) into EBX and addresses every internal datum as
# `(label - ap_trampoline_start)(%ebx)` — a difference of two symbols in the same
# section, which the assembler folds to a constant page offset no matter where the
# blob was linked or copied to. The BSP patches the parameter block (CR3, stack,
# entry, per-CPU pointer) before each wake; see arch/x86_64/smp.zig.
#
# It lives in .rodata (not .text): it is data to be copied out and executed
# elsewhere, never run at its link address, so it must not be a normal code segment.
.section .rodata
.balign 16
.code16
.global ap_trampoline_start
ap_trampoline_start:
cli
cld
# Linear base of this page (CS << 4) into EBX; all data is addressed off it.
xorl %eax, %eax
mov %cs, %ax
shll $4, %eax
movl %eax, %ebx
mov %cs, %ax # DS = CS, so we address our data as DS:(label - start):
mov %ax, %ds # the segment base (CS<<4) already supplies the page base,
# so data operands use the page *offset*, not EBX.
# Relocate the pointers whose absolute (linear) targets depend on where we were
# copied: the GDT base and the two far-jump targets = EBX + their page offsets.
# EBX supplies the base for the *value* (via leal); the store address is DS-rel.
leal (gdt32 - ap_trampoline_start)(%ebx), %eax
movl %eax, gdtr32_base - ap_trampoline_start
leal (prot_entry - ap_trampoline_start)(%ebx), %eax
movl %eax, jmp32_off - ap_trampoline_start
leal (long_entry - ap_trampoline_start)(%ebx), %eax
movl %eax, jmp64_off - ap_trampoline_start
lgdtl gdtr32 - ap_trampoline_start
movl %cr0, %eax # enter protected mode (CR0.PE)
orl $1, %eax
movl %eax, %cr0
ljmpl *(jmp32_ptr - ap_trampoline_start) # -> prot_entry, CS = 0x08
.code32
prot_entry:
movw $0x10, %ax # flat 32-bit data segments
movw %ax, %ds
movw %ax, %es
movw %ax, %ss
movw %ax, %fs
movw %ax, %gs
movl %cr4, %eax # PAE on (CR4.PAE) — required for long mode
orl $(1 << 5), %eax
movl %eax, %cr4
movl (param_cr3 - ap_trampoline_start)(%ebx), %eax # kernel page tables
movl %eax, %cr3
movl $0xC0000080, %ecx # EFER: long mode enable (LME) + NX enable (NXE, since
rdmsr # the kernel's PTEs set the NX bit)
orl $((1 << 8) | (1 << 11)), %eax
wrmsr
movl %cr0, %eax # paging on (CR0.PG) — now in long mode (compat sub-mode)
orl $(1 << 31), %eax
movl %eax, %cr0
ljmpl *(jmp64_ptr - ap_trampoline_start)(%ebx) # -> long_entry, CS = 0x18 (L=1)
.code64
long_entry:
movw $0x10, %ax # sane flat data segments
movw %ax, %ds
movw %ax, %es
movw %ax, %ss
# RBX = EBX (zero-extended) = page base. Load our stack and per-CPU pointer, then
# call the Zig entry — which runs from the kernel image and never returns.
movq (param_stack - ap_trampoline_start)(%rbx), %rsp
movq (param_percpu - ap_trampoline_start)(%rbx), %rdi # SysV arg 0
movq (param_entry - ap_trampoline_start)(%rbx), %rax
callq *%rax
1: hlt # unreachable; guard against a stray return
jmp 1b
# --- data: GDT, far pointers, and the BSP-patched parameter block -----------
.balign 8
gdt32:
.quad 0x0000000000000000 # 0x00 null
.quad 0x00CF9A000000FFFF # 0x08 32-bit code (G, D, present, exec/read)
.quad 0x00CF92000000FFFF # 0x10 data (valid in 32- and 64-bit)
.quad 0x00AF9A000000FFFF # 0x18 64-bit code (L=1)
gdt32_end:
gdtr32:
.word gdt32_end - gdt32 - 1
gdtr32_base:
.long 0 # patched (16-bit code): linear base of gdt32
jmp32_ptr: # indirect far-jump operand: offset then selector
jmp32_off:
.long 0 # patched: linear address of prot_entry
.word 0x08 # 32-bit code selector
jmp64_ptr:
jmp64_off:
.long 0 # patched: linear address of long_entry
.word 0x18 # 64-bit code selector
# The parameter block, filled in by the BSP (smp.zig) before each STARTUP IPI. Global
# so the Zig side can locate each field as (symbol - ap_trampoline_start).
.global ap_tramp_cr3
.global ap_tramp_stack
.global ap_tramp_entry
.global ap_tramp_percpu
param_cr3:
ap_tramp_cr3:
.quad 0 # kernel PML4 physical address (CR3)
param_stack:
ap_tramp_stack:
.quad 0 # top of this AP's kernel stack
param_entry:
ap_tramp_entry:
.quad 0 # address of apEntry (the Zig AP entry)
param_percpu:
ap_tramp_percpu:
.quad 0 # this AP's per-CPU pointer (goes in GS base)
.global ap_trampoline_end
ap_trampoline_end: