From ed7f542006595dba4b8f9da6631c5828dceed532 Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Wed, 8 Jul 2026 12:02:12 +0100 Subject: [PATCH] wake application processors to long mode INIT-SIPI-SIPI plus a self-relocating real-mode trampoline. --- build.zig | 3 + docs/smp.md | 72 ++++++++++++-- src/kernel/arch/x86_64/apic.zig | 37 ++++++++ src/kernel/arch/x86_64/cpu.zig | 22 +++++ src/kernel/arch/x86_64/gdt.zig | 13 ++- src/kernel/arch/x86_64/idt.zig | 7 ++ src/kernel/arch/x86_64/paging.zig | 10 ++ src/kernel/arch/x86_64/smp.zig | 101 ++++++++++++++++++++ src/kernel/arch/x86_64/trampoline.s | 140 ++++++++++++++++++++++++++++ src/kernel/main.zig | 48 ++++++++++ src/kernel/pmm.zig | 18 ++++ src/kernel/scheduler.zig | 19 ++++ 12 files changed, 482 insertions(+), 8 deletions(-) create mode 100644 src/kernel/arch/x86_64/smp.zig create mode 100644 src/kernel/arch/x86_64/trampoline.s diff --git a/build.zig b/build.zig index 11eb011..aad09fb 100644 --- a/build.zig +++ b/build.zig @@ -73,6 +73,9 @@ pub fn build(b: *std.Build) void { // CPU-exception stubs — real assembly, since they need cross-symbol // jumps/calls that Zig inline asm can't express (see the file's header). arch_mod.addAssemblyFile(b.path("src/kernel/arch/x86_64/isr.s")); + // The AP bring-up trampoline: 16-/32-/64-bit mode-switch code that can't be + // inline asm (it runs relocated to a low page, not at its link address). + arch_mod.addAssemblyFile(b.path("src/kernel/arch/x86_64/trampoline.s")); // Firmware-agnostic device discovery. The generic kernel imports this as // "platform" and asks it to enumerate hardware into a backend-neutral device diff --git a/docs/smp.md b/docs/smp.md index 9db00d5..28feea4 100644 --- a/docs/smp.md +++ b/docs/smp.md @@ -16,10 +16,13 @@ danos specifics): - **Real-time** — whether timing is *predictable*. Comes from bounded operations (our O(1) scheduler), not from core count. -danos is uniprocessor today: one global `current` task, one set of ready queues, one -timer. Even on an 8-core CPU, the firmware starts only the **bootstrap processor -(BSP)**; the other cores (**application processors**, APs) sit parked until the -kernel wakes them, which it doesn't yet. +danos is mid-transition. The firmware starts only the **bootstrap processor (BSP)**; +the other cores (**application processors**, APs) sit parked until the kernel wakes +them. As of the SMP work in progress (see [Implementation status](#implementation-status) +below), danos now *does* wake the APs — each climbs to 64-bit long mode and reports +in — and the shared kernel state (scheduler queues, IPC) is already serialised behind +a big kernel lock. What's not done yet is letting the woken APs actually run tasks; +`current` is per-CPU but the run loop is still BSP-only. ## The common microkernel instinct: don't share kernel state @@ -119,12 +122,15 @@ Whatever the top goal, the *sequence* is the same and seL4 validates starting si and `platform.cpus()` returns the list (see [discovery.md](discovery.md)). The boot log reports the count; the ARM (device-tree) path still needs it. 2. **Wake the APs** — INIT–SIPI–SIPI on x86; PSCI/spin-tables on ARM. Each core brings - up its own tables, timer, and idle task. + up its own tables, timer, and idle task. **In progress on x86** — the cores reach + long mode and park; the per-core tables/timer/scheduler entry is the next step + ([status](#implementation-status)). 3. **Start with a big kernel lock.** It's a legitimate first design, not a shortcut — philosophically aligned with a tiny kernel, and it lets the single-core correctness model you already have (the interrupt-flag discipline in [scheduling.md](scheduling.md)) stay largely intact: one lock around kernel entry - instead of rethinking every critical section. + instead of rethinking every critical section. **Done** — see + `src/kernel/sync.zig`. 4. **Later, if contention bites,** evolve toward **per-core run queues + explicit affinity** (the Fiasco.OC direction) — also the more real-time-predictable model. 5. **Placement stays a user-space policy** — the kernel runs a thread on the core it's @@ -133,6 +139,60 @@ Whatever the top goal, the *sequence* is the same and seL4 validates starting si Big-lock-first → per-core-later. The affinity/MCS depth is only worth it if real-time turns out to be the actual goal. +## Implementation status + +The "wake + schedule" build (real parallel task execution) is going in as a sequence +of green checkpoints — each step keeps the single-core test suite passing before the +next lands. + +**Done:** + +- **Core enumeration** — the MADT parse records every usable Local APIC (with its + `apic_id`, which an AP wake targets); `platform.cpus()` returns the list. See + [discovery.md](discovery.md). +- **The big kernel lock** (`src/kernel/sync.zig`) — one coarse spinlock guarding the + scheduler queues and IPC, always held with local interrupts disabled. It is held + *across* a context switch and released by whichever task resumes (the hand-off + rule); `task_trampoline` releases it for a freshly-spawned task. `scheduler.zig` and + `ipc.zig` run every critical section under it. Uncontended on one core, so behaviour + is identical to the old interrupt-flag model. +- **Per-CPU state** — a `PerCpu` struct (running task, idle task, APIC id) per core, + its pointer kept in the x86 **GS base** (`IA32_GS_BASE`; no `swapgs`, since there's + no user mode yet). The old global `current` is now `thisCpu().current`. The ready + queues stay **global** under the lock — work-conserving, so any idle core will pull + the highest-priority ready task; per-core queues are a later optimisation. +- **AP wake to long mode** — `arch.startSecondary` drives INIT–SIPI–SIPI (via the + LAPIC ICR) to wake each parked core one at a time. A woken core starts in 16-bit + real mode at a low page and runs the [trampoline](../src/kernel/arch/x86_64/trampoline.s) + up through protected mode into 64-bit long mode, then lands in `smp.zig:apEntry`, + publishes its per-CPU pointer, and reports in. Verified in QEMU with `-smp 4`: + all four cores report `online`. + + The trampoline earns its complexity from three hardware facts: + - a STARTUP IPI vectors a core to physical `vector << 12` (a *byte* vector), so the + trampoline must live **below 1 MiB** — the kernel reserves that page from the frame + allocator at boot, before paging/heap draw down the scarce low frames; + - the blanket RAM identity map is **NX** (W^X), but the AP fetches the trampoline + from it under paging, so that one page is made executable for bring-up; + - the blob is copied to a page whose address isn't known at link time, so it is + **position-independent**: it derives its own base from `CS` and, crucially, + addresses data *segment-relative in real mode* (where the segment base already + supplies the page base) but *base-register-relative in protected/long mode* (flat + segments, base 0). Getting that distinction wrong was the first bug found. + +**Next:** + +- **Per-core descriptor tables + scheduler entry** — each AP needs its own TSS (its + own IST/`rsp0` stack) and to load the kernel GDT/IDT, enable its LAPIC timer, and + enter the scheduler run loop under the big lock. (The 3a checkpoint deliberately + parks the APs on the trampoline's tables with interrupts off; loading the shared + kernel GDT on an AP faulted, and the per-core-TSS work is where that's resolved.) +- **A parallelism test** — a case where N cores drive N counters at once, proving work + runs truly in parallel rather than just that the APs booted. +- **IPIs** (deferred) — cross-core wake/preempt. Not needed for correctness: an idle + core wakes on its own timer tick and pulls ready work then; IPIs only cut that + latency from ≤1 ms to near-instant. + ## Further reading **Microkernel SMP & scheduling** diff --git a/src/kernel/arch/x86_64/apic.zig b/src/kernel/arch/x86_64/apic.zig index d9d83c8..e18d9b7 100644 --- a/src/kernel/arch/x86_64/apic.zig +++ b/src/kernel/arch/x86_64/apic.zig @@ -44,11 +44,15 @@ const spurious_vector = 47; // LAPIC register offsets. const reg_spurious = 0x0F0; const reg_eoi = 0x0B0; +const reg_icr_low = 0x300; // interrupt command register, low dword (writing it sends) +const reg_icr_high = 0x310; // ICR high dword (destination APIC id in bits 24-31) const reg_lvt_timer = 0x320; const reg_timer_initial = 0x380; const reg_timer_current = 0x390; const reg_timer_divide = 0x3E0; +const icr_delivery_pending = 1 << 12; // ICR low bit 12: a previous IPI is still in flight + const lvt_masked = 1 << 16; const lvt_periodic = 1 << 17; const timer_divide_16 = 0x3; @@ -120,6 +124,39 @@ pub fn init() void { write(reg_spurious, 0x100 | spurious_vector); // bit 8 = software enable } +/// Software-enable *this* core's Local APIC — the application-processor counterpart +/// of `init`, minus the one-time PIC remap (the BSP already masked it) and minus +/// calibration (the timer rate is a shared hardware constant, measured once). Each +/// core has its own LAPIC at the same MMIO address, so no per-core base is needed. +pub fn initSecondary() void { + const msr = io.rdmsr(ia32_apic_base_msr); + io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable + write(reg_spurious, 0x100 | spurious_vector); // software enable +} + +// --- application-processor wakeup (INIT–SIPI–SIPI) -------------------------- + +/// Send an INIT IPI to the core with Local APIC id `apic_id` — the first step of +/// the wake sequence. Blocks until the LAPIC reports the IPI was delivered. +pub fn sendInit(apic_id: u32) void { + write(reg_icr_high, apic_id << 24); + write(reg_icr_low, 0x4500); // INIT, physical destination, assert, edge-triggered + waitIcrIdle(); +} + +/// Send a STARTUP IPI (SIPI) telling the target core to begin executing at physical +/// address `vector << 12` (in real mode). Per the Intel bring-up protocol this is +/// sent twice after the INIT; both calls block until delivery completes. +pub fn sendStartup(apic_id: u32, vector: u8) void { + write(reg_icr_high, apic_id << 24); + write(reg_icr_low, 0x4600 | @as(u32, vector)); // STARTUP with the page vector + waitIcrIdle(); +} + +fn waitIcrIdle() void { + while (read(reg_icr_low) & icr_delivery_pending != 0) {} +} + /// The calibration window: we time everything against a 10 ms reference interval. const calib_ms = 10; diff --git a/src/kernel/arch/x86_64/cpu.zig b/src/kernel/arch/x86_64/cpu.zig index 9757635..b3adb8c 100644 --- a/src/kernel/arch/x86_64/cpu.zig +++ b/src/kernel/arch/x86_64/cpu.zig @@ -13,6 +13,7 @@ const serial = @import("serial.zig"); const apic = @import("apic.zig"); const ioapic = @import("ioapic.zig"); const io = @import("io.zig"); +const smp = @import("smp.zig"); /// The saved register/trap frame passed to a fault handler. pub const CpuState = idt.CpuState; @@ -99,6 +100,27 @@ pub fn cpuLocal() usize { return io.rdmsr(ia32_gs_base); } +// --- SMP: application-processor bring-up ---------------------------------- + +/// Make a low RAM page executable (clear its NX bit) — the AP trampoline is fetched +/// from it under paging. Delegates to the VMM; see paging.setExecutable. +pub fn setPageExecutable(phys: u64) void { + paging.setExecutable(phys); +} + +/// Copy the AP trampoline into its low page (allocated + made executable by the +/// caller). Run once before waking any application processor. +pub fn prepareSecondaries(tramp_phys: u64) void { + smp.prepare(tramp_phys); +} + +/// Wake the core with Local APIC id `apic_id`, giving it `stack_top` and its per-CPU +/// pointer `percpu`; it adopts the current (kernel) page tables. Returns false if it +/// doesn't come online within the timeout. Blocks until the core reports in. +pub fn startSecondary(apic_id: u32, stack_top: usize, percpu: usize) bool { + return smp.startAp(apic_id, stack_top, percpu, readCr3()); +} + /// Kernel tick rate: 1000 Hz (1 ms), the scheduler's time quantum. pub const timer_hz = 1000; diff --git a/src/kernel/arch/x86_64/gdt.zig b/src/kernel/arch/x86_64/gdt.zig index 16e2708..4d2772b 100644 --- a/src/kernel/arch/x86_64/gdt.zig +++ b/src/kernel/arch/x86_64/gdt.zig @@ -44,11 +44,20 @@ const Descriptor = packed struct { /// isr.s — it uses the selectors 0x08 (code) and 0x10 (data) that match `table`. extern fn gdt_flush(descriptor: *const Descriptor) callconv(.c) void; -/// Install our GDT and switch onto its segments. -pub fn init() void { +/// Load our GDT on the current core and switch onto its segments. The table is +/// shared across all cores (the descriptors are flat and read-only); each core just +/// needs to point its GDTR at it. Called by the BSP in `init` and by every AP during +/// bring-up. Note this reloads the segment registers, which zeroes the GS base — so +/// a core must publish its per-CPU pointer (setCpuLocal) *after* calling this. +pub fn loadOnThisCpu() void { const descriptor = Descriptor{ .limit = @sizeOf(@TypeOf(table)) - 1, .base = @intFromPtr(&table), }; gdt_flush(&descriptor); } + +/// Install our GDT and switch onto its segments (bootstrap processor). +pub fn init() void { + loadOnThisCpu(); +} diff --git a/src/kernel/arch/x86_64/idt.zig b/src/kernel/arch/x86_64/idt.zig index a3bbabc..da5b02a 100644 --- a/src/kernel/arch/x86_64/idt.zig +++ b/src/kernel/arch/x86_64/idt.zig @@ -128,6 +128,13 @@ pub fn init() void { // Run the double-fault handler (vector 8) on IST1: a #DF usually means the // current stack is unusable, so it needs a guaranteed-good one. See tss.zig. idt[8].ist = tss.double_fault_ist; + loadOnThisCpu(); +} + +/// Load the (shared, already-populated) IDT on the current core. The gate table is +/// read-only after `init`, so every core points its IDTR at the same one. Called by +/// the BSP via `init` and by each AP during bring-up. +pub fn loadOnThisCpu() void { const descriptor = Descriptor{ .limit = @sizeOf(@TypeOf(idt)) - 1, .base = @intFromPtr(&idt), diff --git a/src/kernel/arch/x86_64/paging.zig b/src/kernel/arch/x86_64/paging.zig index ae450aa..05318bb 100644 --- a/src/kernel/arch/x86_64/paging.zig +++ b/src/kernel/arch/x86_64/paging.zig @@ -128,6 +128,16 @@ pub fn map(virt: u64, phys: u64, writable_page: bool) void { invalidate(virt); } +/// Make an already-identity-mapped RAM page **executable** (clear its NX bit), +/// leaving it present and writable. The blanket RAM mapping is NX for W^X, but the +/// application processors fetch the AP trampoline from a low RAM page under paging — +/// so that one page must be executable. A deliberate, temporary W^X exception for a +/// single bring-up page; the caller frees it once every AP is up. +pub fn setExecutable(phys: u64) void { + mapPage(kernel_pml4, phys, phys, present | writable); // note: no no_execute + invalidate(phys); +} + /// Remove a mapping and flush it from the TLB. pub fn unmap(virt: u64) void { const pml4e = tableAt(kernel_pml4)[(virt >> 39) & 0x1FF]; diff --git a/src/kernel/arch/x86_64/smp.zig b/src/kernel/arch/x86_64/smp.zig new file mode 100644 index 0000000..2ebaac5 --- /dev/null +++ b/src/kernel/arch/x86_64/smp.zig @@ -0,0 +1,101 @@ +//! Application-processor (AP) bring-up: waking the cores the firmware left parked. +//! +//! The firmware starts only the bootstrap processor (BSP); the others sit idle until +//! the kernel wakes them with an INIT–SIPI–SIPI sequence (Intel SDM Vol.3, "MP +//! Initialization"). A woken core begins in 16-bit real mode at a low physical page, +//! runs the [trampoline](trampoline.s) up into 64-bit long mode, and lands in +//! `apEntry` here. This module copies the trampoline into place, patches its +//! per-AP parameters, drives the wake IPIs, and waits for each core to report in. +//! +//! Cores are brought up **one at a time**: a single trampoline page and parameter +//! block are reused, so the BSP patches, wakes, and waits for one AP before the +//! next. The mechanism-vs-policy split matches the rest of the kernel — the generic +//! scheduler decides *what* runs where; this just gets a core executing 64-bit code. +//! +//! This is step 3a: an AP climbs to long mode, publishes its per-CPU pointer, marks +//! itself alive, and parks. Entering the scheduler (its own TSS, LAPIC timer, and +//! the run loop) is the next step. + +const io = @import("io.zig"); +const apic = @import("apic.zig"); + +/// IA32_GS_BASE — the per-CPU data pointer (see cpu.zig; kept in sync here so the AP +/// path doesn't depend on cpu.zig and risk an import cycle). +const ia32_gs_base = 0xC000_0101; + +/// Physical address of the trampoline page (page-aligned, below 1 MiB). Set by +/// `prepare`; the low 20 bits are always zero, so `phys >> 12` is the SIPI vector. +var tramp_phys: u64 = 0; + +/// Set to 1 by a freshly-woken AP once it reaches `apEntry` and finishes its own +/// bring-up. The BSP clears it before each wake and polls it afterwards — a simple +/// one-at-a-time handshake (only one AP is being started at any moment). +var ap_alive: u32 = 0; + +/// Copy the trampoline blob to its low page. Call once, after the page has been +/// allocated and made executable, before waking any AP. +pub fn prepare(phys: u64) void { + tramp_phys = phys; + const start = @extern([*]const u8, .{ .name = "ap_trampoline_start" }); + const end = @extern([*]const u8, .{ .name = "ap_trampoline_end" }); + const len = @intFromPtr(end) - @intFromPtr(start); + const dst: [*]u8 = @ptrFromInt(phys); + @memcpy(dst[0..len], start[0..len]); +} + +/// Address of a patchable trampoline parameter, by symbol name: the copied blob's +/// base plus the field's offset within it (a same-section symbol difference). The +/// pointer is `align(1)` — the fields aren't 8-aligned within the blob, and x86 +/// tolerates unaligned stores, so we don't force layout constraints on the asm. +fn param(comptime name: []const u8) *align(1) volatile u64 { + const start = @intFromPtr(@extern([*]const u8, .{ .name = "ap_trampoline_start" })); + const sym = @intFromPtr(@extern([*]const u8, .{ .name = name })); + return @ptrFromInt(tramp_phys + (sym - start)); +} + +/// Wake the core with Local APIC id `apic_id`, hand it `stack_top` and `percpu` (its +/// per-CPU pointer), and wait for it to come alive. Returns false if it doesn't +/// report in within the timeout (left parked, no harm to the running system). +/// `cr3` is the kernel page tables the AP adopts. Precondition: `prepare` has run. +pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, cr3: u64) bool { + param("ap_tramp_cr3").* = cr3; + param("ap_tramp_stack").* = stack_top; + param("ap_tramp_entry").* = @intFromPtr(&apEntry); + param("ap_tramp_percpu").* = percpu; + + @atomicStore(u32, &ap_alive, 0, .seq_cst); + + const vector: u8 = @intCast(tramp_phys >> 12); + apic.sendInit(apic_id); + delayMicros(10_000); // 10 ms INIT settle + apic.sendStartup(apic_id, vector); + delayMicros(200); + apic.sendStartup(apic_id, vector); + + // Wait up to 100 ms for the AP to reach apEntry and set the flag. + const deadline = apic.millis() + 100; + while (apic.millis() < deadline) { + if (@atomicLoad(u32, &ap_alive, .acquire) != 0) return true; + asm volatile ("pause"); + } + return false; +} + +/// Busy-wait `us` microseconds against the calibrated TSC clock (the AP wake happens +/// after the timer is up, so the clock is available). +fn delayMicros(us: u64) void { + const start = apic.micros(); + while (apic.micros() - start < us) asm volatile ("pause"); +} + +/// The 64-bit entry every AP lands on, called from the trampoline with its per-CPU +/// pointer in RDI. Adopts the shared descriptor tables, publishes its per-CPU +/// pointer, signals the BSP it's alive, and (for now) parks. Never returns. +fn apEntry(percpu: usize) callconv(.c) noreturn { + // Step 3a: minimal. The core keeps the trampoline's descriptor tables, publishes + // its per-CPU pointer, signals the BSP, and parks with interrupts off. Loading + // this core's own kernel GDT/IDT/TSS and entering the scheduler is step 3b. + io.wrmsr(ia32_gs_base, percpu); // publish per-CPU pointer (GS base) + @atomicStore(u32, &ap_alive, 1, .release); // "I'm up" — BSP is polling this + while (true) asm volatile ("hlt"); // parked (3b enters the scheduler here) +} diff --git a/src/kernel/arch/x86_64/trampoline.s b/src/kernel/arch/x86_64/trampoline.s new file mode 100644 index 0000000..1cf7f39 --- /dev/null +++ b/src/kernel/arch/x86_64/trampoline.s @@ -0,0 +1,140 @@ +# AP trampoline: brings a waking application processor from the 16-bit real mode it +# starts in (after INIT-SIPI-SIPI) up through protected mode into 64-bit long mode, +# then jumps to the Zig AP entry (arch/x86_64/smp.zig:apEntry). +# +# A STARTUP IPI vectors a core to physical address `vector << 12` in real mode, so +# this blob is copied to a low (<1 MiB) page and started there; at entry CS = that +# page >> 4 and IP = 0. It is fully **position-independent**: it derives its own +# linear base (CS << 4) into EBX and addresses every internal datum as +# `(label - ap_trampoline_start)(%ebx)` — a difference of two symbols in the same +# section, which the assembler folds to a constant page offset no matter where the +# blob was linked or copied to. The BSP patches the parameter block (CR3, stack, +# entry, per-CPU pointer) before each wake; see arch/x86_64/smp.zig. +# +# It lives in .rodata (not .text): it is data to be copied out and executed +# elsewhere, never run at its link address, so it must not be a normal code segment. + +.section .rodata +.balign 16 +.code16 +.global ap_trampoline_start +ap_trampoline_start: + cli + cld + + # Linear base of this page (CS << 4) into EBX; all data is addressed off it. + xorl %eax, %eax + mov %cs, %ax + shll $4, %eax + movl %eax, %ebx + + mov %cs, %ax # DS = CS, so we address our data as DS:(label - start): + mov %ax, %ds # the segment base (CS<<4) already supplies the page base, + # so data operands use the page *offset*, not EBX. + + # Relocate the pointers whose absolute (linear) targets depend on where we were + # copied: the GDT base and the two far-jump targets = EBX + their page offsets. + # EBX supplies the base for the *value* (via leal); the store address is DS-rel. + leal (gdt32 - ap_trampoline_start)(%ebx), %eax + movl %eax, gdtr32_base - ap_trampoline_start + leal (prot_entry - ap_trampoline_start)(%ebx), %eax + movl %eax, jmp32_off - ap_trampoline_start + leal (long_entry - ap_trampoline_start)(%ebx), %eax + movl %eax, jmp64_off - ap_trampoline_start + + lgdtl gdtr32 - ap_trampoline_start + + movl %cr0, %eax # enter protected mode (CR0.PE) + orl $1, %eax + movl %eax, %cr0 + + ljmpl *(jmp32_ptr - ap_trampoline_start) # -> prot_entry, CS = 0x08 + +.code32 +prot_entry: + movw $0x10, %ax # flat 32-bit data segments + movw %ax, %ds + movw %ax, %es + movw %ax, %ss + movw %ax, %fs + movw %ax, %gs + + movl %cr4, %eax # PAE on (CR4.PAE) — required for long mode + orl $(1 << 5), %eax + movl %eax, %cr4 + + movl (param_cr3 - ap_trampoline_start)(%ebx), %eax # kernel page tables + movl %eax, %cr3 + + movl $0xC0000080, %ecx # EFER: long mode enable (LME) + NX enable (NXE, since + rdmsr # the kernel's PTEs set the NX bit) + orl $((1 << 8) | (1 << 11)), %eax + wrmsr + + movl %cr0, %eax # paging on (CR0.PG) — now in long mode (compat sub-mode) + orl $(1 << 31), %eax + movl %eax, %cr0 + + ljmpl *(jmp64_ptr - ap_trampoline_start)(%ebx) # -> long_entry, CS = 0x18 (L=1) + +.code64 +long_entry: + movw $0x10, %ax # sane flat data segments + movw %ax, %ds + movw %ax, %es + movw %ax, %ss + + # RBX = EBX (zero-extended) = page base. Load our stack and per-CPU pointer, then + # call the Zig entry — which runs from the kernel image and never returns. + movq (param_stack - ap_trampoline_start)(%rbx), %rsp + movq (param_percpu - ap_trampoline_start)(%rbx), %rdi # SysV arg 0 + movq (param_entry - ap_trampoline_start)(%rbx), %rax + callq *%rax +1: hlt # unreachable; guard against a stray return + jmp 1b + +# --- data: GDT, far pointers, and the BSP-patched parameter block ----------- +.balign 8 +gdt32: + .quad 0x0000000000000000 # 0x00 null + .quad 0x00CF9A000000FFFF # 0x08 32-bit code (G, D, present, exec/read) + .quad 0x00CF92000000FFFF # 0x10 data (valid in 32- and 64-bit) + .quad 0x00AF9A000000FFFF # 0x18 64-bit code (L=1) +gdt32_end: + +gdtr32: + .word gdt32_end - gdt32 - 1 +gdtr32_base: + .long 0 # patched (16-bit code): linear base of gdt32 + +jmp32_ptr: # indirect far-jump operand: offset then selector +jmp32_off: + .long 0 # patched: linear address of prot_entry + .word 0x08 # 32-bit code selector + +jmp64_ptr: +jmp64_off: + .long 0 # patched: linear address of long_entry + .word 0x18 # 64-bit code selector + +# The parameter block, filled in by the BSP (smp.zig) before each STARTUP IPI. Global +# so the Zig side can locate each field as (symbol - ap_trampoline_start). +.global ap_tramp_cr3 +.global ap_tramp_stack +.global ap_tramp_entry +.global ap_tramp_percpu +param_cr3: +ap_tramp_cr3: + .quad 0 # kernel PML4 physical address (CR3) +param_stack: +ap_tramp_stack: + .quad 0 # top of this AP's kernel stack +param_entry: +ap_tramp_entry: + .quad 0 # address of apEntry (the Zig AP entry) +param_percpu: +ap_tramp_percpu: + .quad 0 # this AP's per-CPU pointer (goes in GS base) + +.global ap_trampoline_end +ap_trampoline_end: diff --git a/src/kernel/main.zig b/src/kernel/main.zig index 31cc79c..07b8807 100644 --- a/src/kernel/main.zig +++ b/src/kernel/main.zig @@ -30,6 +30,11 @@ const cp_running = 0x70; const cp_exception = 0xE0; const cp_panic = 0xEE; +/// Physical address of the low page reserved at boot for the AP trampoline (0 = none +/// was available). Claimed right after the frame allocator comes up, before paging +/// and the heap consume the scarce sub-1 MiB frames. +var ap_trampoline_page: u64 = 0; + /// Kernel entry point. The bootloader jumps here after `ExitBootServices` with a /// pointer to the handoff data. There is no runtime, no stack unwinding, and no /// caller to return to, so this never returns. @@ -98,6 +103,10 @@ fn kmain(boot_info: *const BootInfo) noreturn { // Bring up the physical frame allocator over that map, and prove it works: // allocate three frames, then hand them back. pmm.init(boot_info.memory_map); + // Claim the AP trampoline's low (<1 MiB) page *now*, before paging and the heap + // draw down sub-1 MiB frames (the allocator scans upward from frame 0). Held + // until SMP bring-up; 0 means none was available (we stay uniprocessor). + ap_trampoline_page = pmm.allocBelow(0x100000) orelse 0; const s1 = pmm.stats(); log.print("\ndanos: frame allocator online\n", .{}); log.print(" free frames: {d} ({d} MiB)\n", .{ s1.free_frames, mib(s1.free_frames) }); @@ -221,6 +230,10 @@ fn kmain(boot_info: *const BootInfo) noreturn { log.checkpoint(cp_timer); log.print("danos: timer online ({d} Hz tick; LAPIC {d} MHz, TSC {d} MHz; calibrated via {s})\n", .{ arch.timer_hz, arch.lapicHz() / 1_000_000, arch.tscHz() / 1_000_000, arch.timerCalibrationSource() }); + // Wake the other cores (application processors). A no-op on a single-core + // machine; on SMP each AP climbs to long mode and reports in (docs/smp.md). + bringUpSecondaries(); + // In a test build (`zig build -Dtest-case=`), run that case and stop. // Normal builds fall through to the idle halt. if (build_options.test_case) |case| { @@ -238,6 +251,41 @@ fn kmain(boot_info: *const BootInfo) noreturn { arch.halt(); } +/// Wake the application processors the firmware left parked. Allocates the low +/// trampoline page (and makes it executable), then wakes each non-boot core in turn, +/// handing it a fresh kernel stack and its per-CPU slot. Cores that don't report in +/// are left parked — the running system is unaffected. See docs/smp.md. +fn bringUpSecondaries() void { + const cores = platform.cpus(); + if (cores.len <= 1) return; + + // The trampoline page was reserved below 1 MiB at boot (a real-mode SIPI vector + // addresses it). Make it executable — the blanket RAM mapping is NX (W^X). + if (ap_trampoline_page == 0) { + log.write("danos: smp: no low page for the AP trampoline; staying uniprocessor\n"); + return; + } + arch.setPageExecutable(ap_trampoline_page); + arch.prepareSecondaries(ap_trampoline_page); + + log.print("\ndanos: bringing up {d} application processor(s)\n", .{cores.len - 1}); + for (cores[1..], 1..) |core, index| { + const stack = heap.allocator().alloc(u8, 16 * 1024) catch { + log.print(" cpu apic_id {d}: no stack; skipped\n", .{core.apic_id}); + continue; + }; + const stack_top = (@intFromPtr(stack.ptr) + stack.len) & ~@as(usize, 15); + const pc = scheduler.prepareSecondary(index, core.apic_id); + if (arch.startSecondary(core.apic_id, stack_top, @intFromPtr(pc))) { + pc.online = true; + log.print(" cpu apic_id {d}: online\n", .{core.apic_id}); + } else { + log.print(" cpu apic_id {d}: no response (parked)\n", .{core.apic_id}); + } + } + log.print("danos: {d}/{d} cores online\n", .{ scheduler.onlineCount(), cores.len }); +} + /// A user-facing status line: to the diagnostic `log` *and* the on-screen console /// (if a framebuffer is present). The verbose log uses `log.*` directly and never /// touches the framebuffer. diff --git a/src/kernel/pmm.zig b/src/kernel/pmm.zig index 6b34049..4349a56 100644 --- a/src/kernel/pmm.zig +++ b/src/kernel/pmm.zig @@ -141,6 +141,24 @@ pub fn alloc() ?u64 { return null; // out of physical memory } +/// Allocate one free frame whose physical address is below `limit`, or null if +/// none is free down there. The AP trampoline needs this: an x86 STARTUP IPI vectors +/// a waking core to physical `vector << 12`, and `vector` is a byte — so the +/// trampoline must live under 1 MiB. A short linear scan of the low frames; only run +/// a handful of times at boot, so it needn't be fast. +pub fn allocBelow(limit: u64) ?u64 { + const cap = @min(total_frames, @as(usize, @intCast(limit / page_size))); + var f: usize = 1; // frame 0 stays reserved as the "none" address + while (f < cap) : (f += 1) { + if (!isUsed(f)) { + setUsed(f); + used_frames += 1; + return @as(u64, f) * page_size; + } + } + return null; +} + /// Return a frame obtained from alloc() to the pool. Bogus or double frees are /// ignored rather than corrupting the count. pub fn free(addr: u64) void { diff --git a/src/kernel/scheduler.zig b/src/kernel/scheduler.zig index 39ccd2c..550846a 100644 --- a/src/kernel/scheduler.zig +++ b/src/kernel/scheduler.zig @@ -102,6 +102,25 @@ fn idle() void { while (true) asm volatile ("hlt"); } +/// Reserve and initialise the per-CPU slot for an application processor at dense +/// `index` (1-based; 0 is the BSP) with Local APIC id `apic_id`, and return a +/// pointer the arch bring-up hands to the core (it publishes it in its GS base). +/// Called on the BSP before waking each AP; the AP marks itself `online`. +pub fn prepareSecondary(index: usize, apic_id: u32) *PerCpu { + const pc = &cpus[index]; + pc.* = .{ .index = @intCast(index), .apic_id = apic_id, .online = false }; + return pc; +} + +/// Number of cores that have finished bring-up (the BSP plus every online AP). +pub fn onlineCount() usize { + var n: usize = 0; + for (&cpus) |*pc| { + if (pc.online) n += 1; + } + return n; +} + fn enqueue(t: *Task) void { t.next = null; const p: usize = t.priority;