keep the AP trampoline inert between wakes, and retry failed cores

Arm the low frame (copy blob, make executable) only while a core climbs, then zero it and restore RW+NX; the frame stays reserved so cores can be re-woken. startSecondary is one re-runnable attempt; boot retries a non-responding core 3x. Validated the retry path by forcing a first-attempt failure.
This commit is contained in:
Daniel Samson
2026-07-08 13:32:34 +01:00
parent dba3939a0f
commit debe815a5c
4 changed files with 68 additions and 28 deletions
+10
View File
@@ -198,6 +198,16 @@ next lands.
- **`single_threaded` off** — the kernel was built `single_threaded = true`, which - **`single_threaded` off** — the kernel was built `single_threaded = true`, which
compiles `std.atomic` down to plain non-atomic ops. Harmless on one core, but it compiles `std.atomic` down to plain non-atomic ops. Harmless on one core, but it
quietly breaks the big kernel lock across cores; it's now `false`. quietly breaks the big kernel lock across cores; it's now `false`.
- **Re-armable wake + retry** — the trampoline frame is reserved for the system's
life, but kept **inert between wakes**: zeroed and non-executable, armed (blob
copied in, page made executable) only for the moment a core is actually climbing,
then disarmed again. So there's never a dormant executable page, and a core can be
(re)woken at any time — `arch.startSecondary` is one self-contained attempt (arm →
INIT–SIPI–SIPI → disarm), and its `INIT` resets a wedged core, so retrying just
works. Boot retries a non-responding core up to three times; the same primitive is
the groundwork a future **power manager** would drive to bring cores up (and,
eventually, its counterpart to take them offline — which additionally needs the
core's tasks migrated off first).
**Next (refinement, not first-light):** **Next (refinement, not first-light):**
+6 -10
View File
@@ -102,16 +102,12 @@ pub fn cpuLocal() usize {
// --- SMP: application-processor bring-up ---------------------------------- // --- SMP: application-processor bring-up ----------------------------------
/// Make a low RAM page executable (clear its NX bit) — the AP trampoline is fetched /// Record the low (<1 MiB) frame reserved for the AP trampoline. Run once at boot.
/// from it under paging. Delegates to the VMM; see paging.setExecutable. /// The frame stays inert (zeroed, non-executable) between wakes and is armed only
pub fn setPageExecutable(phys: u64) void { /// while a core is climbing — so a core can be (re)woken at any time (retry, or a
paging.setExecutable(phys); /// future power manager) without leaving an executable page resident. See smp.zig.
} pub fn setTrampolinePage(phys: u64) void {
smp.setTrampolinePage(phys);
/// Copy the AP trampoline into its low page (allocated + made executable by the
/// caller). Run once before waking any application processor.
pub fn prepareSecondaries(tramp_phys: u64) void {
smp.prepare(tramp_phys);
} }
/// Wake the core with Local APIC id `apic_id` as dense CPU `index`, giving it /// Wake the core with Local APIC id `apic_id` as dense CPU `index`, giving it
+38 -9
View File
@@ -18,13 +18,19 @@ const gdt = @import("gdt.zig");
const tss = @import("tss.zig"); const tss = @import("tss.zig");
const idt = @import("idt.zig"); const idt = @import("idt.zig");
const apic = @import("apic.zig"); const apic = @import("apic.zig");
const paging = @import("paging.zig");
/// IA32_GS_BASE — the per-CPU data pointer (see cpu.zig; kept in sync here so the AP /// IA32_GS_BASE — the per-CPU data pointer (see cpu.zig; kept in sync here so the AP
/// path doesn't depend on cpu.zig and risk an import cycle). /// path doesn't depend on cpu.zig and risk an import cycle).
const ia32_gs_base = 0xC000_0101; const ia32_gs_base = 0xC000_0101;
const page_size = 0x1000;
/// Physical address of the trampoline page (page-aligned, below 1 MiB). Set by /// Physical address of the low (<1 MiB) frame reserved for the trampoline. Held for
/// `prepare`; the low 20 bits are always zero, so `phys >> 12` is the SIPI vector. /// the life of the system so any core can be (re)woken on demand — a retry, or a
/// future power manager bringing a core back online. The frame is kept **inert**
/// between wakes (zeroed and non-executable) and only armed for the brief moment a
/// core is actually climbing. Its low 20 bits are zero, so `phys >> 12` is the SIPI
/// vector.
var tramp_phys: u64 = 0; var tramp_phys: u64 = 0;
/// Set to 1 by a freshly-woken AP once it reaches `apEntry` and finishes its own /// Set to 1 by a freshly-woken AP once it reaches `apEntry` and finishes its own
@@ -45,17 +51,34 @@ pub fn setSecondaryEntry(entry: *const fn () callconv(.c) noreturn) void {
secondary_entry = entry; secondary_entry = entry;
} }
/// Copy the trampoline blob to its low page. Call once, after the page has been /// Record the reserved low frame the trampoline uses. Call once at boot. The frame
/// allocated and made executable, before waking any AP. /// starts inert (identity-mapped RW+NX like all RAM); each wake arms it and disarms
pub fn prepare(phys: u64) void { /// it again, so it's only ever executable while a core is climbing.
pub fn setTrampolinePage(phys: u64) void {
tramp_phys = phys; tramp_phys = phys;
}
/// Arm the trampoline for a wake: make its page executable (W^X exception for the
/// duration of the climb) and copy the blob in.
fn arm() void {
paging.setExecutable(tramp_phys);
const start = @extern([*]const u8, .{ .name = "ap_trampoline_start" }); const start = @extern([*]const u8, .{ .name = "ap_trampoline_start" });
const end = @extern([*]const u8, .{ .name = "ap_trampoline_end" }); const end = @extern([*]const u8, .{ .name = "ap_trampoline_end" });
const len = @intFromPtr(end) - @intFromPtr(start); const len = @intFromPtr(end) - @intFromPtr(start);
const dst: [*]u8 = @ptrFromInt(phys); const dst: [*]u8 = @ptrFromInt(tramp_phys);
@memcpy(dst[0..len], start[0..len]); @memcpy(dst[0..len], start[0..len]);
} }
/// Disarm after a wake: wipe the page and restore it to inert RW+NX, so no
/// executable code (nor any stale bytes) lingers between wakes. Safe to run once the
/// woken core has reported in — it's long past the trampoline by then, in the kernel
/// image; a core that never answered is dead and can't be mid-climb.
fn disarm() void {
const dst: [*]u8 = @ptrFromInt(tramp_phys);
@memset(dst[0..page_size], 0);
paging.map(tramp_phys, tramp_phys, true); // RW + NX, like every other RAM frame
}
/// Address of a patchable trampoline parameter, by symbol name: the copied blob's /// Address of a patchable trampoline parameter, by symbol name: the copied blob's
/// base plus the field's offset within it (a same-section symbol difference). The /// base plus the field's offset within it (a same-section symbol difference). The
/// pointer is `align(1)` — the fields aren't 8-aligned within the blob, and x86 /// pointer is `align(1)` — the fields aren't 8-aligned within the blob, and x86
@@ -67,11 +90,17 @@ fn param(comptime name: []const u8) *align(1) volatile u64 {
} }
/// Wake the core with Local APIC id `apic_id` as dense CPU `index`, hand it /// Wake the core with Local APIC id `apic_id` as dense CPU `index`, hand it
/// `stack_top` and its per-CPU pointer `percpu`, and wait for it to come alive. /// `stack_top` and its per-CPU pointer `percpu`, and wait for it to come alive. This
/// Returns false if it doesn't report in within the timeout (left parked, no harm to /// is one self-contained attempt: it arms the trampoline, drives INIT–SIPI–SIPI, and
/// disarms again before returning — so it's safe to call repeatedly (a retry, or a
/// power manager re-waking a core; the INIT resets a core that was wedged). Returns
/// false if the core doesn't report in within the timeout (left parked, no harm to
/// the running system). `cr3` is the kernel page tables the AP adopts. Precondition: /// the running system). `cr3` is the kernel page tables the AP adopts. Precondition:
/// `prepare` has run. /// `setTrampolinePage` has run.
pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, index: usize, cr3: u64) bool { pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, index: usize, cr3: u64) bool {
arm();
defer disarm();
boot_index = index; boot_index = index;
param("ap_tramp_cr3").* = cr3; param("ap_tramp_cr3").* = cr3;
param("ap_tramp_stack").* = stack_top; param("ap_tramp_stack").* = stack_top;
+12 -7
View File
@@ -259,17 +259,18 @@ fn bringUpSecondaries() void {
const cores = platform.cpus(); const cores = platform.cpus();
if (cores.len <= 1) return; if (cores.len <= 1) return;
// The trampoline page was reserved below 1 MiB at boot (a real-mode SIPI vector // A low (<1 MiB) frame was reserved at boot for the real-mode trampoline (a SIPI
// addresses it). Make it executable — the blanket RAM mapping is NX (W^X). // vector addresses it). It's kept for the system's life — armed only during a
// wake, inert (zeroed, non-executable) otherwise — so cores can be re-woken later.
if (ap_trampoline_page == 0) { if (ap_trampoline_page == 0) {
log.write("danos: smp: no low page for the AP trampoline; staying uniprocessor\n"); log.write("danos: smp: no low page for the AP trampoline; staying uniprocessor\n");
return; return;
} }
arch.setPageExecutable(ap_trampoline_page); arch.setTrampolinePage(ap_trampoline_page);
arch.prepareSecondaries(ap_trampoline_page);
arch.setSecondaryEntry(scheduler.secondaryMain); // where a woken core joins the run loop arch.setSecondaryEntry(scheduler.secondaryMain); // where a woken core joins the run loop
log.print("\ndanos: bringing up {d} application processor(s)\n", .{cores.len - 1}); log.print("\ndanos: bringing up {d} application processor(s)\n", .{cores.len - 1});
const max_wake_attempts = 3; // a core that misses the first INIT-SIPI-SIPI gets retried
for (cores[1..], 1..) |core, index| { for (cores[1..], 1..) |core, index| {
const stack = heap.allocator().alloc(u8, 16 * 1024) catch { const stack = heap.allocator().alloc(u8, 16 * 1024) catch {
log.print(" cpu apic_id {d}: no stack; skipped\n", .{core.apic_id}); log.print(" cpu apic_id {d}: no stack; skipped\n", .{core.apic_id});
@@ -277,11 +278,15 @@ fn bringUpSecondaries() void {
}; };
const stack_top = (@intFromPtr(stack.ptr) + stack.len) & ~@as(usize, 15); const stack_top = (@intFromPtr(stack.ptr) + stack.len) & ~@as(usize, 15);
const pc = scheduler.prepareSecondary(index, core.apic_id); const pc = scheduler.prepareSecondary(index, core.apic_id);
var attempt: u32 = 1;
while (attempt <= max_wake_attempts) : (attempt += 1) {
if (arch.startSecondary(core.apic_id, stack_top, @intFromPtr(pc), index)) { if (arch.startSecondary(core.apic_id, stack_top, @intFromPtr(pc), index)) {
pc.online = true; pc.online = true;
log.print(" cpu apic_id {d}: online\n", .{core.apic_id}); log.print(" cpu apic_id {d}: online (attempt {d})\n", .{ core.apic_id, attempt });
} else { break;
log.print(" cpu apic_id {d}: no response (parked)\n", .{core.apic_id}); }
if (attempt == max_wake_attempts)
log.print(" cpu apic_id {d}: no response after {d} attempts (parked)\n", .{ core.apic_id, max_wake_attempts });
} }
} }
log.print("danos: {d}/{d} cores online\n", .{ scheduler.onlineCount(), cores.len }); log.print("danos: {d}/{d} cores online\n", .{ scheduler.onlineCount(), cores.len });