diff --git a/docs/smp.md b/docs/smp.md index 2c8a803..36bba61 100644 --- a/docs/smp.md +++ b/docs/smp.md @@ -198,6 +198,16 @@ next lands. - **`single_threaded` off** — the kernel was built `single_threaded = true`, which compiles `std.atomic` down to plain non-atomic ops. Harmless on one core, but it quietly breaks the big kernel lock across cores; it's now `false`. +- **Re-armable wake + retry** — the trampoline frame is reserved for the system's + life, but kept **inert between wakes**: zeroed and non-executable, armed (blob + copied in, page made executable) only for the moment a core is actually climbing, + then disarmed again. So there's never a dormant executable page, and a core can be + (re)woken at any time — `arch.startSecondary` is one self-contained attempt (arm → + INIT–SIPI–SIPI → disarm), and its `INIT` resets a wedged core, so retrying just + works. Boot retries a non-responding core up to three times; the same primitive is + the groundwork a future **power manager** would drive to bring cores up (and, + eventually, its counterpart to take them offline — which additionally needs the + core's tasks migrated off first). **Next (refinement, not first-light):** diff --git a/src/kernel/arch/x86_64/cpu.zig b/src/kernel/arch/x86_64/cpu.zig index df42eee..a86d5b5 100644 --- a/src/kernel/arch/x86_64/cpu.zig +++ b/src/kernel/arch/x86_64/cpu.zig @@ -102,16 +102,12 @@ pub fn cpuLocal() usize { // --- SMP: application-processor bring-up ---------------------------------- -/// Make a low RAM page executable (clear its NX bit) — the AP trampoline is fetched -/// from it under paging. Delegates to the VMM; see paging.setExecutable. -pub fn setPageExecutable(phys: u64) void { - paging.setExecutable(phys); -} - -/// Copy the AP trampoline into its low page (allocated + made executable by the -/// caller). Run once before waking any application processor. -pub fn prepareSecondaries(tramp_phys: u64) void { - smp.prepare(tramp_phys); +/// Record the low (<1 MiB) frame reserved for the AP trampoline. Run once at boot. +/// The frame stays inert (zeroed, non-executable) between wakes and is armed only +/// while a core is climbing — so a core can be (re)woken at any time (retry, or a +/// future power manager) without leaving an executable page resident. See smp.zig. +pub fn setTrampolinePage(phys: u64) void { + smp.setTrampolinePage(phys); } /// Wake the core with Local APIC id `apic_id` as dense CPU `index`, giving it diff --git a/src/kernel/arch/x86_64/smp.zig b/src/kernel/arch/x86_64/smp.zig index 86b2dce..f342814 100644 --- a/src/kernel/arch/x86_64/smp.zig +++ b/src/kernel/arch/x86_64/smp.zig @@ -18,13 +18,19 @@ const gdt = @import("gdt.zig"); const tss = @import("tss.zig"); const idt = @import("idt.zig"); const apic = @import("apic.zig"); +const paging = @import("paging.zig"); /// IA32_GS_BASE — the per-CPU data pointer (see cpu.zig; kept in sync here so the AP /// path doesn't depend on cpu.zig and risk an import cycle). const ia32_gs_base = 0xC000_0101; +const page_size = 0x1000; -/// Physical address of the trampoline page (page-aligned, below 1 MiB). Set by -/// `prepare`; the low 20 bits are always zero, so `phys >> 12` is the SIPI vector. +/// Physical address of the low (<1 MiB) frame reserved for the trampoline. Held for +/// the life of the system so any core can be (re)woken on demand — a retry, or a +/// future power manager bringing a core back online. The frame is kept **inert** +/// between wakes (zeroed and non-executable) and only armed for the brief moment a +/// core is actually climbing. Its low 20 bits are zero, so `phys >> 12` is the SIPI +/// vector. var tramp_phys: u64 = 0; /// Set to 1 by a freshly-woken AP once it reaches `apEntry` and finishes its own @@ -45,17 +51,34 @@ pub fn setSecondaryEntry(entry: *const fn () callconv(.c) noreturn) void { secondary_entry = entry; } -/// Copy the trampoline blob to its low page. Call once, after the page has been -/// allocated and made executable, before waking any AP. -pub fn prepare(phys: u64) void { +/// Record the reserved low frame the trampoline uses. Call once at boot. The frame +/// starts inert (identity-mapped RW+NX like all RAM); each wake arms it and disarms +/// it again, so it's only ever executable while a core is climbing. +pub fn setTrampolinePage(phys: u64) void { tramp_phys = phys; +} + +/// Arm the trampoline for a wake: make its page executable (W^X exception for the +/// duration of the climb) and copy the blob in. +fn arm() void { + paging.setExecutable(tramp_phys); const start = @extern([*]const u8, .{ .name = "ap_trampoline_start" }); const end = @extern([*]const u8, .{ .name = "ap_trampoline_end" }); const len = @intFromPtr(end) - @intFromPtr(start); - const dst: [*]u8 = @ptrFromInt(phys); + const dst: [*]u8 = @ptrFromInt(tramp_phys); @memcpy(dst[0..len], start[0..len]); } +/// Disarm after a wake: wipe the page and restore it to inert RW+NX, so no +/// executable code (nor any stale bytes) lingers between wakes. Safe to run once the +/// woken core has reported in — it's long past the trampoline by then, in the kernel +/// image; a core that never answered is dead and can't be mid-climb. +fn disarm() void { + const dst: [*]u8 = @ptrFromInt(tramp_phys); + @memset(dst[0..page_size], 0); + paging.map(tramp_phys, tramp_phys, true); // RW + NX, like every other RAM frame +} + /// Address of a patchable trampoline parameter, by symbol name: the copied blob's /// base plus the field's offset within it (a same-section symbol difference). The /// pointer is `align(1)` — the fields aren't 8-aligned within the blob, and x86 @@ -67,11 +90,17 @@ fn param(comptime name: []const u8) *align(1) volatile u64 { } /// Wake the core with Local APIC id `apic_id` as dense CPU `index`, hand it -/// `stack_top` and its per-CPU pointer `percpu`, and wait for it to come alive. -/// Returns false if it doesn't report in within the timeout (left parked, no harm to +/// `stack_top` and its per-CPU pointer `percpu`, and wait for it to come alive. This +/// is one self-contained attempt: it arms the trampoline, drives INIT–SIPI–SIPI, and +/// disarms again before returning — so it's safe to call repeatedly (a retry, or a +/// power manager re-waking a core; the INIT resets a core that was wedged). Returns +/// false if the core doesn't report in within the timeout (left parked, no harm to /// the running system). `cr3` is the kernel page tables the AP adopts. Precondition: -/// `prepare` has run. +/// `setTrampolinePage` has run. pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, index: usize, cr3: u64) bool { + arm(); + defer disarm(); + boot_index = index; param("ap_tramp_cr3").* = cr3; param("ap_tramp_stack").* = stack_top; diff --git a/src/kernel/main.zig b/src/kernel/main.zig index 97840b6..ddd3c7d 100644 --- a/src/kernel/main.zig +++ b/src/kernel/main.zig @@ -259,17 +259,18 @@ fn bringUpSecondaries() void { const cores = platform.cpus(); if (cores.len <= 1) return; - // The trampoline page was reserved below 1 MiB at boot (a real-mode SIPI vector - // addresses it). Make it executable — the blanket RAM mapping is NX (W^X). + // A low (<1 MiB) frame was reserved at boot for the real-mode trampoline (a SIPI + // vector addresses it). It's kept for the system's life — armed only during a + // wake, inert (zeroed, non-executable) otherwise — so cores can be re-woken later. if (ap_trampoline_page == 0) { log.write("danos: smp: no low page for the AP trampoline; staying uniprocessor\n"); return; } - arch.setPageExecutable(ap_trampoline_page); - arch.prepareSecondaries(ap_trampoline_page); + arch.setTrampolinePage(ap_trampoline_page); arch.setSecondaryEntry(scheduler.secondaryMain); // where a woken core joins the run loop log.print("\ndanos: bringing up {d} application processor(s)\n", .{cores.len - 1}); + const max_wake_attempts = 3; // a core that misses the first INIT-SIPI-SIPI gets retried for (cores[1..], 1..) |core, index| { const stack = heap.allocator().alloc(u8, 16 * 1024) catch { log.print(" cpu apic_id {d}: no stack; skipped\n", .{core.apic_id}); @@ -277,11 +278,15 @@ fn bringUpSecondaries() void { }; const stack_top = (@intFromPtr(stack.ptr) + stack.len) & ~@as(usize, 15); const pc = scheduler.prepareSecondary(index, core.apic_id); - if (arch.startSecondary(core.apic_id, stack_top, @intFromPtr(pc), index)) { - pc.online = true; - log.print(" cpu apic_id {d}: online\n", .{core.apic_id}); - } else { - log.print(" cpu apic_id {d}: no response (parked)\n", .{core.apic_id}); + var attempt: u32 = 1; + while (attempt <= max_wake_attempts) : (attempt += 1) { + if (arch.startSecondary(core.apic_id, stack_top, @intFromPtr(pc), index)) { + pc.online = true; + log.print(" cpu apic_id {d}: online (attempt {d})\n", .{ core.apic_id, attempt }); + break; + } + if (attempt == max_wake_attempts) + log.print(" cpu apic_id {d}: no response after {d} attempts (parked)\n", .{ core.apic_id, max_wake_attempts }); } } log.print("danos: {d}/{d} cores online\n", .{ scheduler.onlineCount(), cores.len });