From 43afe6bf2e995187c2bba14e15e2a4139daf9eee Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Wed, 8 Jul 2026 12:28:31 +0100 Subject: [PATCH] schedule tasks across all cores Per-core GDT/TSS and AP scheduler entry; fix AP SSE + single_threaded. --- build.zig | 2 +- docs/smp.md | 60 ++++++++++++++++++----------- src/kernel/arch/x86_64/cpu.zig | 17 +++++--- src/kernel/arch/x86_64/gdt.zig | 51 ++++++++++++++---------- src/kernel/arch/x86_64/smp.zig | 60 ++++++++++++++++++++--------- src/kernel/arch/x86_64/trampoline.s | 14 ++++++- src/kernel/arch/x86_64/tss.zig | 33 +++++++++++----- src/kernel/main.zig | 3 +- src/kernel/scheduler.zig | 28 ++++++++++++++ src/kernel/tests.zig | 45 ++++++++++++++++++++++ test/qemu_test.py | 7 ++++ 11 files changed, 239 insertions(+), 81 deletions(-) diff --git a/build.zig b/build.zig index aad09fb..369e906 100644 --- a/build.zig +++ b/build.zig @@ -114,7 +114,7 @@ pub fn build(b: *std.Build) void { .optimize = optimize, .code_model = .small, // kernel is linked in the low 2 GiB (see image_base) .red_zone = false, // interrupts would corrupt the SysV red zone - .single_threaded = true, // no scheduler yet; avoids pulling in TLS/atomics + .single_threaded = false, // SMP: the big kernel lock's atomics must be real across cores .sanitize_c = .off, // the UBSan runtime needs f128/SSE support we don't provide .stack_check = false, // stack-probe calls have no runtime to land in .stack_protector = false, diff --git a/docs/smp.md b/docs/smp.md index 28feea4..2c8a803 100644 --- a/docs/smp.md +++ b/docs/smp.md @@ -16,13 +16,14 @@ danos specifics): - **Real-time** — whether timing is *predictable*. Comes from bounded operations (our O(1) scheduler), not from core count. -danos is mid-transition. The firmware starts only the **bootstrap processor (BSP)**; -the other cores (**application processors**, APs) sit parked until the kernel wakes -them. As of the SMP work in progress (see [Implementation status](#implementation-status) -below), danos now *does* wake the APs — each climbs to 64-bit long mode and reports -in — and the shared kernel state (scheduler queues, IPC) is already serialised behind -a big kernel lock. What's not done yet is letting the woken APs actually run tasks; -`current` is per-CPU but the run loop is still BSP-only. +danos now runs on multiple cores. The firmware starts only the **bootstrap processor +(BSP)**; the kernel wakes the other cores (**application processors**, APs) with +INIT–SIPI–SIPI, brings each up into 64-bit long mode with its own descriptor tables, +LAPIC and timer, and drops it into the scheduler. Tasks run **genuinely in parallel** — +the `smp` self-test confirms worker tasks executing on all four cores at once under +QEMU `-smp 4`. Shared kernel state (scheduler queues, IPC) is serialised behind a big +kernel lock. What's left is refinement, not first-light: per-core run queues, IPIs, +and thread-to-core affinity (see [Implementation status](#implementation-status)). ## The common microkernel instinct: don't share kernel state @@ -122,9 +123,9 @@ Whatever the top goal, the *sequence* is the same and seL4 validates starting si and `platform.cpus()` returns the list (see [discovery.md](discovery.md)). The boot log reports the count; the ARM (device-tree) path still needs it. 2. **Wake the APs** — INIT–SIPI–SIPI on x86; PSCI/spin-tables on ARM. Each core brings - up its own tables, timer, and idle task. **In progress on x86** — the cores reach - long mode and park; the per-core tables/timer/scheduler entry is the next step - ([status](#implementation-status)). + up its own tables, timer, and idle task. **Done on x86** — cores climb to long mode, + set up their own GDT/TSS, and enter the scheduler; tasks run in parallel across all + cores ([status](#implementation-status)). 3. **Start with a big kernel lock.** It's a legitimate first design, not a shortcut — philosophically aligned with a tiny kernel, and it lets the single-core correctness model you already have (the interrupt-flag discipline in @@ -168,7 +169,7 @@ next lands. publishes its per-CPU pointer, and reports in. Verified in QEMU with `-smp 4`: all four cores report `online`. - The trampoline earns its complexity from three hardware facts: + The trampoline earns its complexity from four hardware facts: - a STARTUP IPI vectors a core to physical `vector << 12` (a *byte* vector), so the trampoline must live **below 1 MiB** — the kernel reserves that page from the frame allocator at boot, before paging/heap draw down the scarce low frames; @@ -178,20 +179,33 @@ next lands. **position-independent**: it derives its own base from `CS` and, crucially, addresses data *segment-relative in real mode* (where the segment base already supplies the page base) but *base-register-relative in protected/long mode* (flat - segments, base 0). Getting that distinction wrong was the first bug found. + segments, base 0). Getting that distinction wrong was the first bug found; + - an AP starts with a bare `CR0`/`CR4`, but the kernel is built **with SSE** (the + x86_64 baseline) and the compiler emits SSE for things as ordinary as a struct + copy — so the trampoline must set `CR4.OSFXSR`/`OSXMMEXCPT` and fix `CR0.EM`/`MP`, + or the first SSE instruction on the AP `#UD`s. The BSP inherited those bits from + UEFI; the AP has to set them itself. This was the second bug — it masqueraded as a + fault in `lgdt` (the first kernel code after entry that the compiler vectorised). -**Next:** +- **Per-core tables + scheduler entry** — each AP loads **its own GDT** (with its own + TSS descriptor) and **its own TSS** (its own IST/`rsp0` stack), loads the shared + IDT, enables its LAPIC and timer, then calls the generic `secondaryMain`: it turns + its bring-up context into the core's idle task (as task 0 is for the BSP), marks the + core online, and enters the run loop. With interrupts on, each core's own timer tick + preempts its idle context into whatever the global ready queue offers — so all cores + pull real work in parallel. The `smp` test spawns CPU-bound workers and confirms they + execute on all four cores at once. +- **`single_threaded` off** — the kernel was built `single_threaded = true`, which + compiles `std.atomic` down to plain non-atomic ops. Harmless on one core, but it + quietly breaks the big kernel lock across cores; it's now `false`. -- **Per-core descriptor tables + scheduler entry** — each AP needs its own TSS (its - own IST/`rsp0` stack) and to load the kernel GDT/IDT, enable its LAPIC timer, and - enter the scheduler run loop under the big lock. (The 3a checkpoint deliberately - parks the APs on the trampoline's tables with interrupts off; loading the shared - kernel GDT on an AP faulted, and the per-core-TSS work is where that's resolved.) -- **A parallelism test** — a case where N cores drive N counters at once, proving work - runs truly in parallel rather than just that the APs booted. -- **IPIs** (deferred) — cross-core wake/preempt. Not needed for correctness: an idle - core wakes on its own timer tick and pulls ready work then; IPIs only cut that - latency from ≤1 ms to near-instant. +**Next (refinement, not first-light):** + +- **IPIs** — cross-core wake/preempt. Not needed for correctness: an idle core wakes + on its own timer tick and pulls ready work then; IPIs only cut that latency from + ≤1 ms to near-instant. +- **Per-core run queues + thread affinity** — the Fiasco.OC direction, if the single + global queue's lock contention ever bites (and the more real-time-predictable model). ## Further reading diff --git a/src/kernel/arch/x86_64/cpu.zig b/src/kernel/arch/x86_64/cpu.zig index b3adb8c..df42eee 100644 --- a/src/kernel/arch/x86_64/cpu.zig +++ b/src/kernel/arch/x86_64/cpu.zig @@ -114,11 +114,18 @@ pub fn prepareSecondaries(tramp_phys: u64) void { smp.prepare(tramp_phys); } -/// Wake the core with Local APIC id `apic_id`, giving it `stack_top` and its per-CPU -/// pointer `percpu`; it adopts the current (kernel) page tables. Returns false if it -/// doesn't come online within the timeout. Blocks until the core reports in. -pub fn startSecondary(apic_id: u32, stack_top: usize, percpu: usize) bool { - return smp.startAp(apic_id, stack_top, percpu, readCr3()); +/// Wake the core with Local APIC id `apic_id` as dense CPU `index`, giving it +/// `stack_top` and its per-CPU pointer `percpu`; it adopts the current (kernel) page +/// tables. Returns false if it doesn't come online within the timeout. Blocks until +/// the core reports in. +pub fn startSecondary(apic_id: u32, stack_top: usize, percpu: usize, index: usize) bool { + return smp.startAp(apic_id, stack_top, percpu, index, readCr3()); +} + +/// Register the generic entry a woken AP jumps to once its arch state is up (its own +/// descriptor tables, LAPIC, and timer). The kernel passes its scheduler entry here. +pub fn setSecondaryEntry(entry: *const fn () callconv(.c) noreturn) void { + smp.setSecondaryEntry(entry); } /// Kernel tick rate: 1000 Hz (1 ms), the scheduler's time quantum. diff --git a/src/kernel/arch/x86_64/gdt.zig b/src/kernel/arch/x86_64/gdt.zig index 4d2772b..bd07907 100644 --- a/src/kernel/arch/x86_64/gdt.zig +++ b/src/kernel/arch/x86_64/gdt.zig @@ -3,18 +3,25 @@ //! reference a code selector — so we install our own flat GDT with known //! selectors (0x08 kernel code, 0x10 kernel data) rather than trusting whatever //! the firmware left in place. +//! +//! The code/data descriptors are identical on every core, but the **TSS descriptor +//! is per-core** (each core needs its own TSS — its own interrupt/fault stacks; see +//! tss.zig). Two cores can't share one TSS descriptor slot, so each core gets its +//! own copy of the table with its own TSS descriptor. Slot 0 is the BSP. -/// Selectors into the table below (index * 8). +/// Selectors into the table (index * 8). Same on every core's GDT. pub const kernel_code = 0x08; pub const kernel_data = 0x10; pub const tss_selector = 0x18; -/// Flat 64-bit descriptors. Base/limit are ignored in long mode; what matters is -/// the access byte and, for code, the long-mode (L) flag. +const max_cpus = 64; // matches the scheduler / discovery pool +const entries = 5; // null, code, data, TSS-low, TSS-high + +/// The shared descriptors (slots 0-2); slots 3-4 hold this core's TSS descriptor, +/// filled in per core by `setTssFor`. /// code: present, ring 0, executable, readable, L=1 -> 0x00AF9A00_0000FFFF /// data: present, ring 0, writable -> 0x00CF9200_0000FFFF -/// The last two slots hold one 16-byte TSS descriptor, filled in by setTss. -var table = [_]u64{ +const template = [entries]u64{ 0, // null descriptor (required) 0x00AF9A000000FFFF, // kernel code (0x08) 0x00CF92000000FFFF, // kernel data (0x10) @@ -22,16 +29,20 @@ var table = [_]u64{ 0, // TSS descriptor high }; -/// Fill the 64-bit TSS system descriptor (two GDT slots) so the task register can -/// point at our TSS. Type 0x89 = present, ring 0, available 64-bit TSS. -pub fn setTss(base: u64, limit: u64) void { - table[3] = (limit & 0xFFFF) | +/// One GDT per core (each a copy of the template, differing only in its TSS slot). +var gdts = [_][entries]u64{template} ** max_cpus; + +/// Fill core `cpu`'s 64-bit TSS system descriptor (two GDT slots) so its task +/// register can point at its own TSS. Type 0x89 = present, ring 0, available 64-bit +/// TSS. Write it into that core's GDT before it loads the TSS selector. +pub fn setTssFor(cpu: usize, base: u64, limit: u64) void { + gdts[cpu][3] = (limit & 0xFFFF) | ((base & 0xFFFF) << 16) | (((base >> 16) & 0xFF) << 32) | (@as(u64, 0x89) << 40) | (((limit >> 16) & 0xF) << 48) | (((base >> 24) & 0xFF) << 56); - table[4] = (base >> 32) & 0xFFFFFFFF; + gdts[cpu][4] = (base >> 32) & 0xFFFFFFFF; } /// The operand `lgdt` wants: table byte-length minus one, then its address. @@ -41,23 +52,21 @@ const Descriptor = packed struct { }; /// Loads the GDT and reloads the segment registers (including CS). Defined in -/// isr.s — it uses the selectors 0x08 (code) and 0x10 (data) that match `table`. +/// isr.s — it uses the selectors 0x08 (code) and 0x10 (data) that match the table. extern fn gdt_flush(descriptor: *const Descriptor) callconv(.c) void; -/// Load our GDT on the current core and switch onto its segments. The table is -/// shared across all cores (the descriptors are flat and read-only); each core just -/// needs to point its GDTR at it. Called by the BSP in `init` and by every AP during -/// bring-up. Note this reloads the segment registers, which zeroes the GS base — so -/// a core must publish its per-CPU pointer (setCpuLocal) *after* calling this. -pub fn loadOnThisCpu() void { +/// Load core `cpu`'s GDT and switch onto its segments. Note this reloads the segment +/// registers, which zeroes the GS base — so a core must publish its per-CPU pointer +/// (setCpuLocal) *after* calling this. +pub fn loadOnThisCpu(cpu: usize) void { const descriptor = Descriptor{ - .limit = @sizeOf(@TypeOf(table)) - 1, - .base = @intFromPtr(&table), + .limit = @sizeOf([entries]u64) - 1, + .base = @intFromPtr(&gdts[cpu]), }; gdt_flush(&descriptor); } -/// Install our GDT and switch onto its segments (bootstrap processor). +/// Install the bootstrap processor's GDT (slot 0) and switch onto its segments. pub fn init() void { - loadOnThisCpu(); + loadOnThisCpu(0); } diff --git a/src/kernel/arch/x86_64/smp.zig b/src/kernel/arch/x86_64/smp.zig index 2ebaac5..86b2dce 100644 --- a/src/kernel/arch/x86_64/smp.zig +++ b/src/kernel/arch/x86_64/smp.zig @@ -9,14 +9,14 @@ //! //! Cores are brought up **one at a time**: a single trampoline page and parameter //! block are reused, so the BSP patches, wakes, and waits for one AP before the -//! next. The mechanism-vs-policy split matches the rest of the kernel — the generic -//! scheduler decides *what* runs where; this just gets a core executing 64-bit code. -//! -//! This is step 3a: an AP climbs to long mode, publishes its per-CPU pointer, marks -//! itself alive, and parks. Entering the scheduler (its own TSS, LAPIC timer, and -//! the run loop) is the next step. +//! next. That also lets `apEntry` pick up its dense CPU index from a plain global. +//! Once a core has its own descriptor tables, LAPIC, and timer, it calls the generic +//! scheduler entry and joins the run loop — mechanism here, policy there. const io = @import("io.zig"); +const gdt = @import("gdt.zig"); +const tss = @import("tss.zig"); +const idt = @import("idt.zig"); const apic = @import("apic.zig"); /// IA32_GS_BASE — the per-CPU data pointer (see cpu.zig; kept in sync here so the AP @@ -32,6 +32,19 @@ var tramp_phys: u64 = 0; /// one-at-a-time handshake (only one AP is being started at any moment). var ap_alive: u32 = 0; +/// The dense CPU index of the AP currently being started. Set by the BSP before the +/// wake, read by `apEntry` (safe because bring-up is strictly one core at a time). +var boot_index: usize = 0; + +/// The generic scheduler entry a woken core jumps to once its arch state is up. Set +/// by the kernel via `setSecondaryEntry`; never returns. +var secondary_entry: ?*const fn () callconv(.c) noreturn = null; + +/// Register the generic entry an AP calls once its per-CPU tables/LAPIC/timer are up. +pub fn setSecondaryEntry(entry: *const fn () callconv(.c) noreturn) void { + secondary_entry = entry; +} + /// Copy the trampoline blob to its low page. Call once, after the page has been /// allocated and made executable, before waking any AP. pub fn prepare(phys: u64) void { @@ -53,11 +66,13 @@ fn param(comptime name: []const u8) *align(1) volatile u64 { return @ptrFromInt(tramp_phys + (sym - start)); } -/// Wake the core with Local APIC id `apic_id`, hand it `stack_top` and `percpu` (its -/// per-CPU pointer), and wait for it to come alive. Returns false if it doesn't -/// report in within the timeout (left parked, no harm to the running system). -/// `cr3` is the kernel page tables the AP adopts. Precondition: `prepare` has run. -pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, cr3: u64) bool { +/// Wake the core with Local APIC id `apic_id` as dense CPU `index`, hand it +/// `stack_top` and its per-CPU pointer `percpu`, and wait for it to come alive. +/// Returns false if it doesn't report in within the timeout (left parked, no harm to +/// the running system). `cr3` is the kernel page tables the AP adopts. Precondition: +/// `prepare` has run. +pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, index: usize, cr3: u64) bool { + boot_index = index; param("ap_tramp_cr3").* = cr3; param("ap_tramp_stack").* = stack_top; param("ap_tramp_entry").* = @intFromPtr(&apEntry); @@ -89,13 +104,20 @@ fn delayMicros(us: u64) void { } /// The 64-bit entry every AP lands on, called from the trampoline with its per-CPU -/// pointer in RDI. Adopts the shared descriptor tables, publishes its per-CPU -/// pointer, signals the BSP it's alive, and (for now) parks. Never returns. +/// pointer in RDI. Brings up this core's own descriptor tables, LAPIC and timer, +/// signals the BSP, then jumps to the generic scheduler entry. Never returns. fn apEntry(percpu: usize) callconv(.c) noreturn { - // Step 3a: minimal. The core keeps the trampoline's descriptor tables, publishes - // its per-CPU pointer, signals the BSP, and parks with interrupts off. Loading - // this core's own kernel GDT/IDT/TSS and entering the scheduler is step 3b. - io.wrmsr(ia32_gs_base, percpu); // publish per-CPU pointer (GS base) - @atomicStore(u32, &ap_alive, 1, .release); // "I'm up" — BSP is polling this - while (true) asm volatile ("hlt"); // parked (3b enters the scheduler here) + const cpu = boot_index; + gdt.loadOnThisCpu(cpu); // this core's GDT (with its own TSS slot) + tss.setupThisCpu(cpu); // this core's TSS + IST stack, loaded into TR + idt.loadOnThisCpu(); // the shared IDT + io.wrmsr(ia32_gs_base, percpu); // per-CPU pointer — *after* the GDT reload + + apic.initSecondary(); // software-enable this core's LAPIC + apic.initTimer(apic.frequencyHz()); // arm its timer (still masked: interrupts off) + + @atomicStore(u32, &ap_alive, 1, .release); // "arch state up" — BSP is polling this + + if (secondary_entry) |enterScheduler| enterScheduler(); // joins the run loop + while (true) asm volatile ("hlt"); // (only if no entry was registered) } diff --git a/src/kernel/arch/x86_64/trampoline.s b/src/kernel/arch/x86_64/trampoline.s index 1cf7f39..6195ff5 100644 --- a/src/kernel/arch/x86_64/trampoline.s +++ b/src/kernel/arch/x86_64/trampoline.s @@ -59,10 +59,20 @@ prot_entry: movw %ax, %fs movw %ax, %gs - movl %cr4, %eax # PAE on (CR4.PAE) — required for long mode - orl $(1 << 5), %eax + # CR4: PAE (required for long mode) + OSFXSR/OSXMMEXCPT. The kernel is built with + # SSE (part of the x86_64 baseline), and the compiler emits SSE for things as + # ordinary as a struct copy — without OSFXSR those instructions #UD. The BSP got + # these bits from UEFI; an AP starts fresh, so we must set them ourselves. + movl %cr4, %eax + orl $((1 << 5) | (1 << 9) | (1 << 10)), %eax movl %eax, %cr4 + # CR0: clear EM (no x87 emulation) and set MP, so SSE/x87 don't fault. + movl %cr0, %eax + andl $~(1 << 2), %eax # ~EM + orl $(1 << 1), %eax # MP + movl %eax, %cr0 + movl (param_cr3 - ap_trampoline_start)(%ebx), %eax # kernel page tables movl %eax, %cr3 diff --git a/src/kernel/arch/x86_64/tss.zig b/src/kernel/arch/x86_64/tss.zig index 4e1f6c1..ca92983 100644 --- a/src/kernel/arch/x86_64/tss.zig +++ b/src/kernel/arch/x86_64/tss.zig @@ -4,6 +4,10 @@ //! interrupted stack was. We use IST1 for the double-fault handler, so a fault //! that happens *because* the current stack is unusable still lands on solid //! ground instead of triple-faulting. +//! +//! Each core needs **its own TSS** (its own IST stack): two cores taking a fault at +//! once can't share one fault stack. So the TSS and its IST stack are per-core, +//! indexed by CPU number; slot 0 is the BSP. const gdt = @import("gdt.zig"); @@ -30,19 +34,30 @@ const Tss = packed struct { /// The IST slot (1-based, as the IDT gate encodes it) used for critical faults. pub const double_fault_ist = 1; -var tss: Tss align(16) = .{}; +const max_cpus = 64; // matches gdt.zig / the scheduler +const ist_stack_size = 16 * 1024; -/// Dedicated stack for IST1. Static so it needs no allocator and is always valid. -var ist1_stack: [16 * 1024]u8 align(16) = undefined; +/// One TSS per core, and one IST1 stack per core. Static, so they need no allocator +/// and are always valid. (max_cpus × 16 KiB of BSS for the IST stacks.) +var tss_table = [_]Tss{.{}} ** max_cpus; +var ist_stacks: [max_cpus][ist_stack_size]u8 align(16) = undefined; /// Loads the task register with the TSS selector. Defined in isr.s. extern fn load_tr(selector: u16) callconv(.c) void; -/// Point IST1 at its stack, publish the TSS through the GDT, and load it into the -/// task register. Requires the GDT to already be loaded (gdt.init first). -pub fn init() void { - tss.ist1 = @intFromPtr(&ist1_stack) + ist1_stack.len; // stacks grow down - tss.iomap_base = @sizeOf(Tss); // == limit: no I/O permission bitmap - gdt.setTss(@intFromPtr(&tss), @sizeOf(Tss) - 1); +/// Set up core `cpu`'s TSS: point IST1 at that core's stack, install the TSS +/// descriptor into that core's GDT, and load it into the task register. Requires the +/// core's GDT to already be loaded (gdt.loadOnThisCpu first). +pub fn setupThisCpu(cpu: usize) void { + const t = &tss_table[cpu]; + t.* = .{}; + t.ist1 = @intFromPtr(&ist_stacks[cpu]) + ist_stack_size; // stacks grow down + t.iomap_base = @sizeOf(Tss); // == limit: no I/O permission bitmap + gdt.setTssFor(cpu, @intFromPtr(t), @sizeOf(Tss) - 1); load_tr(gdt.tss_selector); } + +/// Set up the bootstrap processor's TSS (slot 0). Requires gdt.init first. +pub fn init() void { + setupThisCpu(0); +} diff --git a/src/kernel/main.zig b/src/kernel/main.zig index 07b8807..97840b6 100644 --- a/src/kernel/main.zig +++ b/src/kernel/main.zig @@ -267,6 +267,7 @@ fn bringUpSecondaries() void { } arch.setPageExecutable(ap_trampoline_page); arch.prepareSecondaries(ap_trampoline_page); + arch.setSecondaryEntry(scheduler.secondaryMain); // where a woken core joins the run loop log.print("\ndanos: bringing up {d} application processor(s)\n", .{cores.len - 1}); for (cores[1..], 1..) |core, index| { @@ -276,7 +277,7 @@ fn bringUpSecondaries() void { }; const stack_top = (@intFromPtr(stack.ptr) + stack.len) & ~@as(usize, 15); const pc = scheduler.prepareSecondary(index, core.apic_id); - if (arch.startSecondary(core.apic_id, stack_top, @intFromPtr(pc))) { + if (arch.startSecondary(core.apic_id, stack_top, @intFromPtr(pc), index)) { pc.online = true; log.print(" cpu apic_id {d}: online\n", .{core.apic_id}); } else { diff --git a/src/kernel/scheduler.zig b/src/kernel/scheduler.zig index 550846a..3cb81b3 100644 --- a/src/kernel/scheduler.zig +++ b/src/kernel/scheduler.zig @@ -112,6 +112,27 @@ pub fn prepareSecondary(index: usize, apic_id: u32) *PerCpu { return pc; } +/// Entry for an application processor once the arch layer has set up its per-CPU +/// tables, LAPIC, and timer. It turns this bring-up context into the core's idle task +/// (as task 0 is for the BSP), marks the core online, and enters the run loop: with +/// interrupts enabled the timer preempts this idle context into whatever the global +/// ready queue offers, so the core runs real work in parallel with the others. The +/// `.c` calling convention lets the arch trampoline path jump here. Never returns. +pub fn secondaryMain() callconv(.c) noreturn { + const flags = sync.enter(); + const pc = thisCpu(); + const t = freeSlot() orelse @panic("sched: task table full (AP idle task)"); + t.* = .{ .id = next_id, .state = .running, .priority = 0 }; + next_id += 1; + pc.current = t; + pc.idle = t; + pc.online = true; + sync.leave(flags); + + arch.enableInterrupts(); // the timer now preempts this idle context into work + while (true) asm volatile ("hlt"); // idle when this core has nothing ready +} + /// Number of cores that have finished bring-up (the BSP plus every online AP). pub fn onlineCount() usize { var n: usize = 0; @@ -333,6 +354,13 @@ pub fn currentId() u32 { return cur().id; } +/// The dense index of the core this task is currently running on (0 = BSP). Reads +/// per-CPU state, so a task calling it on different cores sees different values — +/// which is how a test can prove work is running in parallel. +pub fn currentCpuIndex() u32 { + return thisCpu().index; +} + /// Change the running task's priority (takes effect next time it's enqueued). pub fn setPriority(p: Priority) void { cur().priority = p; diff --git a/src/kernel/tests.zig b/src/kernel/tests.zig index 05470d5..99229af 100644 --- a/src/kernel/tests.zig +++ b/src/kernel/tests.zig @@ -68,6 +68,8 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void { eventTest(); } else if (eql(case, "ipc")) { ipcTest(); + } else if (eql(case, "smp")) { + smpTest(); } else if (eql(case, "fault-ud")) { faultInvalidOpcode(); } else if (eql(case, "fault-pf")) { @@ -437,6 +439,49 @@ fn sleepTest() void { result(); } +// --- SMP parallelism ------------------------------------------------------ + +var seen_core = [_]bool{false} ** 8; +var smp_running: bool = true; + +/// A worker that, while running, records which core it's executing on. Spread across +/// spawned workers and idle APs, these should land on more than one core. +fn smpWorker() void { + const p: *volatile bool = &smp_running; + while (p.*) { + const c = sched.currentCpuIndex(); + if (c < seen_core.len) seen_core[c] = true; + } + sched.exit(); +} + +/// Prove tasks run **in parallel** on multiple cores (not just interleaved on one). +/// Spawn several CPU-bound workers; each stamps the core it runs on into `seen_core`. +/// With the application processors online, more than one core should show up — which +/// can only happen if work is genuinely running at the same time on different cores. +/// (Run with QEMU `-smp N`; on a single core this would see just one and fail.) +fn smpTest() void { + log("DANOS-TEST-BEGIN: smp\n", .{}); + seen_core = .{false} ** 8; + smp_running = true; + + var i: usize = 0; + while (i < 4) : (i += 1) sched.spawn(smpWorker, 4); + + // Let the workers run across cores for a stretch of real time. + var spins: u64 = 0; + while (spins < 2_000_000_000) spins +%= 1; + smp_running = false; + + var cores_seen: u32 = 0; + for (seen_core) |s| { + if (s) cores_seen += 1; + } + log("DANOS-SMP: workers ran on {d} distinct core(s)\n", .{cores_seen}); + check("tasks ran on multiple cores in parallel", cores_seen >= 2); + result(); +} + fn faultInvalidOpcode() void { log("DANOS-TEST-BEGIN: fault-ud\n", .{}); asm volatile ("ud2"); diff --git a/test/qemu_test.py b/test/qemu_test.py index e8b0b23..952043c 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -108,6 +108,11 @@ CASES = [ {"name": "ipc", "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, + # Parallelism: needs more than one core, so this case boots with -smp 4. + {"name": "smp", + "smp": 4, + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, {"name": "fault-ud", "expect": r"invalid opcode \(vector 6\)"}, {"name": "fault-pf", "expect": r"page fault \(vector 14\)"}, {"name": "fault-df", "expect": r"double fault \(vector 8\)"}, @@ -175,6 +180,8 @@ def run_case(arch, case): fail = re.compile(case["fail"]) if case.get("fail") else None cmd = [arch["qemu"]] + arch["qemu_args"](arch, esp, vars_fd, serial) + if case.get("smp"): # some cases need more than one core (e.g. parallelism) + cmd += ["-smp", str(case["smp"])] qemu = subprocess.Popen(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) try: deadline = time.monotonic() + TIMEOUT