schedule tasks across all cores

Per-core GDT/TSS and AP scheduler entry; fix AP SSE + single_threaded.
This commit is contained in:
Daniel Samson
2026-07-08 12:35:30 +01:00
parent ed7f542006
commit 43afe6bf2e
11 changed files with 239 additions and 81 deletions
+12 -5
View File
@@ -114,11 +114,18 @@ pub fn prepareSecondaries(tramp_phys: u64) void {
smp.prepare(tramp_phys);
}
/// Wake the core with Local APIC id `apic_id`, giving it `stack_top` and its per-CPU
/// pointer `percpu`; it adopts the current (kernel) page tables. Returns false if it
/// doesn't come online within the timeout. Blocks until the core reports in.
pub fn startSecondary(apic_id: u32, stack_top: usize, percpu: usize) bool {
return smp.startAp(apic_id, stack_top, percpu, readCr3());
/// Wake the core with Local APIC id `apic_id` as dense CPU `index`, giving it
/// `stack_top` and its per-CPU pointer `percpu`; it adopts the current (kernel) page
/// tables. Returns false if it doesn't come online within the timeout. Blocks until
/// the core reports in.
pub fn startSecondary(apic_id: u32, stack_top: usize, percpu: usize, index: usize) bool {
return smp.startAp(apic_id, stack_top, percpu, index, readCr3());
}
/// Register the generic entry a woken AP jumps to once its arch state is up (its own
/// descriptor tables, LAPIC, and timer). The kernel passes its scheduler entry here.
pub fn setSecondaryEntry(entry: *const fn () callconv(.c) noreturn) void {
smp.setSecondaryEntry(entry);
}
/// Kernel tick rate: 1000 Hz (1 ms), the scheduler's time quantum.
+30 -21
View File
@@ -3,18 +3,25 @@
//! reference a code selector — so we install our own flat GDT with known
//! selectors (0x08 kernel code, 0x10 kernel data) rather than trusting whatever
//! the firmware left in place.
//!
//! The code/data descriptors are identical on every core, but the **TSS descriptor
//! is per-core** (each core needs its own TSS — its own interrupt/fault stacks; see
//! tss.zig). Two cores can't share one TSS descriptor slot, so each core gets its
//! own copy of the table with its own TSS descriptor. Slot 0 is the BSP.
/// Selectors into the table below (index * 8).
/// Selectors into the table (index * 8). Same on every core's GDT.
pub const kernel_code = 0x08;
pub const kernel_data = 0x10;
pub const tss_selector = 0x18;
/// Flat 64-bit descriptors. Base/limit are ignored in long mode; what matters is
/// the access byte and, for code, the long-mode (L) flag.
const max_cpus = 64; // matches the scheduler / discovery pool
const entries = 5; // null, code, data, TSS-low, TSS-high
/// The shared descriptors (slots 0-2); slots 3-4 hold this core's TSS descriptor,
/// filled in per core by `setTssFor`.
/// code: present, ring 0, executable, readable, L=1 -> 0x00AF9A00_0000FFFF
/// data: present, ring 0, writable -> 0x00CF9200_0000FFFF
/// The last two slots hold one 16-byte TSS descriptor, filled in by setTss.
var table = [_]u64{
const template = [entries]u64{
0, // null descriptor (required)
0x00AF9A000000FFFF, // kernel code (0x08)
0x00CF92000000FFFF, // kernel data (0x10)
@@ -22,16 +29,20 @@ var table = [_]u64{
0, // TSS descriptor high
};
/// Fill the 64-bit TSS system descriptor (two GDT slots) so the task register can
/// point at our TSS. Type 0x89 = present, ring 0, available 64-bit TSS.
pub fn setTss(base: u64, limit: u64) void {
table[3] = (limit & 0xFFFF) |
/// One GDT per core (each a copy of the template, differing only in its TSS slot).
var gdts = [_][entries]u64{template} ** max_cpus;
/// Fill core `cpu`'s 64-bit TSS system descriptor (two GDT slots) so its task
/// register can point at its own TSS. Type 0x89 = present, ring 0, available 64-bit
/// TSS. Write it into that core's GDT before it loads the TSS selector.
pub fn setTssFor(cpu: usize, base: u64, limit: u64) void {
gdts[cpu][3] = (limit & 0xFFFF) |
((base & 0xFFFF) << 16) |
(((base >> 16) & 0xFF) << 32) |
(@as(u64, 0x89) << 40) |
(((limit >> 16) & 0xF) << 48) |
(((base >> 24) & 0xFF) << 56);
table[4] = (base >> 32) & 0xFFFFFFFF;
gdts[cpu][4] = (base >> 32) & 0xFFFFFFFF;
}
/// The operand `lgdt` wants: table byte-length minus one, then its address.
@@ -41,23 +52,21 @@ const Descriptor = packed struct {
};
/// Loads the GDT and reloads the segment registers (including CS). Defined in
/// isr.s — it uses the selectors 0x08 (code) and 0x10 (data) that match `table`.
/// isr.s — it uses the selectors 0x08 (code) and 0x10 (data) that match the table.
extern fn gdt_flush(descriptor: *const Descriptor) callconv(.c) void;
/// Load our GDT on the current core and switch onto its segments. The table is
/// shared across all cores (the descriptors are flat and read-only); each core just
/// needs to point its GDTR at it. Called by the BSP in `init` and by every AP during
/// bring-up. Note this reloads the segment registers, which zeroes the GS base — so
/// a core must publish its per-CPU pointer (setCpuLocal) *after* calling this.
pub fn loadOnThisCpu() void {
/// Load core `cpu`'s GDT and switch onto its segments. Note this reloads the segment
/// registers, which zeroes the GS base — so a core must publish its per-CPU pointer
/// (setCpuLocal) *after* calling this.
pub fn loadOnThisCpu(cpu: usize) void {
const descriptor = Descriptor{
.limit = @sizeOf(@TypeOf(table)) - 1,
.base = @intFromPtr(&table),
.limit = @sizeOf([entries]u64) - 1,
.base = @intFromPtr(&gdts[cpu]),
};
gdt_flush(&descriptor);
}
/// Install our GDT and switch onto its segments (bootstrap processor).
/// Install the bootstrap processor's GDT (slot 0) and switch onto its segments.
pub fn init() void {
loadOnThisCpu();
loadOnThisCpu(0);
}
+41 -19
View File
@@ -9,14 +9,14 @@
//!
//! Cores are brought up **one at a time**: a single trampoline page and parameter
//! block are reused, so the BSP patches, wakes, and waits for one AP before the
//! next. The mechanism-vs-policy split matches the rest of the kernel — the generic
//! scheduler decides *what* runs where; this just gets a core executing 64-bit code.
//!
//! This is step 3a: an AP climbs to long mode, publishes its per-CPU pointer, marks
//! itself alive, and parks. Entering the scheduler (its own TSS, LAPIC timer, and
//! the run loop) is the next step.
//! next. That also lets `apEntry` pick up its dense CPU index from a plain global.
//! Once a core has its own descriptor tables, LAPIC, and timer, it calls the generic
//! scheduler entry and joins the run loop — mechanism here, policy there.
const io = @import("io.zig");
const gdt = @import("gdt.zig");
const tss = @import("tss.zig");
const idt = @import("idt.zig");
const apic = @import("apic.zig");
/// IA32_GS_BASE — the per-CPU data pointer (see cpu.zig; kept in sync here so the AP
@@ -32,6 +32,19 @@ var tramp_phys: u64 = 0;
/// one-at-a-time handshake (only one AP is being started at any moment).
var ap_alive: u32 = 0;
/// The dense CPU index of the AP currently being started. Set by the BSP before the
/// wake, read by `apEntry` (safe because bring-up is strictly one core at a time).
var boot_index: usize = 0;
/// The generic scheduler entry a woken core jumps to once its arch state is up. Set
/// by the kernel via `setSecondaryEntry`; never returns.
var secondary_entry: ?*const fn () callconv(.c) noreturn = null;
/// Register the generic entry an AP calls once its per-CPU tables/LAPIC/timer are up.
pub fn setSecondaryEntry(entry: *const fn () callconv(.c) noreturn) void {
secondary_entry = entry;
}
/// Copy the trampoline blob to its low page. Call once, after the page has been
/// allocated and made executable, before waking any AP.
pub fn prepare(phys: u64) void {
@@ -53,11 +66,13 @@ fn param(comptime name: []const u8) *align(1) volatile u64 {
return @ptrFromInt(tramp_phys + (sym - start));
}
/// Wake the core with Local APIC id `apic_id`, hand it `stack_top` and `percpu` (its
/// per-CPU pointer), and wait for it to come alive. Returns false if it doesn't
/// report in within the timeout (left parked, no harm to the running system).
/// `cr3` is the kernel page tables the AP adopts. Precondition: `prepare` has run.
pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, cr3: u64) bool {
/// Wake the core with Local APIC id `apic_id` as dense CPU `index`, hand it
/// `stack_top` and its per-CPU pointer `percpu`, and wait for it to come alive.
/// Returns false if it doesn't report in within the timeout (left parked, no harm to
/// the running system). `cr3` is the kernel page tables the AP adopts. Precondition:
/// `prepare` has run.
pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, index: usize, cr3: u64) bool {
boot_index = index;
param("ap_tramp_cr3").* = cr3;
param("ap_tramp_stack").* = stack_top;
param("ap_tramp_entry").* = @intFromPtr(&apEntry);
@@ -89,13 +104,20 @@ fn delayMicros(us: u64) void {
}
/// The 64-bit entry every AP lands on, called from the trampoline with its per-CPU
/// pointer in RDI. Adopts the shared descriptor tables, publishes its per-CPU
/// pointer, signals the BSP it's alive, and (for now) parks. Never returns.
/// pointer in RDI. Brings up this core's own descriptor tables, LAPIC and timer,
/// signals the BSP, then jumps to the generic scheduler entry. Never returns.
fn apEntry(percpu: usize) callconv(.c) noreturn {
// Step 3a: minimal. The core keeps the trampoline's descriptor tables, publishes
// its per-CPU pointer, signals the BSP, and parks with interrupts off. Loading
// this core's own kernel GDT/IDT/TSS and entering the scheduler is step 3b.
io.wrmsr(ia32_gs_base, percpu); // publish per-CPU pointer (GS base)
@atomicStore(u32, &ap_alive, 1, .release); // "I'm up" — BSP is polling this
while (true) asm volatile ("hlt"); // parked (3b enters the scheduler here)
const cpu = boot_index;
gdt.loadOnThisCpu(cpu); // this core's GDT (with its own TSS slot)
tss.setupThisCpu(cpu); // this core's TSS + IST stack, loaded into TR
idt.loadOnThisCpu(); // the shared IDT
io.wrmsr(ia32_gs_base, percpu); // per-CPU pointer — *after* the GDT reload
apic.initSecondary(); // software-enable this core's LAPIC
apic.initTimer(apic.frequencyHz()); // arm its timer (still masked: interrupts off)
@atomicStore(u32, &ap_alive, 1, .release); // "arch state up" — BSP is polling this
if (secondary_entry) |enterScheduler| enterScheduler(); // joins the run loop
while (true) asm volatile ("hlt"); // (only if no entry was registered)
}
+12 -2
View File
@@ -59,10 +59,20 @@ prot_entry:
movw %ax, %fs
movw %ax, %gs
movl %cr4, %eax # PAE on (CR4.PAE) — required for long mode
orl $(1 << 5), %eax
# CR4: PAE (required for long mode) + OSFXSR/OSXMMEXCPT. The kernel is built with
# SSE (part of the x86_64 baseline), and the compiler emits SSE for things as
# ordinary as a struct copy — without OSFXSR those instructions #UD. The BSP got
# these bits from UEFI; an AP starts fresh, so we must set them ourselves.
movl %cr4, %eax
orl $((1 << 5) | (1 << 9) | (1 << 10)), %eax
movl %eax, %cr4
# CR0: clear EM (no x87 emulation) and set MP, so SSE/x87 don't fault.
movl %cr0, %eax
andl $~(1 << 2), %eax # ~EM
orl $(1 << 1), %eax # MP
movl %eax, %cr0
movl (param_cr3 - ap_trampoline_start)(%ebx), %eax # kernel page tables
movl %eax, %cr3
+24 -9
View File
@@ -4,6 +4,10 @@
//! interrupted stack was. We use IST1 for the double-fault handler, so a fault
//! that happens *because* the current stack is unusable still lands on solid
//! ground instead of triple-faulting.
//!
//! Each core needs **its own TSS** (its own IST stack): two cores taking a fault at
//! once can't share one fault stack. So the TSS and its IST stack are per-core,
//! indexed by CPU number; slot 0 is the BSP.
const gdt = @import("gdt.zig");
@@ -30,19 +34,30 @@ const Tss = packed struct {
/// The IST slot (1-based, as the IDT gate encodes it) used for critical faults.
pub const double_fault_ist = 1;
var tss: Tss align(16) = .{};
const max_cpus = 64; // matches gdt.zig / the scheduler
const ist_stack_size = 16 * 1024;
/// Dedicated stack for IST1. Static so it needs no allocator and is always valid.
var ist1_stack: [16 * 1024]u8 align(16) = undefined;
/// One TSS per core, and one IST1 stack per core. Static, so they need no allocator
/// and are always valid. (max_cpus × 16 KiB of BSS for the IST stacks.)
var tss_table = [_]Tss{.{}} ** max_cpus;
var ist_stacks: [max_cpus][ist_stack_size]u8 align(16) = undefined;
/// Loads the task register with the TSS selector. Defined in isr.s.
extern fn load_tr(selector: u16) callconv(.c) void;
/// Point IST1 at its stack, publish the TSS through the GDT, and load it into the
/// task register. Requires the GDT to already be loaded (gdt.init first).
pub fn init() void {
tss.ist1 = @intFromPtr(&ist1_stack) + ist1_stack.len; // stacks grow down
tss.iomap_base = @sizeOf(Tss); // == limit: no I/O permission bitmap
gdt.setTss(@intFromPtr(&tss), @sizeOf(Tss) - 1);
/// Set up core `cpu`'s TSS: point IST1 at that core's stack, install the TSS
/// descriptor into that core's GDT, and load it into the task register. Requires the
/// core's GDT to already be loaded (gdt.loadOnThisCpu first).
pub fn setupThisCpu(cpu: usize) void {
const t = &tss_table[cpu];
t.* = .{};
t.ist1 = @intFromPtr(&ist_stacks[cpu]) + ist_stack_size; // stacks grow down
t.iomap_base = @sizeOf(Tss); // == limit: no I/O permission bitmap
gdt.setTssFor(cpu, @intFromPtr(t), @sizeOf(Tss) - 1);
load_tr(gdt.tss_selector);
}
/// Set up the bootstrap processor's TSS (slot 0). Requires gdt.init first.
pub fn init() void {
setupThisCpu(0);
}
+2 -1
View File
@@ -267,6 +267,7 @@ fn bringUpSecondaries() void {
}
arch.setPageExecutable(ap_trampoline_page);
arch.prepareSecondaries(ap_trampoline_page);
arch.setSecondaryEntry(scheduler.secondaryMain); // where a woken core joins the run loop
log.print("\ndanos: bringing up {d} application processor(s)\n", .{cores.len - 1});
for (cores[1..], 1..) |core, index| {
@@ -276,7 +277,7 @@ fn bringUpSecondaries() void {
};
const stack_top = (@intFromPtr(stack.ptr) + stack.len) & ~@as(usize, 15);
const pc = scheduler.prepareSecondary(index, core.apic_id);
if (arch.startSecondary(core.apic_id, stack_top, @intFromPtr(pc))) {
if (arch.startSecondary(core.apic_id, stack_top, @intFromPtr(pc), index)) {
pc.online = true;
log.print(" cpu apic_id {d}: online\n", .{core.apic_id});
} else {
+28
View File
@@ -112,6 +112,27 @@ pub fn prepareSecondary(index: usize, apic_id: u32) *PerCpu {
return pc;
}
/// Entry for an application processor once the arch layer has set up its per-CPU
/// tables, LAPIC, and timer. It turns this bring-up context into the core's idle task
/// (as task 0 is for the BSP), marks the core online, and enters the run loop: with
/// interrupts enabled the timer preempts this idle context into whatever the global
/// ready queue offers, so the core runs real work in parallel with the others. The
/// `.c` calling convention lets the arch trampoline path jump here. Never returns.
pub fn secondaryMain() callconv(.c) noreturn {
const flags = sync.enter();
const pc = thisCpu();
const t = freeSlot() orelse @panic("sched: task table full (AP idle task)");
t.* = .{ .id = next_id, .state = .running, .priority = 0 };
next_id += 1;
pc.current = t;
pc.idle = t;
pc.online = true;
sync.leave(flags);
arch.enableInterrupts(); // the timer now preempts this idle context into work
while (true) asm volatile ("hlt"); // idle when this core has nothing ready
}
/// Number of cores that have finished bring-up (the BSP plus every online AP).
pub fn onlineCount() usize {
var n: usize = 0;
@@ -333,6 +354,13 @@ pub fn currentId() u32 {
return cur().id;
}
/// The dense index of the core this task is currently running on (0 = BSP). Reads
/// per-CPU state, so a task calling it on different cores sees different values —
/// which is how a test can prove work is running in parallel.
pub fn currentCpuIndex() u32 {
return thisCpu().index;
}
/// Change the running task's priority (takes effect next time it's enqueued).
pub fn setPriority(p: Priority) void {
cur().priority = p;
+45
View File
@@ -68,6 +68,8 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void {
eventTest();
} else if (eql(case, "ipc")) {
ipcTest();
} else if (eql(case, "smp")) {
smpTest();
} else if (eql(case, "fault-ud")) {
faultInvalidOpcode();
} else if (eql(case, "fault-pf")) {
@@ -437,6 +439,49 @@ fn sleepTest() void {
result();
}
// --- SMP parallelism ------------------------------------------------------
var seen_core = [_]bool{false} ** 8;
var smp_running: bool = true;
/// A worker that, while running, records which core it's executing on. Spread across
/// spawned workers and idle APs, these should land on more than one core.
fn smpWorker() void {
const p: *volatile bool = &smp_running;
while (p.*) {
const c = sched.currentCpuIndex();
if (c < seen_core.len) seen_core[c] = true;
}
sched.exit();
}
/// Prove tasks run **in parallel** on multiple cores (not just interleaved on one).
/// Spawn several CPU-bound workers; each stamps the core it runs on into `seen_core`.
/// With the application processors online, more than one core should show up — which
/// can only happen if work is genuinely running at the same time on different cores.
/// (Run with QEMU `-smp N`; on a single core this would see just one and fail.)
fn smpTest() void {
log("DANOS-TEST-BEGIN: smp\n", .{});
seen_core = .{false} ** 8;
smp_running = true;
var i: usize = 0;
while (i < 4) : (i += 1) sched.spawn(smpWorker, 4);
// Let the workers run across cores for a stretch of real time.
var spins: u64 = 0;
while (spins < 2_000_000_000) spins +%= 1;
smp_running = false;
var cores_seen: u32 = 0;
for (seen_core) |s| {
if (s) cores_seen += 1;
}
log("DANOS-SMP: workers ran on {d} distinct core(s)\n", .{cores_seen});
check("tasks ran on multiple cores in parallel", cores_seen >= 2);
result();
}
fn faultInvalidOpcode() void {
log("DANOS-TEST-BEGIN: fault-ud\n", .{});
asm volatile ("ud2");