schedule tasks across all cores
Per-core GDT/TSS and AP scheduler entry; fix AP SSE + single_threaded.
This commit is contained in:
@@ -114,11 +114,18 @@ pub fn prepareSecondaries(tramp_phys: u64) void {
|
||||
smp.prepare(tramp_phys);
|
||||
}
|
||||
|
||||
/// Wake the core with Local APIC id `apic_id`, giving it `stack_top` and its per-CPU
|
||||
/// pointer `percpu`; it adopts the current (kernel) page tables. Returns false if it
|
||||
/// doesn't come online within the timeout. Blocks until the core reports in.
|
||||
pub fn startSecondary(apic_id: u32, stack_top: usize, percpu: usize) bool {
|
||||
return smp.startAp(apic_id, stack_top, percpu, readCr3());
|
||||
/// Wake the core with Local APIC id `apic_id` as dense CPU `index`, giving it
|
||||
/// `stack_top` and its per-CPU pointer `percpu`; it adopts the current (kernel) page
|
||||
/// tables. Returns false if it doesn't come online within the timeout. Blocks until
|
||||
/// the core reports in.
|
||||
pub fn startSecondary(apic_id: u32, stack_top: usize, percpu: usize, index: usize) bool {
|
||||
return smp.startAp(apic_id, stack_top, percpu, index, readCr3());
|
||||
}
|
||||
|
||||
/// Register the generic entry a woken AP jumps to once its arch state is up (its own
|
||||
/// descriptor tables, LAPIC, and timer). The kernel passes its scheduler entry here.
|
||||
pub fn setSecondaryEntry(entry: *const fn () callconv(.c) noreturn) void {
|
||||
smp.setSecondaryEntry(entry);
|
||||
}
|
||||
|
||||
/// Kernel tick rate: 1000 Hz (1 ms), the scheduler's time quantum.
|
||||
|
||||
@@ -3,18 +3,25 @@
|
||||
//! reference a code selector — so we install our own flat GDT with known
|
||||
//! selectors (0x08 kernel code, 0x10 kernel data) rather than trusting whatever
|
||||
//! the firmware left in place.
|
||||
//!
|
||||
//! The code/data descriptors are identical on every core, but the **TSS descriptor
|
||||
//! is per-core** (each core needs its own TSS — its own interrupt/fault stacks; see
|
||||
//! tss.zig). Two cores can't share one TSS descriptor slot, so each core gets its
|
||||
//! own copy of the table with its own TSS descriptor. Slot 0 is the BSP.
|
||||
|
||||
/// Selectors into the table below (index * 8).
|
||||
/// Selectors into the table (index * 8). Same on every core's GDT.
|
||||
pub const kernel_code = 0x08;
|
||||
pub const kernel_data = 0x10;
|
||||
pub const tss_selector = 0x18;
|
||||
|
||||
/// Flat 64-bit descriptors. Base/limit are ignored in long mode; what matters is
|
||||
/// the access byte and, for code, the long-mode (L) flag.
|
||||
const max_cpus = 64; // matches the scheduler / discovery pool
|
||||
const entries = 5; // null, code, data, TSS-low, TSS-high
|
||||
|
||||
/// The shared descriptors (slots 0-2); slots 3-4 hold this core's TSS descriptor,
|
||||
/// filled in per core by `setTssFor`.
|
||||
/// code: present, ring 0, executable, readable, L=1 -> 0x00AF9A00_0000FFFF
|
||||
/// data: present, ring 0, writable -> 0x00CF9200_0000FFFF
|
||||
/// The last two slots hold one 16-byte TSS descriptor, filled in by setTss.
|
||||
var table = [_]u64{
|
||||
const template = [entries]u64{
|
||||
0, // null descriptor (required)
|
||||
0x00AF9A000000FFFF, // kernel code (0x08)
|
||||
0x00CF92000000FFFF, // kernel data (0x10)
|
||||
@@ -22,16 +29,20 @@ var table = [_]u64{
|
||||
0, // TSS descriptor high
|
||||
};
|
||||
|
||||
/// Fill the 64-bit TSS system descriptor (two GDT slots) so the task register can
|
||||
/// point at our TSS. Type 0x89 = present, ring 0, available 64-bit TSS.
|
||||
pub fn setTss(base: u64, limit: u64) void {
|
||||
table[3] = (limit & 0xFFFF) |
|
||||
/// One GDT per core (each a copy of the template, differing only in its TSS slot).
|
||||
var gdts = [_][entries]u64{template} ** max_cpus;
|
||||
|
||||
/// Fill core `cpu`'s 64-bit TSS system descriptor (two GDT slots) so its task
|
||||
/// register can point at its own TSS. Type 0x89 = present, ring 0, available 64-bit
|
||||
/// TSS. Write it into that core's GDT before it loads the TSS selector.
|
||||
pub fn setTssFor(cpu: usize, base: u64, limit: u64) void {
|
||||
gdts[cpu][3] = (limit & 0xFFFF) |
|
||||
((base & 0xFFFF) << 16) |
|
||||
(((base >> 16) & 0xFF) << 32) |
|
||||
(@as(u64, 0x89) << 40) |
|
||||
(((limit >> 16) & 0xF) << 48) |
|
||||
(((base >> 24) & 0xFF) << 56);
|
||||
table[4] = (base >> 32) & 0xFFFFFFFF;
|
||||
gdts[cpu][4] = (base >> 32) & 0xFFFFFFFF;
|
||||
}
|
||||
|
||||
/// The operand `lgdt` wants: table byte-length minus one, then its address.
|
||||
@@ -41,23 +52,21 @@ const Descriptor = packed struct {
|
||||
};
|
||||
|
||||
/// Loads the GDT and reloads the segment registers (including CS). Defined in
|
||||
/// isr.s — it uses the selectors 0x08 (code) and 0x10 (data) that match `table`.
|
||||
/// isr.s — it uses the selectors 0x08 (code) and 0x10 (data) that match the table.
|
||||
extern fn gdt_flush(descriptor: *const Descriptor) callconv(.c) void;
|
||||
|
||||
/// Load our GDT on the current core and switch onto its segments. The table is
|
||||
/// shared across all cores (the descriptors are flat and read-only); each core just
|
||||
/// needs to point its GDTR at it. Called by the BSP in `init` and by every AP during
|
||||
/// bring-up. Note this reloads the segment registers, which zeroes the GS base — so
|
||||
/// a core must publish its per-CPU pointer (setCpuLocal) *after* calling this.
|
||||
pub fn loadOnThisCpu() void {
|
||||
/// Load core `cpu`'s GDT and switch onto its segments. Note this reloads the segment
|
||||
/// registers, which zeroes the GS base — so a core must publish its per-CPU pointer
|
||||
/// (setCpuLocal) *after* calling this.
|
||||
pub fn loadOnThisCpu(cpu: usize) void {
|
||||
const descriptor = Descriptor{
|
||||
.limit = @sizeOf(@TypeOf(table)) - 1,
|
||||
.base = @intFromPtr(&table),
|
||||
.limit = @sizeOf([entries]u64) - 1,
|
||||
.base = @intFromPtr(&gdts[cpu]),
|
||||
};
|
||||
gdt_flush(&descriptor);
|
||||
}
|
||||
|
||||
/// Install our GDT and switch onto its segments (bootstrap processor).
|
||||
/// Install the bootstrap processor's GDT (slot 0) and switch onto its segments.
|
||||
pub fn init() void {
|
||||
loadOnThisCpu();
|
||||
loadOnThisCpu(0);
|
||||
}
|
||||
|
||||
@@ -9,14 +9,14 @@
|
||||
//!
|
||||
//! Cores are brought up **one at a time**: a single trampoline page and parameter
|
||||
//! block are reused, so the BSP patches, wakes, and waits for one AP before the
|
||||
//! next. The mechanism-vs-policy split matches the rest of the kernel — the generic
|
||||
//! scheduler decides *what* runs where; this just gets a core executing 64-bit code.
|
||||
//!
|
||||
//! This is step 3a: an AP climbs to long mode, publishes its per-CPU pointer, marks
|
||||
//! itself alive, and parks. Entering the scheduler (its own TSS, LAPIC timer, and
|
||||
//! the run loop) is the next step.
|
||||
//! next. That also lets `apEntry` pick up its dense CPU index from a plain global.
|
||||
//! Once a core has its own descriptor tables, LAPIC, and timer, it calls the generic
|
||||
//! scheduler entry and joins the run loop — mechanism here, policy there.
|
||||
|
||||
const io = @import("io.zig");
|
||||
const gdt = @import("gdt.zig");
|
||||
const tss = @import("tss.zig");
|
||||
const idt = @import("idt.zig");
|
||||
const apic = @import("apic.zig");
|
||||
|
||||
/// IA32_GS_BASE — the per-CPU data pointer (see cpu.zig; kept in sync here so the AP
|
||||
@@ -32,6 +32,19 @@ var tramp_phys: u64 = 0;
|
||||
/// one-at-a-time handshake (only one AP is being started at any moment).
|
||||
var ap_alive: u32 = 0;
|
||||
|
||||
/// The dense CPU index of the AP currently being started. Set by the BSP before the
|
||||
/// wake, read by `apEntry` (safe because bring-up is strictly one core at a time).
|
||||
var boot_index: usize = 0;
|
||||
|
||||
/// The generic scheduler entry a woken core jumps to once its arch state is up. Set
|
||||
/// by the kernel via `setSecondaryEntry`; never returns.
|
||||
var secondary_entry: ?*const fn () callconv(.c) noreturn = null;
|
||||
|
||||
/// Register the generic entry an AP calls once its per-CPU tables/LAPIC/timer are up.
|
||||
pub fn setSecondaryEntry(entry: *const fn () callconv(.c) noreturn) void {
|
||||
secondary_entry = entry;
|
||||
}
|
||||
|
||||
/// Copy the trampoline blob to its low page. Call once, after the page has been
|
||||
/// allocated and made executable, before waking any AP.
|
||||
pub fn prepare(phys: u64) void {
|
||||
@@ -53,11 +66,13 @@ fn param(comptime name: []const u8) *align(1) volatile u64 {
|
||||
return @ptrFromInt(tramp_phys + (sym - start));
|
||||
}
|
||||
|
||||
/// Wake the core with Local APIC id `apic_id`, hand it `stack_top` and `percpu` (its
|
||||
/// per-CPU pointer), and wait for it to come alive. Returns false if it doesn't
|
||||
/// report in within the timeout (left parked, no harm to the running system).
|
||||
/// `cr3` is the kernel page tables the AP adopts. Precondition: `prepare` has run.
|
||||
pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, cr3: u64) bool {
|
||||
/// Wake the core with Local APIC id `apic_id` as dense CPU `index`, hand it
|
||||
/// `stack_top` and its per-CPU pointer `percpu`, and wait for it to come alive.
|
||||
/// Returns false if it doesn't report in within the timeout (left parked, no harm to
|
||||
/// the running system). `cr3` is the kernel page tables the AP adopts. Precondition:
|
||||
/// `prepare` has run.
|
||||
pub fn startAp(apic_id: u32, stack_top: usize, percpu: usize, index: usize, cr3: u64) bool {
|
||||
boot_index = index;
|
||||
param("ap_tramp_cr3").* = cr3;
|
||||
param("ap_tramp_stack").* = stack_top;
|
||||
param("ap_tramp_entry").* = @intFromPtr(&apEntry);
|
||||
@@ -89,13 +104,20 @@ fn delayMicros(us: u64) void {
|
||||
}
|
||||
|
||||
/// The 64-bit entry every AP lands on, called from the trampoline with its per-CPU
|
||||
/// pointer in RDI. Adopts the shared descriptor tables, publishes its per-CPU
|
||||
/// pointer, signals the BSP it's alive, and (for now) parks. Never returns.
|
||||
/// pointer in RDI. Brings up this core's own descriptor tables, LAPIC and timer,
|
||||
/// signals the BSP, then jumps to the generic scheduler entry. Never returns.
|
||||
fn apEntry(percpu: usize) callconv(.c) noreturn {
|
||||
// Step 3a: minimal. The core keeps the trampoline's descriptor tables, publishes
|
||||
// its per-CPU pointer, signals the BSP, and parks with interrupts off. Loading
|
||||
// this core's own kernel GDT/IDT/TSS and entering the scheduler is step 3b.
|
||||
io.wrmsr(ia32_gs_base, percpu); // publish per-CPU pointer (GS base)
|
||||
@atomicStore(u32, &ap_alive, 1, .release); // "I'm up" — BSP is polling this
|
||||
while (true) asm volatile ("hlt"); // parked (3b enters the scheduler here)
|
||||
const cpu = boot_index;
|
||||
gdt.loadOnThisCpu(cpu); // this core's GDT (with its own TSS slot)
|
||||
tss.setupThisCpu(cpu); // this core's TSS + IST stack, loaded into TR
|
||||
idt.loadOnThisCpu(); // the shared IDT
|
||||
io.wrmsr(ia32_gs_base, percpu); // per-CPU pointer — *after* the GDT reload
|
||||
|
||||
apic.initSecondary(); // software-enable this core's LAPIC
|
||||
apic.initTimer(apic.frequencyHz()); // arm its timer (still masked: interrupts off)
|
||||
|
||||
@atomicStore(u32, &ap_alive, 1, .release); // "arch state up" — BSP is polling this
|
||||
|
||||
if (secondary_entry) |enterScheduler| enterScheduler(); // joins the run loop
|
||||
while (true) asm volatile ("hlt"); // (only if no entry was registered)
|
||||
}
|
||||
|
||||
@@ -59,10 +59,20 @@ prot_entry:
|
||||
movw %ax, %fs
|
||||
movw %ax, %gs
|
||||
|
||||
movl %cr4, %eax # PAE on (CR4.PAE) — required for long mode
|
||||
orl $(1 << 5), %eax
|
||||
# CR4: PAE (required for long mode) + OSFXSR/OSXMMEXCPT. The kernel is built with
|
||||
# SSE (part of the x86_64 baseline), and the compiler emits SSE for things as
|
||||
# ordinary as a struct copy — without OSFXSR those instructions #UD. The BSP got
|
||||
# these bits from UEFI; an AP starts fresh, so we must set them ourselves.
|
||||
movl %cr4, %eax
|
||||
orl $((1 << 5) | (1 << 9) | (1 << 10)), %eax
|
||||
movl %eax, %cr4
|
||||
|
||||
# CR0: clear EM (no x87 emulation) and set MP, so SSE/x87 don't fault.
|
||||
movl %cr0, %eax
|
||||
andl $~(1 << 2), %eax # ~EM
|
||||
orl $(1 << 1), %eax # MP
|
||||
movl %eax, %cr0
|
||||
|
||||
movl (param_cr3 - ap_trampoline_start)(%ebx), %eax # kernel page tables
|
||||
movl %eax, %cr3
|
||||
|
||||
|
||||
@@ -4,6 +4,10 @@
|
||||
//! interrupted stack was. We use IST1 for the double-fault handler, so a fault
|
||||
//! that happens *because* the current stack is unusable still lands on solid
|
||||
//! ground instead of triple-faulting.
|
||||
//!
|
||||
//! Each core needs **its own TSS** (its own IST stack): two cores taking a fault at
|
||||
//! once can't share one fault stack. So the TSS and its IST stack are per-core,
|
||||
//! indexed by CPU number; slot 0 is the BSP.
|
||||
|
||||
const gdt = @import("gdt.zig");
|
||||
|
||||
@@ -30,19 +34,30 @@ const Tss = packed struct {
|
||||
/// The IST slot (1-based, as the IDT gate encodes it) used for critical faults.
|
||||
pub const double_fault_ist = 1;
|
||||
|
||||
var tss: Tss align(16) = .{};
|
||||
const max_cpus = 64; // matches gdt.zig / the scheduler
|
||||
const ist_stack_size = 16 * 1024;
|
||||
|
||||
/// Dedicated stack for IST1. Static so it needs no allocator and is always valid.
|
||||
var ist1_stack: [16 * 1024]u8 align(16) = undefined;
|
||||
/// One TSS per core, and one IST1 stack per core. Static, so they need no allocator
|
||||
/// and are always valid. (max_cpus × 16 KiB of BSS for the IST stacks.)
|
||||
var tss_table = [_]Tss{.{}} ** max_cpus;
|
||||
var ist_stacks: [max_cpus][ist_stack_size]u8 align(16) = undefined;
|
||||
|
||||
/// Loads the task register with the TSS selector. Defined in isr.s.
|
||||
extern fn load_tr(selector: u16) callconv(.c) void;
|
||||
|
||||
/// Point IST1 at its stack, publish the TSS through the GDT, and load it into the
|
||||
/// task register. Requires the GDT to already be loaded (gdt.init first).
|
||||
pub fn init() void {
|
||||
tss.ist1 = @intFromPtr(&ist1_stack) + ist1_stack.len; // stacks grow down
|
||||
tss.iomap_base = @sizeOf(Tss); // == limit: no I/O permission bitmap
|
||||
gdt.setTss(@intFromPtr(&tss), @sizeOf(Tss) - 1);
|
||||
/// Set up core `cpu`'s TSS: point IST1 at that core's stack, install the TSS
|
||||
/// descriptor into that core's GDT, and load it into the task register. Requires the
|
||||
/// core's GDT to already be loaded (gdt.loadOnThisCpu first).
|
||||
pub fn setupThisCpu(cpu: usize) void {
|
||||
const t = &tss_table[cpu];
|
||||
t.* = .{};
|
||||
t.ist1 = @intFromPtr(&ist_stacks[cpu]) + ist_stack_size; // stacks grow down
|
||||
t.iomap_base = @sizeOf(Tss); // == limit: no I/O permission bitmap
|
||||
gdt.setTssFor(cpu, @intFromPtr(t), @sizeOf(Tss) - 1);
|
||||
load_tr(gdt.tss_selector);
|
||||
}
|
||||
|
||||
/// Set up the bootstrap processor's TSS (slot 0). Requires gdt.init first.
|
||||
pub fn init() void {
|
||||
setupThisCpu(0);
|
||||
}
|
||||
|
||||
+2
-1
@@ -267,6 +267,7 @@ fn bringUpSecondaries() void {
|
||||
}
|
||||
arch.setPageExecutable(ap_trampoline_page);
|
||||
arch.prepareSecondaries(ap_trampoline_page);
|
||||
arch.setSecondaryEntry(scheduler.secondaryMain); // where a woken core joins the run loop
|
||||
|
||||
log.print("\ndanos: bringing up {d} application processor(s)\n", .{cores.len - 1});
|
||||
for (cores[1..], 1..) |core, index| {
|
||||
@@ -276,7 +277,7 @@ fn bringUpSecondaries() void {
|
||||
};
|
||||
const stack_top = (@intFromPtr(stack.ptr) + stack.len) & ~@as(usize, 15);
|
||||
const pc = scheduler.prepareSecondary(index, core.apic_id);
|
||||
if (arch.startSecondary(core.apic_id, stack_top, @intFromPtr(pc))) {
|
||||
if (arch.startSecondary(core.apic_id, stack_top, @intFromPtr(pc), index)) {
|
||||
pc.online = true;
|
||||
log.print(" cpu apic_id {d}: online\n", .{core.apic_id});
|
||||
} else {
|
||||
|
||||
@@ -112,6 +112,27 @@ pub fn prepareSecondary(index: usize, apic_id: u32) *PerCpu {
|
||||
return pc;
|
||||
}
|
||||
|
||||
/// Entry for an application processor once the arch layer has set up its per-CPU
|
||||
/// tables, LAPIC, and timer. It turns this bring-up context into the core's idle task
|
||||
/// (as task 0 is for the BSP), marks the core online, and enters the run loop: with
|
||||
/// interrupts enabled the timer preempts this idle context into whatever the global
|
||||
/// ready queue offers, so the core runs real work in parallel with the others. The
|
||||
/// `.c` calling convention lets the arch trampoline path jump here. Never returns.
|
||||
pub fn secondaryMain() callconv(.c) noreturn {
|
||||
const flags = sync.enter();
|
||||
const pc = thisCpu();
|
||||
const t = freeSlot() orelse @panic("sched: task table full (AP idle task)");
|
||||
t.* = .{ .id = next_id, .state = .running, .priority = 0 };
|
||||
next_id += 1;
|
||||
pc.current = t;
|
||||
pc.idle = t;
|
||||
pc.online = true;
|
||||
sync.leave(flags);
|
||||
|
||||
arch.enableInterrupts(); // the timer now preempts this idle context into work
|
||||
while (true) asm volatile ("hlt"); // idle when this core has nothing ready
|
||||
}
|
||||
|
||||
/// Number of cores that have finished bring-up (the BSP plus every online AP).
|
||||
pub fn onlineCount() usize {
|
||||
var n: usize = 0;
|
||||
@@ -333,6 +354,13 @@ pub fn currentId() u32 {
|
||||
return cur().id;
|
||||
}
|
||||
|
||||
/// The dense index of the core this task is currently running on (0 = BSP). Reads
|
||||
/// per-CPU state, so a task calling it on different cores sees different values —
|
||||
/// which is how a test can prove work is running in parallel.
|
||||
pub fn currentCpuIndex() u32 {
|
||||
return thisCpu().index;
|
||||
}
|
||||
|
||||
/// Change the running task's priority (takes effect next time it's enqueued).
|
||||
pub fn setPriority(p: Priority) void {
|
||||
cur().priority = p;
|
||||
|
||||
@@ -68,6 +68,8 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void {
|
||||
eventTest();
|
||||
} else if (eql(case, "ipc")) {
|
||||
ipcTest();
|
||||
} else if (eql(case, "smp")) {
|
||||
smpTest();
|
||||
} else if (eql(case, "fault-ud")) {
|
||||
faultInvalidOpcode();
|
||||
} else if (eql(case, "fault-pf")) {
|
||||
@@ -437,6 +439,49 @@ fn sleepTest() void {
|
||||
result();
|
||||
}
|
||||
|
||||
// --- SMP parallelism ------------------------------------------------------
|
||||
|
||||
var seen_core = [_]bool{false} ** 8;
|
||||
var smp_running: bool = true;
|
||||
|
||||
/// A worker that, while running, records which core it's executing on. Spread across
|
||||
/// spawned workers and idle APs, these should land on more than one core.
|
||||
fn smpWorker() void {
|
||||
const p: *volatile bool = &smp_running;
|
||||
while (p.*) {
|
||||
const c = sched.currentCpuIndex();
|
||||
if (c < seen_core.len) seen_core[c] = true;
|
||||
}
|
||||
sched.exit();
|
||||
}
|
||||
|
||||
/// Prove tasks run **in parallel** on multiple cores (not just interleaved on one).
|
||||
/// Spawn several CPU-bound workers; each stamps the core it runs on into `seen_core`.
|
||||
/// With the application processors online, more than one core should show up — which
|
||||
/// can only happen if work is genuinely running at the same time on different cores.
|
||||
/// (Run with QEMU `-smp N`; on a single core this would see just one and fail.)
|
||||
fn smpTest() void {
|
||||
log("DANOS-TEST-BEGIN: smp\n", .{});
|
||||
seen_core = .{false} ** 8;
|
||||
smp_running = true;
|
||||
|
||||
var i: usize = 0;
|
||||
while (i < 4) : (i += 1) sched.spawn(smpWorker, 4);
|
||||
|
||||
// Let the workers run across cores for a stretch of real time.
|
||||
var spins: u64 = 0;
|
||||
while (spins < 2_000_000_000) spins +%= 1;
|
||||
smp_running = false;
|
||||
|
||||
var cores_seen: u32 = 0;
|
||||
for (seen_core) |s| {
|
||||
if (s) cores_seen += 1;
|
||||
}
|
||||
log("DANOS-SMP: workers ran on {d} distinct core(s)\n", .{cores_seen});
|
||||
check("tasks ran on multiple cores in parallel", cores_seen >= 2);
|
||||
result();
|
||||
}
|
||||
|
||||
fn faultInvalidOpcode() void {
|
||||
log("DANOS-TEST-BEGIN: fault-ud\n", .{});
|
||||
asm volatile ("ud2");
|
||||
|
||||
Reference in New Issue
Block a user