move the running task into per-CPU state

PerCpu.current via the GS base; ready queues stay global.
This commit is contained in:
Daniel Samson 2026-07-08 11:31:53 +01:00
parent 941ab091db
commit 36c29d2d6d
No known key found for this signature in database
GPG Key ID: A9EB2589C60F7268
2 changed files with 95 additions and 23 deletions

View File

@ -81,6 +81,24 @@ pub fn readCr3() u64 {
); );
} }
/// IA32_GS_BASE: the hidden base of the GS segment. We repurpose it as the per-CPU
/// data pointer (there's no user mode yet, so no `swapgs` dance GS base is always
/// the running core's per-CPU block). Set once per core during bring-up, after the
/// GDT is loaded (loading a GS *selector* would otherwise clobber this base).
const ia32_gs_base = 0xC000_0101;
/// Publish this core's per-CPU data pointer so `cpuLocal` can retrieve it. Each
/// core calls this once, after its GDT is in place.
pub fn setCpuLocal(ptr: usize) void {
io.wrmsr(ia32_gs_base, ptr);
}
/// This core's per-CPU data pointer (the value `setCpuLocal` stored). Reads the GS
/// base MSR a per-core register, so each core sees its own without any locking.
pub fn cpuLocal() usize {
return io.rdmsr(ia32_gs_base);
}
/// Kernel tick rate: 1000 Hz (1 ms), the scheduler's time quantum. /// Kernel tick rate: 1000 Hz (1 ms), the scheduler's time quantum.
pub const timer_hz = 1000; pub const timer_hz = 1000;

View File

@ -42,22 +42,57 @@ const Task = struct {
}; };
var tasks = [_]Task{.{}} ** max_tasks; var tasks = [_]Task{.{}} ** max_tasks;
var current: *Task = undefined;
var next_id: u32 = 1; var next_id: u32 = 1;
// Per-priority FIFO ready queues, and a bitmap of which levels are non-empty. /// Per-CPU scheduler state: the task each core is running, plus its own idle task.
/// One entry per core; the arch layer stashes a pointer to the *running* core's
/// entry in the GS base, so `thisCpu()` fetches it with a single read and no lock.
/// This is the only state that's genuinely per-core the ready queues below stay
/// **global** under the big kernel lock, so any idle core pulls the highest-priority
/// ready task (work-conserving). Per-core ready queues are a later optimisation if
/// the global queue's lock contention ever bites (docs/smp.md).
pub const PerCpu = struct {
current: *Task = undefined, // the task running on this core
idle: *Task = undefined, // this core's idle task (always ready, lowest priority)
apic_id: u32 = 0, // the core's Local APIC id
index: u32 = 0, // dense 0-based core index
online: bool = false, // has this core finished bring-up?
};
const max_cpus = 64; // matches the discovery pool (src/device/acpi.zig)
var cpus = [_]PerCpu{.{}} ** max_cpus;
/// This core's per-CPU state, via the arch layer's GS-base pointer. Valid only once
/// this core has run its scheduler bring-up (BSP in `init`, AP in `secondaryInit`).
inline fn thisCpu() *PerCpu {
return @ptrFromInt(arch.cpuLocal());
}
/// The task running on this core the per-CPU replacement for the old global
/// `current`. A convenience reader; writes go through `thisCpu().current`.
inline fn cur() *Task {
return thisCpu().current;
}
// Per-priority FIFO ready queues, and a bitmap of which levels are non-empty. These
// are shared across all cores and mutated only under the big kernel lock.
var ready_head: [num_priorities]?*Task = .{null} ** num_priorities; var ready_head: [num_priorities]?*Task = .{null} ** num_priorities;
var ready_tail: [num_priorities]?*Task = .{null} ** num_priorities; var ready_tail: [num_priorities]?*Task = .{null} ** num_priorities;
var ready_bitmap: u8 = 0; var ready_bitmap: u8 = 0;
var preemption_enabled = true; var preemption_enabled = true;
/// Register the currently-running kernel context as the first task, spawn the /// Bring up scheduling on the bootstrap processor: register the currently-running
/// idle task, and hook the timer for preemption. /// kernel context as task 0, publish this core's per-CPU state (via the GS base),
/// give the core an idle task, and hook the timer for preemption. Runs once, at
/// boot, before interrupts are enabled so no lock is needed here.
pub fn init(boot_priority: Priority) void { pub fn init(boot_priority: Priority) void {
const pc = &cpus[0];
pc.* = .{ .index = 0, .online = true };
arch.setCpuLocal(@intFromPtr(pc));
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority }; tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
current = &tasks[0]; pc.current = &tasks[0];
spawn(idle, 0); // lowest priority, always runnable runs when nothing else is pc.idle = create(idle, 0); // this core's idle task: always ready, lowest priority
arch.setTickHook(tick); arch.setTickHook(tick);
} }
@ -93,7 +128,18 @@ fn levelBit(p: Priority) u8 {
} }
/// Create a task that runs `entry` at `priority`. It becomes ready immediately. /// Create a task that runs `entry` at `priority`. It becomes ready immediately.
/// Takes the kernel lock: it mutates the shared task table and ready queues and
/// allocates from the (non-thread-safe) heap, so on SMP it must be serialised.
pub fn spawn(entry: *const fn () void, priority: Priority) void { pub fn spawn(entry: *const fn () void, priority: Priority) void {
const flags = sync.enter();
_ = create(entry, priority);
sync.leave(flags);
}
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be
/// the single-threaded boot path). Returns the new task so a core can keep a handle
/// to its idle task.
fn create(entry: *const fn () void, priority: Priority) *Task {
const t = freeSlot() orelse @panic("sched: task table full"); const t = freeSlot() orelse @panic("sched: task table full");
const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack"); const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack");
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack }; t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack };
@ -101,6 +147,7 @@ pub fn spawn(entry: *const fn () void, priority: Priority) void {
const top = @intFromPtr(stack.ptr) + stack.len; const top = @intFromPtr(stack.ptr) + stack.len;
t.rsp = arch.initTaskStack(top, @intFromPtr(entry)); t.rsp = arch.initTaskStack(top, @intFromPtr(entry));
enqueue(t); enqueue(t);
return t;
} }
fn freeSlot() ?*Task { fn freeSlot() ?*Task {
@ -110,10 +157,14 @@ fn freeSlot() ?*Task {
return null; return null;
} }
/// Pick the highest-priority ready task and switch to it. Interrupts must be /// Pick the highest-priority ready task and switch this core to it. The big kernel
/// disabled by the caller. /// lock must be held by the caller (which also keeps local interrupts disabled);
/// it serialises every core's scheduling, so no other core can touch the shared
/// queues while we requeue `prev` and dequeue `next`. A dequeued task is `.ready`,
/// never running elsewhere, so two cores never run the same task.
fn schedule() void { fn schedule() void {
const prev = current; const pc = thisCpu();
const prev = pc.current;
if (prev.state == .running) { if (prev.state == .running) {
prev.state = .ready; prev.state = .ready;
enqueue(prev); // back of its level's queue (round-robin) enqueue(prev); // back of its level's queue (round-robin)
@ -123,7 +174,7 @@ fn schedule() void {
return; return;
}; };
next.state = .running; next.state = .running;
current = next; pc.current = next;
if (next != prev) arch.switchContext(&prev.rsp, next.rsp); if (next != prev) arch.switchContext(&prev.rsp, next.rsp);
} }
@ -138,8 +189,9 @@ pub fn yield() void {
/// again. The idle task (or other work) runs in the meantime. /// again. The idle task (or other work) runs in the meantime.
pub fn sleep(ms: u64) void { pub fn sleep(ms: u64) void {
const flags = sync.enter(); const flags = sync.enter();
current.wake_at = arch.millis() + ms; const t = cur();
current.state = .blocked; t.wake_at = arch.millis() + ms;
t.state = .blocked;
schedule(); // current is blocked, so schedule() won't re-enqueue it schedule(); // current is blocked, so schedule() won't re-enqueue it
sync.leave(flags); sync.leave(flags);
} }
@ -160,9 +212,10 @@ pub const WaitQueue = struct {
/// it also keeps local interrupts disabled). On return when woken the lock is /// it also keeps local interrupts disabled). On return when woken the lock is
/// still held. /// still held.
pub fn waitLocked(wq: *WaitQueue) void { pub fn waitLocked(wq: *WaitQueue) void {
current.state = .blocked; const t = cur();
current.next = wq.head; t.state = .blocked;
wq.head = current; t.next = wq.head;
wq.head = t;
schedule(); schedule();
} }
@ -173,10 +226,10 @@ pub fn wakeLocked(wq: *WaitQueue) void {
var best_prev: ?*Task = null; var best_prev: ?*Task = null;
var best: ?*Task = null; var best: ?*Task = null;
var prev: ?*Task = null; var prev: ?*Task = null;
var cur = wq.head; var node = wq.head;
while (cur) |t| : ({ while (node) |t| : ({
prev = t; prev = t;
cur = t.next; node = t.next;
}) { }) {
if (best == null or t.priority > best.?.priority) { if (best == null or t.priority > best.?.priority) {
best = t; best = t;
@ -202,7 +255,7 @@ pub fn wake(wq: *WaitQueue) void {
wakeLocked(wq); wakeLocked(wq);
// If a higher-priority task is now ready, run it immediately. // If a higher-priority task is now ready, run it immediately.
if (highestReadyPriority()) |p| { if (highestReadyPriority()) |p| {
if (p > current.priority) schedule(); if (p > cur().priority) schedule();
} }
sync.leave(flags); sync.leave(flags);
} }
@ -247,20 +300,21 @@ pub fn setPreemption(enabled: bool) void {
/// the task we switch into (which releases it) this frame never returns to leave. /// the task we switch into (which releases it) this frame never returns to leave.
pub fn exit() noreturn { pub fn exit() noreturn {
_ = sync.enter(); _ = sync.enter();
current.state = .free; const pc = thisCpu();
pc.current.state = .free;
const next = dequeueHighest() orelse @panic("sched: no task left to run"); const next = dequeueHighest() orelse @panic("sched: no task left to run");
next.state = .running; next.state = .running;
current = next; pc.current = next;
var discard: usize = 0; var discard: usize = 0;
arch.switchContext(&discard, next.rsp); arch.switchContext(&discard, next.rsp);
unreachable; unreachable;
} }
pub fn currentId() u32 { pub fn currentId() u32 {
return current.id; return cur().id;
} }
/// Change the running task's priority (takes effect next time it's enqueued). /// Change the running task's priority (takes effect next time it's enqueued).
pub fn setPriority(p: Priority) void { pub fn setPriority(p: Priority) void {
current.priority = p; cur().priority = p;
} }