Files
danos/src/kernel/scheduler.zig
T
Daniel SamsonandClaude Fable 5 8a7235c725 M3 step 2: swapgs discipline + arch per-CPU block
The GS base now points at an arch-owned ArchPerCpu (percpu.zig) holding
the kernel RSP and a scratch slot (for the coming syscall stub, at fixed
%gs offsets) plus the scheduler pointer. isr_common conditionally
swapgs's on entry/exit when the interrupted frame was ring 3, and
enter_user swapgs's before dropping to ring 3 — so kernel code always
sees the kernel GS base and a ring-3 `mov %ax,%gs` can no longer poison
cpuLocal(). No swapgs on ring-0 interrupts (the common case). Suite
27/27, including int 0x80 from ring 3 and faults.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-08 23:09:19 +01:00

449 lines
18 KiB
Zig

//! The scheduler: fixed-priority preemptive multitasking.
//!
//! Tasks are kernel threads (ring 0, each with its own stack). The **highest-
//! priority ready task always runs**; within a priority level, tasks round-robin.
//! Selection is O(1) — a bitmap of non-empty priority levels plus a FIFO queue per
//! level — which keeps scheduling deterministic, as a real-time kernel needs (see
//! docs/vision.md).
//!
//! Switching happens both cooperatively (`yield`) and preemptively (the timer
//! calls `tick`). See docs/scheduling.md for the interrupt-flag discipline that
//! makes those two paths coexist.
//!
//! Cross-core safety is the **big kernel lock** (`sync.zig`): every critical
//! section here runs under it, and it is held across a context switch and released
//! by the task that resumes (see sync.zig's hand-off rule). On a single core the
//! lock is never contended, so the behaviour is exactly the old interrupt-flag
//! model; it's what lets a second core enter `schedule()` without corrupting the
//! shared queues.
const std = @import("std");
const config = @import("config");
const arch = @import("arch");
const heap = @import("heap.zig");
const sync = @import("sync.zig");
/// Priority level: 0 (lowest) .. 7 (highest). 8 levels total.
pub const Priority = u3;
const num_priorities = 8;
const stack_size = config.kernel_stack_size; // each task's kernel stack
const max_tasks = config.max_tasks; // maximum tasks alive at once (static pool)
const State = enum { free, ready, running, blocked };
const Task = struct {
id: u32 = 0,
state: State = .free,
priority: Priority = 0,
rsp: usize = 0, // saved stack pointer, valid while not running
stack: []u8 = &.{},
kstack_top: usize = 0, // top of `stack` (== TSS.rsp0 for a user task); 0 = none
wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping
affinity: ?u32 = null, // null = runs on any core; else the index of its pinned core
// Physical PML4 of this task's address space, or 0 for a kernel task (which
// runs on the shared kernel page tables). A user task carries its own.
pml4: u64 = 0,
next: ?*Task = null, // ready-queue link
};
var tasks = [_]Task{.{}} ** max_tasks;
var next_id: u32 = 1;
/// Per-CPU scheduler state: the task each core is running, its own idle task, and a
/// queue of tasks **pinned** to it. One entry per core; the arch layer stashes a
/// pointer to the *running* core's entry in the GS base, so `thisCpu()` fetches it
/// with a single read and no lock.
///
/// Most work stays in the **global** ready queue (below), which any idle core pulls
/// from — work-conserving. A task given an *affinity* instead goes to that core's
/// `pinned_*` queue and is only ever run there (no surprise migration — the more
/// real-time-predictable model, docs/smp.md). The two queues are merged at selection
/// time. Both are still mutated only under the big kernel lock, so one core enqueuing
/// into another core's pinned queue is safe.
pub const PerCpu = struct {
current: *Task = undefined, // the task running on this core
idle: *Task = undefined, // this core's idle task (always ready, lowest priority)
apic_id: u32 = 0, // the core's Local APIC id
index: u32 = 0, // dense 0-based core index
online: bool = false, // has this core finished bring-up?
loaded_pml4: u64 = 0, // the address space (CR3) currently loaded on this core
// Tasks pinned to this core (affinity == index), per priority level + bitmap.
pinned_head: [num_priorities]?*Task = .{null} ** num_priorities,
pinned_tail: [num_priorities]?*Task = .{null} ** num_priorities,
pinned_bitmap: u8 = 0,
};
const max_cpus = config.max_cpus;
var cpus = [_]PerCpu{.{}} ** max_cpus;
/// This core's per-CPU state, via the arch layer's GS-base pointer. Valid only once
/// this core has run its scheduler bring-up (BSP in `init`, AP in `secondaryInit`).
inline fn thisCpu() *PerCpu {
return @ptrFromInt(arch.cpuLocal());
}
/// The task running on this core — the per-CPU replacement for the old global
/// `current`. A convenience reader; writes go through `thisCpu().current`.
inline fn cur() *Task {
return thisCpu().current;
}
// Per-priority FIFO ready queues, and a bitmap of which levels are non-empty. These
// are shared across all cores and mutated only under the big kernel lock.
var ready_head: [num_priorities]?*Task = .{null} ** num_priorities;
var ready_tail: [num_priorities]?*Task = .{null} ** num_priorities;
var ready_bitmap: u8 = 0;
var preemption_enabled = true;
/// Bring up scheduling on the bootstrap processor: register the currently-running
/// kernel context as task 0, publish this core's per-CPU state (via the GS base),
/// give the core an idle task, and hook the timer for preemption. Runs once, at
/// boot, before interrupts are enabled — so no lock is needed here.
pub fn init(boot_priority: Priority) void {
const pc = &cpus[0];
pc.* = .{ .index = 0, .online = true, .loaded_pml4 = arch.kernelPageTable() };
arch.setCpuLocal(0, @intFromPtr(pc));
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
pc.current = &tasks[0];
pc.idle = create(idle, 0, null); // this core's idle task: always ready, lowest priority
arch.setTickHook(tick);
}
/// The idle task: run when every other task is blocked or sleeping. `hlt` waits
/// for the next interrupt at near-zero power (see docs/halting.md).
fn idle() void {
while (true) asm volatile ("hlt");
}
/// Reserve and initialise the per-CPU slot for an application processor at dense
/// `index` (1-based; 0 is the BSP) with Local APIC id `apic_id`, and return a
/// pointer the arch bring-up hands to the core (it publishes it in its GS base).
/// Called on the BSP before waking each AP; the AP marks itself `online`.
pub fn prepareSecondary(index: usize, apic_id: u32) *PerCpu {
const pc = &cpus[index];
pc.* = .{ .index = @intCast(index), .apic_id = apic_id, .online = false };
return pc;
}
/// Entry for an application processor once the arch layer has set up its per-CPU
/// tables, LAPIC, and timer. It turns this bring-up context into the core's idle task
/// (as task 0 is for the BSP), marks the core online, and enters the run loop: with
/// interrupts enabled the timer preempts this idle context into whatever the global
/// ready queue offers, so the core runs real work in parallel with the others. The
/// `.c` calling convention lets the arch trampoline path jump here. Never returns.
pub fn secondaryMain() callconv(.c) noreturn {
const flags = sync.enter();
const pc = thisCpu();
const t = freeSlot() orelse @panic("sched: task table full (AP idle task)");
t.* = .{ .id = next_id, .state = .running, .priority = 0 };
next_id += 1;
pc.current = t;
pc.idle = t;
pc.online = true;
pc.loaded_pml4 = arch.kernelPageTable(); // the AP adopted the kernel tables at bring-up
sync.leave(flags);
arch.enableInterrupts(); // the timer now preempts this idle context into work
while (true) asm volatile ("hlt"); // idle when this core has nothing ready
}
/// Number of cores that have finished bring-up (the BSP plus every online AP).
pub fn onlineCount() usize {
var n: usize = 0;
for (&cpus) |*pc| {
if (pc.online) n += 1;
}
return n;
}
/// Make `t` ready. A pinned task (affinity set) goes to that core's pinned queue;
/// everything else goes to the shared global queue.
fn enqueue(t: *Task) void {
if (t.affinity) |cpu| {
const pc = &cpus[cpu];
enqueueTo(&pc.pinned_head, &pc.pinned_tail, &pc.pinned_bitmap, t);
} else {
enqueueTo(&ready_head, &ready_tail, &ready_bitmap, t);
}
}
fn enqueueTo(head: *[num_priorities]?*Task, tail: *[num_priorities]?*Task, bitmap: *u8, t: *Task) void {
t.next = null;
const p: usize = t.priority;
if (tail[p]) |tl| tl.next = t else head[p] = t;
tail[p] = t;
bitmap.* |= @as(u8, 1) << t.priority;
}
/// The highest non-empty priority level in a bitmap, or -1 if empty.
fn topLevel(bitmap: u8) i32 {
if (bitmap == 0) return -1;
return @as(i32, num_priorities - 1) - @as(i32, @clz(bitmap));
}
/// Pick the highest-priority ready task for core `pc`: the better of the global queue
/// and this core's pinned queue. Still O(1) (two `clz` and a compare). A pinned task
/// wins an equal-priority tie, so it can't be starved by global work at its level.
fn dequeueHighest(pc: *PerCpu) ?*Task {
const g = topLevel(ready_bitmap);
const p = topLevel(pc.pinned_bitmap);
if (g < 0 and p < 0) return null;
if (p >= g) return dequeueFrom(&pc.pinned_head, &pc.pinned_tail, &pc.pinned_bitmap, @intCast(p));
return dequeueFrom(&ready_head, &ready_tail, &ready_bitmap, @intCast(g));
}
fn dequeueFrom(head: *[num_priorities]?*Task, tail: *[num_priorities]?*Task, bitmap: *u8, level: usize) ?*Task {
const t = head[level].?;
head[level] = t.next;
if (head[level] == null) {
tail[level] = null;
bitmap.* &= ~(@as(u8, 1) << @intCast(level));
}
t.next = null;
return t;
}
/// Create a task that runs `entry` at `priority`, runnable on any core. It becomes
/// ready immediately. Takes the kernel lock: it mutates the shared task table and
/// ready queues and allocates from the (non-thread-safe) heap, so on SMP it must be
/// serialised.
pub fn spawn(entry: *const fn () void, priority: Priority) void {
const flags = sync.enter();
_ = create(entry, priority, null);
sync.leave(flags);
}
/// Like `spawn`, but **pins** the task to core `cpu` — it will only ever run there.
/// Returns true if pinned; false if `cpu` isn't a valid, online core, in which case
/// the task is still created but left unpinned (so it runs *somewhere* rather than
/// stranding in a queue no core services). Callers that require the pin (e.g. tests)
/// should check the result.
pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
const flags = sync.enter();
defer sync.leave(flags);
const ok = cpu < max_cpus and cpus[cpu].online;
_ = create(entry, priority, if (ok) cpu else null);
return ok;
}
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the
/// single-threaded boot path). `affinity` pins the task to a core (null = any).
/// Returns the new task so a core can keep a handle to its idle task.
fn create(entry: *const fn () void, priority: Priority, affinity: ?u32) *Task {
const t = freeSlot() orelse @panic("sched: task table full");
const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack");
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack, .affinity = affinity };
next_id += 1;
const top = @intFromPtr(stack.ptr) + stack.len;
t.kstack_top = top;
t.rsp = arch.initTaskStack(top, @intFromPtr(entry));
enqueue(t);
return t;
}
fn freeSlot() ?*Task {
for (&tasks) |*t| {
if (t.state == .free) return t;
}
return null;
}
/// Pick the highest-priority ready task and switch this core to it. The big kernel
/// lock must be held by the caller (which also keeps local interrupts disabled);
/// it serialises every core's scheduling, so no other core can touch the shared
/// queues while we requeue `prev` and dequeue `next`. A dequeued task is `.ready`,
/// never running elsewhere, so two cores never run the same task.
fn schedule() void {
const pc = thisCpu();
const prev = pc.current;
if (prev.state == .running) {
prev.state = .ready;
enqueue(prev); // back of its level's queue (round-robin)
}
const next = dequeueHighest(pc) orelse {
prev.state = .running; // nothing else ready — keep running
return;
};
next.state = .running;
pc.current = next;
if (next != prev) switchTo(pc, &prev.rsp, next);
}
/// Make `next` this core's running task: publish its kernel stack (TSS.rsp0, so a
/// ring-3 interrupt lands on a good stack) and its address space (CR3, only when
/// it differs from what's loaded — every CR3 write is a full TLB flush), then
/// switch registers/stacks. Kernel tasks (pml4 == 0, no kstack_top used from
/// ring 3) resolve to the shared kernel page tables and skip the rsp0 write, so
/// this is a no-op beyond the register switch for a pure-kernel workload. The
/// big kernel lock is held and interrupts are off throughout, so no interrupt
/// can observe a half-updated (rsp0, CR3) pair. `save_rsp` receives the outgoing
/// task's stack pointer.
fn switchTo(pc: *PerCpu, save_rsp: *usize, next: *Task) void {
if (next.kstack_top != 0) arch.setKernelStack(pc.index, next.kstack_top);
const want = if (next.pml4 != 0) next.pml4 else arch.kernelPageTable();
if (want != pc.loaded_pml4) {
arch.loadPageTable(want);
pc.loaded_pml4 = want;
}
arch.switchContext(save_rsp, next.rsp);
}
/// Voluntarily give up the CPU to the next ready task.
pub fn yield() void {
const flags = sync.enter();
schedule();
sync.leave(flags);
}
/// Block the current task for `ms` milliseconds, then let it become runnable
/// again. The idle task (or other work) runs in the meantime.
pub fn sleep(ms: u64) void {
const flags = sync.enter();
const t = cur();
t.wake_at = arch.millis() + ms;
t.state = .blocked;
schedule(); // current is blocked, so schedule() won't re-enqueue it
sync.leave(flags);
}
// --- event-based blocking -------------------------------------------------
//
// A WaitQueue is a set of tasks blocked waiting for something (a resource, a
// message). Tasks link into it through the same `next` field the ready queues
// use — a task is in exactly one queue at a time. These are the primitive locks,
// semaphores and IPC channels are built on.
pub const WaitQueue = struct {
head: ?*Task = null,
};
/// Block the current task on `wq` and switch away. Precondition: the big kernel
/// lock is held (so a condition can be checked and the block committed atomically;
/// it also keeps local interrupts disabled). On return — when woken — the lock is
/// still held.
pub fn waitLocked(wq: *WaitQueue) void {
const t = cur();
t.state = .blocked;
t.next = wq.head;
wq.head = t;
schedule();
}
/// Move the highest-priority waiter on `wq` (if any) to the ready queue.
/// Precondition: the big kernel lock is held. Does not preempt — the caller decides.
pub fn wakeLocked(wq: *WaitQueue) void {
// Find the highest-priority waiter (bounded scan) and unlink it.
var best_prev: ?*Task = null;
var best: ?*Task = null;
var prev: ?*Task = null;
var node = wq.head;
while (node) |t| : ({
prev = t;
node = t.next;
}) {
if (best == null or t.priority > best.?.priority) {
best = t;
best_prev = prev;
}
}
const t = best orelse return;
if (best_prev) |p| p.next = t.next else wq.head = t.next;
t.state = .ready;
enqueue(t);
}
/// Block on `wq` (a self-contained critical section).
pub fn wait(wq: *WaitQueue) void {
const flags = sync.enter();
waitLocked(wq);
sync.leave(flags);
}
/// Wake the highest-priority waiter on `wq`, preempting if it outranks us.
pub fn wake(wq: *WaitQueue) void {
const flags = sync.enter();
const pc = thisCpu();
wakeLocked(wq);
// If a task this core would now pick outranks the running one, run it at once.
// (A waiter pinned to *another* core isn't counted — that core picks it up on its
// next tick; this core doesn't preempt for work it can't run.)
if (highestReadyPriority(pc)) |p| {
if (p > pc.current.priority) schedule();
}
sync.leave(flags);
}
/// The highest-priority task core `pc` could run right now — the better of the global
/// queue and this core's pinned queue — or null if it would fall back to idle.
fn highestReadyPriority(pc: *PerCpu) ?Priority {
const top = @max(topLevel(ready_bitmap), topLevel(pc.pinned_bitmap));
if (top < 0) return null;
return @intCast(top);
}
/// Wake any sleeping task whose deadline has passed. Bounded by the task count,
/// so it stays deterministic. Called from the timer tick (interrupts disabled).
fn wakeExpired() void {
const now = arch.millis();
for (&tasks) |*t| {
if (t.state == .blocked and t.wake_at != 0 and now >= t.wake_at) {
t.wake_at = 0;
t.state = .ready;
enqueue(t);
}
}
}
/// Called from the timer interrupt (interrupts already disabled): wake due
/// sleepers, then preempt. Takes the kernel lock like any other critical section,
/// but releases it *without* touching the interrupt flag — the handler's `iretq`
/// restores the interrupted context's flags, so re-enabling here would open a
/// nested-interrupt window before the return.
pub fn tick() void {
_ = sync.enter();
wakeExpired();
if (preemption_enabled) schedule();
sync.leaveIsr();
}
/// Enable or disable timer-driven preemption (cooperative-only when off).
pub fn setPreemption(enabled: bool) void {
preemption_enabled = enabled;
}
/// End the current task and switch away for good; never returns. The task's stack
/// is leaked for now (no reaper yet). Acquires the kernel lock and hands it off to
/// the task we switch into (which releases it) — this frame never returns to leave.
pub fn exit() noreturn {
_ = sync.enter();
const pc = thisCpu();
pc.current.state = .free;
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
next.state = .running;
pc.current = next;
var discard: usize = 0;
switchTo(pc, &discard, next);
unreachable;
}
pub fn currentId() u32 {
return cur().id;
}
/// The dense index of the core this task is currently running on (0 = BSP). Reads
/// per-CPU state, so a task calling it on different cores sees different values —
/// which is how a test can prove work is running in parallel. Returns 0 if the GS
/// base isn't published yet (a fault in very early boot, before `init`), so a fault
/// reporter can call it unconditionally without a second fault.
pub fn currentCpuIndex() u32 {
if (arch.cpuLocal() == 0) return 0;
return thisCpu().index;
}
/// Change the running task's priority (takes effect next time it's enqueued).
pub fn setPriority(p: Priority) void {
cur().priority = p;
}