368 lines
14 KiB
Zig
368 lines
14 KiB
Zig
//! The scheduler: fixed-priority preemptive multitasking.
|
|
//!
|
|
//! Tasks are kernel threads (ring 0, each with its own stack). The **highest-
|
|
//! priority ready task always runs**; within a priority level, tasks round-robin.
|
|
//! Selection is O(1) — a bitmap of non-empty priority levels plus a FIFO queue per
|
|
//! level — which keeps scheduling deterministic, as a real-time kernel needs (see
|
|
//! docs/vision.md).
|
|
//!
|
|
//! Switching happens both cooperatively (`yield`) and preemptively (the timer
|
|
//! calls `tick`). See docs/scheduling.md for the interrupt-flag discipline that
|
|
//! makes those two paths coexist.
|
|
//!
|
|
//! Cross-core safety is the **big kernel lock** (`sync.zig`): every critical
|
|
//! section here runs under it, and it is held across a context switch and released
|
|
//! by the task that resumes (see sync.zig's hand-off rule). On a single core the
|
|
//! lock is never contended, so the behaviour is exactly the old interrupt-flag
|
|
//! model; it's what lets a second core enter `schedule()` without corrupting the
|
|
//! shared queues.
|
|
|
|
const std = @import("std");
|
|
const arch = @import("arch");
|
|
const heap = @import("heap.zig");
|
|
const sync = @import("sync.zig");
|
|
|
|
/// Priority level: 0 (lowest) .. 7 (highest). 8 levels total.
|
|
pub const Priority = u3;
|
|
const num_priorities = 8;
|
|
|
|
const stack_size = 16 * 1024; // each task's kernel stack is 16 KiB
|
|
const max_tasks = 16; // the maximum number of tasks alive at once is 16 in a static sized pool
|
|
|
|
const State = enum { free, ready, running, blocked };
|
|
|
|
const Task = struct {
|
|
id: u32 = 0,
|
|
state: State = .free,
|
|
priority: Priority = 0,
|
|
rsp: usize = 0, // saved stack pointer, valid while not running
|
|
stack: []u8 = &.{},
|
|
wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping
|
|
next: ?*Task = null, // ready-queue link
|
|
};
|
|
|
|
var tasks = [_]Task{.{}} ** max_tasks;
|
|
var next_id: u32 = 1;
|
|
|
|
/// Per-CPU scheduler state: the task each core is running, plus its own idle task.
|
|
/// One entry per core; the arch layer stashes a pointer to the *running* core's
|
|
/// entry in the GS base, so `thisCpu()` fetches it with a single read and no lock.
|
|
/// This is the only state that's genuinely per-core — the ready queues below stay
|
|
/// **global** under the big kernel lock, so any idle core pulls the highest-priority
|
|
/// ready task (work-conserving). Per-core ready queues are a later optimisation if
|
|
/// the global queue's lock contention ever bites (docs/smp.md).
|
|
pub const PerCpu = struct {
|
|
current: *Task = undefined, // the task running on this core
|
|
idle: *Task = undefined, // this core's idle task (always ready, lowest priority)
|
|
apic_id: u32 = 0, // the core's Local APIC id
|
|
index: u32 = 0, // dense 0-based core index
|
|
online: bool = false, // has this core finished bring-up?
|
|
};
|
|
|
|
const max_cpus = 64; // matches the discovery pool (src/device/acpi.zig)
|
|
var cpus = [_]PerCpu{.{}} ** max_cpus;
|
|
|
|
/// This core's per-CPU state, via the arch layer's GS-base pointer. Valid only once
|
|
/// this core has run its scheduler bring-up (BSP in `init`, AP in `secondaryInit`).
|
|
inline fn thisCpu() *PerCpu {
|
|
return @ptrFromInt(arch.cpuLocal());
|
|
}
|
|
|
|
/// The task running on this core — the per-CPU replacement for the old global
|
|
/// `current`. A convenience reader; writes go through `thisCpu().current`.
|
|
inline fn cur() *Task {
|
|
return thisCpu().current;
|
|
}
|
|
|
|
// Per-priority FIFO ready queues, and a bitmap of which levels are non-empty. These
|
|
// are shared across all cores and mutated only under the big kernel lock.
|
|
var ready_head: [num_priorities]?*Task = .{null} ** num_priorities;
|
|
var ready_tail: [num_priorities]?*Task = .{null} ** num_priorities;
|
|
var ready_bitmap: u8 = 0;
|
|
|
|
var preemption_enabled = true;
|
|
|
|
/// Bring up scheduling on the bootstrap processor: register the currently-running
|
|
/// kernel context as task 0, publish this core's per-CPU state (via the GS base),
|
|
/// give the core an idle task, and hook the timer for preemption. Runs once, at
|
|
/// boot, before interrupts are enabled — so no lock is needed here.
|
|
pub fn init(boot_priority: Priority) void {
|
|
const pc = &cpus[0];
|
|
pc.* = .{ .index = 0, .online = true };
|
|
arch.setCpuLocal(@intFromPtr(pc));
|
|
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
|
|
pc.current = &tasks[0];
|
|
pc.idle = create(idle, 0); // this core's idle task: always ready, lowest priority
|
|
arch.setTickHook(tick);
|
|
}
|
|
|
|
/// The idle task: run when every other task is blocked or sleeping. `hlt` waits
|
|
/// for the next interrupt at near-zero power (see docs/halting.md).
|
|
fn idle() void {
|
|
while (true) asm volatile ("hlt");
|
|
}
|
|
|
|
/// Reserve and initialise the per-CPU slot for an application processor at dense
|
|
/// `index` (1-based; 0 is the BSP) with Local APIC id `apic_id`, and return a
|
|
/// pointer the arch bring-up hands to the core (it publishes it in its GS base).
|
|
/// Called on the BSP before waking each AP; the AP marks itself `online`.
|
|
pub fn prepareSecondary(index: usize, apic_id: u32) *PerCpu {
|
|
const pc = &cpus[index];
|
|
pc.* = .{ .index = @intCast(index), .apic_id = apic_id, .online = false };
|
|
return pc;
|
|
}
|
|
|
|
/// Entry for an application processor once the arch layer has set up its per-CPU
|
|
/// tables, LAPIC, and timer. It turns this bring-up context into the core's idle task
|
|
/// (as task 0 is for the BSP), marks the core online, and enters the run loop: with
|
|
/// interrupts enabled the timer preempts this idle context into whatever the global
|
|
/// ready queue offers, so the core runs real work in parallel with the others. The
|
|
/// `.c` calling convention lets the arch trampoline path jump here. Never returns.
|
|
pub fn secondaryMain() callconv(.c) noreturn {
|
|
const flags = sync.enter();
|
|
const pc = thisCpu();
|
|
const t = freeSlot() orelse @panic("sched: task table full (AP idle task)");
|
|
t.* = .{ .id = next_id, .state = .running, .priority = 0 };
|
|
next_id += 1;
|
|
pc.current = t;
|
|
pc.idle = t;
|
|
pc.online = true;
|
|
sync.leave(flags);
|
|
|
|
arch.enableInterrupts(); // the timer now preempts this idle context into work
|
|
while (true) asm volatile ("hlt"); // idle when this core has nothing ready
|
|
}
|
|
|
|
/// Number of cores that have finished bring-up (the BSP plus every online AP).
|
|
pub fn onlineCount() usize {
|
|
var n: usize = 0;
|
|
for (&cpus) |*pc| {
|
|
if (pc.online) n += 1;
|
|
}
|
|
return n;
|
|
}
|
|
|
|
fn enqueue(t: *Task) void {
|
|
t.next = null;
|
|
const p: usize = t.priority;
|
|
if (ready_tail[p]) |tail| tail.next = t else ready_head[p] = t;
|
|
ready_tail[p] = t;
|
|
ready_bitmap |= levelBit(t.priority);
|
|
}
|
|
|
|
fn dequeueHighest() ?*Task {
|
|
if (ready_bitmap == 0) return null;
|
|
const level: Priority = @intCast(num_priorities - 1 - @clz(ready_bitmap));
|
|
const t = ready_head[level].?;
|
|
ready_head[level] = t.next;
|
|
if (ready_head[level] == null) {
|
|
ready_tail[level] = null;
|
|
ready_bitmap &= ~levelBit(level);
|
|
}
|
|
t.next = null;
|
|
return t;
|
|
}
|
|
|
|
fn levelBit(p: Priority) u8 {
|
|
return @as(u8, 1) << p;
|
|
}
|
|
|
|
/// Create a task that runs `entry` at `priority`. It becomes ready immediately.
|
|
/// Takes the kernel lock: it mutates the shared task table and ready queues and
|
|
/// allocates from the (non-thread-safe) heap, so on SMP it must be serialised.
|
|
pub fn spawn(entry: *const fn () void, priority: Priority) void {
|
|
const flags = sync.enter();
|
|
_ = create(entry, priority);
|
|
sync.leave(flags);
|
|
}
|
|
|
|
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be
|
|
/// the single-threaded boot path). Returns the new task so a core can keep a handle
|
|
/// to its idle task.
|
|
fn create(entry: *const fn () void, priority: Priority) *Task {
|
|
const t = freeSlot() orelse @panic("sched: task table full");
|
|
const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack");
|
|
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack };
|
|
next_id += 1;
|
|
const top = @intFromPtr(stack.ptr) + stack.len;
|
|
t.rsp = arch.initTaskStack(top, @intFromPtr(entry));
|
|
enqueue(t);
|
|
return t;
|
|
}
|
|
|
|
fn freeSlot() ?*Task {
|
|
for (&tasks) |*t| {
|
|
if (t.state == .free) return t;
|
|
}
|
|
return null;
|
|
}
|
|
|
|
/// Pick the highest-priority ready task and switch this core to it. The big kernel
|
|
/// lock must be held by the caller (which also keeps local interrupts disabled);
|
|
/// it serialises every core's scheduling, so no other core can touch the shared
|
|
/// queues while we requeue `prev` and dequeue `next`. A dequeued task is `.ready`,
|
|
/// never running elsewhere, so two cores never run the same task.
|
|
fn schedule() void {
|
|
const pc = thisCpu();
|
|
const prev = pc.current;
|
|
if (prev.state == .running) {
|
|
prev.state = .ready;
|
|
enqueue(prev); // back of its level's queue (round-robin)
|
|
}
|
|
const next = dequeueHighest() orelse {
|
|
prev.state = .running; // nothing else ready — keep running
|
|
return;
|
|
};
|
|
next.state = .running;
|
|
pc.current = next;
|
|
if (next != prev) arch.switchContext(&prev.rsp, next.rsp);
|
|
}
|
|
|
|
/// Voluntarily give up the CPU to the next ready task.
|
|
pub fn yield() void {
|
|
const flags = sync.enter();
|
|
schedule();
|
|
sync.leave(flags);
|
|
}
|
|
|
|
/// Block the current task for `ms` milliseconds, then let it become runnable
|
|
/// again. The idle task (or other work) runs in the meantime.
|
|
pub fn sleep(ms: u64) void {
|
|
const flags = sync.enter();
|
|
const t = cur();
|
|
t.wake_at = arch.millis() + ms;
|
|
t.state = .blocked;
|
|
schedule(); // current is blocked, so schedule() won't re-enqueue it
|
|
sync.leave(flags);
|
|
}
|
|
|
|
// --- event-based blocking -------------------------------------------------
|
|
//
|
|
// A WaitQueue is a set of tasks blocked waiting for something (a resource, a
|
|
// message). Tasks link into it through the same `next` field the ready queues
|
|
// use — a task is in exactly one queue at a time. These are the primitive locks,
|
|
// semaphores and IPC channels are built on.
|
|
|
|
pub const WaitQueue = struct {
|
|
head: ?*Task = null,
|
|
};
|
|
|
|
/// Block the current task on `wq` and switch away. Precondition: the big kernel
|
|
/// lock is held (so a condition can be checked and the block committed atomically;
|
|
/// it also keeps local interrupts disabled). On return — when woken — the lock is
|
|
/// still held.
|
|
pub fn waitLocked(wq: *WaitQueue) void {
|
|
const t = cur();
|
|
t.state = .blocked;
|
|
t.next = wq.head;
|
|
wq.head = t;
|
|
schedule();
|
|
}
|
|
|
|
/// Move the highest-priority waiter on `wq` (if any) to the ready queue.
|
|
/// Precondition: the big kernel lock is held. Does not preempt — the caller decides.
|
|
pub fn wakeLocked(wq: *WaitQueue) void {
|
|
// Find the highest-priority waiter (bounded scan) and unlink it.
|
|
var best_prev: ?*Task = null;
|
|
var best: ?*Task = null;
|
|
var prev: ?*Task = null;
|
|
var node = wq.head;
|
|
while (node) |t| : ({
|
|
prev = t;
|
|
node = t.next;
|
|
}) {
|
|
if (best == null or t.priority > best.?.priority) {
|
|
best = t;
|
|
best_prev = prev;
|
|
}
|
|
}
|
|
const t = best orelse return;
|
|
if (best_prev) |p| p.next = t.next else wq.head = t.next;
|
|
t.state = .ready;
|
|
enqueue(t);
|
|
}
|
|
|
|
/// Block on `wq` (a self-contained critical section).
|
|
pub fn wait(wq: *WaitQueue) void {
|
|
const flags = sync.enter();
|
|
waitLocked(wq);
|
|
sync.leave(flags);
|
|
}
|
|
|
|
/// Wake the highest-priority waiter on `wq`, preempting if it outranks us.
|
|
pub fn wake(wq: *WaitQueue) void {
|
|
const flags = sync.enter();
|
|
wakeLocked(wq);
|
|
// If a higher-priority task is now ready, run it immediately.
|
|
if (highestReadyPriority()) |p| {
|
|
if (p > cur().priority) schedule();
|
|
}
|
|
sync.leave(flags);
|
|
}
|
|
|
|
fn highestReadyPriority() ?Priority {
|
|
if (ready_bitmap == 0) return null;
|
|
return @intCast(num_priorities - 1 - @clz(ready_bitmap));
|
|
}
|
|
|
|
/// Wake any sleeping task whose deadline has passed. Bounded by the task count,
|
|
/// so it stays deterministic. Called from the timer tick (interrupts disabled).
|
|
fn wakeExpired() void {
|
|
const now = arch.millis();
|
|
for (&tasks) |*t| {
|
|
if (t.state == .blocked and t.wake_at != 0 and now >= t.wake_at) {
|
|
t.wake_at = 0;
|
|
t.state = .ready;
|
|
enqueue(t);
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Called from the timer interrupt (interrupts already disabled): wake due
|
|
/// sleepers, then preempt. Takes the kernel lock like any other critical section,
|
|
/// but releases it *without* touching the interrupt flag — the handler's `iretq`
|
|
/// restores the interrupted context's flags, so re-enabling here would open a
|
|
/// nested-interrupt window before the return.
|
|
pub fn tick() void {
|
|
_ = sync.enter();
|
|
wakeExpired();
|
|
if (preemption_enabled) schedule();
|
|
sync.leaveIsr();
|
|
}
|
|
|
|
/// Enable or disable timer-driven preemption (cooperative-only when off).
|
|
pub fn setPreemption(enabled: bool) void {
|
|
preemption_enabled = enabled;
|
|
}
|
|
|
|
/// End the current task and switch away for good; never returns. The task's stack
|
|
/// is leaked for now (no reaper yet). Acquires the kernel lock and hands it off to
|
|
/// the task we switch into (which releases it) — this frame never returns to leave.
|
|
pub fn exit() noreturn {
|
|
_ = sync.enter();
|
|
const pc = thisCpu();
|
|
pc.current.state = .free;
|
|
const next = dequeueHighest() orelse @panic("sched: no task left to run");
|
|
next.state = .running;
|
|
pc.current = next;
|
|
var discard: usize = 0;
|
|
arch.switchContext(&discard, next.rsp);
|
|
unreachable;
|
|
}
|
|
|
|
pub fn currentId() u32 {
|
|
return cur().id;
|
|
}
|
|
|
|
/// The dense index of the core this task is currently running on (0 = BSP). Reads
|
|
/// per-CPU state, so a task calling it on different cores sees different values —
|
|
/// which is how a test can prove work is running in parallel.
|
|
pub fn currentCpuIndex() u32 {
|
|
return thisCpu().index;
|
|
}
|
|
|
|
/// Change the running task's priority (takes effect next time it's enqueued).
|
|
pub fn setPriority(p: Priority) void {
|
|
cur().priority = p;
|
|
}
|