//! The scheduler: fixed-priority preemptive multitasking. //! //! Tasks are kernel threads (ring 0, each with its own stack). The **highest- //! priority ready task always runs**; within a priority level, tasks round-robin. //! Selection is O(1) — a bitmap of non-empty priority levels plus a FIFO queue per //! level — which keeps scheduling deterministic, as a real-time kernel needs (see //! docs/vision.md). //! //! Switching happens both cooperatively (`yield`) and preemptively (the timer //! calls `tick`). See docs/scheduling.md for the interrupt-flag discipline that //! makes those two paths coexist. //! //! Cross-core safety is the **big kernel lock** (`sync.zig`): every critical //! section here runs under it, and it is held across a context switch and released //! by the task that resumes (see sync.zig's hand-off rule). On a single core the //! lock is never contended, so the behaviour is exactly the old interrupt-flag //! model; it's what lets a second core enter `schedule()` without corrupting the //! shared queues. const std = @import("std"); const arch = @import("arch"); const heap = @import("heap.zig"); const sync = @import("sync.zig"); /// Priority level: 0 (lowest) .. 7 (highest). 8 levels total. pub const Priority = u3; const num_priorities = 8; const stack_size = 16 * 1024; // each task's kernel stack is 16 KiB const max_tasks = 16; // the maximum number of tasks alive at once is 16 in a static sized pool const State = enum { free, ready, running, blocked }; const Task = struct { id: u32 = 0, state: State = .free, priority: Priority = 0, rsp: usize = 0, // saved stack pointer, valid while not running stack: []u8 = &.{}, wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping next: ?*Task = null, // ready-queue link }; var tasks = [_]Task{.{}} ** max_tasks; var next_id: u32 = 1; /// Per-CPU scheduler state: the task each core is running, plus its own idle task. /// One entry per core; the arch layer stashes a pointer to the *running* core's /// entry in the GS base, so `thisCpu()` fetches it with a single read and no lock. /// This is the only state that's genuinely per-core — the ready queues below stay /// **global** under the big kernel lock, so any idle core pulls the highest-priority /// ready task (work-conserving). Per-core ready queues are a later optimisation if /// the global queue's lock contention ever bites (docs/smp.md). pub const PerCpu = struct { current: *Task = undefined, // the task running on this core idle: *Task = undefined, // this core's idle task (always ready, lowest priority) apic_id: u32 = 0, // the core's Local APIC id index: u32 = 0, // dense 0-based core index online: bool = false, // has this core finished bring-up? }; const max_cpus = 64; // matches the discovery pool (src/device/acpi.zig) var cpus = [_]PerCpu{.{}} ** max_cpus; /// This core's per-CPU state, via the arch layer's GS-base pointer. Valid only once /// this core has run its scheduler bring-up (BSP in `init`, AP in `secondaryInit`). inline fn thisCpu() *PerCpu { return @ptrFromInt(arch.cpuLocal()); } /// The task running on this core — the per-CPU replacement for the old global /// `current`. A convenience reader; writes go through `thisCpu().current`. inline fn cur() *Task { return thisCpu().current; } // Per-priority FIFO ready queues, and a bitmap of which levels are non-empty. These // are shared across all cores and mutated only under the big kernel lock. var ready_head: [num_priorities]?*Task = .{null} ** num_priorities; var ready_tail: [num_priorities]?*Task = .{null} ** num_priorities; var ready_bitmap: u8 = 0; var preemption_enabled = true; /// Bring up scheduling on the bootstrap processor: register the currently-running /// kernel context as task 0, publish this core's per-CPU state (via the GS base), /// give the core an idle task, and hook the timer for preemption. Runs once, at /// boot, before interrupts are enabled — so no lock is needed here. pub fn init(boot_priority: Priority) void { const pc = &cpus[0]; pc.* = .{ .index = 0, .online = true }; arch.setCpuLocal(@intFromPtr(pc)); tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority }; pc.current = &tasks[0]; pc.idle = create(idle, 0); // this core's idle task: always ready, lowest priority arch.setTickHook(tick); } /// The idle task: run when every other task is blocked or sleeping. `hlt` waits /// for the next interrupt at near-zero power (see docs/halting.md). fn idle() void { while (true) asm volatile ("hlt"); } /// Reserve and initialise the per-CPU slot for an application processor at dense /// `index` (1-based; 0 is the BSP) with Local APIC id `apic_id`, and return a /// pointer the arch bring-up hands to the core (it publishes it in its GS base). /// Called on the BSP before waking each AP; the AP marks itself `online`. pub fn prepareSecondary(index: usize, apic_id: u32) *PerCpu { const pc = &cpus[index]; pc.* = .{ .index = @intCast(index), .apic_id = apic_id, .online = false }; return pc; } /// Entry for an application processor once the arch layer has set up its per-CPU /// tables, LAPIC, and timer. It turns this bring-up context into the core's idle task /// (as task 0 is for the BSP), marks the core online, and enters the run loop: with /// interrupts enabled the timer preempts this idle context into whatever the global /// ready queue offers, so the core runs real work in parallel with the others. The /// `.c` calling convention lets the arch trampoline path jump here. Never returns. pub fn secondaryMain() callconv(.c) noreturn { const flags = sync.enter(); const pc = thisCpu(); const t = freeSlot() orelse @panic("sched: task table full (AP idle task)"); t.* = .{ .id = next_id, .state = .running, .priority = 0 }; next_id += 1; pc.current = t; pc.idle = t; pc.online = true; sync.leave(flags); arch.enableInterrupts(); // the timer now preempts this idle context into work while (true) asm volatile ("hlt"); // idle when this core has nothing ready } /// Number of cores that have finished bring-up (the BSP plus every online AP). pub fn onlineCount() usize { var n: usize = 0; for (&cpus) |*pc| { if (pc.online) n += 1; } return n; } fn enqueue(t: *Task) void { t.next = null; const p: usize = t.priority; if (ready_tail[p]) |tail| tail.next = t else ready_head[p] = t; ready_tail[p] = t; ready_bitmap |= levelBit(t.priority); } fn dequeueHighest() ?*Task { if (ready_bitmap == 0) return null; const level: Priority = @intCast(num_priorities - 1 - @clz(ready_bitmap)); const t = ready_head[level].?; ready_head[level] = t.next; if (ready_head[level] == null) { ready_tail[level] = null; ready_bitmap &= ~levelBit(level); } t.next = null; return t; } fn levelBit(p: Priority) u8 { return @as(u8, 1) << p; } /// Create a task that runs `entry` at `priority`. It becomes ready immediately. /// Takes the kernel lock: it mutates the shared task table and ready queues and /// allocates from the (non-thread-safe) heap, so on SMP it must be serialised. pub fn spawn(entry: *const fn () void, priority: Priority) void { const flags = sync.enter(); _ = create(entry, priority); sync.leave(flags); } /// The unlocked task-creation primitive. Caller must hold the kernel lock (or be /// the single-threaded boot path). Returns the new task so a core can keep a handle /// to its idle task. fn create(entry: *const fn () void, priority: Priority) *Task { const t = freeSlot() orelse @panic("sched: task table full"); const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack"); t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack }; next_id += 1; const top = @intFromPtr(stack.ptr) + stack.len; t.rsp = arch.initTaskStack(top, @intFromPtr(entry)); enqueue(t); return t; } fn freeSlot() ?*Task { for (&tasks) |*t| { if (t.state == .free) return t; } return null; } /// Pick the highest-priority ready task and switch this core to it. The big kernel /// lock must be held by the caller (which also keeps local interrupts disabled); /// it serialises every core's scheduling, so no other core can touch the shared /// queues while we requeue `prev` and dequeue `next`. A dequeued task is `.ready`, /// never running elsewhere, so two cores never run the same task. fn schedule() void { const pc = thisCpu(); const prev = pc.current; if (prev.state == .running) { prev.state = .ready; enqueue(prev); // back of its level's queue (round-robin) } const next = dequeueHighest() orelse { prev.state = .running; // nothing else ready — keep running return; }; next.state = .running; pc.current = next; if (next != prev) arch.switchContext(&prev.rsp, next.rsp); } /// Voluntarily give up the CPU to the next ready task. pub fn yield() void { const flags = sync.enter(); schedule(); sync.leave(flags); } /// Block the current task for `ms` milliseconds, then let it become runnable /// again. The idle task (or other work) runs in the meantime. pub fn sleep(ms: u64) void { const flags = sync.enter(); const t = cur(); t.wake_at = arch.millis() + ms; t.state = .blocked; schedule(); // current is blocked, so schedule() won't re-enqueue it sync.leave(flags); } // --- event-based blocking ------------------------------------------------- // // A WaitQueue is a set of tasks blocked waiting for something (a resource, a // message). Tasks link into it through the same `next` field the ready queues // use — a task is in exactly one queue at a time. These are the primitive locks, // semaphores and IPC channels are built on. pub const WaitQueue = struct { head: ?*Task = null, }; /// Block the current task on `wq` and switch away. Precondition: the big kernel /// lock is held (so a condition can be checked and the block committed atomically; /// it also keeps local interrupts disabled). On return — when woken — the lock is /// still held. pub fn waitLocked(wq: *WaitQueue) void { const t = cur(); t.state = .blocked; t.next = wq.head; wq.head = t; schedule(); } /// Move the highest-priority waiter on `wq` (if any) to the ready queue. /// Precondition: the big kernel lock is held. Does not preempt — the caller decides. pub fn wakeLocked(wq: *WaitQueue) void { // Find the highest-priority waiter (bounded scan) and unlink it. var best_prev: ?*Task = null; var best: ?*Task = null; var prev: ?*Task = null; var node = wq.head; while (node) |t| : ({ prev = t; node = t.next; }) { if (best == null or t.priority > best.?.priority) { best = t; best_prev = prev; } } const t = best orelse return; if (best_prev) |p| p.next = t.next else wq.head = t.next; t.state = .ready; enqueue(t); } /// Block on `wq` (a self-contained critical section). pub fn wait(wq: *WaitQueue) void { const flags = sync.enter(); waitLocked(wq); sync.leave(flags); } /// Wake the highest-priority waiter on `wq`, preempting if it outranks us. pub fn wake(wq: *WaitQueue) void { const flags = sync.enter(); wakeLocked(wq); // If a higher-priority task is now ready, run it immediately. if (highestReadyPriority()) |p| { if (p > cur().priority) schedule(); } sync.leave(flags); } fn highestReadyPriority() ?Priority { if (ready_bitmap == 0) return null; return @intCast(num_priorities - 1 - @clz(ready_bitmap)); } /// Wake any sleeping task whose deadline has passed. Bounded by the task count, /// so it stays deterministic. Called from the timer tick (interrupts disabled). fn wakeExpired() void { const now = arch.millis(); for (&tasks) |*t| { if (t.state == .blocked and t.wake_at != 0 and now >= t.wake_at) { t.wake_at = 0; t.state = .ready; enqueue(t); } } } /// Called from the timer interrupt (interrupts already disabled): wake due /// sleepers, then preempt. Takes the kernel lock like any other critical section, /// but releases it *without* touching the interrupt flag — the handler's `iretq` /// restores the interrupted context's flags, so re-enabling here would open a /// nested-interrupt window before the return. pub fn tick() void { _ = sync.enter(); wakeExpired(); if (preemption_enabled) schedule(); sync.leaveIsr(); } /// Enable or disable timer-driven preemption (cooperative-only when off). pub fn setPreemption(enabled: bool) void { preemption_enabled = enabled; } /// End the current task and switch away for good; never returns. The task's stack /// is leaked for now (no reaper yet). Acquires the kernel lock and hands it off to /// the task we switch into (which releases it) — this frame never returns to leave. pub fn exit() noreturn { _ = sync.enter(); const pc = thisCpu(); pc.current.state = .free; const next = dequeueHighest() orelse @panic("sched: no task left to run"); next.state = .running; pc.current = next; var discard: usize = 0; arch.switchContext(&discard, next.rsp); unreachable; } pub fn currentId() u32 { return cur().id; } /// The dense index of the core this task is currently running on (0 = BSP). Reads /// per-CPU state, so a task calling it on different cores sees different values — /// which is how a test can prove work is running in parallel. pub fn currentCpuIndex() u32 { return thisCpu().index; } /// Change the running task's priority (takes effect next time it's enqueued). pub fn setPriority(p: Priority) void { cur().priority = p; }