diff --git a/docs/scheduling.md b/docs/scheduling.md index 0c5741f..5e03a77 100644 --- a/docs/scheduling.md +++ b/docs/scheduling.md @@ -67,6 +67,27 @@ exist, which is what a real-time scheduler needs. - **Round-robin within a level.** When a task is descheduled it goes to the *back* of its level's queue, so equal-priority tasks share the CPU fairly. +## Affinity: pinning a task to a core + +By default a task runs on **any** core — the ready queue above is global, and any +idle core pulls the highest-priority task from it (work-conserving; see +[smp.md](smp.md)). A task can instead be **pinned** to one core with +`spawnOn(entry, priority, cpu)`, giving it an *affinity*: it will only ever run +there, never migrating. + +Mechanically, each core has its **own** pinned queue (same 8-level FIFO + bitmap) +alongside the global one. A pinned task is enqueued only into its core's pinned +queue; selection compares the top of the global queue and the running core's pinned +queue and takes the higher priority (still O(1) — two bit-scans and a compare), with +a pinned task winning an equal-priority tie so it can't be starved by global work. +Because every queue is mutated under the [big kernel lock](smp.md), one core enqueuing +into another core's pinned queue is safe. + +This is the *explicit-affinity* model (no surprise migration mid-deadline), which is +the more real-time-predictable direction. `spawnOn` refuses to pin to an offline or +out-of-range core — it creates the task unpinned instead, so it still runs somewhere +rather than stranding in a queue no core services, and returns whether the pin took. + ## Sleeping and the idle task A task can **block** — give up the CPU until an event, rather than busy-wait diff --git a/src/kernel/scheduler.zig b/src/kernel/scheduler.zig index 3cb81b3..a70b06c 100644 --- a/src/kernel/scheduler.zig +++ b/src/kernel/scheduler.zig @@ -38,25 +38,34 @@ const Task = struct { rsp: usize = 0, // saved stack pointer, valid while not running stack: []u8 = &.{}, wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping + affinity: ?u32 = null, // null = runs on any core; else the index of its pinned core next: ?*Task = null, // ready-queue link }; var tasks = [_]Task{.{}} ** max_tasks; var next_id: u32 = 1; -/// Per-CPU scheduler state: the task each core is running, plus its own idle task. -/// One entry per core; the arch layer stashes a pointer to the *running* core's -/// entry in the GS base, so `thisCpu()` fetches it with a single read and no lock. -/// This is the only state that's genuinely per-core — the ready queues below stay -/// **global** under the big kernel lock, so any idle core pulls the highest-priority -/// ready task (work-conserving). Per-core ready queues are a later optimisation if -/// the global queue's lock contention ever bites (docs/smp.md). +/// Per-CPU scheduler state: the task each core is running, its own idle task, and a +/// queue of tasks **pinned** to it. One entry per core; the arch layer stashes a +/// pointer to the *running* core's entry in the GS base, so `thisCpu()` fetches it +/// with a single read and no lock. +/// +/// Most work stays in the **global** ready queue (below), which any idle core pulls +/// from — work-conserving. A task given an *affinity* instead goes to that core's +/// `pinned_*` queue and is only ever run there (no surprise migration — the more +/// real-time-predictable model, docs/smp.md). The two queues are merged at selection +/// time. Both are still mutated only under the big kernel lock, so one core enqueuing +/// into another core's pinned queue is safe. pub const PerCpu = struct { current: *Task = undefined, // the task running on this core idle: *Task = undefined, // this core's idle task (always ready, lowest priority) apic_id: u32 = 0, // the core's Local APIC id index: u32 = 0, // dense 0-based core index online: bool = false, // has this core finished bring-up? + // Tasks pinned to this core (affinity == index), per priority level + bitmap. + pinned_head: [num_priorities]?*Task = .{null} ** num_priorities, + pinned_tail: [num_priorities]?*Task = .{null} ** num_priorities, + pinned_bitmap: u8 = 0, }; const max_cpus = 64; // matches the discovery pool (src/device/acpi.zig) @@ -92,7 +101,7 @@ pub fn init(boot_priority: Priority) void { arch.setCpuLocal(@intFromPtr(pc)); tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority }; pc.current = &tasks[0]; - pc.idle = create(idle, 0); // this core's idle task: always ready, lowest priority + pc.idle = create(idle, 0, null); // this core's idle task: always ready, lowest priority arch.setTickHook(tick); } @@ -142,47 +151,83 @@ pub fn onlineCount() usize { return n; } +/// Make `t` ready. A pinned task (affinity set) goes to that core's pinned queue; +/// everything else goes to the shared global queue. fn enqueue(t: *Task) void { - t.next = null; - const p: usize = t.priority; - if (ready_tail[p]) |tail| tail.next = t else ready_head[p] = t; - ready_tail[p] = t; - ready_bitmap |= levelBit(t.priority); + if (t.affinity) |cpu| { + const pc = &cpus[cpu]; + enqueueTo(&pc.pinned_head, &pc.pinned_tail, &pc.pinned_bitmap, t); + } else { + enqueueTo(&ready_head, &ready_tail, &ready_bitmap, t); + } } -fn dequeueHighest() ?*Task { - if (ready_bitmap == 0) return null; - const level: Priority = @intCast(num_priorities - 1 - @clz(ready_bitmap)); - const t = ready_head[level].?; - ready_head[level] = t.next; - if (ready_head[level] == null) { - ready_tail[level] = null; - ready_bitmap &= ~levelBit(level); +fn enqueueTo(head: *[num_priorities]?*Task, tail: *[num_priorities]?*Task, bitmap: *u8, t: *Task) void { + t.next = null; + const p: usize = t.priority; + if (tail[p]) |tl| tl.next = t else head[p] = t; + tail[p] = t; + bitmap.* |= @as(u8, 1) << t.priority; +} + +/// The highest non-empty priority level in a bitmap, or -1 if empty. +fn topLevel(bitmap: u8) i32 { + if (bitmap == 0) return -1; + return @as(i32, num_priorities - 1) - @as(i32, @clz(bitmap)); +} + +/// Pick the highest-priority ready task for core `pc`: the better of the global queue +/// and this core's pinned queue. Still O(1) (two `clz` and a compare). A pinned task +/// wins an equal-priority tie, so it can't be starved by global work at its level. +fn dequeueHighest(pc: *PerCpu) ?*Task { + const g = topLevel(ready_bitmap); + const p = topLevel(pc.pinned_bitmap); + if (g < 0 and p < 0) return null; + if (p >= g) return dequeueFrom(&pc.pinned_head, &pc.pinned_tail, &pc.pinned_bitmap, @intCast(p)); + return dequeueFrom(&ready_head, &ready_tail, &ready_bitmap, @intCast(g)); +} + +fn dequeueFrom(head: *[num_priorities]?*Task, tail: *[num_priorities]?*Task, bitmap: *u8, level: usize) ?*Task { + const t = head[level].?; + head[level] = t.next; + if (head[level] == null) { + tail[level] = null; + bitmap.* &= ~(@as(u8, 1) << @intCast(level)); } t.next = null; return t; } -fn levelBit(p: Priority) u8 { - return @as(u8, 1) << p; -} - -/// Create a task that runs `entry` at `priority`. It becomes ready immediately. -/// Takes the kernel lock: it mutates the shared task table and ready queues and -/// allocates from the (non-thread-safe) heap, so on SMP it must be serialised. +/// Create a task that runs `entry` at `priority`, runnable on any core. It becomes +/// ready immediately. Takes the kernel lock: it mutates the shared task table and +/// ready queues and allocates from the (non-thread-safe) heap, so on SMP it must be +/// serialised. pub fn spawn(entry: *const fn () void, priority: Priority) void { const flags = sync.enter(); - _ = create(entry, priority); + _ = create(entry, priority, null); sync.leave(flags); } -/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be -/// the single-threaded boot path). Returns the new task so a core can keep a handle -/// to its idle task. -fn create(entry: *const fn () void, priority: Priority) *Task { +/// Like `spawn`, but **pins** the task to core `cpu` — it will only ever run there. +/// Returns true if pinned; false if `cpu` isn't a valid, online core, in which case +/// the task is still created but left unpinned (so it runs *somewhere* rather than +/// stranding in a queue no core services). Callers that require the pin (e.g. tests) +/// should check the result. +pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool { + const flags = sync.enter(); + defer sync.leave(flags); + const ok = cpu < max_cpus and cpus[cpu].online; + _ = create(entry, priority, if (ok) cpu else null); + return ok; +} + +/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the +/// single-threaded boot path). `affinity` pins the task to a core (null = any). +/// Returns the new task so a core can keep a handle to its idle task. +fn create(entry: *const fn () void, priority: Priority, affinity: ?u32) *Task { const t = freeSlot() orelse @panic("sched: task table full"); const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack"); - t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack }; + t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack, .affinity = affinity }; next_id += 1; const top = @intFromPtr(stack.ptr) + stack.len; t.rsp = arch.initTaskStack(top, @intFromPtr(entry)); @@ -209,7 +254,7 @@ fn schedule() void { prev.state = .ready; enqueue(prev); // back of its level's queue (round-robin) } - const next = dequeueHighest() orelse { + const next = dequeueHighest(pc) orelse { prev.state = .running; // nothing else ready — keep running return; }; @@ -292,17 +337,23 @@ pub fn wait(wq: *WaitQueue) void { /// Wake the highest-priority waiter on `wq`, preempting if it outranks us. pub fn wake(wq: *WaitQueue) void { const flags = sync.enter(); + const pc = thisCpu(); wakeLocked(wq); - // If a higher-priority task is now ready, run it immediately. - if (highestReadyPriority()) |p| { - if (p > cur().priority) schedule(); + // If a task this core would now pick outranks the running one, run it at once. + // (A waiter pinned to *another* core isn't counted — that core picks it up on its + // next tick; this core doesn't preempt for work it can't run.) + if (highestReadyPriority(pc)) |p| { + if (p > pc.current.priority) schedule(); } sync.leave(flags); } -fn highestReadyPriority() ?Priority { - if (ready_bitmap == 0) return null; - return @intCast(num_priorities - 1 - @clz(ready_bitmap)); +/// The highest-priority task core `pc` could run right now — the better of the global +/// queue and this core's pinned queue — or null if it would fall back to idle. +fn highestReadyPriority(pc: *PerCpu) ?Priority { + const top = @max(topLevel(ready_bitmap), topLevel(pc.pinned_bitmap)); + if (top < 0) return null; + return @intCast(top); } /// Wake any sleeping task whose deadline has passed. Bounded by the task count, @@ -342,7 +393,7 @@ pub fn exit() noreturn { _ = sync.enter(); const pc = thisCpu(); pc.current.state = .free; - const next = dequeueHighest() orelse @panic("sched: no task left to run"); + const next = dequeueHighest(pc) orelse @panic("sched: no task left to run"); next.state = .running; pc.current = next; var discard: usize = 0; @@ -356,8 +407,11 @@ pub fn currentId() u32 { /// The dense index of the core this task is currently running on (0 = BSP). Reads /// per-CPU state, so a task calling it on different cores sees different values — -/// which is how a test can prove work is running in parallel. +/// which is how a test can prove work is running in parallel. Returns 0 if the GS +/// base isn't published yet (a fault in very early boot, before `init`), so a fault +/// reporter can call it unconditionally without a second fault. pub fn currentCpuIndex() u32 { + if (arch.cpuLocal() == 0) return 0; return thisCpu().index; }