add thread affinity: pin a task to a core

spawnOn(entry, priority, cpu) routes to a per-core pinned queue, merged with the global queue at O(1) selection. Falls back to unpinned for an offline/invalid core.
This commit is contained in:
Daniel Samson
2026-07-08 14:45:28 +01:00
parent f02259cae0
commit 37f72a4df8
2 changed files with 118 additions and 43 deletions
+21
View File
@@ -67,6 +67,27 @@ exist, which is what a real-time scheduler needs.
- **Round-robin within a level.** When a task is descheduled it goes to the *back* - **Round-robin within a level.** When a task is descheduled it goes to the *back*
of its level's queue, so equal-priority tasks share the CPU fairly. of its level's queue, so equal-priority tasks share the CPU fairly.
## Affinity: pinning a task to a core
By default a task runs on **any** core — the ready queue above is global, and any
idle core pulls the highest-priority task from it (work-conserving; see
[smp.md](smp.md)). A task can instead be **pinned** to one core with
`spawnOn(entry, priority, cpu)`, giving it an *affinity*: it will only ever run
there, never migrating.
Mechanically, each core has its **own** pinned queue (same 8-level FIFO + bitmap)
alongside the global one. A pinned task is enqueued only into its core's pinned
queue; selection compares the top of the global queue and the running core's pinned
queue and takes the higher priority (still O(1) — two bit-scans and a compare), with
a pinned task winning an equal-priority tie so it can't be starved by global work.
Because every queue is mutated under the [big kernel lock](smp.md), one core enqueuing
into another core's pinned queue is safe.
This is the *explicit-affinity* model (no surprise migration mid-deadline), which is
the more real-time-predictable direction. `spawnOn` refuses to pin to an offline or
out-of-range core — it creates the task unpinned instead, so it still runs somewhere
rather than stranding in a queue no core services, and returns whether the pin took.
## Sleeping and the idle task ## Sleeping and the idle task
A task can **block** — give up the CPU until an event, rather than busy-wait A task can **block** — give up the CPU until an event, rather than busy-wait
+97 -43
View File
@@ -38,25 +38,34 @@ const Task = struct {
rsp: usize = 0, // saved stack pointer, valid while not running rsp: usize = 0, // saved stack pointer, valid while not running
stack: []u8 = &.{}, stack: []u8 = &.{},
wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping
affinity: ?u32 = null, // null = runs on any core; else the index of its pinned core
next: ?*Task = null, // ready-queue link next: ?*Task = null, // ready-queue link
}; };
var tasks = [_]Task{.{}} ** max_tasks; var tasks = [_]Task{.{}} ** max_tasks;
var next_id: u32 = 1; var next_id: u32 = 1;
/// Per-CPU scheduler state: the task each core is running, plus its own idle task. /// Per-CPU scheduler state: the task each core is running, its own idle task, and a
/// One entry per core; the arch layer stashes a pointer to the *running* core's /// queue of tasks **pinned** to it. One entry per core; the arch layer stashes a
/// entry in the GS base, so `thisCpu()` fetches it with a single read and no lock. /// pointer to the *running* core's entry in the GS base, so `thisCpu()` fetches it
/// This is the only state that's genuinely per-core — the ready queues below stay /// with a single read and no lock.
/// **global** under the big kernel lock, so any idle core pulls the highest-priority ///
/// ready task (work-conserving). Per-core ready queues are a later optimisation if /// Most work stays in the **global** ready queue (below), which any idle core pulls
/// the global queue's lock contention ever bites (docs/smp.md). /// from — work-conserving. A task given an *affinity* instead goes to that core's
/// `pinned_*` queue and is only ever run there (no surprise migration — the more
/// real-time-predictable model, docs/smp.md). The two queues are merged at selection
/// time. Both are still mutated only under the big kernel lock, so one core enqueuing
/// into another core's pinned queue is safe.
pub const PerCpu = struct { pub const PerCpu = struct {
current: *Task = undefined, // the task running on this core current: *Task = undefined, // the task running on this core
idle: *Task = undefined, // this core's idle task (always ready, lowest priority) idle: *Task = undefined, // this core's idle task (always ready, lowest priority)
apic_id: u32 = 0, // the core's Local APIC id apic_id: u32 = 0, // the core's Local APIC id
index: u32 = 0, // dense 0-based core index index: u32 = 0, // dense 0-based core index
online: bool = false, // has this core finished bring-up? online: bool = false, // has this core finished bring-up?
// Tasks pinned to this core (affinity == index), per priority level + bitmap.
pinned_head: [num_priorities]?*Task = .{null} ** num_priorities,
pinned_tail: [num_priorities]?*Task = .{null} ** num_priorities,
pinned_bitmap: u8 = 0,
}; };
const max_cpus = 64; // matches the discovery pool (src/device/acpi.zig) const max_cpus = 64; // matches the discovery pool (src/device/acpi.zig)
@@ -92,7 +101,7 @@ pub fn init(boot_priority: Priority) void {
arch.setCpuLocal(@intFromPtr(pc)); arch.setCpuLocal(@intFromPtr(pc));
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority }; tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
pc.current = &tasks[0]; pc.current = &tasks[0];
pc.idle = create(idle, 0); // this core's idle task: always ready, lowest priority pc.idle = create(idle, 0, null); // this core's idle task: always ready, lowest priority
arch.setTickHook(tick); arch.setTickHook(tick);
} }
@@ -142,47 +151,83 @@ pub fn onlineCount() usize {
return n; return n;
} }
/// Make `t` ready. A pinned task (affinity set) goes to that core's pinned queue;
/// everything else goes to the shared global queue.
fn enqueue(t: *Task) void { fn enqueue(t: *Task) void {
t.next = null; if (t.affinity) |cpu| {
const p: usize = t.priority; const pc = &cpus[cpu];
if (ready_tail[p]) |tail| tail.next = t else ready_head[p] = t; enqueueTo(&pc.pinned_head, &pc.pinned_tail, &pc.pinned_bitmap, t);
ready_tail[p] = t; } else {
ready_bitmap |= levelBit(t.priority); enqueueTo(&ready_head, &ready_tail, &ready_bitmap, t);
}
} }
fn dequeueHighest() ?*Task { fn enqueueTo(head: *[num_priorities]?*Task, tail: *[num_priorities]?*Task, bitmap: *u8, t: *Task) void {
if (ready_bitmap == 0) return null; t.next = null;
const level: Priority = @intCast(num_priorities - 1 - @clz(ready_bitmap)); const p: usize = t.priority;
const t = ready_head[level].?; if (tail[p]) |tl| tl.next = t else head[p] = t;
ready_head[level] = t.next; tail[p] = t;
if (ready_head[level] == null) { bitmap.* |= @as(u8, 1) << t.priority;
ready_tail[level] = null; }
ready_bitmap &= ~levelBit(level);
/// The highest non-empty priority level in a bitmap, or -1 if empty.
fn topLevel(bitmap: u8) i32 {
if (bitmap == 0) return -1;
return @as(i32, num_priorities - 1) - @as(i32, @clz(bitmap));
}
/// Pick the highest-priority ready task for core `pc`: the better of the global queue
/// and this core's pinned queue. Still O(1) (two `clz` and a compare). A pinned task
/// wins an equal-priority tie, so it can't be starved by global work at its level.
fn dequeueHighest(pc: *PerCpu) ?*Task {
const g = topLevel(ready_bitmap);
const p = topLevel(pc.pinned_bitmap);
if (g < 0 and p < 0) return null;
if (p >= g) return dequeueFrom(&pc.pinned_head, &pc.pinned_tail, &pc.pinned_bitmap, @intCast(p));
return dequeueFrom(&ready_head, &ready_tail, &ready_bitmap, @intCast(g));
}
fn dequeueFrom(head: *[num_priorities]?*Task, tail: *[num_priorities]?*Task, bitmap: *u8, level: usize) ?*Task {
const t = head[level].?;
head[level] = t.next;
if (head[level] == null) {
tail[level] = null;
bitmap.* &= ~(@as(u8, 1) << @intCast(level));
} }
t.next = null; t.next = null;
return t; return t;
} }
fn levelBit(p: Priority) u8 { /// Create a task that runs `entry` at `priority`, runnable on any core. It becomes
return @as(u8, 1) << p; /// ready immediately. Takes the kernel lock: it mutates the shared task table and
} /// ready queues and allocates from the (non-thread-safe) heap, so on SMP it must be
/// serialised.
/// Create a task that runs `entry` at `priority`. It becomes ready immediately.
/// Takes the kernel lock: it mutates the shared task table and ready queues and
/// allocates from the (non-thread-safe) heap, so on SMP it must be serialised.
pub fn spawn(entry: *const fn () void, priority: Priority) void { pub fn spawn(entry: *const fn () void, priority: Priority) void {
const flags = sync.enter(); const flags = sync.enter();
_ = create(entry, priority); _ = create(entry, priority, null);
sync.leave(flags); sync.leave(flags);
} }
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be /// Like `spawn`, but **pins** the task to core `cpu` — it will only ever run there.
/// the single-threaded boot path). Returns the new task so a core can keep a handle /// Returns true if pinned; false if `cpu` isn't a valid, online core, in which case
/// to its idle task. /// the task is still created but left unpinned (so it runs *somewhere* rather than
fn create(entry: *const fn () void, priority: Priority) *Task { /// stranding in a queue no core services). Callers that require the pin (e.g. tests)
/// should check the result.
pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
const flags = sync.enter();
defer sync.leave(flags);
const ok = cpu < max_cpus and cpus[cpu].online;
_ = create(entry, priority, if (ok) cpu else null);
return ok;
}
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the
/// single-threaded boot path). `affinity` pins the task to a core (null = any).
/// Returns the new task so a core can keep a handle to its idle task.
fn create(entry: *const fn () void, priority: Priority, affinity: ?u32) *Task {
const t = freeSlot() orelse @panic("sched: task table full"); const t = freeSlot() orelse @panic("sched: task table full");
const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack"); const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack");
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack }; t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack, .affinity = affinity };
next_id += 1; next_id += 1;
const top = @intFromPtr(stack.ptr) + stack.len; const top = @intFromPtr(stack.ptr) + stack.len;
t.rsp = arch.initTaskStack(top, @intFromPtr(entry)); t.rsp = arch.initTaskStack(top, @intFromPtr(entry));
@@ -209,7 +254,7 @@ fn schedule() void {
prev.state = .ready; prev.state = .ready;
enqueue(prev); // back of its level's queue (round-robin) enqueue(prev); // back of its level's queue (round-robin)
} }
const next = dequeueHighest() orelse { const next = dequeueHighest(pc) orelse {
prev.state = .running; // nothing else ready — keep running prev.state = .running; // nothing else ready — keep running
return; return;
}; };
@@ -292,17 +337,23 @@ pub fn wait(wq: *WaitQueue) void {
/// Wake the highest-priority waiter on `wq`, preempting if it outranks us. /// Wake the highest-priority waiter on `wq`, preempting if it outranks us.
pub fn wake(wq: *WaitQueue) void { pub fn wake(wq: *WaitQueue) void {
const flags = sync.enter(); const flags = sync.enter();
const pc = thisCpu();
wakeLocked(wq); wakeLocked(wq);
// If a higher-priority task is now ready, run it immediately. // If a task this core would now pick outranks the running one, run it at once.
if (highestReadyPriority()) |p| { // (A waiter pinned to *another* core isn't counted — that core picks it up on its
if (p > cur().priority) schedule(); // next tick; this core doesn't preempt for work it can't run.)
if (highestReadyPriority(pc)) |p| {
if (p > pc.current.priority) schedule();
} }
sync.leave(flags); sync.leave(flags);
} }
fn highestReadyPriority() ?Priority { /// The highest-priority task core `pc` could run right now — the better of the global
if (ready_bitmap == 0) return null; /// queue and this core's pinned queue — or null if it would fall back to idle.
return @intCast(num_priorities - 1 - @clz(ready_bitmap)); fn highestReadyPriority(pc: *PerCpu) ?Priority {
const top = @max(topLevel(ready_bitmap), topLevel(pc.pinned_bitmap));
if (top < 0) return null;
return @intCast(top);
} }
/// Wake any sleeping task whose deadline has passed. Bounded by the task count, /// Wake any sleeping task whose deadline has passed. Bounded by the task count,
@@ -342,7 +393,7 @@ pub fn exit() noreturn {
_ = sync.enter(); _ = sync.enter();
const pc = thisCpu(); const pc = thisCpu();
pc.current.state = .free; pc.current.state = .free;
const next = dequeueHighest() orelse @panic("sched: no task left to run"); const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
next.state = .running; next.state = .running;
pc.current = next; pc.current = next;
var discard: usize = 0; var discard: usize = 0;
@@ -356,8 +407,11 @@ pub fn currentId() u32 {
/// The dense index of the core this task is currently running on (0 = BSP). Reads /// The dense index of the core this task is currently running on (0 = BSP). Reads
/// per-CPU state, so a task calling it on different cores sees different values — /// per-CPU state, so a task calling it on different cores sees different values —
/// which is how a test can prove work is running in parallel. /// which is how a test can prove work is running in parallel. Returns 0 if the GS
/// base isn't published yet (a fault in very early boot, before `init`), so a fault
/// reporter can call it unconditionally without a second fault.
pub fn currentCpuIndex() u32 { pub fn currentCpuIndex() u32 {
if (arch.cpuLocal() == 0) return 0;
return thisCpu().index; return thisCpu().index;
} }