add thread affinity: pin a task to a core
spawnOn(entry, priority, cpu) routes to a per-core pinned queue, merged with the global queue at O(1) selection. Falls back to unpinned for an offline/invalid core.
This commit is contained in:
@@ -67,6 +67,27 @@ exist, which is what a real-time scheduler needs.
|
|||||||
- **Round-robin within a level.** When a task is descheduled it goes to the *back*
|
- **Round-robin within a level.** When a task is descheduled it goes to the *back*
|
||||||
of its level's queue, so equal-priority tasks share the CPU fairly.
|
of its level's queue, so equal-priority tasks share the CPU fairly.
|
||||||
|
|
||||||
|
## Affinity: pinning a task to a core
|
||||||
|
|
||||||
|
By default a task runs on **any** core — the ready queue above is global, and any
|
||||||
|
idle core pulls the highest-priority task from it (work-conserving; see
|
||||||
|
[smp.md](smp.md)). A task can instead be **pinned** to one core with
|
||||||
|
`spawnOn(entry, priority, cpu)`, giving it an *affinity*: it will only ever run
|
||||||
|
there, never migrating.
|
||||||
|
|
||||||
|
Mechanically, each core has its **own** pinned queue (same 8-level FIFO + bitmap)
|
||||||
|
alongside the global one. A pinned task is enqueued only into its core's pinned
|
||||||
|
queue; selection compares the top of the global queue and the running core's pinned
|
||||||
|
queue and takes the higher priority (still O(1) — two bit-scans and a compare), with
|
||||||
|
a pinned task winning an equal-priority tie so it can't be starved by global work.
|
||||||
|
Because every queue is mutated under the [big kernel lock](smp.md), one core enqueuing
|
||||||
|
into another core's pinned queue is safe.
|
||||||
|
|
||||||
|
This is the *explicit-affinity* model (no surprise migration mid-deadline), which is
|
||||||
|
the more real-time-predictable direction. `spawnOn` refuses to pin to an offline or
|
||||||
|
out-of-range core — it creates the task unpinned instead, so it still runs somewhere
|
||||||
|
rather than stranding in a queue no core services, and returns whether the pin took.
|
||||||
|
|
||||||
## Sleeping and the idle task
|
## Sleeping and the idle task
|
||||||
|
|
||||||
A task can **block** — give up the CPU until an event, rather than busy-wait
|
A task can **block** — give up the CPU until an event, rather than busy-wait
|
||||||
|
|||||||
+97
-43
@@ -38,25 +38,34 @@ const Task = struct {
|
|||||||
rsp: usize = 0, // saved stack pointer, valid while not running
|
rsp: usize = 0, // saved stack pointer, valid while not running
|
||||||
stack: []u8 = &.{},
|
stack: []u8 = &.{},
|
||||||
wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping
|
wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping
|
||||||
|
affinity: ?u32 = null, // null = runs on any core; else the index of its pinned core
|
||||||
next: ?*Task = null, // ready-queue link
|
next: ?*Task = null, // ready-queue link
|
||||||
};
|
};
|
||||||
|
|
||||||
var tasks = [_]Task{.{}} ** max_tasks;
|
var tasks = [_]Task{.{}} ** max_tasks;
|
||||||
var next_id: u32 = 1;
|
var next_id: u32 = 1;
|
||||||
|
|
||||||
/// Per-CPU scheduler state: the task each core is running, plus its own idle task.
|
/// Per-CPU scheduler state: the task each core is running, its own idle task, and a
|
||||||
/// One entry per core; the arch layer stashes a pointer to the *running* core's
|
/// queue of tasks **pinned** to it. One entry per core; the arch layer stashes a
|
||||||
/// entry in the GS base, so `thisCpu()` fetches it with a single read and no lock.
|
/// pointer to the *running* core's entry in the GS base, so `thisCpu()` fetches it
|
||||||
/// This is the only state that's genuinely per-core — the ready queues below stay
|
/// with a single read and no lock.
|
||||||
/// **global** under the big kernel lock, so any idle core pulls the highest-priority
|
///
|
||||||
/// ready task (work-conserving). Per-core ready queues are a later optimisation if
|
/// Most work stays in the **global** ready queue (below), which any idle core pulls
|
||||||
/// the global queue's lock contention ever bites (docs/smp.md).
|
/// from — work-conserving. A task given an *affinity* instead goes to that core's
|
||||||
|
/// `pinned_*` queue and is only ever run there (no surprise migration — the more
|
||||||
|
/// real-time-predictable model, docs/smp.md). The two queues are merged at selection
|
||||||
|
/// time. Both are still mutated only under the big kernel lock, so one core enqueuing
|
||||||
|
/// into another core's pinned queue is safe.
|
||||||
pub const PerCpu = struct {
|
pub const PerCpu = struct {
|
||||||
current: *Task = undefined, // the task running on this core
|
current: *Task = undefined, // the task running on this core
|
||||||
idle: *Task = undefined, // this core's idle task (always ready, lowest priority)
|
idle: *Task = undefined, // this core's idle task (always ready, lowest priority)
|
||||||
apic_id: u32 = 0, // the core's Local APIC id
|
apic_id: u32 = 0, // the core's Local APIC id
|
||||||
index: u32 = 0, // dense 0-based core index
|
index: u32 = 0, // dense 0-based core index
|
||||||
online: bool = false, // has this core finished bring-up?
|
online: bool = false, // has this core finished bring-up?
|
||||||
|
// Tasks pinned to this core (affinity == index), per priority level + bitmap.
|
||||||
|
pinned_head: [num_priorities]?*Task = .{null} ** num_priorities,
|
||||||
|
pinned_tail: [num_priorities]?*Task = .{null} ** num_priorities,
|
||||||
|
pinned_bitmap: u8 = 0,
|
||||||
};
|
};
|
||||||
|
|
||||||
const max_cpus = 64; // matches the discovery pool (src/device/acpi.zig)
|
const max_cpus = 64; // matches the discovery pool (src/device/acpi.zig)
|
||||||
@@ -92,7 +101,7 @@ pub fn init(boot_priority: Priority) void {
|
|||||||
arch.setCpuLocal(@intFromPtr(pc));
|
arch.setCpuLocal(@intFromPtr(pc));
|
||||||
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
|
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
|
||||||
pc.current = &tasks[0];
|
pc.current = &tasks[0];
|
||||||
pc.idle = create(idle, 0); // this core's idle task: always ready, lowest priority
|
pc.idle = create(idle, 0, null); // this core's idle task: always ready, lowest priority
|
||||||
arch.setTickHook(tick);
|
arch.setTickHook(tick);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -142,47 +151,83 @@ pub fn onlineCount() usize {
|
|||||||
return n;
|
return n;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Make `t` ready. A pinned task (affinity set) goes to that core's pinned queue;
|
||||||
|
/// everything else goes to the shared global queue.
|
||||||
fn enqueue(t: *Task) void {
|
fn enqueue(t: *Task) void {
|
||||||
t.next = null;
|
if (t.affinity) |cpu| {
|
||||||
const p: usize = t.priority;
|
const pc = &cpus[cpu];
|
||||||
if (ready_tail[p]) |tail| tail.next = t else ready_head[p] = t;
|
enqueueTo(&pc.pinned_head, &pc.pinned_tail, &pc.pinned_bitmap, t);
|
||||||
ready_tail[p] = t;
|
} else {
|
||||||
ready_bitmap |= levelBit(t.priority);
|
enqueueTo(&ready_head, &ready_tail, &ready_bitmap, t);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
fn dequeueHighest() ?*Task {
|
fn enqueueTo(head: *[num_priorities]?*Task, tail: *[num_priorities]?*Task, bitmap: *u8, t: *Task) void {
|
||||||
if (ready_bitmap == 0) return null;
|
t.next = null;
|
||||||
const level: Priority = @intCast(num_priorities - 1 - @clz(ready_bitmap));
|
const p: usize = t.priority;
|
||||||
const t = ready_head[level].?;
|
if (tail[p]) |tl| tl.next = t else head[p] = t;
|
||||||
ready_head[level] = t.next;
|
tail[p] = t;
|
||||||
if (ready_head[level] == null) {
|
bitmap.* |= @as(u8, 1) << t.priority;
|
||||||
ready_tail[level] = null;
|
}
|
||||||
ready_bitmap &= ~levelBit(level);
|
|
||||||
|
/// The highest non-empty priority level in a bitmap, or -1 if empty.
|
||||||
|
fn topLevel(bitmap: u8) i32 {
|
||||||
|
if (bitmap == 0) return -1;
|
||||||
|
return @as(i32, num_priorities - 1) - @as(i32, @clz(bitmap));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Pick the highest-priority ready task for core `pc`: the better of the global queue
|
||||||
|
/// and this core's pinned queue. Still O(1) (two `clz` and a compare). A pinned task
|
||||||
|
/// wins an equal-priority tie, so it can't be starved by global work at its level.
|
||||||
|
fn dequeueHighest(pc: *PerCpu) ?*Task {
|
||||||
|
const g = topLevel(ready_bitmap);
|
||||||
|
const p = topLevel(pc.pinned_bitmap);
|
||||||
|
if (g < 0 and p < 0) return null;
|
||||||
|
if (p >= g) return dequeueFrom(&pc.pinned_head, &pc.pinned_tail, &pc.pinned_bitmap, @intCast(p));
|
||||||
|
return dequeueFrom(&ready_head, &ready_tail, &ready_bitmap, @intCast(g));
|
||||||
|
}
|
||||||
|
|
||||||
|
fn dequeueFrom(head: *[num_priorities]?*Task, tail: *[num_priorities]?*Task, bitmap: *u8, level: usize) ?*Task {
|
||||||
|
const t = head[level].?;
|
||||||
|
head[level] = t.next;
|
||||||
|
if (head[level] == null) {
|
||||||
|
tail[level] = null;
|
||||||
|
bitmap.* &= ~(@as(u8, 1) << @intCast(level));
|
||||||
}
|
}
|
||||||
t.next = null;
|
t.next = null;
|
||||||
return t;
|
return t;
|
||||||
}
|
}
|
||||||
|
|
||||||
fn levelBit(p: Priority) u8 {
|
/// Create a task that runs `entry` at `priority`, runnable on any core. It becomes
|
||||||
return @as(u8, 1) << p;
|
/// ready immediately. Takes the kernel lock: it mutates the shared task table and
|
||||||
}
|
/// ready queues and allocates from the (non-thread-safe) heap, so on SMP it must be
|
||||||
|
/// serialised.
|
||||||
/// Create a task that runs `entry` at `priority`. It becomes ready immediately.
|
|
||||||
/// Takes the kernel lock: it mutates the shared task table and ready queues and
|
|
||||||
/// allocates from the (non-thread-safe) heap, so on SMP it must be serialised.
|
|
||||||
pub fn spawn(entry: *const fn () void, priority: Priority) void {
|
pub fn spawn(entry: *const fn () void, priority: Priority) void {
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
_ = create(entry, priority);
|
_ = create(entry, priority, null);
|
||||||
sync.leave(flags);
|
sync.leave(flags);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be
|
/// Like `spawn`, but **pins** the task to core `cpu` — it will only ever run there.
|
||||||
/// the single-threaded boot path). Returns the new task so a core can keep a handle
|
/// Returns true if pinned; false if `cpu` isn't a valid, online core, in which case
|
||||||
/// to its idle task.
|
/// the task is still created but left unpinned (so it runs *somewhere* rather than
|
||||||
fn create(entry: *const fn () void, priority: Priority) *Task {
|
/// stranding in a queue no core services). Callers that require the pin (e.g. tests)
|
||||||
|
/// should check the result.
|
||||||
|
pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
|
const ok = cpu < max_cpus and cpus[cpu].online;
|
||||||
|
_ = create(entry, priority, if (ok) cpu else null);
|
||||||
|
return ok;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the
|
||||||
|
/// single-threaded boot path). `affinity` pins the task to a core (null = any).
|
||||||
|
/// Returns the new task so a core can keep a handle to its idle task.
|
||||||
|
fn create(entry: *const fn () void, priority: Priority, affinity: ?u32) *Task {
|
||||||
const t = freeSlot() orelse @panic("sched: task table full");
|
const t = freeSlot() orelse @panic("sched: task table full");
|
||||||
const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack");
|
const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack");
|
||||||
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack };
|
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack, .affinity = affinity };
|
||||||
next_id += 1;
|
next_id += 1;
|
||||||
const top = @intFromPtr(stack.ptr) + stack.len;
|
const top = @intFromPtr(stack.ptr) + stack.len;
|
||||||
t.rsp = arch.initTaskStack(top, @intFromPtr(entry));
|
t.rsp = arch.initTaskStack(top, @intFromPtr(entry));
|
||||||
@@ -209,7 +254,7 @@ fn schedule() void {
|
|||||||
prev.state = .ready;
|
prev.state = .ready;
|
||||||
enqueue(prev); // back of its level's queue (round-robin)
|
enqueue(prev); // back of its level's queue (round-robin)
|
||||||
}
|
}
|
||||||
const next = dequeueHighest() orelse {
|
const next = dequeueHighest(pc) orelse {
|
||||||
prev.state = .running; // nothing else ready — keep running
|
prev.state = .running; // nothing else ready — keep running
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
@@ -292,17 +337,23 @@ pub fn wait(wq: *WaitQueue) void {
|
|||||||
/// Wake the highest-priority waiter on `wq`, preempting if it outranks us.
|
/// Wake the highest-priority waiter on `wq`, preempting if it outranks us.
|
||||||
pub fn wake(wq: *WaitQueue) void {
|
pub fn wake(wq: *WaitQueue) void {
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
|
const pc = thisCpu();
|
||||||
wakeLocked(wq);
|
wakeLocked(wq);
|
||||||
// If a higher-priority task is now ready, run it immediately.
|
// If a task this core would now pick outranks the running one, run it at once.
|
||||||
if (highestReadyPriority()) |p| {
|
// (A waiter pinned to *another* core isn't counted — that core picks it up on its
|
||||||
if (p > cur().priority) schedule();
|
// next tick; this core doesn't preempt for work it can't run.)
|
||||||
|
if (highestReadyPriority(pc)) |p| {
|
||||||
|
if (p > pc.current.priority) schedule();
|
||||||
}
|
}
|
||||||
sync.leave(flags);
|
sync.leave(flags);
|
||||||
}
|
}
|
||||||
|
|
||||||
fn highestReadyPriority() ?Priority {
|
/// The highest-priority task core `pc` could run right now — the better of the global
|
||||||
if (ready_bitmap == 0) return null;
|
/// queue and this core's pinned queue — or null if it would fall back to idle.
|
||||||
return @intCast(num_priorities - 1 - @clz(ready_bitmap));
|
fn highestReadyPriority(pc: *PerCpu) ?Priority {
|
||||||
|
const top = @max(topLevel(ready_bitmap), topLevel(pc.pinned_bitmap));
|
||||||
|
if (top < 0) return null;
|
||||||
|
return @intCast(top);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Wake any sleeping task whose deadline has passed. Bounded by the task count,
|
/// Wake any sleeping task whose deadline has passed. Bounded by the task count,
|
||||||
@@ -342,7 +393,7 @@ pub fn exit() noreturn {
|
|||||||
_ = sync.enter();
|
_ = sync.enter();
|
||||||
const pc = thisCpu();
|
const pc = thisCpu();
|
||||||
pc.current.state = .free;
|
pc.current.state = .free;
|
||||||
const next = dequeueHighest() orelse @panic("sched: no task left to run");
|
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
|
||||||
next.state = .running;
|
next.state = .running;
|
||||||
pc.current = next;
|
pc.current = next;
|
||||||
var discard: usize = 0;
|
var discard: usize = 0;
|
||||||
@@ -356,8 +407,11 @@ pub fn currentId() u32 {
|
|||||||
|
|
||||||
/// The dense index of the core this task is currently running on (0 = BSP). Reads
|
/// The dense index of the core this task is currently running on (0 = BSP). Reads
|
||||||
/// per-CPU state, so a task calling it on different cores sees different values —
|
/// per-CPU state, so a task calling it on different cores sees different values —
|
||||||
/// which is how a test can prove work is running in parallel.
|
/// which is how a test can prove work is running in parallel. Returns 0 if the GS
|
||||||
|
/// base isn't published yet (a fault in very early boot, before `init`), so a fault
|
||||||
|
/// reporter can call it unconditionally without a second fault.
|
||||||
pub fn currentCpuIndex() u32 {
|
pub fn currentCpuIndex() u32 {
|
||||||
|
if (arch.cpuLocal() == 0) return 0;
|
||||||
return thisCpu().index;
|
return thisCpu().index;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user