add thread affinity: pin a task to a core
spawnOn(entry, priority, cpu) routes to a per-core pinned queue, merged with the global queue at O(1) selection. Falls back to unpinned for an offline/invalid core.
This commit is contained in:
+97
-43
@@ -38,25 +38,34 @@ const Task = struct {
|
||||
rsp: usize = 0, // saved stack pointer, valid while not running
|
||||
stack: []u8 = &.{},
|
||||
wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping
|
||||
affinity: ?u32 = null, // null = runs on any core; else the index of its pinned core
|
||||
next: ?*Task = null, // ready-queue link
|
||||
};
|
||||
|
||||
var tasks = [_]Task{.{}} ** max_tasks;
|
||||
var next_id: u32 = 1;
|
||||
|
||||
/// Per-CPU scheduler state: the task each core is running, plus its own idle task.
|
||||
/// One entry per core; the arch layer stashes a pointer to the *running* core's
|
||||
/// entry in the GS base, so `thisCpu()` fetches it with a single read and no lock.
|
||||
/// This is the only state that's genuinely per-core — the ready queues below stay
|
||||
/// **global** under the big kernel lock, so any idle core pulls the highest-priority
|
||||
/// ready task (work-conserving). Per-core ready queues are a later optimisation if
|
||||
/// the global queue's lock contention ever bites (docs/smp.md).
|
||||
/// Per-CPU scheduler state: the task each core is running, its own idle task, and a
|
||||
/// queue of tasks **pinned** to it. One entry per core; the arch layer stashes a
|
||||
/// pointer to the *running* core's entry in the GS base, so `thisCpu()` fetches it
|
||||
/// with a single read and no lock.
|
||||
///
|
||||
/// Most work stays in the **global** ready queue (below), which any idle core pulls
|
||||
/// from — work-conserving. A task given an *affinity* instead goes to that core's
|
||||
/// `pinned_*` queue and is only ever run there (no surprise migration — the more
|
||||
/// real-time-predictable model, docs/smp.md). The two queues are merged at selection
|
||||
/// time. Both are still mutated only under the big kernel lock, so one core enqueuing
|
||||
/// into another core's pinned queue is safe.
|
||||
pub const PerCpu = struct {
|
||||
current: *Task = undefined, // the task running on this core
|
||||
idle: *Task = undefined, // this core's idle task (always ready, lowest priority)
|
||||
apic_id: u32 = 0, // the core's Local APIC id
|
||||
index: u32 = 0, // dense 0-based core index
|
||||
online: bool = false, // has this core finished bring-up?
|
||||
// Tasks pinned to this core (affinity == index), per priority level + bitmap.
|
||||
pinned_head: [num_priorities]?*Task = .{null} ** num_priorities,
|
||||
pinned_tail: [num_priorities]?*Task = .{null} ** num_priorities,
|
||||
pinned_bitmap: u8 = 0,
|
||||
};
|
||||
|
||||
const max_cpus = 64; // matches the discovery pool (src/device/acpi.zig)
|
||||
@@ -92,7 +101,7 @@ pub fn init(boot_priority: Priority) void {
|
||||
arch.setCpuLocal(@intFromPtr(pc));
|
||||
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
|
||||
pc.current = &tasks[0];
|
||||
pc.idle = create(idle, 0); // this core's idle task: always ready, lowest priority
|
||||
pc.idle = create(idle, 0, null); // this core's idle task: always ready, lowest priority
|
||||
arch.setTickHook(tick);
|
||||
}
|
||||
|
||||
@@ -142,47 +151,83 @@ pub fn onlineCount() usize {
|
||||
return n;
|
||||
}
|
||||
|
||||
/// Make `t` ready. A pinned task (affinity set) goes to that core's pinned queue;
|
||||
/// everything else goes to the shared global queue.
|
||||
fn enqueue(t: *Task) void {
|
||||
t.next = null;
|
||||
const p: usize = t.priority;
|
||||
if (ready_tail[p]) |tail| tail.next = t else ready_head[p] = t;
|
||||
ready_tail[p] = t;
|
||||
ready_bitmap |= levelBit(t.priority);
|
||||
if (t.affinity) |cpu| {
|
||||
const pc = &cpus[cpu];
|
||||
enqueueTo(&pc.pinned_head, &pc.pinned_tail, &pc.pinned_bitmap, t);
|
||||
} else {
|
||||
enqueueTo(&ready_head, &ready_tail, &ready_bitmap, t);
|
||||
}
|
||||
}
|
||||
|
||||
fn dequeueHighest() ?*Task {
|
||||
if (ready_bitmap == 0) return null;
|
||||
const level: Priority = @intCast(num_priorities - 1 - @clz(ready_bitmap));
|
||||
const t = ready_head[level].?;
|
||||
ready_head[level] = t.next;
|
||||
if (ready_head[level] == null) {
|
||||
ready_tail[level] = null;
|
||||
ready_bitmap &= ~levelBit(level);
|
||||
fn enqueueTo(head: *[num_priorities]?*Task, tail: *[num_priorities]?*Task, bitmap: *u8, t: *Task) void {
|
||||
t.next = null;
|
||||
const p: usize = t.priority;
|
||||
if (tail[p]) |tl| tl.next = t else head[p] = t;
|
||||
tail[p] = t;
|
||||
bitmap.* |= @as(u8, 1) << t.priority;
|
||||
}
|
||||
|
||||
/// The highest non-empty priority level in a bitmap, or -1 if empty.
|
||||
fn topLevel(bitmap: u8) i32 {
|
||||
if (bitmap == 0) return -1;
|
||||
return @as(i32, num_priorities - 1) - @as(i32, @clz(bitmap));
|
||||
}
|
||||
|
||||
/// Pick the highest-priority ready task for core `pc`: the better of the global queue
|
||||
/// and this core's pinned queue. Still O(1) (two `clz` and a compare). A pinned task
|
||||
/// wins an equal-priority tie, so it can't be starved by global work at its level.
|
||||
fn dequeueHighest(pc: *PerCpu) ?*Task {
|
||||
const g = topLevel(ready_bitmap);
|
||||
const p = topLevel(pc.pinned_bitmap);
|
||||
if (g < 0 and p < 0) return null;
|
||||
if (p >= g) return dequeueFrom(&pc.pinned_head, &pc.pinned_tail, &pc.pinned_bitmap, @intCast(p));
|
||||
return dequeueFrom(&ready_head, &ready_tail, &ready_bitmap, @intCast(g));
|
||||
}
|
||||
|
||||
fn dequeueFrom(head: *[num_priorities]?*Task, tail: *[num_priorities]?*Task, bitmap: *u8, level: usize) ?*Task {
|
||||
const t = head[level].?;
|
||||
head[level] = t.next;
|
||||
if (head[level] == null) {
|
||||
tail[level] = null;
|
||||
bitmap.* &= ~(@as(u8, 1) << @intCast(level));
|
||||
}
|
||||
t.next = null;
|
||||
return t;
|
||||
}
|
||||
|
||||
fn levelBit(p: Priority) u8 {
|
||||
return @as(u8, 1) << p;
|
||||
}
|
||||
|
||||
/// Create a task that runs `entry` at `priority`. It becomes ready immediately.
|
||||
/// Takes the kernel lock: it mutates the shared task table and ready queues and
|
||||
/// allocates from the (non-thread-safe) heap, so on SMP it must be serialised.
|
||||
/// Create a task that runs `entry` at `priority`, runnable on any core. It becomes
|
||||
/// ready immediately. Takes the kernel lock: it mutates the shared task table and
|
||||
/// ready queues and allocates from the (non-thread-safe) heap, so on SMP it must be
|
||||
/// serialised.
|
||||
pub fn spawn(entry: *const fn () void, priority: Priority) void {
|
||||
const flags = sync.enter();
|
||||
_ = create(entry, priority);
|
||||
_ = create(entry, priority, null);
|
||||
sync.leave(flags);
|
||||
}
|
||||
|
||||
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be
|
||||
/// the single-threaded boot path). Returns the new task so a core can keep a handle
|
||||
/// to its idle task.
|
||||
fn create(entry: *const fn () void, priority: Priority) *Task {
|
||||
/// Like `spawn`, but **pins** the task to core `cpu` — it will only ever run there.
|
||||
/// Returns true if pinned; false if `cpu` isn't a valid, online core, in which case
|
||||
/// the task is still created but left unpinned (so it runs *somewhere* rather than
|
||||
/// stranding in a queue no core services). Callers that require the pin (e.g. tests)
|
||||
/// should check the result.
|
||||
pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const ok = cpu < max_cpus and cpus[cpu].online;
|
||||
_ = create(entry, priority, if (ok) cpu else null);
|
||||
return ok;
|
||||
}
|
||||
|
||||
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the
|
||||
/// single-threaded boot path). `affinity` pins the task to a core (null = any).
|
||||
/// Returns the new task so a core can keep a handle to its idle task.
|
||||
fn create(entry: *const fn () void, priority: Priority, affinity: ?u32) *Task {
|
||||
const t = freeSlot() orelse @panic("sched: task table full");
|
||||
const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack");
|
||||
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack };
|
||||
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack, .affinity = affinity };
|
||||
next_id += 1;
|
||||
const top = @intFromPtr(stack.ptr) + stack.len;
|
||||
t.rsp = arch.initTaskStack(top, @intFromPtr(entry));
|
||||
@@ -209,7 +254,7 @@ fn schedule() void {
|
||||
prev.state = .ready;
|
||||
enqueue(prev); // back of its level's queue (round-robin)
|
||||
}
|
||||
const next = dequeueHighest() orelse {
|
||||
const next = dequeueHighest(pc) orelse {
|
||||
prev.state = .running; // nothing else ready — keep running
|
||||
return;
|
||||
};
|
||||
@@ -292,17 +337,23 @@ pub fn wait(wq: *WaitQueue) void {
|
||||
/// Wake the highest-priority waiter on `wq`, preempting if it outranks us.
|
||||
pub fn wake(wq: *WaitQueue) void {
|
||||
const flags = sync.enter();
|
||||
const pc = thisCpu();
|
||||
wakeLocked(wq);
|
||||
// If a higher-priority task is now ready, run it immediately.
|
||||
if (highestReadyPriority()) |p| {
|
||||
if (p > cur().priority) schedule();
|
||||
// If a task this core would now pick outranks the running one, run it at once.
|
||||
// (A waiter pinned to *another* core isn't counted — that core picks it up on its
|
||||
// next tick; this core doesn't preempt for work it can't run.)
|
||||
if (highestReadyPriority(pc)) |p| {
|
||||
if (p > pc.current.priority) schedule();
|
||||
}
|
||||
sync.leave(flags);
|
||||
}
|
||||
|
||||
fn highestReadyPriority() ?Priority {
|
||||
if (ready_bitmap == 0) return null;
|
||||
return @intCast(num_priorities - 1 - @clz(ready_bitmap));
|
||||
/// The highest-priority task core `pc` could run right now — the better of the global
|
||||
/// queue and this core's pinned queue — or null if it would fall back to idle.
|
||||
fn highestReadyPriority(pc: *PerCpu) ?Priority {
|
||||
const top = @max(topLevel(ready_bitmap), topLevel(pc.pinned_bitmap));
|
||||
if (top < 0) return null;
|
||||
return @intCast(top);
|
||||
}
|
||||
|
||||
/// Wake any sleeping task whose deadline has passed. Bounded by the task count,
|
||||
@@ -342,7 +393,7 @@ pub fn exit() noreturn {
|
||||
_ = sync.enter();
|
||||
const pc = thisCpu();
|
||||
pc.current.state = .free;
|
||||
const next = dequeueHighest() orelse @panic("sched: no task left to run");
|
||||
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
|
||||
next.state = .running;
|
||||
pc.current = next;
|
||||
var discard: usize = 0;
|
||||
@@ -356,8 +407,11 @@ pub fn currentId() u32 {
|
||||
|
||||
/// The dense index of the core this task is currently running on (0 = BSP). Reads
|
||||
/// per-CPU state, so a task calling it on different cores sees different values —
|
||||
/// which is how a test can prove work is running in parallel.
|
||||
/// which is how a test can prove work is running in parallel. Returns 0 if the GS
|
||||
/// base isn't published yet (a fault in very early boot, before `init`), so a fault
|
||||
/// reporter can call it unconditionally without a second fault.
|
||||
pub fn currentCpuIndex() u32 {
|
||||
if (arch.cpuLocal() == 0) return 0;
|
||||
return thisCpu().index;
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user