ipc_call and ipc_reply_wait grow a `send_cap` argument (r9) and a `received_cap` return (r8): an endpoint travels alongside a message, installed into the receiver's handle table. The transfer is a share, not a move — the endpoint's refcount is bumped and the sender keeps its handle. If the receiver's table is full the call fails -ENOSPC and the message is NOT delivered (a half-delivered capability is worse than a failed send); a bad handle fails -EBADF. Both directions carry a cap: a client's call hands one to the server (seen in the server's replyWait), and the server's reply hands one back (seen in the client's call return). This is the "open" primitive the driver model was blocked on: a bus driver mints a per-device endpoint and hands it to a class driver, giving it a private channel to one device without the 8-slot global name registry. Kernel: shareCapability in ipc-synchronous.zig at both copy points; new setSystemCallResult3 (r8, saved/restored by the syscall stub); Task gains ipc_send_cap / ipc_received_cap. Runtime: callCap + Reply, replyWait gains send_cap and Received.cap; plain call/replyWait delegate with no_cap. New abi.no_cap. New ipc-cap test (two kernel tasks exercise both directions, each verifying the endpoint it received is the same object shared, refcount bumped to 2). No class driver consumes callCap yet — it lands with the first one. Suite 37/37 plus host tests.
564 lines
24 KiB
Zig
564 lines
24 KiB
Zig
//! The scheduler: fixed-priority preemptive multitasking.
|
|
//!
|
|
//! Tasks are kernel threads (ring 0, each with its own stack). The **highest-
|
|
//! priority ready task always runs**; within a priority level, tasks round-robin.
|
|
//! Selection is O(1) — a bitmap of non-empty priority levels plus a FIFO queue per
|
|
//! level — which keeps scheduling deterministic, as a real-time kernel needs (see
|
|
//! docs/vision.md).
|
|
//!
|
|
//! Switching happens both cooperatively (`yield`) and preemptively (the timer
|
|
//! calls `tick`). See docs/scheduling.md for the interrupt-flag discipline that
|
|
//! makes those two paths coexist.
|
|
//!
|
|
//! Cross-core safety is the **big kernel lock** (`sync.zig`): every critical
|
|
//! section here runs under it, and it is held across a context switch and released
|
|
//! by the task that resumes (see sync.zig's hand-off rule). On a single core the
|
|
//! lock is never contended, so the behaviour is exactly the old interrupt-flag
|
|
//! model; it's what lets a second core enter `schedule()` without corrupting the
|
|
//! shared queues.
|
|
|
|
const std = @import("std");
|
|
const parameters = @import("parameters");
|
|
const architecture = @import("architecture");
|
|
const heap = @import("heap.zig");
|
|
const sync = @import("sync.zig");
|
|
|
|
/// Priority level: 0 (lowest) .. 7 (highest). 8 levels total.
|
|
pub const Priority = u3;
|
|
const number_priorities = 8;
|
|
|
|
const stack_size = parameters.kernel_stack_size; // each task's kernel stack
|
|
const maximum_tasks = parameters.maximum_tasks; // maximum tasks alive at once (static pool)
|
|
|
|
const State = enum { free, ready, running, blocked };
|
|
|
|
pub const Task = struct {
|
|
id: u32 = 0,
|
|
state: State = .free,
|
|
priority: Priority = 0,
|
|
sp: usize = 0, // saved stack pointer, valid while not running
|
|
stack: []u8 = &.{},
|
|
kstack_top: usize = 0, // top of `stack` (== TSS.rsp0 for a user task); 0 = none
|
|
wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping
|
|
affinity: ?u32 = null, // null = runs on any core; else the index of its pinned core
|
|
// Physical root of this task's address space, or 0 for a kernel task (which
|
|
// runs on the shared kernel page tables). A user task carries its own.
|
|
aspace: u64 = 0,
|
|
user_ip: u64 = 0, // user-mode entry point (user task only)
|
|
user_sp: u64 = 0, // user-mode stack pointer (user task only)
|
|
// Next free virtual address in this task's mmap grant arena (0 = uninitialised;
|
|
// process.zig lazily seeds it to the arena base on the first mmap). Bumped up
|
|
// as the user heap grows; user task only.
|
|
heap_next: u64 = 0,
|
|
// Next free virtual address in this task's MMIO-grant arena (PML4[226]; 0 =
|
|
// uninitialised, process.zig seeds it on the first mmio_map). User task only.
|
|
device_map_next: u64 = 0,
|
|
// --- synchronous IPC (ipc_sync.zig) ---
|
|
// Per-process handle table: small-int handle -> *ipc_sync.Endpoint, kept
|
|
// opaque here so the scheduler and IPC modules don't import each other.
|
|
handles: [ipc_maximum_handles]?*anyopaque = .{null} ** ipc_maximum_handles,
|
|
// A server holds the caller it currently owes a reply to (set by ReplyWait's
|
|
// receive, cleared when it replies). A client, while blocked in Call, records
|
|
// its message + reply buffers here and its result lands in `ipc_status`.
|
|
ipc_client: ?*Task = null,
|
|
ipc_send_ptr: u64 = 0, // client: outgoing message (vaddr in this task's AS)
|
|
ipc_send_len: u64 = 0,
|
|
ipc_reply_ptr: u64 = 0, // client: reply buffer (vaddr)
|
|
ipc_reply_cap: u64 = 0,
|
|
ipc_status: i64 = 0, // client: reply length / -errno, written by the replier
|
|
ipc_send_cap: u64 = ~@as(u64, 0), // handle to transfer with this message (abi.no_cap = none)
|
|
ipc_received_cap: u64 = ~@as(u64, 0), // client: handle the reply's transferred cap landed at (abi.no_cap = none)
|
|
next: ?*Task = null, // ready-queue link (also the endpoint sender-FIFO link)
|
|
};
|
|
|
|
/// Size of each task's IPC handle table. Kept here (not in ipc_sync.zig) because
|
|
/// it dimensions a field of `Task`; ipc_sync.zig re-exports it.
|
|
pub const ipc_maximum_handles = 16;
|
|
|
|
var tasks = [_]Task{.{}} ** maximum_tasks;
|
|
var next_id: u32 = 1;
|
|
|
|
/// Per-CPU scheduler state: the task each core is running, its own idle task, and a
|
|
/// queue of tasks **pinned** to it. One entry per core; the architecture layer stashes a
|
|
/// pointer to the *running* core's entry in the GS base, so `thisCpu()` fetches it
|
|
/// with a single read and no lock.
|
|
///
|
|
/// Most work stays in the **global** ready queue (below), which any idle core pulls
|
|
/// from — work-conserving. A task given an *affinity* instead goes to that core's
|
|
/// `pinned_*` queue and is only ever run there (no surprise migration — the more
|
|
/// real-time-predictable model, docs/smp.md). The two queues are merged at selection
|
|
/// time. Both are still mutated only under the big kernel lock, so one core enqueuing
|
|
/// into another core's pinned queue is safe.
|
|
pub const PerCpu = struct {
|
|
current: *Task = undefined, // the task running on this core
|
|
idle: *Task = undefined, // this core's idle task (always ready, lowest priority)
|
|
hw_id: u32 = 0, // the core's hardware id (Local APIC id on x86_64)
|
|
index: u32 = 0, // dense 0-based core index
|
|
online: bool = false, // has this core finished bring-up?
|
|
loaded_aspace: u64 = 0, // the address-space root currently loaded on this core
|
|
// Tasks pinned to this core (affinity == index), per priority level + bitmap.
|
|
pinned_head: [number_priorities]?*Task = .{null} ** number_priorities,
|
|
pinned_tail: [number_priorities]?*Task = .{null} ** number_priorities,
|
|
pinned_bitmap: u8 = 0,
|
|
};
|
|
|
|
const maximum_cpus = parameters.maximum_cpus;
|
|
var cpus = [_]PerCpu{.{}} ** maximum_cpus;
|
|
|
|
/// This core's per-CPU state, via the architecture layer's GS-base pointer. Valid only once
|
|
/// this core has run its scheduler bring-up (BSP in `init`, AP in `secondaryInit`).
|
|
inline fn thisCpu() *PerCpu {
|
|
return @ptrFromInt(architecture.cpuLocal());
|
|
}
|
|
|
|
/// The task running on this core — the per-CPU replacement for the old global
|
|
/// `current`. A convenience reader; writes go through `thisCpu().current`.
|
|
pub inline fn current() *Task {
|
|
return thisCpu().current;
|
|
}
|
|
|
|
// Per-priority FIFO ready queues, and a bitmap of which levels are non-empty. These
|
|
// are shared across all cores and mutated only under the big kernel lock.
|
|
var ready_head: [number_priorities]?*Task = .{null} ** number_priorities;
|
|
var ready_tail: [number_priorities]?*Task = .{null} ** number_priorities;
|
|
var ready_bitmap: u8 = 0;
|
|
|
|
var preemption_enabled = true;
|
|
|
|
/// Bring up scheduling on the bootstrap processor: register the currently-running
|
|
/// kernel context as task 0, publish this core's per-CPU state (via the GS base),
|
|
/// give the core an idle task, and hook the timer for preemption. Runs once, at
|
|
/// boot, before interrupts are enabled — so no lock is needed here.
|
|
pub fn init(boot_priority: Priority) void {
|
|
const pc = &cpus[0];
|
|
pc.* = .{ .index = 0, .online = true, .loaded_aspace = architecture.kernelPageTable() };
|
|
architecture.setCpuLocal(0, @intFromPtr(pc));
|
|
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
|
|
pc.current = &tasks[0];
|
|
pc.idle = create(idle, 0, null); // this core's idle task: always ready, lowest priority
|
|
architecture.setTickHook(tick);
|
|
}
|
|
|
|
/// The idle task: run when every other task is blocked or sleeping. `hlt` waits
|
|
/// for the next interrupt at near-zero power (see docs/halting.md).
|
|
fn idle() void {
|
|
while (true) asm volatile ("hlt");
|
|
}
|
|
|
|
/// Reserve and initialise the per-CPU slot for an application processor at dense
|
|
/// `index` (1-based; 0 is the BSP) with hardware id `hw_id`, and return a
|
|
/// pointer the architecture bring-up hands to the core (it publishes it in its GS base).
|
|
/// Called on the BSP before waking each AP; the AP marks itself `online`.
|
|
pub fn prepareSecondary(index: usize, hw_id: u32) *PerCpu {
|
|
const pc = &cpus[index];
|
|
pc.* = .{ .index = @intCast(index), .hw_id = hw_id, .online = false };
|
|
return pc;
|
|
}
|
|
|
|
/// Entry for an application processor once the architecture layer has set up its per-CPU
|
|
/// tables, LAPIC, and timer. It turns this bring-up context into the core's idle task
|
|
/// (as task 0 is for the BSP), marks the core online, and enters the run loop: with
|
|
/// interrupts enabled the timer preempts this idle context into whatever the global
|
|
/// ready queue offers, so the core runs real work in parallel with the others. The
|
|
/// `.c` calling convention lets the architecture trampoline path jump here. Never returns.
|
|
pub fn secondaryMain() callconv(.c) noreturn {
|
|
const flags = sync.enter();
|
|
const pc = thisCpu();
|
|
const t = freeSlot() orelse @panic("sched: task table full (AP idle task)");
|
|
t.* = .{ .id = next_id, .state = .running, .priority = 0 };
|
|
next_id += 1;
|
|
pc.current = t;
|
|
pc.idle = t;
|
|
pc.online = true;
|
|
pc.loaded_aspace = architecture.kernelPageTable(); // the AP adopted the kernel tables at bring-up
|
|
sync.leave(flags);
|
|
|
|
architecture.enableInterrupts(); // the timer now preempts this idle context into work
|
|
while (true) asm volatile ("hlt"); // idle when this core has nothing ready
|
|
}
|
|
|
|
/// Number of cores that have finished bring-up (the BSP plus every online AP).
|
|
pub fn onlineCount() usize {
|
|
var n: usize = 0;
|
|
for (&cpus) |*pc| {
|
|
if (pc.online) n += 1;
|
|
}
|
|
return n;
|
|
}
|
|
|
|
/// Make `t` ready. A pinned task (affinity set) goes to that core's pinned queue;
|
|
/// everything else goes to the shared global queue.
|
|
fn enqueue(t: *Task) void {
|
|
if (t.affinity) |cpu| {
|
|
const pc = &cpus[cpu];
|
|
enqueueTo(&pc.pinned_head, &pc.pinned_tail, &pc.pinned_bitmap, t);
|
|
} else {
|
|
enqueueTo(&ready_head, &ready_tail, &ready_bitmap, t);
|
|
}
|
|
}
|
|
|
|
fn enqueueTo(head: *[number_priorities]?*Task, tail: *[number_priorities]?*Task, bitmap: *u8, t: *Task) void {
|
|
t.next = null;
|
|
const p: usize = t.priority;
|
|
if (tail[p]) |tl| tl.next = t else head[p] = t;
|
|
tail[p] = t;
|
|
bitmap.* |= @as(u8, 1) << t.priority;
|
|
}
|
|
|
|
/// The highest non-empty priority level in a bitmap, or -1 if empty.
|
|
fn topLevel(bitmap: u8) i32 {
|
|
if (bitmap == 0) return -1;
|
|
return @as(i32, number_priorities - 1) - @as(i32, @clz(bitmap));
|
|
}
|
|
|
|
/// Pick the highest-priority ready task for core `pc`: the better of the global queue
|
|
/// and this core's pinned queue. Still O(1) (two `clz` and a compare). A pinned task
|
|
/// wins an equal-priority tie, so it can't be starved by global work at its level.
|
|
fn dequeueHighest(pc: *PerCpu) ?*Task {
|
|
const g = topLevel(ready_bitmap);
|
|
const p = topLevel(pc.pinned_bitmap);
|
|
if (g < 0 and p < 0) return null;
|
|
if (p >= g) return dequeueFrom(&pc.pinned_head, &pc.pinned_tail, &pc.pinned_bitmap, @intCast(p));
|
|
return dequeueFrom(&ready_head, &ready_tail, &ready_bitmap, @intCast(g));
|
|
}
|
|
|
|
fn dequeueFrom(head: *[number_priorities]?*Task, tail: *[number_priorities]?*Task, bitmap: *u8, level: usize) ?*Task {
|
|
const t = head[level].?;
|
|
head[level] = t.next;
|
|
if (head[level] == null) {
|
|
tail[level] = null;
|
|
bitmap.* &= ~(@as(u8, 1) << @intCast(level));
|
|
}
|
|
t.next = null;
|
|
return t;
|
|
}
|
|
|
|
/// Create a task that runs `entry` at `priority`, runnable on any core. It becomes
|
|
/// ready immediately. Takes the kernel lock: it mutates the shared task table and
|
|
/// ready queues and allocates from the (non-thread-safe) heap, so on SMP it must be
|
|
/// serialised.
|
|
pub fn spawn(entry: *const fn () void, priority: Priority) void {
|
|
const flags = sync.enter();
|
|
_ = create(entry, priority, null);
|
|
sync.leave(flags);
|
|
}
|
|
|
|
/// Like `spawn`, but **pins** the task to core `cpu` — it will only ever run there.
|
|
/// Returns true if pinned; false if `cpu` isn't a valid, online core, in which case
|
|
/// the task is still created but left unpinned (so it runs *somewhere* rather than
|
|
/// stranding in a queue no core services). Callers that require the pin (e.g. tests)
|
|
/// should check the result.
|
|
pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
|
|
const flags = sync.enter();
|
|
defer sync.leave(flags);
|
|
const ok = cpu < maximum_cpus and cpus[cpu].online;
|
|
_ = create(entry, priority, if (ok) cpu else null);
|
|
return ok;
|
|
}
|
|
|
|
/// Spawn a **user** task: a task with its own address space (`aspace`) that starts
|
|
/// in user mode at `entry` on `user_sp`. It gets a fresh kernel stack for
|
|
/// syscalls/interrupts, and its first switch-in lands in `user_task_trampoline`.
|
|
/// Returns false (creating nothing) if the table is full or out of memory.
|
|
/// **Caller must hold the kernel lock** (the loader that builds `aspace` holds it
|
|
/// across the whole spawn, so the address space and the task appear atomically).
|
|
pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority) bool {
|
|
const t = freeSlot() orelse return false;
|
|
const stack = heap.allocator().alloc(u8, stack_size) catch return false;
|
|
t.* = .{
|
|
.id = next_id,
|
|
.state = .ready,
|
|
.priority = priority,
|
|
.stack = stack,
|
|
.aspace = aspace,
|
|
.user_ip = entry,
|
|
.user_sp = user_sp,
|
|
};
|
|
next_id += 1;
|
|
const top = @intFromPtr(stack.ptr) + stack.len;
|
|
t.kstack_top = top;
|
|
// First switch-in lands in startUserTask (no register smuggling — it reads
|
|
// the user entry/stack from the Task itself).
|
|
t.sp = architecture.initTaskStack(top, @intFromPtr(&startUserTask));
|
|
enqueue(t);
|
|
return true;
|
|
}
|
|
|
|
/// The first thing a fresh user task runs (in ring 0, via task_trampoline). It
|
|
/// drops to ring 3 at the task's recorded entry/stack. Reading them from the
|
|
/// Task avoids smuggling values through callee-saved registers across the
|
|
/// context switch and lock release.
|
|
fn startUserTask() void {
|
|
const t = current();
|
|
var buffer: [96]u8 = undefined;
|
|
architecture.serialWrite(std.fmt.bufPrint(&buffer, "DBG startUserTask ip=0x{x} sp=0x{x} aspace=0x{x} kstack=0x{x}\n", .{ t.user_ip, t.user_sp, t.aspace, t.kstack_top }) catch "");
|
|
architecture.jumpToUser(t.user_ip, t.user_sp); // noreturn
|
|
}
|
|
|
|
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the
|
|
/// single-threaded boot path). `affinity` pins the task to a core (null = any).
|
|
/// Returns the new task so a core can keep a handle to its idle task.
|
|
fn create(entry: *const fn () void, priority: Priority, affinity: ?u32) *Task {
|
|
const t = freeSlot() orelse @panic("sched: task table full");
|
|
const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack");
|
|
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack, .affinity = affinity };
|
|
next_id += 1;
|
|
const top = @intFromPtr(stack.ptr) + stack.len;
|
|
t.kstack_top = top;
|
|
t.sp = architecture.initTaskStack(top, @intFromPtr(entry));
|
|
enqueue(t);
|
|
return t;
|
|
}
|
|
|
|
fn freeSlot() ?*Task {
|
|
for (&tasks) |*t| {
|
|
if (t.state == .free) return t;
|
|
}
|
|
return null;
|
|
}
|
|
|
|
/// Pick the highest-priority ready task and switch this core to it. The big kernel
|
|
/// lock must be held by the caller (which also keeps local interrupts disabled);
|
|
/// it serialises every core's scheduling, so no other core can touch the shared
|
|
/// queues while we requeue `previous` and dequeue `next`. A dequeued task is `.ready`,
|
|
/// never running elsewhere, so two cores never run the same task.
|
|
fn schedule() void {
|
|
const pc = thisCpu();
|
|
const previous = pc.current;
|
|
if (previous.state == .running) {
|
|
previous.state = .ready;
|
|
enqueue(previous); // back of its level's queue (round-robin)
|
|
}
|
|
const next = dequeueHighest(pc) orelse {
|
|
previous.state = .running; // nothing else ready — keep running
|
|
return;
|
|
};
|
|
next.state = .running;
|
|
pc.current = next;
|
|
if (next != previous) switchTo(pc, &previous.sp, next);
|
|
}
|
|
|
|
/// Make `next` this core's running task: publish its kernel stack (TSS.rsp0, so a
|
|
/// user-mode interrupt lands on a good stack) and its address space (only when
|
|
/// it differs from what's loaded — every page-table switch is a full TLB flush),
|
|
/// then switch registers/stacks. Kernel tasks (aspace == 0, no kstack_top used
|
|
/// from user mode) resolve to the shared kernel page tables and skip the kernel-
|
|
/// stack write, so this is a no-op beyond the register switch for a pure-kernel
|
|
/// workload. The big kernel lock is held and interrupts are off throughout, so no
|
|
/// interrupt can observe a half-updated (kernel stack, address space) pair.
|
|
/// `save_sp` receives the outgoing task's stack pointer.
|
|
fn switchTo(pc: *PerCpu, save_sp: *usize, next: *Task) void {
|
|
if (next.kstack_top != 0) architecture.setKernelStack(pc.index, next.kstack_top);
|
|
const want = if (next.aspace != 0) next.aspace else architecture.kernelPageTable();
|
|
if (want != pc.loaded_aspace) {
|
|
architecture.loadPageTable(want);
|
|
pc.loaded_aspace = want;
|
|
}
|
|
architecture.switchContext(save_sp, next.sp);
|
|
}
|
|
|
|
/// Voluntarily give up the CPU to the next ready task.
|
|
pub fn yield() void {
|
|
const flags = sync.enter();
|
|
schedule();
|
|
sync.leave(flags);
|
|
}
|
|
|
|
/// Block the current task for `ms` milliseconds, then let it become runnable
|
|
/// again. The idle task (or other work) runs in the meantime.
|
|
pub fn sleep(ms: u64) void {
|
|
const flags = sync.enter();
|
|
const t = current();
|
|
t.wake_at = architecture.millis() + ms;
|
|
t.state = .blocked;
|
|
schedule(); // current is blocked, so schedule() won't re-enqueue it
|
|
sync.leave(flags);
|
|
}
|
|
|
|
// --- event-based blocking -------------------------------------------------
|
|
//
|
|
// A WaitQueue is a set of tasks blocked waiting for something (a resource, a
|
|
// message). Tasks link into it through the same `next` field the ready queues
|
|
// use — a task is in exactly one queue at a time. These are the primitive locks,
|
|
// semaphores and IPC channels are built on.
|
|
|
|
pub const WaitQueue = struct {
|
|
head: ?*Task = null,
|
|
};
|
|
|
|
/// Block the current task on `wait_queue` and switch away. Precondition: the big kernel
|
|
/// lock is held (so a condition can be checked and the block committed atomically;
|
|
/// it also keeps local interrupts disabled). On return — when woken — the lock is
|
|
/// still held.
|
|
pub fn waitLocked(wait_queue: *WaitQueue) void {
|
|
const t = current();
|
|
t.state = .blocked;
|
|
t.next = wait_queue.head;
|
|
wait_queue.head = t;
|
|
schedule();
|
|
}
|
|
|
|
/// Move the highest-priority waiter on `wait_queue` (if any) to the ready queue.
|
|
/// Precondition: the big kernel lock is held. Does not preempt — the caller decides.
|
|
pub fn wakeLocked(wait_queue: *WaitQueue) void {
|
|
// Find the highest-priority waiter (bounded scan) and unlink it.
|
|
var best_previous: ?*Task = null;
|
|
var best: ?*Task = null;
|
|
var previous: ?*Task = null;
|
|
var node = wait_queue.head;
|
|
while (node) |t| : ({
|
|
previous = t;
|
|
node = t.next;
|
|
}) {
|
|
if (best == null or t.priority > best.?.priority) {
|
|
best = t;
|
|
best_previous = previous;
|
|
}
|
|
}
|
|
const t = best orelse return;
|
|
if (best_previous) |p| p.next = t.next else wait_queue.head = t.next;
|
|
t.state = .ready;
|
|
enqueue(t);
|
|
}
|
|
|
|
/// Block the current task and switch away, without putting it on any wait queue —
|
|
/// the caller has already linked it wherever it belongs (e.g. an endpoint's sender
|
|
/// FIFO). Precondition: the big kernel lock is held; still held on return (when the
|
|
/// task is made ready again). The IPC layer's counterpart to `waitLocked`.
|
|
pub fn blockCurrentLocked() void {
|
|
current().state = .blocked;
|
|
schedule();
|
|
}
|
|
|
|
/// Make a specific (currently blocked) task ready to run again. Precondition: the
|
|
/// big kernel lock is held. Used by the IPC layer to wake a specific caller/server
|
|
/// rather than "some waiter on a queue".
|
|
pub fn readyLocked(t: *Task) void {
|
|
t.state = .ready;
|
|
enqueue(t);
|
|
}
|
|
|
|
/// Block on `wait_queue` (a self-contained critical section).
|
|
pub fn wait(wait_queue: *WaitQueue) void {
|
|
const flags = sync.enter();
|
|
waitLocked(wait_queue);
|
|
sync.leave(flags);
|
|
}
|
|
|
|
/// Wake the highest-priority waiter on `wait_queue`, preempting if it outranks us.
|
|
pub fn wake(wait_queue: *WaitQueue) void {
|
|
const flags = sync.enter();
|
|
const pc = thisCpu();
|
|
wakeLocked(wait_queue);
|
|
// If a task this core would now pick outranks the running one, run it at once.
|
|
// (A waiter pinned to *another* core isn't counted — that core picks it up on its
|
|
// next tick; this core doesn't preempt for work it can't run.)
|
|
if (highestReadyPriority(pc)) |p| {
|
|
if (p > pc.current.priority) schedule();
|
|
}
|
|
sync.leave(flags);
|
|
}
|
|
|
|
/// The highest-priority task core `pc` could run right now — the better of the global
|
|
/// queue and this core's pinned queue — or null if it would fall back to idle.
|
|
fn highestReadyPriority(pc: *PerCpu) ?Priority {
|
|
const top = @max(topLevel(ready_bitmap), topLevel(pc.pinned_bitmap));
|
|
if (top < 0) return null;
|
|
return @intCast(top);
|
|
}
|
|
|
|
/// Wake any sleeping task whose deadline has passed. Bounded by the task count,
|
|
/// so it stays deterministic. Called from the timer tick (interrupts disabled).
|
|
fn wakeExpired() void {
|
|
const now = architecture.millis();
|
|
for (&tasks) |*t| {
|
|
if (t.state == .blocked and t.wake_at != 0 and now >= t.wake_at) {
|
|
t.wake_at = 0;
|
|
t.state = .ready;
|
|
enqueue(t);
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Called from the timer interrupt (interrupts already disabled): wake due
|
|
/// sleepers, then preempt. Takes the kernel lock like any other critical section,
|
|
/// but releases it *without* touching the interrupt flag — the handler's `iretq`
|
|
/// restores the interrupted context's flags, so re-enabling here would open a
|
|
/// nested-interrupt window before the return.
|
|
pub fn tick() void {
|
|
_ = sync.enter();
|
|
wakeExpired();
|
|
if (preemption_enabled) schedule();
|
|
sync.leaveIsr();
|
|
}
|
|
|
|
/// Enable or disable timer-driven preemption (cooperative-only when off).
|
|
pub fn setPreemption(enabled: bool) void {
|
|
preemption_enabled = enabled;
|
|
}
|
|
|
|
/// End the current task and switch away for good; never returns. The task's stack
|
|
/// is leaked for now (no reaper yet). Acquires the kernel lock and hands it off to
|
|
/// the task we switch into (which releases it) — this frame never returns to leave.
|
|
pub fn exit() noreturn {
|
|
_ = sync.enter();
|
|
const pc = thisCpu();
|
|
pc.current.state = .free;
|
|
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
|
|
next.state = .running;
|
|
pc.current = next;
|
|
var discard: usize = 0;
|
|
switchTo(pc, &discard, next);
|
|
unreachable;
|
|
}
|
|
|
|
/// End the current **user** task: free its address space, then exit. Runs on the
|
|
/// dying task's kernel stack (in the shared kernel half, so it survives the CR3
|
|
/// switch to the kernel tables that must happen before we free the process's own
|
|
/// tables — we can't free the page tables we're standing on). The kernel stack
|
|
/// itself is leaked, as in `exit` (no reaper yet). Never returns.
|
|
pub fn exitUser() noreturn {
|
|
_ = sync.enter();
|
|
const pc = thisCpu();
|
|
const dying = pc.current;
|
|
const as = dying.aspace;
|
|
if (as != 0) {
|
|
const kroot = architecture.kernelPageTable();
|
|
architecture.loadPageTable(kroot); // off the process tables before freeing them
|
|
pc.loaded_aspace = kroot;
|
|
architecture.destroyAddressSpace(as);
|
|
}
|
|
dying.state = .free;
|
|
dying.aspace = 0;
|
|
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
|
|
next.state = .running;
|
|
pc.current = next;
|
|
var discard: usize = 0;
|
|
switchTo(pc, &discard, next);
|
|
unreachable;
|
|
}
|
|
|
|
/// Whether the running task is a user process (has its own address space).
|
|
pub fn currentIsUserProcess() bool {
|
|
return current().aspace != 0;
|
|
}
|
|
|
|
pub fn currentId() u32 {
|
|
return current().id;
|
|
}
|
|
|
|
/// The dense index of the core this task is currently running on (0 = BSP). Reads
|
|
/// per-CPU state, so a task calling it on different cores sees different values —
|
|
/// which is how a test can prove work is running in parallel. Returns 0 if the GS
|
|
/// base isn't published yet (a fault in very early boot, before `init`), so a fault
|
|
/// reporter can call it unconditionally without a second fault.
|
|
pub fn currentCpuIndex() u32 {
|
|
if (architecture.cpuLocal() == 0) return 0;
|
|
return thisCpu().index;
|
|
}
|
|
|
|
/// Change the running task's priority (takes effect next time it's enqueued).
|
|
pub fn setPriority(p: Priority) void {
|
|
current().priority = p;
|
|
}
|