Fixed-priority preemptive scheduler
This commit is contained in:
@@ -139,9 +139,19 @@ pub fn eoi() void {
|
||||
write(reg_eoi, 0);
|
||||
}
|
||||
|
||||
/// The timer interrupt handler: just count ticks for now.
|
||||
/// Optional callback run each tick (the scheduler registers it for preemption).
|
||||
var on_tick: ?*const fn () void = null;
|
||||
|
||||
pub fn setTickHook(hook: *const fn () void) void {
|
||||
on_tick = hook;
|
||||
}
|
||||
|
||||
/// The timer interrupt handler: advance the monotonic tick count, then run the
|
||||
/// tick hook (which may switch tasks). The interrupt is already acknowledged by
|
||||
/// the dispatcher before we get here, so a task switch here doesn't stall it.
|
||||
pub fn timerTick() void {
|
||||
tick_count +%= 1;
|
||||
if (on_tick) |hook| hook();
|
||||
}
|
||||
|
||||
/// Number of timer ticks so far. Volatile load: the count is bumped
|
||||
|
||||
@@ -98,6 +98,44 @@ pub fn disableInterrupts() void {
|
||||
asm volatile ("cli");
|
||||
}
|
||||
|
||||
/// Register a callback the timer interrupt invokes each tick (e.g. the scheduler).
|
||||
pub fn setTickHook(hook: *const fn () void) void {
|
||||
apic.setTickHook(hook);
|
||||
}
|
||||
|
||||
// --- context switching (for the scheduler) -------------------------------
|
||||
|
||||
/// Save the current task's registers/stack and resume `new_rsp`; the old stack
|
||||
/// pointer is written to `old_rsp`. Defined in isr.s.
|
||||
extern fn switch_context(old_rsp: *usize, new_rsp: usize) callconv(.c) void;
|
||||
|
||||
pub fn switchContext(old_rsp: *usize, new_rsp: usize) void {
|
||||
switch_context(old_rsp, new_rsp);
|
||||
}
|
||||
|
||||
/// Build the initial stack for a new task so that switching to it lands in
|
||||
/// `task_trampoline`, which then calls `entry`. Returns the saved stack pointer.
|
||||
/// The layout must match switch_context's push order (callee-saved, then the
|
||||
/// return address on top); `entry` is smuggled in via the r15 slot.
|
||||
pub fn initTaskStack(stack_top: usize, entry: usize) usize {
|
||||
const trampoline = @extern(*const anyopaque, .{ .name = "task_trampoline" });
|
||||
var sp = stack_top;
|
||||
const push = struct {
|
||||
fn f(p: *usize, value: usize) void {
|
||||
p.* -= @sizeOf(usize);
|
||||
@as(*usize, @ptrFromInt(p.*)).* = value;
|
||||
}
|
||||
}.f;
|
||||
push(&sp, @intFromPtr(trampoline)); // return address for switch_context's `ret`
|
||||
push(&sp, 0); // rbx
|
||||
push(&sp, 0); // rbp
|
||||
push(&sp, 0); // r12
|
||||
push(&sp, 0); // r13
|
||||
push(&sp, 0); // r14
|
||||
push(&sp, entry); // r15 -> task entry, read by task_trampoline
|
||||
return sp;
|
||||
}
|
||||
|
||||
/// Route CPU exceptions to `handler`, which receives the trap frame and does not
|
||||
/// return. Until set, faults just halt the core.
|
||||
pub fn setFaultHandler(handler: *const fn (*const CpuState) noreturn) void {
|
||||
|
||||
@@ -142,8 +142,12 @@ export fn interruptDispatch(state: *const CpuState) callconv(.c) void {
|
||||
if (state.vector < 32) {
|
||||
on_fault(state); // CPU exception — never returns
|
||||
} else if (handlers[state.vector]) |handler| {
|
||||
handler();
|
||||
// Acknowledge before running the handler: a handler that switches tasks
|
||||
// (the scheduler) may not return promptly, and the LAPIC mustn't wait on
|
||||
// it to deliver the next interrupt. Fine for edge-triggered sources like
|
||||
// the timer; a level-triggered device would need EOI after handling.
|
||||
apic.eoi();
|
||||
handler();
|
||||
}
|
||||
// else: spurious/unhandled device interrupt — don't acknowledge it
|
||||
}
|
||||
|
||||
@@ -39,6 +39,38 @@ load_tr:
|
||||
ltr %di
|
||||
ret
|
||||
|
||||
# switch_context(rdi = &old_task.rsp, rsi = new_task.rsp)
|
||||
# Cooperative context switch: save the callee-saved registers on the current
|
||||
# stack, stash the stack pointer in the old task, load the new task's stack
|
||||
# pointer, restore its callee-saved registers, and return into it. Caller-saved
|
||||
# registers are the compiler's responsibility (this looks like a normal call).
|
||||
.global switch_context
|
||||
switch_context:
|
||||
push %rbx
|
||||
push %rbp
|
||||
push %r12
|
||||
push %r13
|
||||
push %r14
|
||||
push %r15
|
||||
mov %rsp, (%rdi) # save old stack pointer into old_task.rsp
|
||||
mov %rsi, %rsp # switch to the new task's stack
|
||||
pop %r15
|
||||
pop %r14
|
||||
pop %r13
|
||||
pop %r12
|
||||
pop %rbp
|
||||
pop %rbx
|
||||
ret # return into the new task's saved instruction pointer
|
||||
|
||||
# task_trampoline: the first thing a freshly-spawned task runs. init_task_stack
|
||||
# leaves its entry function in r15. New tasks start with interrupts enabled.
|
||||
.global task_trampoline
|
||||
task_trampoline:
|
||||
sti
|
||||
call *%r15 # call the task entry (fn() void)
|
||||
1: hlt # if the entry returns, idle (still preemptible)
|
||||
jmp 1b
|
||||
|
||||
# Stub for a vector the CPU does NOT push an error code for: push a dummy 0.
|
||||
.macro STUB_NOERR vec
|
||||
.global isr\vec
|
||||
|
||||
+7
-1
@@ -4,6 +4,7 @@ const arch = @import("arch");
|
||||
const console = @import("console.zig");
|
||||
const pmm = @import("pmm.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const sched = @import("sched.zig");
|
||||
const tests = @import("tests.zig");
|
||||
const build_options = @import("build_options");
|
||||
const BootInfo = danos.BootInfo;
|
||||
@@ -97,7 +98,12 @@ fn kmain(boot_info: *const BootInfo) noreturn {
|
||||
heap.init();
|
||||
con.write("\ndanos: kernel heap online\n");
|
||||
|
||||
// Start the timer and unmask interrupts — the kernel now has a heartbeat.
|
||||
// Register the current context as the first task before enabling preemption.
|
||||
sched.init(4);
|
||||
con.write("danos: scheduler online\n");
|
||||
|
||||
// Start the timer and unmask interrupts — the kernel now has a heartbeat, and
|
||||
// the timer preempts among tasks.
|
||||
arch.startTimer();
|
||||
arch.enableInterrupts();
|
||||
con.print("danos: timer online ({d} Hz tick, LAPIC {d} MHz measured)\n", .{ arch.timer_hz, arch.lapicHz() / 1_000_000 });
|
||||
|
||||
+151
@@ -0,0 +1,151 @@
|
||||
//! The scheduler: fixed-priority preemptive multitasking.
|
||||
//!
|
||||
//! Tasks are kernel threads (ring 0, each with its own stack). The **highest-
|
||||
//! priority ready task always runs**; within a priority level, tasks round-robin.
|
||||
//! Selection is O(1) — a bitmap of non-empty priority levels plus a FIFO queue per
|
||||
//! level — which keeps scheduling deterministic, as a real-time kernel needs (see
|
||||
//! docs/vision.md).
|
||||
//!
|
||||
//! Switching happens both cooperatively (`yield`) and preemptively (the timer
|
||||
//! calls `tick`). See docs/scheduling.md for the interrupt-flag discipline that
|
||||
//! makes those two paths coexist.
|
||||
|
||||
const std = @import("std");
|
||||
const arch = @import("arch");
|
||||
const heap = @import("heap.zig");
|
||||
|
||||
/// Priority level: 0 (lowest) .. 7 (highest). 8 levels total.
|
||||
pub const Priority = u3;
|
||||
const num_priorities = 8;
|
||||
|
||||
const stack_size = 16 * 1024; // per-task kernel stack
|
||||
const max_tasks = 16;
|
||||
|
||||
const State = enum { free, ready, running };
|
||||
|
||||
const Task = struct {
|
||||
id: u32 = 0,
|
||||
state: State = .free,
|
||||
priority: Priority = 0,
|
||||
rsp: usize = 0, // saved stack pointer, valid while not running
|
||||
stack: []u8 = &.{},
|
||||
next: ?*Task = null, // ready-queue link
|
||||
};
|
||||
|
||||
var tasks = [_]Task{.{}} ** max_tasks;
|
||||
var current: *Task = undefined;
|
||||
var next_id: u32 = 1;
|
||||
|
||||
// Per-priority FIFO ready queues, and a bitmap of which levels are non-empty.
|
||||
var ready_head: [num_priorities]?*Task = .{null} ** num_priorities;
|
||||
var ready_tail: [num_priorities]?*Task = .{null} ** num_priorities;
|
||||
var ready_bitmap: u8 = 0;
|
||||
|
||||
var preemption_enabled = true;
|
||||
|
||||
/// Register the currently-running kernel context as the first task, and hook the
|
||||
/// timer for preemption.
|
||||
pub fn init(boot_priority: Priority) void {
|
||||
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
|
||||
current = &tasks[0];
|
||||
arch.setTickHook(tick);
|
||||
}
|
||||
|
||||
fn enqueue(t: *Task) void {
|
||||
t.next = null;
|
||||
const p: usize = t.priority;
|
||||
if (ready_tail[p]) |tail| tail.next = t else ready_head[p] = t;
|
||||
ready_tail[p] = t;
|
||||
ready_bitmap |= levelBit(t.priority);
|
||||
}
|
||||
|
||||
fn dequeueHighest() ?*Task {
|
||||
if (ready_bitmap == 0) return null;
|
||||
const level: Priority = @intCast(num_priorities - 1 - @clz(ready_bitmap));
|
||||
const t = ready_head[level].?;
|
||||
ready_head[level] = t.next;
|
||||
if (ready_head[level] == null) {
|
||||
ready_tail[level] = null;
|
||||
ready_bitmap &= ~levelBit(level);
|
||||
}
|
||||
t.next = null;
|
||||
return t;
|
||||
}
|
||||
|
||||
fn levelBit(p: Priority) u8 {
|
||||
return @as(u8, 1) << p;
|
||||
}
|
||||
|
||||
/// Create a task that runs `entry` at `priority`. It becomes ready immediately.
|
||||
pub fn spawn(entry: *const fn () void, priority: Priority) void {
|
||||
const t = freeSlot() orelse @panic("sched: task table full");
|
||||
const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack");
|
||||
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack };
|
||||
next_id += 1;
|
||||
const top = @intFromPtr(stack.ptr) + stack.len;
|
||||
t.rsp = arch.initTaskStack(top, @intFromPtr(entry));
|
||||
enqueue(t);
|
||||
}
|
||||
|
||||
fn freeSlot() ?*Task {
|
||||
for (&tasks) |*t| {
|
||||
if (t.state == .free) return t;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Pick the highest-priority ready task and switch to it. Interrupts must be
|
||||
/// disabled by the caller.
|
||||
fn schedule() void {
|
||||
const prev = current;
|
||||
if (prev.state == .running) {
|
||||
prev.state = .ready;
|
||||
enqueue(prev); // back of its level's queue (round-robin)
|
||||
}
|
||||
const next = dequeueHighest() orelse {
|
||||
prev.state = .running; // nothing else ready — keep running
|
||||
return;
|
||||
};
|
||||
next.state = .running;
|
||||
current = next;
|
||||
if (next != prev) arch.switchContext(&prev.rsp, next.rsp);
|
||||
}
|
||||
|
||||
/// Voluntarily give up the CPU to the next ready task.
|
||||
pub fn yield() void {
|
||||
arch.disableInterrupts();
|
||||
schedule();
|
||||
arch.enableInterrupts();
|
||||
}
|
||||
|
||||
/// Called from the timer interrupt (interrupts already disabled) to preempt.
|
||||
pub fn tick() void {
|
||||
if (preemption_enabled) schedule();
|
||||
}
|
||||
|
||||
/// Enable or disable timer-driven preemption (cooperative-only when off).
|
||||
pub fn setPreemption(enabled: bool) void {
|
||||
preemption_enabled = enabled;
|
||||
}
|
||||
|
||||
/// End the current task and switch away for good; never returns. The task's stack
|
||||
/// is leaked for now (no reaper yet).
|
||||
pub fn exit() noreturn {
|
||||
arch.disableInterrupts();
|
||||
current.state = .free;
|
||||
const next = dequeueHighest() orelse @panic("sched: no task left to run");
|
||||
next.state = .running;
|
||||
current = next;
|
||||
var discard: usize = 0;
|
||||
arch.switchContext(&discard, next.rsp);
|
||||
unreachable;
|
||||
}
|
||||
|
||||
pub fn currentId() u32 {
|
||||
return current.id;
|
||||
}
|
||||
|
||||
/// Change the running task's priority (takes effect next time it's enqueued).
|
||||
pub fn setPriority(p: Priority) void {
|
||||
current.priority = p;
|
||||
}
|
||||
@@ -14,6 +14,7 @@ const danos = @import("danos");
|
||||
const arch = @import("arch");
|
||||
const pmm = @import("pmm.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const sched = @import("sched.zig");
|
||||
|
||||
/// Formatted write straight to serial, independent of the framebuffer console.
|
||||
fn log(comptime fmt: []const u8, args: anytype) void {
|
||||
@@ -55,6 +56,10 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void {
|
||||
vmm();
|
||||
} else if (eql(case, "heap")) {
|
||||
heapTest();
|
||||
} else if (eql(case, "sched")) {
|
||||
schedTest();
|
||||
} else if (eql(case, "priority")) {
|
||||
priorityTest();
|
||||
} else if (eql(case, "fault-ud")) {
|
||||
faultInvalidOpcode();
|
||||
} else if (eql(case, "fault-pf")) {
|
||||
@@ -227,6 +232,82 @@ fn clock() void {
|
||||
result();
|
||||
}
|
||||
|
||||
// --- scheduler tests ------------------------------------------------------
|
||||
|
||||
var counters = [_]u64{0} ** 3;
|
||||
|
||||
fn spin0() void {
|
||||
const p: *volatile u64 = &counters[0];
|
||||
while (true) p.* = p.* +% 1;
|
||||
}
|
||||
fn spin1() void {
|
||||
const p: *volatile u64 = &counters[1];
|
||||
while (true) p.* = p.* +% 1;
|
||||
}
|
||||
fn spin2() void {
|
||||
const p: *volatile u64 = &counters[2];
|
||||
while (true) p.* = p.* +% 1;
|
||||
}
|
||||
|
||||
/// Preemption: spawn three tasks that busy-loop *without* yielding. If they all
|
||||
/// make progress, the timer must be preempting between them (and the context
|
||||
/// switch works) — because nothing yields voluntarily.
|
||||
fn schedTest() void {
|
||||
log("DANOS-TEST-BEGIN: sched\n", .{});
|
||||
counters = .{ 0, 0, 0 };
|
||||
sched.spawn(spin0, 4);
|
||||
sched.spawn(spin1, 4);
|
||||
sched.spawn(spin2, 4);
|
||||
|
||||
const c0: *volatile u64 = &counters[0];
|
||||
const c1: *volatile u64 = &counters[1];
|
||||
const c2: *volatile u64 = &counters[2];
|
||||
var spins: u64 = 0;
|
||||
while ((c0.* == 0 or c1.* == 0 or c2.* == 0) and spins < 5_000_000_000) spins +%= 1;
|
||||
|
||||
check("all three non-yielding tasks made progress (preemption)", c0.* > 0 and c1.* > 0 and c2.* > 0);
|
||||
result();
|
||||
}
|
||||
|
||||
var run_order = [_]u8{0} ** 4;
|
||||
var run_n: usize = 0;
|
||||
|
||||
fn recordExit(priority: u8) void {
|
||||
run_order[run_n] = priority;
|
||||
run_n += 1;
|
||||
sched.exit();
|
||||
}
|
||||
fn taskHigh() void {
|
||||
recordExit(6);
|
||||
}
|
||||
fn taskMid() void {
|
||||
recordExit(4);
|
||||
}
|
||||
fn taskLow() void {
|
||||
recordExit(2);
|
||||
}
|
||||
|
||||
/// Fixed priority: with preemption off (deterministic), spawn tasks at three
|
||||
/// priorities and let them run cooperatively. They must run highest-first.
|
||||
fn priorityTest() void {
|
||||
log("DANOS-TEST-BEGIN: priority\n", .{});
|
||||
sched.setPreemption(false);
|
||||
sched.setPriority(0); // run this observer task last, after all workers
|
||||
run_n = 0;
|
||||
|
||||
sched.spawn(taskLow, 2);
|
||||
sched.spawn(taskMid, 4);
|
||||
sched.spawn(taskHigh, 6);
|
||||
|
||||
while (run_n < 3) sched.yield(); // regain control only once the workers are done
|
||||
|
||||
check("tasks ran highest-priority first", run_order[0] == 6 and run_order[1] == 4 and run_order[2] == 2);
|
||||
|
||||
sched.setPriority(4);
|
||||
sched.setPreemption(true);
|
||||
result();
|
||||
}
|
||||
|
||||
fn faultInvalidOpcode() void {
|
||||
log("DANOS-TEST-BEGIN: fault-ud\n", .{});
|
||||
asm volatile ("ud2");
|
||||
|
||||
Reference in New Issue
Block a user