M3 step 1: per-switch rsp0 + CR3 plumbing

Task gains kstack_top and pml4 (0 = kernel task); PerCpu tracks the
loaded CR3. A shared switchTo() publishes the incoming task's kernel
stack (TSS.rsp0) and address space (CR3, only when it changes — every
write is a full TLB flush), used by both schedule() and exit(). APs
adopt the kernel page tables explicitly rather than the caller's live
CR3. All tasks are kernel tasks today, so this is a no-op beyond the
register switch. Suite 27/27.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Daniel Samson
2026-07-08 22:59:57 +01:00
co-authored by Claude Fable 5
parent bb7597ea0b
commit f4590fcc19
3 changed files with 63 additions and 4 deletions
+29 -3
View File
@@ -38,8 +38,12 @@ const Task = struct {
priority: Priority = 0,
rsp: usize = 0, // saved stack pointer, valid while not running
stack: []u8 = &.{},
kstack_top: usize = 0, // top of `stack` (== TSS.rsp0 for a user task); 0 = none
wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping
affinity: ?u32 = null, // null = runs on any core; else the index of its pinned core
// Physical PML4 of this task's address space, or 0 for a kernel task (which
// runs on the shared kernel page tables). A user task carries its own.
pml4: u64 = 0,
next: ?*Task = null, // ready-queue link
};
@@ -63,6 +67,7 @@ pub const PerCpu = struct {
apic_id: u32 = 0, // the core's Local APIC id
index: u32 = 0, // dense 0-based core index
online: bool = false, // has this core finished bring-up?
loaded_pml4: u64 = 0, // the address space (CR3) currently loaded on this core
// Tasks pinned to this core (affinity == index), per priority level + bitmap.
pinned_head: [num_priorities]?*Task = .{null} ** num_priorities,
pinned_tail: [num_priorities]?*Task = .{null} ** num_priorities,
@@ -98,7 +103,7 @@ var preemption_enabled = true;
/// boot, before interrupts are enabled — so no lock is needed here.
pub fn init(boot_priority: Priority) void {
const pc = &cpus[0];
pc.* = .{ .index = 0, .online = true };
pc.* = .{ .index = 0, .online = true, .loaded_pml4 = arch.kernelPageTable() };
arch.setCpuLocal(@intFromPtr(pc));
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
pc.current = &tasks[0];
@@ -137,6 +142,7 @@ pub fn secondaryMain() callconv(.c) noreturn {
pc.current = t;
pc.idle = t;
pc.online = true;
pc.loaded_pml4 = arch.kernelPageTable(); // the AP adopted the kernel tables at bring-up
sync.leave(flags);
arch.enableInterrupts(); // the timer now preempts this idle context into work
@@ -231,6 +237,7 @@ fn create(entry: *const fn () void, priority: Priority, affinity: ?u32) *Task {
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack, .affinity = affinity };
next_id += 1;
const top = @intFromPtr(stack.ptr) + stack.len;
t.kstack_top = top;
t.rsp = arch.initTaskStack(top, @intFromPtr(entry));
enqueue(t);
return t;
@@ -261,7 +268,26 @@ fn schedule() void {
};
next.state = .running;
pc.current = next;
if (next != prev) arch.switchContext(&prev.rsp, next.rsp);
if (next != prev) switchTo(pc, &prev.rsp, next);
}
/// Make `next` this core's running task: publish its kernel stack (TSS.rsp0, so a
/// ring-3 interrupt lands on a good stack) and its address space (CR3, only when
/// it differs from what's loaded — every CR3 write is a full TLB flush), then
/// switch registers/stacks. Kernel tasks (pml4 == 0, no kstack_top used from
/// ring 3) resolve to the shared kernel page tables and skip the rsp0 write, so
/// this is a no-op beyond the register switch for a pure-kernel workload. The
/// big kernel lock is held and interrupts are off throughout, so no interrupt
/// can observe a half-updated (rsp0, CR3) pair. `save_rsp` receives the outgoing
/// task's stack pointer.
fn switchTo(pc: *PerCpu, save_rsp: *usize, next: *Task) void {
if (next.kstack_top != 0) arch.setKernelStack(pc.index, next.kstack_top);
const want = if (next.pml4 != 0) next.pml4 else arch.kernelPageTable();
if (want != pc.loaded_pml4) {
arch.loadPageTable(want);
pc.loaded_pml4 = want;
}
arch.switchContext(save_rsp, next.rsp);
}
/// Voluntarily give up the CPU to the next ready task.
@@ -398,7 +424,7 @@ pub fn exit() noreturn {
next.state = .running;
pc.current = next;
var discard: usize = 0;
arch.switchContext(&discard, next.rsp);
switchTo(pc, &discard, next);
unreachable;
}