M3 step 1: per-switch rsp0 + CR3 plumbing

Task gains kstack_top and pml4 (0 = kernel task); PerCpu tracks the
loaded CR3. A shared switchTo() publishes the incoming task's kernel
stack (TSS.rsp0) and address space (CR3, only when it changes — every
write is a full TLB flush), used by both schedule() and exit(). APs
adopt the kernel page tables explicitly rather than the caller's live
CR3. All tasks are kernel tasks today, so this is a no-op beyond the
register switch. Suite 27/27.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Daniel Samson
2026-07-08 22:59:57 +01:00
co-authored by Claude Fable 5
parent bb7597ea0b
commit f4590fcc19
3 changed files with 63 additions and 4 deletions
+20 -1
View File
@@ -78,6 +78,22 @@ pub fn mapMmio(phys: u64, len: u64, writable: bool) u64 {
return paging.mapMmio(phys, len, writable); return paging.mapMmio(phys, len, writable);
} }
/// The kernel's page-table root (physical), shared into every address space.
pub fn kernelPageTable() u64 {
return paging.kernelPml4();
}
/// Switch the active address space (load CR3 with a physical PML4).
pub fn loadPageTable(pml4: u64) void {
paging.loadCr3(pml4);
}
/// Set core `cpu`'s kernel stack pointer for ring-3 -> ring-0 transitions
/// (TSS.rsp0), updated by the scheduler when it switches to a user task.
pub fn setKernelStack(cpu: usize, top: usize) void {
tss.rsp0Ptr(cpu).* = top;
}
/// Remove a kernel mapping. /// Remove a kernel mapping.
pub fn unmapPage(virt: u64) void { pub fn unmapPage(virt: u64) void {
paging.unmap(virt); paging.unmap(virt);
@@ -164,7 +180,10 @@ pub fn setTrampolinePage(phys: u64) void {
/// tables. Returns false if it doesn't come online within the timeout. Blocks until /// tables. Returns false if it doesn't come online within the timeout. Blocks until
/// the core reports in. /// the core reports in.
pub fn startSecondary(apic_id: u32, stack_top: usize, percpu: usize, index: usize) bool { pub fn startSecondary(apic_id: u32, stack_top: usize, percpu: usize, index: usize) bool {
return smp.startAp(apic_id, stack_top, percpu, index, readCr3()); // The AP adopts the kernel page tables explicitly — never the caller's live
// CR3, which a future re-wake from a core running a process would make a
// process address space.
return smp.startAp(apic_id, stack_top, percpu, index, paging.kernelPml4());
} }
/// Register the generic entry a woken AP jumps to once its arch state is up (its own /// Register the generic entry a woken AP jumps to once its arch state is up (its own
+14
View File
@@ -166,6 +166,20 @@ pub fn init(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) voi
init_done = true; // the kernel half is fixed from here init_done = true; // the kernel half is fixed from here
} }
/// The kernel's own top-level page table (physical). Every kernel task and every
/// per-process address space shares this table's higher half.
pub fn kernelPml4() u64 {
return kernel_pml4;
}
/// Load CR3 (switch the active address space). `pml4` is a physical frame.
pub fn loadCr3(pml4: u64) void {
asm volatile ("mov %[pml4], %%cr3"
:
: [pml4] "r" (pml4),
: .{ .memory = true });
}
/// Map a page into the kernel address space on demand (for the heap, etc.). /// Map a page into the kernel address space on demand (for the heap, etc.).
/// `writable_page` controls W; pages are always mapped non-executable. /// `writable_page` controls W; pages are always mapped non-executable.
pub fn map(virt: u64, phys: u64, writable_page: bool) void { pub fn map(virt: u64, phys: u64, writable_page: bool) void {
+29 -3
View File
@@ -38,8 +38,12 @@ const Task = struct {
priority: Priority = 0, priority: Priority = 0,
rsp: usize = 0, // saved stack pointer, valid while not running rsp: usize = 0, // saved stack pointer, valid while not running
stack: []u8 = &.{}, stack: []u8 = &.{},
kstack_top: usize = 0, // top of `stack` (== TSS.rsp0 for a user task); 0 = none
wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping
affinity: ?u32 = null, // null = runs on any core; else the index of its pinned core affinity: ?u32 = null, // null = runs on any core; else the index of its pinned core
// Physical PML4 of this task's address space, or 0 for a kernel task (which
// runs on the shared kernel page tables). A user task carries its own.
pml4: u64 = 0,
next: ?*Task = null, // ready-queue link next: ?*Task = null, // ready-queue link
}; };
@@ -63,6 +67,7 @@ pub const PerCpu = struct {
apic_id: u32 = 0, // the core's Local APIC id apic_id: u32 = 0, // the core's Local APIC id
index: u32 = 0, // dense 0-based core index index: u32 = 0, // dense 0-based core index
online: bool = false, // has this core finished bring-up? online: bool = false, // has this core finished bring-up?
loaded_pml4: u64 = 0, // the address space (CR3) currently loaded on this core
// Tasks pinned to this core (affinity == index), per priority level + bitmap. // Tasks pinned to this core (affinity == index), per priority level + bitmap.
pinned_head: [num_priorities]?*Task = .{null} ** num_priorities, pinned_head: [num_priorities]?*Task = .{null} ** num_priorities,
pinned_tail: [num_priorities]?*Task = .{null} ** num_priorities, pinned_tail: [num_priorities]?*Task = .{null} ** num_priorities,
@@ -98,7 +103,7 @@ var preemption_enabled = true;
/// boot, before interrupts are enabled — so no lock is needed here. /// boot, before interrupts are enabled — so no lock is needed here.
pub fn init(boot_priority: Priority) void { pub fn init(boot_priority: Priority) void {
const pc = &cpus[0]; const pc = &cpus[0];
pc.* = .{ .index = 0, .online = true }; pc.* = .{ .index = 0, .online = true, .loaded_pml4 = arch.kernelPageTable() };
arch.setCpuLocal(@intFromPtr(pc)); arch.setCpuLocal(@intFromPtr(pc));
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority }; tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
pc.current = &tasks[0]; pc.current = &tasks[0];
@@ -137,6 +142,7 @@ pub fn secondaryMain() callconv(.c) noreturn {
pc.current = t; pc.current = t;
pc.idle = t; pc.idle = t;
pc.online = true; pc.online = true;
pc.loaded_pml4 = arch.kernelPageTable(); // the AP adopted the kernel tables at bring-up
sync.leave(flags); sync.leave(flags);
arch.enableInterrupts(); // the timer now preempts this idle context into work arch.enableInterrupts(); // the timer now preempts this idle context into work
@@ -231,6 +237,7 @@ fn create(entry: *const fn () void, priority: Priority, affinity: ?u32) *Task {
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack, .affinity = affinity }; t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack, .affinity = affinity };
next_id += 1; next_id += 1;
const top = @intFromPtr(stack.ptr) + stack.len; const top = @intFromPtr(stack.ptr) + stack.len;
t.kstack_top = top;
t.rsp = arch.initTaskStack(top, @intFromPtr(entry)); t.rsp = arch.initTaskStack(top, @intFromPtr(entry));
enqueue(t); enqueue(t);
return t; return t;
@@ -261,7 +268,26 @@ fn schedule() void {
}; };
next.state = .running; next.state = .running;
pc.current = next; pc.current = next;
if (next != prev) arch.switchContext(&prev.rsp, next.rsp); if (next != prev) switchTo(pc, &prev.rsp, next);
}
/// Make `next` this core's running task: publish its kernel stack (TSS.rsp0, so a
/// ring-3 interrupt lands on a good stack) and its address space (CR3, only when
/// it differs from what's loaded — every CR3 write is a full TLB flush), then
/// switch registers/stacks. Kernel tasks (pml4 == 0, no kstack_top used from
/// ring 3) resolve to the shared kernel page tables and skip the rsp0 write, so
/// this is a no-op beyond the register switch for a pure-kernel workload. The
/// big kernel lock is held and interrupts are off throughout, so no interrupt
/// can observe a half-updated (rsp0, CR3) pair. `save_rsp` receives the outgoing
/// task's stack pointer.
fn switchTo(pc: *PerCpu, save_rsp: *usize, next: *Task) void {
if (next.kstack_top != 0) arch.setKernelStack(pc.index, next.kstack_top);
const want = if (next.pml4 != 0) next.pml4 else arch.kernelPageTable();
if (want != pc.loaded_pml4) {
arch.loadPageTable(want);
pc.loaded_pml4 = want;
}
arch.switchContext(save_rsp, next.rsp);
} }
/// Voluntarily give up the CPU to the next ready task. /// Voluntarily give up the CPU to the next ready task.
@@ -398,7 +424,7 @@ pub fn exit() noreturn {
next.state = .running; next.state = .running;
pc.current = next; pc.current = next;
var discard: usize = 0; var discard: usize = 0;
arch.switchContext(&discard, next.rsp); switchTo(pc, &discard, next);
unreachable; unreachable;
} }