From f4590fcc192c648d25b77460b600c369e14ab4cb Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Wed, 8 Jul 2026 22:59:57 +0100 Subject: [PATCH] M3 step 1: per-switch rsp0 + CR3 plumbing MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Task gains kstack_top and pml4 (0 = kernel task); PerCpu tracks the loaded CR3. A shared switchTo() publishes the incoming task's kernel stack (TSS.rsp0) and address space (CR3, only when it changes — every write is a full TLB flush), used by both schedule() and exit(). APs adopt the kernel page tables explicitly rather than the caller's live CR3. All tasks are kernel tasks today, so this is a no-op beyond the register switch. Suite 27/27. Co-Authored-By: Claude Fable 5 --- src/kernel/arch/x86_64/cpu.zig | 21 +++++++++++++++++++- src/kernel/arch/x86_64/paging.zig | 14 ++++++++++++++ src/kernel/scheduler.zig | 32 ++++++++++++++++++++++++++++--- 3 files changed, 63 insertions(+), 4 deletions(-) diff --git a/src/kernel/arch/x86_64/cpu.zig b/src/kernel/arch/x86_64/cpu.zig index a14e988..d6aae2d 100644 --- a/src/kernel/arch/x86_64/cpu.zig +++ b/src/kernel/arch/x86_64/cpu.zig @@ -78,6 +78,22 @@ pub fn mapMmio(phys: u64, len: u64, writable: bool) u64 { return paging.mapMmio(phys, len, writable); } +/// The kernel's page-table root (physical), shared into every address space. +pub fn kernelPageTable() u64 { + return paging.kernelPml4(); +} + +/// Switch the active address space (load CR3 with a physical PML4). +pub fn loadPageTable(pml4: u64) void { + paging.loadCr3(pml4); +} + +/// Set core `cpu`'s kernel stack pointer for ring-3 -> ring-0 transitions +/// (TSS.rsp0), updated by the scheduler when it switches to a user task. +pub fn setKernelStack(cpu: usize, top: usize) void { + tss.rsp0Ptr(cpu).* = top; +} + /// Remove a kernel mapping. pub fn unmapPage(virt: u64) void { paging.unmap(virt); @@ -164,7 +180,10 @@ pub fn setTrampolinePage(phys: u64) void { /// tables. Returns false if it doesn't come online within the timeout. Blocks until /// the core reports in. pub fn startSecondary(apic_id: u32, stack_top: usize, percpu: usize, index: usize) bool { - return smp.startAp(apic_id, stack_top, percpu, index, readCr3()); + // The AP adopts the kernel page tables explicitly — never the caller's live + // CR3, which a future re-wake from a core running a process would make a + // process address space. + return smp.startAp(apic_id, stack_top, percpu, index, paging.kernelPml4()); } /// Register the generic entry a woken AP jumps to once its arch state is up (its own diff --git a/src/kernel/arch/x86_64/paging.zig b/src/kernel/arch/x86_64/paging.zig index 987ea06..aa98e80 100644 --- a/src/kernel/arch/x86_64/paging.zig +++ b/src/kernel/arch/x86_64/paging.zig @@ -166,6 +166,20 @@ pub fn init(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) voi init_done = true; // the kernel half is fixed from here } +/// The kernel's own top-level page table (physical). Every kernel task and every +/// per-process address space shares this table's higher half. +pub fn kernelPml4() u64 { + return kernel_pml4; +} + +/// Load CR3 (switch the active address space). `pml4` is a physical frame. +pub fn loadCr3(pml4: u64) void { + asm volatile ("mov %[pml4], %%cr3" + : + : [pml4] "r" (pml4), + : .{ .memory = true }); +} + /// Map a page into the kernel address space on demand (for the heap, etc.). /// `writable_page` controls W; pages are always mapped non-executable. pub fn map(virt: u64, phys: u64, writable_page: bool) void { diff --git a/src/kernel/scheduler.zig b/src/kernel/scheduler.zig index aa03dbd..671de4b 100644 --- a/src/kernel/scheduler.zig +++ b/src/kernel/scheduler.zig @@ -38,8 +38,12 @@ const Task = struct { priority: Priority = 0, rsp: usize = 0, // saved stack pointer, valid while not running stack: []u8 = &.{}, + kstack_top: usize = 0, // top of `stack` (== TSS.rsp0 for a user task); 0 = none wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping affinity: ?u32 = null, // null = runs on any core; else the index of its pinned core + // Physical PML4 of this task's address space, or 0 for a kernel task (which + // runs on the shared kernel page tables). A user task carries its own. + pml4: u64 = 0, next: ?*Task = null, // ready-queue link }; @@ -63,6 +67,7 @@ pub const PerCpu = struct { apic_id: u32 = 0, // the core's Local APIC id index: u32 = 0, // dense 0-based core index online: bool = false, // has this core finished bring-up? + loaded_pml4: u64 = 0, // the address space (CR3) currently loaded on this core // Tasks pinned to this core (affinity == index), per priority level + bitmap. pinned_head: [num_priorities]?*Task = .{null} ** num_priorities, pinned_tail: [num_priorities]?*Task = .{null} ** num_priorities, @@ -98,7 +103,7 @@ var preemption_enabled = true; /// boot, before interrupts are enabled — so no lock is needed here. pub fn init(boot_priority: Priority) void { const pc = &cpus[0]; - pc.* = .{ .index = 0, .online = true }; + pc.* = .{ .index = 0, .online = true, .loaded_pml4 = arch.kernelPageTable() }; arch.setCpuLocal(@intFromPtr(pc)); tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority }; pc.current = &tasks[0]; @@ -137,6 +142,7 @@ pub fn secondaryMain() callconv(.c) noreturn { pc.current = t; pc.idle = t; pc.online = true; + pc.loaded_pml4 = arch.kernelPageTable(); // the AP adopted the kernel tables at bring-up sync.leave(flags); arch.enableInterrupts(); // the timer now preempts this idle context into work @@ -231,6 +237,7 @@ fn create(entry: *const fn () void, priority: Priority, affinity: ?u32) *Task { t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack, .affinity = affinity }; next_id += 1; const top = @intFromPtr(stack.ptr) + stack.len; + t.kstack_top = top; t.rsp = arch.initTaskStack(top, @intFromPtr(entry)); enqueue(t); return t; @@ -261,7 +268,26 @@ fn schedule() void { }; next.state = .running; pc.current = next; - if (next != prev) arch.switchContext(&prev.rsp, next.rsp); + if (next != prev) switchTo(pc, &prev.rsp, next); +} + +/// Make `next` this core's running task: publish its kernel stack (TSS.rsp0, so a +/// ring-3 interrupt lands on a good stack) and its address space (CR3, only when +/// it differs from what's loaded — every CR3 write is a full TLB flush), then +/// switch registers/stacks. Kernel tasks (pml4 == 0, no kstack_top used from +/// ring 3) resolve to the shared kernel page tables and skip the rsp0 write, so +/// this is a no-op beyond the register switch for a pure-kernel workload. The +/// big kernel lock is held and interrupts are off throughout, so no interrupt +/// can observe a half-updated (rsp0, CR3) pair. `save_rsp` receives the outgoing +/// task's stack pointer. +fn switchTo(pc: *PerCpu, save_rsp: *usize, next: *Task) void { + if (next.kstack_top != 0) arch.setKernelStack(pc.index, next.kstack_top); + const want = if (next.pml4 != 0) next.pml4 else arch.kernelPageTable(); + if (want != pc.loaded_pml4) { + arch.loadPageTable(want); + pc.loaded_pml4 = want; + } + arch.switchContext(save_rsp, next.rsp); } /// Voluntarily give up the CPU to the next ready task. @@ -398,7 +424,7 @@ pub fn exit() noreturn { next.state = .running; pc.current = next; var discard: usize = 0; - arch.switchContext(&discard, next.rsp); + switchTo(pc, &discard, next); unreachable; }