diff --git a/src/kernel/arch/x86_64/cpu.zig b/src/kernel/arch/x86_64/cpu.zig index a14e988..d6aae2d 100644 --- a/src/kernel/arch/x86_64/cpu.zig +++ b/src/kernel/arch/x86_64/cpu.zig @@ -78,6 +78,22 @@ pub fn mapMmio(phys: u64, len: u64, writable: bool) u64 { return paging.mapMmio(phys, len, writable); } +/// The kernel's page-table root (physical), shared into every address space. +pub fn kernelPageTable() u64 { + return paging.kernelPml4(); +} + +/// Switch the active address space (load CR3 with a physical PML4). +pub fn loadPageTable(pml4: u64) void { + paging.loadCr3(pml4); +} + +/// Set core `cpu`'s kernel stack pointer for ring-3 -> ring-0 transitions +/// (TSS.rsp0), updated by the scheduler when it switches to a user task. +pub fn setKernelStack(cpu: usize, top: usize) void { + tss.rsp0Ptr(cpu).* = top; +} + /// Remove a kernel mapping. pub fn unmapPage(virt: u64) void { paging.unmap(virt); @@ -164,7 +180,10 @@ pub fn setTrampolinePage(phys: u64) void { /// tables. Returns false if it doesn't come online within the timeout. Blocks until /// the core reports in. pub fn startSecondary(apic_id: u32, stack_top: usize, percpu: usize, index: usize) bool { - return smp.startAp(apic_id, stack_top, percpu, index, readCr3()); + // The AP adopts the kernel page tables explicitly — never the caller's live + // CR3, which a future re-wake from a core running a process would make a + // process address space. + return smp.startAp(apic_id, stack_top, percpu, index, paging.kernelPml4()); } /// Register the generic entry a woken AP jumps to once its arch state is up (its own diff --git a/src/kernel/arch/x86_64/paging.zig b/src/kernel/arch/x86_64/paging.zig index 987ea06..aa98e80 100644 --- a/src/kernel/arch/x86_64/paging.zig +++ b/src/kernel/arch/x86_64/paging.zig @@ -166,6 +166,20 @@ pub fn init(allocFrame: *const fn () ?u64, boot_info: *const danos.BootInfo) voi init_done = true; // the kernel half is fixed from here } +/// The kernel's own top-level page table (physical). Every kernel task and every +/// per-process address space shares this table's higher half. +pub fn kernelPml4() u64 { + return kernel_pml4; +} + +/// Load CR3 (switch the active address space). `pml4` is a physical frame. +pub fn loadCr3(pml4: u64) void { + asm volatile ("mov %[pml4], %%cr3" + : + : [pml4] "r" (pml4), + : .{ .memory = true }); +} + /// Map a page into the kernel address space on demand (for the heap, etc.). /// `writable_page` controls W; pages are always mapped non-executable. pub fn map(virt: u64, phys: u64, writable_page: bool) void { diff --git a/src/kernel/scheduler.zig b/src/kernel/scheduler.zig index aa03dbd..671de4b 100644 --- a/src/kernel/scheduler.zig +++ b/src/kernel/scheduler.zig @@ -38,8 +38,12 @@ const Task = struct { priority: Priority = 0, rsp: usize = 0, // saved stack pointer, valid while not running stack: []u8 = &.{}, + kstack_top: usize = 0, // top of `stack` (== TSS.rsp0 for a user task); 0 = none wake_at: u64 = 0, // uptime (ms) to wake a sleeping task; 0 = not sleeping affinity: ?u32 = null, // null = runs on any core; else the index of its pinned core + // Physical PML4 of this task's address space, or 0 for a kernel task (which + // runs on the shared kernel page tables). A user task carries its own. + pml4: u64 = 0, next: ?*Task = null, // ready-queue link }; @@ -63,6 +67,7 @@ pub const PerCpu = struct { apic_id: u32 = 0, // the core's Local APIC id index: u32 = 0, // dense 0-based core index online: bool = false, // has this core finished bring-up? + loaded_pml4: u64 = 0, // the address space (CR3) currently loaded on this core // Tasks pinned to this core (affinity == index), per priority level + bitmap. pinned_head: [num_priorities]?*Task = .{null} ** num_priorities, pinned_tail: [num_priorities]?*Task = .{null} ** num_priorities, @@ -98,7 +103,7 @@ var preemption_enabled = true; /// boot, before interrupts are enabled — so no lock is needed here. pub fn init(boot_priority: Priority) void { const pc = &cpus[0]; - pc.* = .{ .index = 0, .online = true }; + pc.* = .{ .index = 0, .online = true, .loaded_pml4 = arch.kernelPageTable() }; arch.setCpuLocal(@intFromPtr(pc)); tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority }; pc.current = &tasks[0]; @@ -137,6 +142,7 @@ pub fn secondaryMain() callconv(.c) noreturn { pc.current = t; pc.idle = t; pc.online = true; + pc.loaded_pml4 = arch.kernelPageTable(); // the AP adopted the kernel tables at bring-up sync.leave(flags); arch.enableInterrupts(); // the timer now preempts this idle context into work @@ -231,6 +237,7 @@ fn create(entry: *const fn () void, priority: Priority, affinity: ?u32) *Task { t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack, .affinity = affinity }; next_id += 1; const top = @intFromPtr(stack.ptr) + stack.len; + t.kstack_top = top; t.rsp = arch.initTaskStack(top, @intFromPtr(entry)); enqueue(t); return t; @@ -261,7 +268,26 @@ fn schedule() void { }; next.state = .running; pc.current = next; - if (next != prev) arch.switchContext(&prev.rsp, next.rsp); + if (next != prev) switchTo(pc, &prev.rsp, next); +} + +/// Make `next` this core's running task: publish its kernel stack (TSS.rsp0, so a +/// ring-3 interrupt lands on a good stack) and its address space (CR3, only when +/// it differs from what's loaded — every CR3 write is a full TLB flush), then +/// switch registers/stacks. Kernel tasks (pml4 == 0, no kstack_top used from +/// ring 3) resolve to the shared kernel page tables and skip the rsp0 write, so +/// this is a no-op beyond the register switch for a pure-kernel workload. The +/// big kernel lock is held and interrupts are off throughout, so no interrupt +/// can observe a half-updated (rsp0, CR3) pair. `save_rsp` receives the outgoing +/// task's stack pointer. +fn switchTo(pc: *PerCpu, save_rsp: *usize, next: *Task) void { + if (next.kstack_top != 0) arch.setKernelStack(pc.index, next.kstack_top); + const want = if (next.pml4 != 0) next.pml4 else arch.kernelPageTable(); + if (want != pc.loaded_pml4) { + arch.loadPageTable(want); + pc.loaded_pml4 = want; + } + arch.switchContext(save_rsp, next.rsp); } /// Voluntarily give up the CPU to the next ready task. @@ -398,7 +424,7 @@ pub fn exit() noreturn { next.state = .running; pc.current = next; var discard: usize = 0; - arch.switchContext(&discard, next.rsp); + switchTo(pc, &discard, next); unreachable; }