diff --git a/src/kernel/arch/x86_64/cpu.zig b/src/kernel/arch/x86_64/cpu.zig index d6aae2d..18df1ae 100644 --- a/src/kernel/arch/x86_64/cpu.zig +++ b/src/kernel/arch/x86_64/cpu.zig @@ -15,6 +15,7 @@ const apic = @import("apic.zig"); const ioapic = @import("ioapic.zig"); const io = @import("io.zig"); const smp = @import("smp.zig"); +const pcpu = @import("percpu.zig"); /// The saved register/trap frame passed to a fault handler. pub const CpuState = idt.CpuState; @@ -88,10 +89,13 @@ pub fn loadPageTable(pml4: u64) void { paging.loadCr3(pml4); } -/// Set core `cpu`'s kernel stack pointer for ring-3 -> ring-0 transitions -/// (TSS.rsp0), updated by the scheduler when it switches to a user task. +/// Set core `cpu`'s kernel stack pointer for ring-3 -> ring-0 transitions: +/// TSS.rsp0 (for interrupts/exceptions, which switch stacks in hardware) and the +/// per-CPU `kernel_rsp` (for the syscall stub, which switches by hand). Updated +/// by the scheduler when it switches to a user task. pub fn setKernelStack(cpu: usize, top: usize) void { tss.rsp0Ptr(cpu).* = top; + pcpu.setKernelRsp(cpu, top); } /// Remove a kernel mapping. @@ -145,24 +149,17 @@ pub fn readCr3() u64 { ); } -/// IA32_GS_BASE: the hidden base of the GS segment. We repurpose it as the per-CPU -/// data pointer. Because it's always read back via `rdmsr` (never gs-relative -/// addressing), no `swapgs` dance is needed even with user mode: the MSR is -/// privileged, ring 3 can't touch it, and its value is unaffected by ring -/// transitions. Set once per core during bring-up, after the GDT is loaded -/// (loading a GS *selector* would otherwise clobber this base). -const ia32_gs_base = 0xC000_0101; - -/// Publish this core's per-CPU data pointer so `cpuLocal` can retrieve it. Each -/// core calls this once, after its GDT is in place. -pub fn setCpuLocal(ptr: usize) void { - io.wrmsr(ia32_gs_base, ptr); +/// Publish core `cpu`'s scheduler pointer via its per-CPU block (GS base). Each +/// core calls this once, after its GDT is in place (a GS *selector* reload would +/// clobber the base). See percpu.zig for the swapgs discipline. +pub fn setCpuLocal(cpu: usize, ptr: usize) void { + pcpu.setLocal(cpu, ptr); } -/// This core's per-CPU data pointer (the value `setCpuLocal` stored). Reads the GS -/// base MSR — a per-core register, so each core sees its own without any locking. +/// This core's scheduler pointer (via the GS base) — a per-core register, so each +/// core sees its own without locking. Valid in any ring-0 context. pub fn cpuLocal() usize { - return io.rdmsr(ia32_gs_base); + return pcpu.sched(); } // --- SMP: application-processor bring-up ---------------------------------- diff --git a/src/kernel/arch/x86_64/isr.s b/src/kernel/arch/x86_64/isr.s index ab39ce8..a6e12ad 100644 --- a/src/kernel/arch/x86_64/isr.s +++ b/src/kernel/arch/x86_64/isr.s @@ -122,6 +122,7 @@ enter_user: push $0x202 # RFLAGS: IF | reserved-1 push $0x23 # user CS (0x20 | RPL 3) push %rdi # user RIP + swapgs # user GS base for ring 3 (isr_common swaps back) iretq # user_exit_to_kernel: abandon the in-flight ring-3 trap frame (it lives in the @@ -264,8 +265,14 @@ STUB_NOERR 128 .extern interruptDispatch # Shared tail. Register push order here defines the CpuState field order. +# If the interrupt came from ring 3 the GS base holds the user's value, so swap +# in the kernel's before anything reads per-CPU data (swapgs discipline; see +# percpu.zig). CS sits at offset 24 here (vector@0, error@8, RIP@16, CS@24). isr_common: - push %rax + testb $3, 24(%rsp) + jz 1f + swapgs +1: push %rax push %rbx push %rcx push %rdx @@ -298,4 +305,9 @@ isr_common: pop %rbx pop %rax add $16, %rsp # drop the vector and error code - iretq + # Symmetric to entry: if returning to ring 3, restore the user GS base. CS is + # now at offset 8 (RIP@0, CS@8). + testb $3, 8(%rsp) + jz 1f + swapgs +1: iretq diff --git a/src/kernel/arch/x86_64/percpu.zig b/src/kernel/arch/x86_64/percpu.zig new file mode 100644 index 0000000..dee8c3c --- /dev/null +++ b/src/kernel/arch/x86_64/percpu.zig @@ -0,0 +1,56 @@ +//! Per-CPU data reached through the GS segment base. The GS base holds a pointer +//! to this core's `ArchPerCpu`, so kernel code gets the running core's block with +//! a single MSR read (`sched()`) and the syscall entry stub gets its kernel stack +//! with a `%gs`-relative load (no usable stack yet at that point). +//! +//! **swapgs discipline.** In ring 0 the GS base points here; in ring 3 it holds +//! the user's own GS (which ring 3 may set freely), and this pointer lives in the +//! KERNEL_GS_BASE MSR instead. Every ring-3 -> ring-0 entry (`swapgs` in the +//! syscall stub and the conditional swapgs in isr_common) brings it back, and +//! every ring-0 -> ring-3 exit swaps it away. Because the very first ring +//! transition is always an exit (the kernel starts in ring 0), the swap pairs +//! keep the invariant without seeding KERNEL_GS_BASE. `sched()` is therefore +//! valid in any ring-0 context and never sees a user-controlled base. + +const std = @import("std"); +const io = @import("io.zig"); +const config = @import("config"); + +const ia32_gs_base = 0xC000_0101; + +/// Layout is load-bearing: the syscall entry stub in isr.s reaches `kernel_rsp` +/// at `%gs:0` and `scratch` at `%gs:8`. Keep those two first; the asserts below +/// pin the offsets. +pub const ArchPerCpu = extern struct { + kernel_rsp: u64 = 0, // %gs:0 — kernel stack top for syscall entry (== TSS.rsp0) + scratch: u64 = 0, // %gs:8 — stashes the user rsp during syscall entry + sched: usize = 0, // the scheduler's PerCpu pointer (what `cpuLocal` returns) +}; + +comptime { + std.debug.assert(@offsetOf(ArchPerCpu, "kernel_rsp") == 0); + std.debug.assert(@offsetOf(ArchPerCpu, "scratch") == 8); +} + +var blocks = [_]ArchPerCpu{.{}} ** config.max_cpus; + +/// Publish core `index`'s per-CPU block: record the scheduler pointer and point +/// the GS base at the block. Called once per core during bring-up, after the GDT +/// is loaded (a GS *selector* reload would clobber the base). +pub fn setLocal(index: usize, sched_ptr: usize) void { + blocks[index].sched = sched_ptr; + io.wrmsr(ia32_gs_base, @intFromPtr(&blocks[index])); +} + +/// The scheduler pointer for the running core (via the GS base). Valid in any +/// ring-0 context under the swapgs discipline. +pub fn sched() usize { + return @as(*const ArchPerCpu, @ptrFromInt(io.rdmsr(ia32_gs_base))).sched; +} + +/// Record core `index`'s kernel stack top, used by the syscall entry stub to +/// switch off the user stack. The scheduler sets this (and TSS.rsp0) whenever it +/// switches to a user task. +pub fn setKernelRsp(index: usize, top: usize) void { + blocks[index].kernel_rsp = top; +} diff --git a/src/kernel/arch/x86_64/smp.zig b/src/kernel/arch/x86_64/smp.zig index 949dd8a..ee54042 100644 --- a/src/kernel/arch/x86_64/smp.zig +++ b/src/kernel/arch/x86_64/smp.zig @@ -20,6 +20,7 @@ const tss = @import("tss.zig"); const idt = @import("idt.zig"); const apic = @import("apic.zig"); const paging = @import("paging.zig"); +const pcpu = @import("percpu.zig"); /// IA32_GS_BASE — the per-CPU data pointer (see cpu.zig; kept in sync here so the AP /// path doesn't depend on cpu.zig and risk an import cycle). @@ -168,7 +169,7 @@ fn apEntry(percpu: usize) callconv(.c) noreturn { gdt.loadOnThisCpu(cpu); // this core's GDT (with its own TSS slot) tss.setupThisCpu(cpu); // this core's TSS + IST stack, loaded into TR idt.loadOnThisCpu(); // the shared IDT - io.wrmsr(ia32_gs_base, percpu); // per-CPU pointer — *after* the GDT reload + pcpu.setLocal(cpu, percpu); // per-CPU block via GS base — *after* the GDT reload apic.initSecondary(); // software-enable this core's LAPIC apic.initTimer(apic.frequencyHz()); // arm its timer (still masked: interrupts off) diff --git a/src/kernel/scheduler.zig b/src/kernel/scheduler.zig index 671de4b..26dd53b 100644 --- a/src/kernel/scheduler.zig +++ b/src/kernel/scheduler.zig @@ -104,7 +104,7 @@ var preemption_enabled = true; pub fn init(boot_priority: Priority) void { const pc = &cpus[0]; pc.* = .{ .index = 0, .online = true, .loaded_pml4 = arch.kernelPageTable() }; - arch.setCpuLocal(@intFromPtr(pc)); + arch.setCpuLocal(0, @intFromPtr(pc)); tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority }; pc.current = &tasks[0]; pc.idle = create(idle, 0, null); // this core's idle task: always ready, lowest priority