M3 step 2: swapgs discipline + arch per-CPU block

The GS base now points at an arch-owned ArchPerCpu (percpu.zig) holding
the kernel RSP and a scratch slot (for the coming syscall stub, at fixed
%gs offsets) plus the scheduler pointer. isr_common conditionally
swapgs's on entry/exit when the interrupted frame was ring 3, and
enter_user swapgs's before dropping to ring 3 — so kernel code always
sees the kernel GS base and a ring-3 `mov %ax,%gs` can no longer poison
cpuLocal(). No swapgs on ring-0 interrupts (the common case). Suite
27/27, including int 0x80 from ring 3 and faults.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Daniel Samson
2026-07-08 23:09:19 +01:00
co-authored by Claude Fable 5
parent f4590fcc19
commit 8a7235c725
5 changed files with 87 additions and 21 deletions
+14 -17
View File
@@ -15,6 +15,7 @@ const apic = @import("apic.zig");
const ioapic = @import("ioapic.zig");
const io = @import("io.zig");
const smp = @import("smp.zig");
const pcpu = @import("percpu.zig");
/// The saved register/trap frame passed to a fault handler.
pub const CpuState = idt.CpuState;
@@ -88,10 +89,13 @@ pub fn loadPageTable(pml4: u64) void {
paging.loadCr3(pml4);
}
/// Set core `cpu`'s kernel stack pointer for ring-3 -> ring-0 transitions
/// (TSS.rsp0), updated by the scheduler when it switches to a user task.
/// Set core `cpu`'s kernel stack pointer for ring-3 -> ring-0 transitions:
/// TSS.rsp0 (for interrupts/exceptions, which switch stacks in hardware) and the
/// per-CPU `kernel_rsp` (for the syscall stub, which switches by hand). Updated
/// by the scheduler when it switches to a user task.
pub fn setKernelStack(cpu: usize, top: usize) void {
tss.rsp0Ptr(cpu).* = top;
pcpu.setKernelRsp(cpu, top);
}
/// Remove a kernel mapping.
@@ -145,24 +149,17 @@ pub fn readCr3() u64 {
);
}
/// IA32_GS_BASE: the hidden base of the GS segment. We repurpose it as the per-CPU
/// data pointer. Because it's always read back via `rdmsr` (never gs-relative
/// addressing), no `swapgs` dance is needed even with user mode: the MSR is
/// privileged, ring 3 can't touch it, and its value is unaffected by ring
/// transitions. Set once per core during bring-up, after the GDT is loaded
/// (loading a GS *selector* would otherwise clobber this base).
const ia32_gs_base = 0xC000_0101;
/// Publish this core's per-CPU data pointer so `cpuLocal` can retrieve it. Each
/// core calls this once, after its GDT is in place.
pub fn setCpuLocal(ptr: usize) void {
io.wrmsr(ia32_gs_base, ptr);
/// Publish core `cpu`'s scheduler pointer via its per-CPU block (GS base). Each
/// core calls this once, after its GDT is in place (a GS *selector* reload would
/// clobber the base). See percpu.zig for the swapgs discipline.
pub fn setCpuLocal(cpu: usize, ptr: usize) void {
pcpu.setLocal(cpu, ptr);
}
/// This core's per-CPU data pointer (the value `setCpuLocal` stored). Reads the GS
/// base MSR — a per-core register, so each core sees its own without any locking.
/// This core's scheduler pointer (via the GS base) — a per-core register, so each
/// core sees its own without locking. Valid in any ring-0 context.
pub fn cpuLocal() usize {
return io.rdmsr(ia32_gs_base);
return pcpu.sched();
}
// --- SMP: application-processor bring-up ----------------------------------
+14 -2
View File
@@ -122,6 +122,7 @@ enter_user:
push $0x202 # RFLAGS: IF | reserved-1
push $0x23 # user CS (0x20 | RPL 3)
push %rdi # user RIP
swapgs # user GS base for ring 3 (isr_common swaps back)
iretq
# user_exit_to_kernel: abandon the in-flight ring-3 trap frame (it lives in the
@@ -264,8 +265,14 @@ STUB_NOERR 128
.extern interruptDispatch
# Shared tail. Register push order here defines the CpuState field order.
# If the interrupt came from ring 3 the GS base holds the user's value, so swap
# in the kernel's before anything reads per-CPU data (swapgs discipline; see
# percpu.zig). CS sits at offset 24 here (vector@0, error@8, RIP@16, CS@24).
isr_common:
push %rax
testb $3, 24(%rsp)
jz 1f
swapgs
1: push %rax
push %rbx
push %rcx
push %rdx
@@ -298,4 +305,9 @@ isr_common:
pop %rbx
pop %rax
add $16, %rsp # drop the vector and error code
iretq
# Symmetric to entry: if returning to ring 3, restore the user GS base. CS is
# now at offset 8 (RIP@0, CS@8).
testb $3, 8(%rsp)
jz 1f
swapgs
1: iretq
+56
View File
@@ -0,0 +1,56 @@
//! Per-CPU data reached through the GS segment base. The GS base holds a pointer
//! to this core's `ArchPerCpu`, so kernel code gets the running core's block with
//! a single MSR read (`sched()`) and the syscall entry stub gets its kernel stack
//! with a `%gs`-relative load (no usable stack yet at that point).
//!
//! **swapgs discipline.** In ring 0 the GS base points here; in ring 3 it holds
//! the user's own GS (which ring 3 may set freely), and this pointer lives in the
//! KERNEL_GS_BASE MSR instead. Every ring-3 -> ring-0 entry (`swapgs` in the
//! syscall stub and the conditional swapgs in isr_common) brings it back, and
//! every ring-0 -> ring-3 exit swaps it away. Because the very first ring
//! transition is always an exit (the kernel starts in ring 0), the swap pairs
//! keep the invariant without seeding KERNEL_GS_BASE. `sched()` is therefore
//! valid in any ring-0 context and never sees a user-controlled base.
const std = @import("std");
const io = @import("io.zig");
const config = @import("config");
const ia32_gs_base = 0xC000_0101;
/// Layout is load-bearing: the syscall entry stub in isr.s reaches `kernel_rsp`
/// at `%gs:0` and `scratch` at `%gs:8`. Keep those two first; the asserts below
/// pin the offsets.
pub const ArchPerCpu = extern struct {
kernel_rsp: u64 = 0, // %gs:0 — kernel stack top for syscall entry (== TSS.rsp0)
scratch: u64 = 0, // %gs:8 — stashes the user rsp during syscall entry
sched: usize = 0, // the scheduler's PerCpu pointer (what `cpuLocal` returns)
};
comptime {
std.debug.assert(@offsetOf(ArchPerCpu, "kernel_rsp") == 0);
std.debug.assert(@offsetOf(ArchPerCpu, "scratch") == 8);
}
var blocks = [_]ArchPerCpu{.{}} ** config.max_cpus;
/// Publish core `index`'s per-CPU block: record the scheduler pointer and point
/// the GS base at the block. Called once per core during bring-up, after the GDT
/// is loaded (a GS *selector* reload would clobber the base).
pub fn setLocal(index: usize, sched_ptr: usize) void {
blocks[index].sched = sched_ptr;
io.wrmsr(ia32_gs_base, @intFromPtr(&blocks[index]));
}
/// The scheduler pointer for the running core (via the GS base). Valid in any
/// ring-0 context under the swapgs discipline.
pub fn sched() usize {
return @as(*const ArchPerCpu, @ptrFromInt(io.rdmsr(ia32_gs_base))).sched;
}
/// Record core `index`'s kernel stack top, used by the syscall entry stub to
/// switch off the user stack. The scheduler sets this (and TSS.rsp0) whenever it
/// switches to a user task.
pub fn setKernelRsp(index: usize, top: usize) void {
blocks[index].kernel_rsp = top;
}
+2 -1
View File
@@ -20,6 +20,7 @@ const tss = @import("tss.zig");
const idt = @import("idt.zig");
const apic = @import("apic.zig");
const paging = @import("paging.zig");
const pcpu = @import("percpu.zig");
/// IA32_GS_BASE — the per-CPU data pointer (see cpu.zig; kept in sync here so the AP
/// path doesn't depend on cpu.zig and risk an import cycle).
@@ -168,7 +169,7 @@ fn apEntry(percpu: usize) callconv(.c) noreturn {
gdt.loadOnThisCpu(cpu); // this core's GDT (with its own TSS slot)
tss.setupThisCpu(cpu); // this core's TSS + IST stack, loaded into TR
idt.loadOnThisCpu(); // the shared IDT
io.wrmsr(ia32_gs_base, percpu); // per-CPU pointer — *after* the GDT reload
pcpu.setLocal(cpu, percpu); // per-CPU block via GS base — *after* the GDT reload
apic.initSecondary(); // software-enable this core's LAPIC
apic.initTimer(apic.frequencyHz()); // arm its timer (still masked: interrupts off)
+1 -1
View File
@@ -104,7 +104,7 @@ var preemption_enabled = true;
pub fn init(boot_priority: Priority) void {
const pc = &cpus[0];
pc.* = .{ .index = 0, .online = true, .loaded_pml4 = arch.kernelPageTable() };
arch.setCpuLocal(@intFromPtr(pc));
arch.setCpuLocal(0, @intFromPtr(pc));
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
pc.current = &tasks[0];
pc.idle = create(idle, 0, null); // this core's idle task: always ready, lowest priority