M3 step 2: swapgs discipline + arch per-CPU block
The GS base now points at an arch-owned ArchPerCpu (percpu.zig) holding the kernel RSP and a scratch slot (for the coming syscall stub, at fixed %gs offsets) plus the scheduler pointer. isr_common conditionally swapgs's on entry/exit when the interrupted frame was ring 3, and enter_user swapgs's before dropping to ring 3 — so kernel code always sees the kernel GS base and a ring-3 `mov %ax,%gs` can no longer poison cpuLocal(). No swapgs on ring-0 interrupts (the common case). Suite 27/27, including int 0x80 from ring 3 and faults. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Fable 5
parent
f4590fcc19
commit
8a7235c725
@@ -15,6 +15,7 @@ const apic = @import("apic.zig");
|
||||
const ioapic = @import("ioapic.zig");
|
||||
const io = @import("io.zig");
|
||||
const smp = @import("smp.zig");
|
||||
const pcpu = @import("percpu.zig");
|
||||
|
||||
/// The saved register/trap frame passed to a fault handler.
|
||||
pub const CpuState = idt.CpuState;
|
||||
@@ -88,10 +89,13 @@ pub fn loadPageTable(pml4: u64) void {
|
||||
paging.loadCr3(pml4);
|
||||
}
|
||||
|
||||
/// Set core `cpu`'s kernel stack pointer for ring-3 -> ring-0 transitions
|
||||
/// (TSS.rsp0), updated by the scheduler when it switches to a user task.
|
||||
/// Set core `cpu`'s kernel stack pointer for ring-3 -> ring-0 transitions:
|
||||
/// TSS.rsp0 (for interrupts/exceptions, which switch stacks in hardware) and the
|
||||
/// per-CPU `kernel_rsp` (for the syscall stub, which switches by hand). Updated
|
||||
/// by the scheduler when it switches to a user task.
|
||||
pub fn setKernelStack(cpu: usize, top: usize) void {
|
||||
tss.rsp0Ptr(cpu).* = top;
|
||||
pcpu.setKernelRsp(cpu, top);
|
||||
}
|
||||
|
||||
/// Remove a kernel mapping.
|
||||
@@ -145,24 +149,17 @@ pub fn readCr3() u64 {
|
||||
);
|
||||
}
|
||||
|
||||
/// IA32_GS_BASE: the hidden base of the GS segment. We repurpose it as the per-CPU
|
||||
/// data pointer. Because it's always read back via `rdmsr` (never gs-relative
|
||||
/// addressing), no `swapgs` dance is needed even with user mode: the MSR is
|
||||
/// privileged, ring 3 can't touch it, and its value is unaffected by ring
|
||||
/// transitions. Set once per core during bring-up, after the GDT is loaded
|
||||
/// (loading a GS *selector* would otherwise clobber this base).
|
||||
const ia32_gs_base = 0xC000_0101;
|
||||
|
||||
/// Publish this core's per-CPU data pointer so `cpuLocal` can retrieve it. Each
|
||||
/// core calls this once, after its GDT is in place.
|
||||
pub fn setCpuLocal(ptr: usize) void {
|
||||
io.wrmsr(ia32_gs_base, ptr);
|
||||
/// Publish core `cpu`'s scheduler pointer via its per-CPU block (GS base). Each
|
||||
/// core calls this once, after its GDT is in place (a GS *selector* reload would
|
||||
/// clobber the base). See percpu.zig for the swapgs discipline.
|
||||
pub fn setCpuLocal(cpu: usize, ptr: usize) void {
|
||||
pcpu.setLocal(cpu, ptr);
|
||||
}
|
||||
|
||||
/// This core's per-CPU data pointer (the value `setCpuLocal` stored). Reads the GS
|
||||
/// base MSR — a per-core register, so each core sees its own without any locking.
|
||||
/// This core's scheduler pointer (via the GS base) — a per-core register, so each
|
||||
/// core sees its own without locking. Valid in any ring-0 context.
|
||||
pub fn cpuLocal() usize {
|
||||
return io.rdmsr(ia32_gs_base);
|
||||
return pcpu.sched();
|
||||
}
|
||||
|
||||
// --- SMP: application-processor bring-up ----------------------------------
|
||||
|
||||
@@ -122,6 +122,7 @@ enter_user:
|
||||
push $0x202 # RFLAGS: IF | reserved-1
|
||||
push $0x23 # user CS (0x20 | RPL 3)
|
||||
push %rdi # user RIP
|
||||
swapgs # user GS base for ring 3 (isr_common swaps back)
|
||||
iretq
|
||||
|
||||
# user_exit_to_kernel: abandon the in-flight ring-3 trap frame (it lives in the
|
||||
@@ -264,8 +265,14 @@ STUB_NOERR 128
|
||||
.extern interruptDispatch
|
||||
|
||||
# Shared tail. Register push order here defines the CpuState field order.
|
||||
# If the interrupt came from ring 3 the GS base holds the user's value, so swap
|
||||
# in the kernel's before anything reads per-CPU data (swapgs discipline; see
|
||||
# percpu.zig). CS sits at offset 24 here (vector@0, error@8, RIP@16, CS@24).
|
||||
isr_common:
|
||||
push %rax
|
||||
testb $3, 24(%rsp)
|
||||
jz 1f
|
||||
swapgs
|
||||
1: push %rax
|
||||
push %rbx
|
||||
push %rcx
|
||||
push %rdx
|
||||
@@ -298,4 +305,9 @@ isr_common:
|
||||
pop %rbx
|
||||
pop %rax
|
||||
add $16, %rsp # drop the vector and error code
|
||||
iretq
|
||||
# Symmetric to entry: if returning to ring 3, restore the user GS base. CS is
|
||||
# now at offset 8 (RIP@0, CS@8).
|
||||
testb $3, 8(%rsp)
|
||||
jz 1f
|
||||
swapgs
|
||||
1: iretq
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
//! Per-CPU data reached through the GS segment base. The GS base holds a pointer
|
||||
//! to this core's `ArchPerCpu`, so kernel code gets the running core's block with
|
||||
//! a single MSR read (`sched()`) and the syscall entry stub gets its kernel stack
|
||||
//! with a `%gs`-relative load (no usable stack yet at that point).
|
||||
//!
|
||||
//! **swapgs discipline.** In ring 0 the GS base points here; in ring 3 it holds
|
||||
//! the user's own GS (which ring 3 may set freely), and this pointer lives in the
|
||||
//! KERNEL_GS_BASE MSR instead. Every ring-3 -> ring-0 entry (`swapgs` in the
|
||||
//! syscall stub and the conditional swapgs in isr_common) brings it back, and
|
||||
//! every ring-0 -> ring-3 exit swaps it away. Because the very first ring
|
||||
//! transition is always an exit (the kernel starts in ring 0), the swap pairs
|
||||
//! keep the invariant without seeding KERNEL_GS_BASE. `sched()` is therefore
|
||||
//! valid in any ring-0 context and never sees a user-controlled base.
|
||||
|
||||
const std = @import("std");
|
||||
const io = @import("io.zig");
|
||||
const config = @import("config");
|
||||
|
||||
const ia32_gs_base = 0xC000_0101;
|
||||
|
||||
/// Layout is load-bearing: the syscall entry stub in isr.s reaches `kernel_rsp`
|
||||
/// at `%gs:0` and `scratch` at `%gs:8`. Keep those two first; the asserts below
|
||||
/// pin the offsets.
|
||||
pub const ArchPerCpu = extern struct {
|
||||
kernel_rsp: u64 = 0, // %gs:0 — kernel stack top for syscall entry (== TSS.rsp0)
|
||||
scratch: u64 = 0, // %gs:8 — stashes the user rsp during syscall entry
|
||||
sched: usize = 0, // the scheduler's PerCpu pointer (what `cpuLocal` returns)
|
||||
};
|
||||
|
||||
comptime {
|
||||
std.debug.assert(@offsetOf(ArchPerCpu, "kernel_rsp") == 0);
|
||||
std.debug.assert(@offsetOf(ArchPerCpu, "scratch") == 8);
|
||||
}
|
||||
|
||||
var blocks = [_]ArchPerCpu{.{}} ** config.max_cpus;
|
||||
|
||||
/// Publish core `index`'s per-CPU block: record the scheduler pointer and point
|
||||
/// the GS base at the block. Called once per core during bring-up, after the GDT
|
||||
/// is loaded (a GS *selector* reload would clobber the base).
|
||||
pub fn setLocal(index: usize, sched_ptr: usize) void {
|
||||
blocks[index].sched = sched_ptr;
|
||||
io.wrmsr(ia32_gs_base, @intFromPtr(&blocks[index]));
|
||||
}
|
||||
|
||||
/// The scheduler pointer for the running core (via the GS base). Valid in any
|
||||
/// ring-0 context under the swapgs discipline.
|
||||
pub fn sched() usize {
|
||||
return @as(*const ArchPerCpu, @ptrFromInt(io.rdmsr(ia32_gs_base))).sched;
|
||||
}
|
||||
|
||||
/// Record core `index`'s kernel stack top, used by the syscall entry stub to
|
||||
/// switch off the user stack. The scheduler sets this (and TSS.rsp0) whenever it
|
||||
/// switches to a user task.
|
||||
pub fn setKernelRsp(index: usize, top: usize) void {
|
||||
blocks[index].kernel_rsp = top;
|
||||
}
|
||||
@@ -20,6 +20,7 @@ const tss = @import("tss.zig");
|
||||
const idt = @import("idt.zig");
|
||||
const apic = @import("apic.zig");
|
||||
const paging = @import("paging.zig");
|
||||
const pcpu = @import("percpu.zig");
|
||||
|
||||
/// IA32_GS_BASE — the per-CPU data pointer (see cpu.zig; kept in sync here so the AP
|
||||
/// path doesn't depend on cpu.zig and risk an import cycle).
|
||||
@@ -168,7 +169,7 @@ fn apEntry(percpu: usize) callconv(.c) noreturn {
|
||||
gdt.loadOnThisCpu(cpu); // this core's GDT (with its own TSS slot)
|
||||
tss.setupThisCpu(cpu); // this core's TSS + IST stack, loaded into TR
|
||||
idt.loadOnThisCpu(); // the shared IDT
|
||||
io.wrmsr(ia32_gs_base, percpu); // per-CPU pointer — *after* the GDT reload
|
||||
pcpu.setLocal(cpu, percpu); // per-CPU block via GS base — *after* the GDT reload
|
||||
|
||||
apic.initSecondary(); // software-enable this core's LAPIC
|
||||
apic.initTimer(apic.frequencyHz()); // arm its timer (still masked: interrupts off)
|
||||
|
||||
@@ -104,7 +104,7 @@ var preemption_enabled = true;
|
||||
pub fn init(boot_priority: Priority) void {
|
||||
const pc = &cpus[0];
|
||||
pc.* = .{ .index = 0, .online = true, .loaded_pml4 = arch.kernelPageTable() };
|
||||
arch.setCpuLocal(@intFromPtr(pc));
|
||||
arch.setCpuLocal(0, @intFromPtr(pc));
|
||||
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
|
||||
pc.current = &tasks[0];
|
||||
pc.idle = create(idle, 0, null); // this core's idle task: always ready, lowest priority
|
||||
|
||||
Reference in New Issue
Block a user