Files
Daniel Samson 7db5fba884 kernel: boot on AMD — SYSRET puts the RPL in the STAR base
A Ryzen 3 3200G triple-faulted on its first timer tick after reaching init.
Three defects in a chain, each hiding the one beneath it.

STAR's SYSRET base was 0x10, so SS came back as base+8 = 0x18 with RPL 0
while CS carried RPL 3. Intel ORs RPL 3 into SS on SYSRET; AMD only does so
for CS. Ring 3 ran fine — RPL is not checked on data access — and died the
moment an interrupt tried to IRETQ back, where SS.RPL must equal CS.RPL.
The base now carries the RPL (0x13), as Linux does.

Two fixes below it, both of which made the first one unreadable:

scheduler() read IA32_GS_BASE and dereferenced it without testing for zero,
so every fault reporter faulted in turn — a panic inside a panic, and the
machine reset before printing anything. Cast after the null test, plus a
re-entrancy guard in the panic handler.

NT is now masked in SFMASK alongside the rest, and isr.s exports
isr_return_iretq at the faulting instruction so a frame dump can say which
IRETQ died and print the CS/SS it was about to load. That dump is what
identified the RPL mismatch.
2026-08-08 11:07:55 +01:00

531 lines
21 KiB
ArmAsm

# x86_64 low-level entry code: the CPU-exception stubs, plus the GDT/IDT load
# helpers. Kept in a dedicated assembly file rather than inline asm because these
# need real labels and cross-symbol jumps/calls (isr_common, interruptDispatch),
# and because `lgdt`/`lidt` memory operands aren't expressible in Zig inline asm.
#
# Each exception vector normalises the stack to a uniform trap frame — a dummy
# error code where the CPU pushes none, then the vector number — and jumps to the
# shared tail, which saves the general registers and calls the Zig handler with a
# pointer to the frame (matching src/arch/x86_64/idt.zig's CpuState).
.text
# _start: the kernel entry. The loader jumps here (higher-half address) with
# boot_info in RDI, still on the loader's low stack. Switch to a kernel-owned
# stack in .bss (the loader stack is a low address that goes away once the low
# half is dropped), keeping RDI, then call the Zig entry. kmainEntry never
# returns; the hlt loop is a belt-and-braces backstop.
.global _start
_start:
leaq bootstrap_stack_top(%rip), %rsp
call kmainEntry
1: hlt
jmp 1b
# The kernel's initial stack (used until the scheduler hands each task its own).
# 64 KiB: kmain's discovery path includes the recursive AML interpreter, so it
# needs more than a token stack. Lives in .bss (zeroed, higher-half).
.section .bss
.balign 16
bootstrap_stack:
.skip 65536
bootstrap_stack_top:
.text
# gdt_flush(rdi = *GDT descriptor): load the GDT, reload the data segment
# registers to the data selector, and reload CS to the code selector. CS can't be
# set with mov, so we far-return through the caller's own return address.
.global gdt_flush
gdt_flush:
lgdt (%rdi)
mov $0x10, %ax # kernel data selector
mov %ax, %ds
mov %ax, %es
mov %ax, %ss
mov %ax, %fs
mov %ax, %gs
pop %rax # caller's return address
push $0x08 # kernel code selector (new CS)
push %rax # return address (new RIP)
lretq
# idt_flush(rdi = *IDT descriptor): load the IDT.
.global idt_flush
idt_flush:
lidt (%rdi)
ret
# load_tr(di = TSS selector): load the task register.
.global load_tr
load_tr:
ltr %di
ret
# switch_context(rdi = &old_task.rsp, rsi = new_task.rsp)
# Cooperative context switch: save the callee-saved registers on the current
# stack, stash the stack pointer in the old task, load the new task's stack
# pointer, restore its callee-saved registers, and return into it. Caller-saved
# registers are the compiler's responsibility (this looks like a normal call).
.global switch_context
switch_context:
push %rbx
push %rbp
push %r12
push %r13
push %r14
push %r15
mov %rsp, (%rdi) # save old stack pointer into old_task.rsp
mov %rsi, %rsp # switch to the new task's stack
pop %r15
pop %r14
pop %r13
pop %r12
pop %rbp
pop %rbx
ret # return into the new task's saved instruction pointer
# task_trampoline: the first thing a freshly-spawned task runs. init_task_stack
# leaves its entry function in r15. A fresh task is switched to with the big kernel
# lock held (the hand-off rule in sync.zig) but has no enter/leave frame of its own,
# so it releases the lock here before running its body. r15 survives the call (it's
# callee-saved). New tasks then start with interrupts enabled.
.extern releaseForFreshTask
.global task_trampoline
task_trampoline:
call releaseForFreshTask # drop the kernel lock we inherited across the switch
sti
call *%r15 # call the task entry (fn() void)
1: hlt # if the entry returns, idle (still preemptible)
jmp 1b
# user_task_trampoline: the first thing a freshly-spawned *user* task runs.
# init_user_task_stack leaves the user entry in r15 and the user stack in r14
# (both callee-saved, so they survive the lock-release call). Like task_trampoline
# it drops the inherited kernel lock, then — instead of calling a kernel fn — it
# builds an iretq frame and drops to ring 3. The scheduler's switchTo already
# loaded this task's address space (CR3) and published its kernel stack
# (TSS.rsp0 + gs kernel_rsp) before switching here, so interrupts and syscalls
# from ring 3 land correctly. IF is set in the pushed RFLAGS (no sti needed).
# jump_to_user(rdi = user rip, rsi = user rsp): drop the current kernel context
# to ring 3, never returning. The scheduler calls this from a fresh user task's
# trampoline (after the lock is released and the entry/stack read from the Task).
# cli guards the swapgs..iretq window: an interrupt there would run in ring 0
# with the user GS base and mis-read per-CPU data. The pushed RFLAGS (IF set)
# re-enables interrupts on the drop to ring 3.
.global jump_to_user
jump_to_user:
cli
push $0x1B # user SS (0x18 | RPL 3)
push %rsi # user RSP
push $0x202 # RFLAGS: IF | reserved-1
push $0x23 # user CS (0x20 | RPL 3)
push %rdi # user RIP
swapgs # user GS base (isr_common/syscall swap back on entry)
iretq
# jump_to_user_arg(rdi = user rip, rsi = user rsp, rdx = user rdi/arg0): as
# jump_to_user, but delivers arg0 in the user's rdi — how a fresh **thread**
# receives its closure pointer (docs/threading.md). rdi carries the rip only until
# it is pushed into the iretq frame, after which we overwrite it with the arg.
.global jump_to_user_arg
jump_to_user_arg:
cli
push $0x1B # user SS (0x18 | RPL 3)
push %rsi # user RSP
push $0x202 # RFLAGS: IF | reserved-1
push $0x23 # user CS (0x20 | RPL 3)
push %rdi # user RIP (consumes rdi)
mov %rdx, %rdi # user rdi = arg0 (the thread's closure pointer)
swapgs # user GS base (isr_common/syscall swap back on entry)
iretq
# --- ring 3 entry/exit ------------------------------------------------------
# enter_user(rdi = user rip, rsi = user rsp, rdx = &TSS.rsp0)
# Drop to ring 3. Saves the callee-saved registers and the kernel stack pointer
# (so user_exit_to_kernel can unwind back here), publishes that stack pointer as
# this core's TSS.rsp0 — everything below it is dead, so ring-3 interrupt frames
# grow safely into it — then builds the 5-word iretq frame with the user
# selectors (RPL 3) and drops privilege. IF is set in the pushed RFLAGS so the
# timer keeps running in user mode.
.global enter_user
enter_user:
push %rbx
push %rbp
push %r12
push %r13
push %r14
push %r15
mov %rsp, user_saved_rsp(%rip) # where user_exit_to_kernel unwinds to
mov %rsp, (%rdx) # TSS.rsp0: ring-3 interrupts stack here
mov %rsp, %gs:0 # kernel_rsp: the syscall stub stacks here too
push $0x1B # user SS (0x18 | RPL 3)
push %rsi # user RSP
push $0x202 # RFLAGS: IF | reserved-1
push $0x23 # user CS (0x20 | RPL 3)
push %rdi # user RIP
swapgs # user GS base for ring 3 (isr_common swaps back)
iretq
# user_exit_to_kernel: abandon the in-flight ring-3 trap frame (it lives in the
# dead zone below user_saved_rsp) and return as if enter_user's call completed.
# Reached from the exit syscall's handler; interrupts are off (interrupt gate)
# and stay off — the Zig caller re-enables.
.global user_exit_to_kernel
user_exit_to_kernel:
mov user_saved_rsp(%rip), %rsp
pop %r15
pop %r14
pop %r13
pop %r12
pop %rbp
pop %rbx
ret
# syscall_entry: the target of the SYSCALL instruction (LSTAR). The CPU does NOT
# switch stacks — it puts the return RIP in RCX, the saved RFLAGS in R11, loads
# CS/SS from STAR, masks RFLAGS with SFMASK (so IF is already clear), and jumps
# here with RSP still the *user* stack. We swap in the kernel GS, switch to the
# task's kernel stack via the per-CPU block, build a CpuState frame identical to
# the interrupt path's, and reuse interruptDispatch (vector 128) — then SYSRET,
# unless the return RIP is non-canonical, in which case the canonical-RIP guard at
# the exit below returns through IRETQ instead (docs/os-development/smep-smap.md).
.global syscall_entry
syscall_entry:
swapgs # kernel GS base
movq %rsp, %gs:8 # stash user rsp in the scratch slot
movq %gs:0, %rsp # switch to this task's kernel stack
# Build the trap frame (same field order as isr_common), highest field first.
pushq $0x1B # ss (user data | 3)
pushq %gs:8 # rsp (user, from scratch)
pushq %r11 # rflags (saved by syscall)
pushq $0x23 # cs (user code | 3)
pushq %rcx # rip (saved by syscall)
pushq $0 # error_code (none for a syscall)
pushq $128 # vector (same as the int 0x80 gate)
push %rax
push %rbx
push %rcx
push %rdx
push %rsi
push %rdi
push %rbp
push %r8
push %r9
push %r10
push %r11
push %r12
push %r13
push %r14
push %r15
mov %rsp, %rdi # trap-frame pointer
# Preserve the caller's SSE/x87 register file across the syscall — see the same
# dance in isr_common. Without it a syscall (or a task the scheduler runs while
# this one blocks) clobbers the caller's live XMM values, which the compiler is
# free to hold across a syscall (its wrappers only clobber rcx/r11/memory).
mov %rsp, %rbx
and $-16, %rsp
sub $512, %rsp
fxsave (%rsp)
call interruptDispatch
fxrstor (%rsp)
mov %rbx, %rsp # back to the trap frame (undo the fxsave scratch)
pop %r15
pop %r14
pop %r13
pop %r12
pop %r11
pop %r10
pop %r9
pop %r8
pop %rbp
pop %rdi
pop %rsi
pop %rdx
pop %rcx
pop %rbx
pop %rax
add $16, %rsp # drop vector + error_code -> rsp at rip
popq %rcx # rip -> RCX (SYSRETQ restores RIP from RCX)
# --- canonical-RIP guard ---
# SYSRETQ with a non-canonical RCX raises #GP *in ring 0* on Intel: on the
# kernel stack, after the swapgs below has already installed the user's GS
# base — a fault in the trusted base, on user-influenced state, which is the
# classic escalation primitive (CVE-2012-0217). Ring 3 gets to choose that RIP
# without any kernel bug: the CPU saves the address of the instruction *after*
# `syscall`, so a program executing `syscall` as the last two bytes of the last
# canonical page returns to 0x0000_8000_0000_0000. Nothing between entry and
# here rewrites the frame's rip (no signal or context-restore path exists that
# could), so this is the whole attack surface — and one compare closes it.
#
# Canonical means bits 63:47 all equal bit 47, so sign-extending from bit 47
# and comparing is the complete test. Bit 47 is the right pivot because danos
# is 4-level only: paging.zig builds a PML4 and nothing anywhere sets CR4.LA57
# (bit 12), so a linear address is 48-bit on every core, on every machine we
# boot. A future 5-level port must pivot on bit 56 instead — and should patch
# this shift pair at boot rather than branch on a feature flag, to keep the
# return path of every syscall in the system free of loads.
#
# Cost: four register-only ALU ops and one forward branch the predictor sees
# taken exactly never (a correct program cannot have a non-canonical return
# address — it could not have executed there). R11 is free scratch: SYSCALL
# already destroyed the user's copy, and the fast path overwrites it with the
# saved RFLAGS two instructions further on.
movq %rcx, %r11
shlq $16, %r11
sarq $16, %r11 # sign-extend from bit 47
cmpq %rcx, %r11 # changed by the round trip => non-canonical
jne .Lnon_canonical_return
addq $8, %rsp # skip the cs slot (SYSRETQ loads CS from STAR)
popq %r11 # rflags -> R11 (SYSRETQ restores RFLAGS from R11)
popq %rsp # user rsp (the ss slot below is abandoned)
swapgs # user GS base
sysretq # -> ring 3: RIP=RCX, RFLAGS=R11, CS/SS from STAR
# The guard's cold path: return through IRETQ, which is safe where SYSRETQ is not.
# IRETQ loads CS — committing the privilege change to ring 3 — before the new RIP
# is fetched, so the #GP arrives *from ring 3*: through the IDT, onto this task's
# kernel stack, with a user CS in the frame, where isr_common swaps GS back and the
# kernel kills the process like any other user fault. That ordering is why IRETQ is
# the standard fallback for this exact case; the sysret-canonical test asserts it
# on the machine we run on (the process dies, the kernel does not).
#
# State handed to ring 3 is identical to what the fast path would have produced.
# IRETQ consumes the same 5-word frame the CPU pushes for an interrupt — rip, cs,
# rflags, rsp, ss — which is exactly the frame syscall_entry built and the fast
# path is part-way through dismantling, so un-popping the rip slot makes it whole:
# same user RIP, same user RSP, same RFLAGS, and CS/SS = 0x23/0x1B, the very
# selectors SYSRETQ would have loaded from STAR. R11 is reloaded from the frame's
# rflags slot so even the register SYSRET synthesizes matches. The swapgs sits in
# the same place relative to the ring change as the fast path's, so the swapgs
# discipline is untouched: kernel GS while we still touch kernel data, user GS for
# the instant before ring 3.
.Lnon_canonical_return:
# Cold-path diagnostic: how many hostile return addresses this boot refused.
# `lock` because every core shares the counter, and it costs nothing here — a
# process that reaches this line is about to die.
lock incq sysret_non_canonical_count(%rip)
movq 8(%rsp), %r11 # rflags -> R11, exactly as the fast path leaves it
subq $8, %rsp # un-pop the rip slot: rsp back at the iretq frame
swapgs # user GS base
iretq # -> ring 3, where the bad RIP faults harmlessly
.section .bss
.balign 8
user_saved_rsp:
.skip 8
# Times the canonical-RIP guard above refused a SYSRETQ this boot. Read through the
# architecture layer (cpu.zig nonCanonicalReturnCount); zero on any machine no
# process has attacked.
.global sysret_non_canonical_count
sysret_non_canonical_count:
.skip 8
.text
# --- user-mode test program --------------------------------------------------
# A hand-assembled ring-3 blob, copied by the kernel onto a user-mapped page and
# entered via enter_user. Position-independent (immediates and short jumps only).
# In .rodata: these bytes are data to the kernel — they only execute at CPL 3
# from the user mapping. (The old hello/ping blob was retired once /sbin/init
# became the real ring-3 exerciser; only the isolation proof remains.)
.section .rodata
# The isolation-proof program: read a kernel-only page from ring 3. The LAPIC
# lives in the kernel's physmap (physmap_base + 0xFEE00000) as a supervisor
# page, so this must take a #PF with error code 0x5 (present | user) before any
# access happens. movabs loads the full 64-bit higher-half address (a disp32
# would sign-extend and miss).
.global user_pf_start
.global user_pf_end
user_pf_start:
movabs $0xFFFF8800FEE00000, %rcx
mov (%rcx), %rax
1: jmp 1b
user_pf_end:
# The SYSRET-guard program: two harmless system calls, the second placed so that
# its *return address* is not canonical. The caller (tests.zig) copies these bytes
# to the very end of the last canonical user page, so the final `syscall` occupies
# the last two bytes of address space ring 3 can execute, and the RIP the CPU saves
# into RCX for it is 0x0000_8000_0000_0000 — the first non-canonical address.
# The first call proves the ordinary SYSRETQ path still works (the program only
# reaches the second instruction pair by returning correctly from the first).
# 39 is abi.SystemCall.current_core: no arguments, no side effects, always
# succeeds; the test asserts the immediate below still matches that enum.
.global user_sysret_start
.global user_sysret_end
user_sysret_start:
mov $39, %eax # current_core
syscall # canonical return address (mid-page): the fast path
mov $39, %eax # current_core
syscall # return address = the end of the page = non-canonical
user_sysret_end:
.text
# Stub for a vector the CPU does NOT push an error code for: push a dummy 0.
.macro STUB_NOERR vec
.global isr\vec
isr\vec:
pushq $0
pushq $\vec
jmp isr_common
.endm
# Stub for a vector the CPU DOES push an error code for: leave it in place.
.macro STUB_ERR vec
.global isr\vec
isr\vec:
pushq $\vec
jmp isr_common
.endm
STUB_NOERR 0
STUB_NOERR 1
STUB_NOERR 2
STUB_NOERR 3
STUB_NOERR 4
STUB_NOERR 5
STUB_NOERR 6
STUB_NOERR 7
STUB_ERR 8
STUB_NOERR 9
STUB_ERR 10
STUB_ERR 11
STUB_ERR 12
STUB_ERR 13
STUB_ERR 14
STUB_NOERR 15
STUB_NOERR 16
STUB_ERR 17
STUB_NOERR 18
STUB_NOERR 19
STUB_NOERR 20
STUB_ERR 21
STUB_NOERR 22
STUB_NOERR 23
STUB_NOERR 24
STUB_NOERR 25
STUB_NOERR 26
STUB_NOERR 27
STUB_NOERR 28
STUB_NOERR 29
STUB_NOERR 30
STUB_NOERR 31
# Device-interrupt vectors (timer, spurious, room for more). None push an error
# code, so they all use the dummy-zero form.
STUB_NOERR 32
STUB_NOERR 33
STUB_NOERR 34
STUB_NOERR 35
STUB_NOERR 36
STUB_NOERR 37
STUB_NOERR 38
STUB_NOERR 39
STUB_NOERR 40
STUB_NOERR 41
STUB_NOERR 42
STUB_NOERR 43
STUB_NOERR 44
STUB_NOERR 45
STUB_NOERR 46
STUB_NOERR 47
# The syscall gate (int $0x80 from ring 3). Same frame shape as every other
# vector; dispatched specially in interruptDispatch.
STUB_NOERR 128
.extern interruptDispatch
# Shared tail. Register push order here defines the CpuState field order.
# If the interrupt came from ring 3 the GS base holds the user's value, so swap
# in the kernel's before anything reads per-CPU data (swapgs discipline; see
# percpu.zig). CS sits at offset 24 here (vector@0, error@8, RIP@16, CS@24).
#
# The first three bytes are the SMAP guard, and they come before everything —
# before the CPL test, before the swapgs. Interrupt delivery does not clear
# EFLAGS.AC (SYSCALL does, through SFMASK; an IDT gate does not), and ring 3 sets
# AC freely with popfq, so without this a hostile process could take an interrupt
# with AC=1 and have the whole handler run with SMAP suspended. Ahead of the CPL
# test because ring 0 inherits AC just as readily: a fault or IRQ nested inside
# kernel code carries whatever AC the interrupted context had, and that context
# may itself be an entry that has not reached its own guard yet. Clearing first,
# unconditionally, means no path into the kernel is ever a path in with AC set.
#
# `clac` is #UD on a CPU without SMAP, so the image ships the 3-byte canonical NOP
# (`nopl (%rax)`) and per-cpu.zig overwrites it with `clac` (0f 01 ca) at boot,
# on the boot processor, only when CPUID says the instruction exists — a one-time
# patch rather than a branch in the hottest path in the kernel. Neither encoding
# touches the flags the `testb` below sets, so the guard is invisible to the code
# that follows it either way.
.global isr_smap_patch
isr_common:
isr_smap_patch:
.byte 0x0f, 0x1f, 0x00 # nopl (%rax) -> patched to `clac` when the CPU has SMAP
testb $3, 24(%rsp)
jz 1f
swapgs
1: push %rax
push %rbx
push %rcx
push %rdx
push %rsi
push %rdi
push %rbp
push %r8
push %r9
push %r10
push %r11
push %r12
push %r13
push %r14
push %r15
mov %rsp, %rdi # first argument: pointer to the trap frame
# Save the interrupted SSE/x87 register file before any kernel code runs, and
# restore it on the way out — the kernel and user both keep live values in XMM
# (a 16-byte struct copy is a movdqu), and the kernel never otherwise preserves
# them, so an interrupt handler (and whatever the scheduler runs in its place)
# would silently clobber the interrupted task's vector registers. rbx bridges the
# exact rsp across the call: it is callee-saved (interruptDispatch and every
# context switch preserve it), so it survives even a blocking dispatch, and the
# `and`/`sub` gives fxsave its required 16-byte-aligned scratch on the kernel stack.
mov %rsp, %rbx
and $-16, %rsp
sub $512, %rsp
fxsave (%rsp)
call interruptDispatch
fxrstor (%rsp)
mov %rbx, %rsp # back to the trap frame (undo the fxsave scratch)
pop %r15
pop %r14
pop %r13
pop %r12
pop %r11
pop %r10
pop %r9
pop %r8
pop %rbp
pop %rdi
pop %rsi
pop %rdx
pop %rcx
pop %rbx
pop %rax
add $16, %rsp # drop the vector and error code
# Symmetric to entry: if returning to ring 3, restore the user GS base. CS is
# now at offset 8 (RIP@0, CS@8).
testb $3, 8(%rsp)
jz 1f
swapgs
# Named so a fault reporter can recognise its own return instruction. A #GP
# here is the frame's fault, not this code's — the five words below RSP are
# what the CPU rejected, and they are the only evidence of why. Note that on
# the ring-3 path the swapgs above has already run, so a fault at this exact
# address re-enters the kernel with the *user's* GS base: per-CPU reads in
# that handler are reading user-controlled state and must not be trusted.
1:
.global isr_return_iretq
isr_return_iretq:
iretq