danos/system/kernel/architecture/x86_64/isr.s

522 lines
21 KiB
ArmAsm

# x86_64 low-level entry code: the CPU-exception stubs, plus the GDT/IDT load
# helpers. Kept in a dedicated assembly file rather than inline asm because these
# need real labels and cross-symbol jumps/calls (isr_common, interruptDispatch),
# and because `lgdt`/`lidt` memory operands aren't expressible in Zig inline asm.
#
# Each exception vector normalises the stack to a uniform trap frame a dummy
# error code where the CPU pushes none, then the vector number and jumps to the
# shared tail, which saves the general registers and calls the Zig handler with a
# pointer to the frame (matching src/arch/x86_64/idt.zig's CpuState).
.text
# _start: the kernel entry. The loader jumps here (higher-half address) with
# boot_info in RDI, still on the loader's low stack. Switch to a kernel-owned
# stack in .bss (the loader stack is a low address that goes away once the low
# half is dropped), keeping RDI, then call the Zig entry. kmainEntry never
# returns; the hlt loop is a belt-and-braces backstop.
.global _start
_start:
leaq bootstrap_stack_top(%rip), %rsp
call kmainEntry
1: hlt
jmp 1b
# The kernel's initial stack (used until the scheduler hands each task its own).
# 64 KiB: kmain's discovery path includes the recursive AML interpreter, so it
# needs more than a token stack. Lives in .bss (zeroed, higher-half).
.section .bss
.balign 16
bootstrap_stack:
.skip 65536
bootstrap_stack_top:
.text
# gdt_flush(rdi = *GDT descriptor): load the GDT, reload the data segment
# registers to the data selector, and reload CS to the code selector. CS can't be
# set with mov, so we far-return through the caller's own return address.
.global gdt_flush
gdt_flush:
lgdt (%rdi)
mov $0x10, %ax # kernel data selector
mov %ax, %ds
mov %ax, %es
mov %ax, %ss
mov %ax, %fs
mov %ax, %gs
pop %rax # caller's return address
push $0x08 # kernel code selector (new CS)
push %rax # return address (new RIP)
lretq
# idt_flush(rdi = *IDT descriptor): load the IDT.
.global idt_flush
idt_flush:
lidt (%rdi)
ret
# load_tr(di = TSS selector): load the task register.
.global load_tr
load_tr:
ltr %di
ret
# switch_context(rdi = &old_task.rsp, rsi = new_task.rsp)
# Cooperative context switch: save the callee-saved registers on the current
# stack, stash the stack pointer in the old task, load the new task's stack
# pointer, restore its callee-saved registers, and return into it. Caller-saved
# registers are the compiler's responsibility (this looks like a normal call).
.global switch_context
switch_context:
push %rbx
push %rbp
push %r12
push %r13
push %r14
push %r15
mov %rsp, (%rdi) # save old stack pointer into old_task.rsp
mov %rsi, %rsp # switch to the new task's stack
pop %r15
pop %r14
pop %r13
pop %r12
pop %rbp
pop %rbx
ret # return into the new task's saved instruction pointer
# task_trampoline: the first thing a freshly-spawned task runs. init_task_stack
# leaves its entry function in r15. A fresh task is switched to with the big kernel
# lock held (the hand-off rule in sync.zig) but has no enter/leave frame of its own,
# so it releases the lock here before running its body. r15 survives the call (it's
# callee-saved). New tasks then start with interrupts enabled.
.extern releaseForFreshTask
.global task_trampoline
task_trampoline:
call releaseForFreshTask # drop the kernel lock we inherited across the switch
sti
call *%r15 # call the task entry (fn() void)
1: hlt # if the entry returns, idle (still preemptible)
jmp 1b
# user_task_trampoline: the first thing a freshly-spawned *user* task runs.
# init_user_task_stack leaves the user entry in r15 and the user stack in r14
# (both callee-saved, so they survive the lock-release call). Like task_trampoline
# it drops the inherited kernel lock, then instead of calling a kernel fn it
# builds an iretq frame and drops to ring 3. The scheduler's switchTo already
# loaded this task's address space (CR3) and published its kernel stack
# (TSS.rsp0 + gs kernel_rsp) before switching here, so interrupts and syscalls
# from ring 3 land correctly. IF is set in the pushed RFLAGS (no sti needed).
# jump_to_user(rdi = user rip, rsi = user rsp): drop the current kernel context
# to ring 3, never returning. The scheduler calls this from a fresh user task's
# trampoline (after the lock is released and the entry/stack read from the Task).
# cli guards the swapgs..iretq window: an interrupt there would run in ring 0
# with the user GS base and mis-read per-CPU data. The pushed RFLAGS (IF set)
# re-enables interrupts on the drop to ring 3.
.global jump_to_user
jump_to_user:
cli
push $0x1B # user SS (0x18 | RPL 3)
push %rsi # user RSP
push $0x202 # RFLAGS: IF | reserved-1
push $0x23 # user CS (0x20 | RPL 3)
push %rdi # user RIP
swapgs # user GS base (isr_common/syscall swap back on entry)
iretq
# jump_to_user_arg(rdi = user rip, rsi = user rsp, rdx = user rdi/arg0): as
# jump_to_user, but delivers arg0 in the user's rdi how a fresh **thread**
# receives its closure pointer (docs/threading.md). rdi carries the rip only until
# it is pushed into the iretq frame, after which we overwrite it with the arg.
.global jump_to_user_arg
jump_to_user_arg:
cli
push $0x1B # user SS (0x18 | RPL 3)
push %rsi # user RSP
push $0x202 # RFLAGS: IF | reserved-1
push $0x23 # user CS (0x20 | RPL 3)
push %rdi # user RIP (consumes rdi)
mov %rdx, %rdi # user rdi = arg0 (the thread's closure pointer)
swapgs # user GS base (isr_common/syscall swap back on entry)
iretq
# --- ring 3 entry/exit ------------------------------------------------------
# enter_user(rdi = user rip, rsi = user rsp, rdx = &TSS.rsp0)
# Drop to ring 3. Saves the callee-saved registers and the kernel stack pointer
# (so user_exit_to_kernel can unwind back here), publishes that stack pointer as
# this core's TSS.rsp0 everything below it is dead, so ring-3 interrupt frames
# grow safely into it then builds the 5-word iretq frame with the user
# selectors (RPL 3) and drops privilege. IF is set in the pushed RFLAGS so the
# timer keeps running in user mode.
.global enter_user
enter_user:
push %rbx
push %rbp
push %r12
push %r13
push %r14
push %r15
mov %rsp, user_saved_rsp(%rip) # where user_exit_to_kernel unwinds to
mov %rsp, (%rdx) # TSS.rsp0: ring-3 interrupts stack here
mov %rsp, %gs:0 # kernel_rsp: the syscall stub stacks here too
push $0x1B # user SS (0x18 | RPL 3)
push %rsi # user RSP
push $0x202 # RFLAGS: IF | reserved-1
push $0x23 # user CS (0x20 | RPL 3)
push %rdi # user RIP
swapgs # user GS base for ring 3 (isr_common swaps back)
iretq
# user_exit_to_kernel: abandon the in-flight ring-3 trap frame (it lives in the
# dead zone below user_saved_rsp) and return as if enter_user's call completed.
# Reached from the exit syscall's handler; interrupts are off (interrupt gate)
# and stay off the Zig caller re-enables.
.global user_exit_to_kernel
user_exit_to_kernel:
mov user_saved_rsp(%rip), %rsp
pop %r15
pop %r14
pop %r13
pop %r12
pop %rbp
pop %rbx
ret
# syscall_entry: the target of the SYSCALL instruction (LSTAR). The CPU does NOT
# switch stacks it puts the return RIP in RCX, the saved RFLAGS in R11, loads
# CS/SS from STAR, masks RFLAGS with SFMASK (so IF is already clear), and jumps
# here with RSP still the *user* stack. We swap in the kernel GS, switch to the
# task's kernel stack via the per-CPU block, build a CpuState frame identical to
# the interrupt path's, and reuse interruptDispatch (vector 128) then SYSRET,
# unless the return RIP is non-canonical, in which case the canonical-RIP guard at
# the exit below returns through IRETQ instead (docs/os-development/smep-smap.md).
.global syscall_entry
syscall_entry:
swapgs # kernel GS base
movq %rsp, %gs:8 # stash user rsp in the scratch slot
movq %gs:0, %rsp # switch to this task's kernel stack
# Build the trap frame (same field order as isr_common), highest field first.
pushq $0x1B # ss (user data | 3)
pushq %gs:8 # rsp (user, from scratch)
pushq %r11 # rflags (saved by syscall)
pushq $0x23 # cs (user code | 3)
pushq %rcx # rip (saved by syscall)
pushq $0 # error_code (none for a syscall)
pushq $128 # vector (same as the int 0x80 gate)
push %rax
push %rbx
push %rcx
push %rdx
push %rsi
push %rdi
push %rbp
push %r8
push %r9
push %r10
push %r11
push %r12
push %r13
push %r14
push %r15
mov %rsp, %rdi # trap-frame pointer
# Preserve the caller's SSE/x87 register file across the syscall see the same
# dance in isr_common. Without it a syscall (or a task the scheduler runs while
# this one blocks) clobbers the caller's live XMM values, which the compiler is
# free to hold across a syscall (its wrappers only clobber rcx/r11/memory).
mov %rsp, %rbx
and $-16, %rsp
sub $512, %rsp
fxsave (%rsp)
call interruptDispatch
fxrstor (%rsp)
mov %rbx, %rsp # back to the trap frame (undo the fxsave scratch)
pop %r15
pop %r14
pop %r13
pop %r12
pop %r11
pop %r10
pop %r9
pop %r8
pop %rbp
pop %rdi
pop %rsi
pop %rdx
pop %rcx
pop %rbx
pop %rax
add $16, %rsp # drop vector + error_code -> rsp at rip
popq %rcx # rip -> RCX (SYSRETQ restores RIP from RCX)
# --- canonical-RIP guard ---
# SYSRETQ with a non-canonical RCX raises #GP *in ring 0* on Intel: on the
# kernel stack, after the swapgs below has already installed the user's GS
# base a fault in the trusted base, on user-influenced state, which is the
# classic escalation primitive (CVE-2012-0217). Ring 3 gets to choose that RIP
# without any kernel bug: the CPU saves the address of the instruction *after*
# `syscall`, so a program executing `syscall` as the last two bytes of the last
# canonical page returns to 0x0000_8000_0000_0000. Nothing between entry and
# here rewrites the frame's rip (no signal or context-restore path exists that
# could), so this is the whole attack surface and one compare closes it.
#
# Canonical means bits 63:47 all equal bit 47, so sign-extending from bit 47
# and comparing is the complete test. Bit 47 is the right pivot because danos
# is 4-level only: paging.zig builds a PML4 and nothing anywhere sets CR4.LA57
# (bit 12), so a linear address is 48-bit on every core, on every machine we
# boot. A future 5-level port must pivot on bit 56 instead and should patch
# this shift pair at boot rather than branch on a feature flag, to keep the
# return path of every syscall in the system free of loads.
#
# Cost: four register-only ALU ops and one forward branch the predictor sees
# taken exactly never (a correct program cannot have a non-canonical return
# address it could not have executed there). R11 is free scratch: SYSCALL
# already destroyed the user's copy, and the fast path overwrites it with the
# saved RFLAGS two instructions further on.
movq %rcx, %r11
shlq $16, %r11
sarq $16, %r11 # sign-extend from bit 47
cmpq %rcx, %r11 # changed by the round trip => non-canonical
jne .Lnon_canonical_return
addq $8, %rsp # skip the cs slot (SYSRETQ loads CS from STAR)
popq %r11 # rflags -> R11 (SYSRETQ restores RFLAGS from R11)
popq %rsp # user rsp (the ss slot below is abandoned)
swapgs # user GS base
sysretq # -> ring 3: RIP=RCX, RFLAGS=R11, CS/SS from STAR
# The guard's cold path: return through IRETQ, which is safe where SYSRETQ is not.
# IRETQ loads CS committing the privilege change to ring 3 before the new RIP
# is fetched, so the #GP arrives *from ring 3*: through the IDT, onto this task's
# kernel stack, with a user CS in the frame, where isr_common swaps GS back and the
# kernel kills the process like any other user fault. That ordering is why IRETQ is
# the standard fallback for this exact case; the sysret-canonical test asserts it
# on the machine we run on (the process dies, the kernel does not).
#
# State handed to ring 3 is identical to what the fast path would have produced.
# IRETQ consumes the same 5-word frame the CPU pushes for an interrupt rip, cs,
# rflags, rsp, ss which is exactly the frame syscall_entry built and the fast
# path is part-way through dismantling, so un-popping the rip slot makes it whole:
# same user RIP, same user RSP, same RFLAGS, and CS/SS = 0x23/0x1B, the very
# selectors SYSRETQ would have loaded from STAR. R11 is reloaded from the frame's
# rflags slot so even the register SYSRET synthesizes matches. The swapgs sits in
# the same place relative to the ring change as the fast path's, so the swapgs
# discipline is untouched: kernel GS while we still touch kernel data, user GS for
# the instant before ring 3.
.Lnon_canonical_return:
# Cold-path diagnostic: how many hostile return addresses this boot refused.
# `lock` because every core shares the counter, and it costs nothing here a
# process that reaches this line is about to die.
lock incq sysret_non_canonical_count(%rip)
movq 8(%rsp), %r11 # rflags -> R11, exactly as the fast path leaves it
subq $8, %rsp # un-pop the rip slot: rsp back at the iretq frame
swapgs # user GS base
iretq # -> ring 3, where the bad RIP faults harmlessly
.section .bss
.balign 8
user_saved_rsp:
.skip 8
# Times the canonical-RIP guard above refused a SYSRETQ this boot. Read through the
# architecture layer (cpu.zig nonCanonicalReturnCount); zero on any machine no
# process has attacked.
.global sysret_non_canonical_count
sysret_non_canonical_count:
.skip 8
.text
# --- user-mode test program --------------------------------------------------
# A hand-assembled ring-3 blob, copied by the kernel onto a user-mapped page and
# entered via enter_user. Position-independent (immediates and short jumps only).
# In .rodata: these bytes are data to the kernel they only execute at CPL 3
# from the user mapping. (The old hello/ping blob was retired once /sbin/init
# became the real ring-3 exerciser; only the isolation proof remains.)
.section .rodata
# The isolation-proof program: read a kernel-only page from ring 3. The LAPIC
# lives in the kernel's physmap (physmap_base + 0xFEE00000) as a supervisor
# page, so this must take a #PF with error code 0x5 (present | user) before any
# access happens. movabs loads the full 64-bit higher-half address (a disp32
# would sign-extend and miss).
.global user_pf_start
.global user_pf_end
user_pf_start:
movabs $0xFFFF8800FEE00000, %rcx
mov (%rcx), %rax
1: jmp 1b
user_pf_end:
# The SYSRET-guard program: two harmless system calls, the second placed so that
# its *return address* is not canonical. The caller (tests.zig) copies these bytes
# to the very end of the last canonical user page, so the final `syscall` occupies
# the last two bytes of address space ring 3 can execute, and the RIP the CPU saves
# into RCX for it is 0x0000_8000_0000_0000 the first non-canonical address.
# The first call proves the ordinary SYSRETQ path still works (the program only
# reaches the second instruction pair by returning correctly from the first).
# 39 is abi.SystemCall.current_core: no arguments, no side effects, always
# succeeds; the test asserts the immediate below still matches that enum.
.global user_sysret_start
.global user_sysret_end
user_sysret_start:
mov $39, %eax # current_core
syscall # canonical return address (mid-page): the fast path
mov $39, %eax # current_core
syscall # return address = the end of the page = non-canonical
user_sysret_end:
.text
# Stub for a vector the CPU does NOT push an error code for: push a dummy 0.
.macro STUB_NOERR vec
.global isr\vec
isr\vec:
pushq $0
pushq $\vec
jmp isr_common
.endm
# Stub for a vector the CPU DOES push an error code for: leave it in place.
.macro STUB_ERR vec
.global isr\vec
isr\vec:
pushq $\vec
jmp isr_common
.endm
STUB_NOERR 0
STUB_NOERR 1
STUB_NOERR 2
STUB_NOERR 3
STUB_NOERR 4
STUB_NOERR 5
STUB_NOERR 6
STUB_NOERR 7
STUB_ERR 8
STUB_NOERR 9
STUB_ERR 10
STUB_ERR 11
STUB_ERR 12
STUB_ERR 13
STUB_ERR 14
STUB_NOERR 15
STUB_NOERR 16
STUB_ERR 17
STUB_NOERR 18
STUB_NOERR 19
STUB_NOERR 20
STUB_ERR 21
STUB_NOERR 22
STUB_NOERR 23
STUB_NOERR 24
STUB_NOERR 25
STUB_NOERR 26
STUB_NOERR 27
STUB_NOERR 28
STUB_NOERR 29
STUB_NOERR 30
STUB_NOERR 31
# Device-interrupt vectors (timer, spurious, room for more). None push an error
# code, so they all use the dummy-zero form.
STUB_NOERR 32
STUB_NOERR 33
STUB_NOERR 34
STUB_NOERR 35
STUB_NOERR 36
STUB_NOERR 37
STUB_NOERR 38
STUB_NOERR 39
STUB_NOERR 40
STUB_NOERR 41
STUB_NOERR 42
STUB_NOERR 43
STUB_NOERR 44
STUB_NOERR 45
STUB_NOERR 46
STUB_NOERR 47
# The syscall gate (int $0x80 from ring 3). Same frame shape as every other
# vector; dispatched specially in interruptDispatch.
STUB_NOERR 128
.extern interruptDispatch
# Shared tail. Register push order here defines the CpuState field order.
# If the interrupt came from ring 3 the GS base holds the user's value, so swap
# in the kernel's before anything reads per-CPU data (swapgs discipline; see
# percpu.zig). CS sits at offset 24 here (vector@0, error@8, RIP@16, CS@24).
#
# The first three bytes are the SMAP guard, and they come before everything
# before the CPL test, before the swapgs. Interrupt delivery does not clear
# EFLAGS.AC (SYSCALL does, through SFMASK; an IDT gate does not), and ring 3 sets
# AC freely with popfq, so without this a hostile process could take an interrupt
# with AC=1 and have the whole handler run with SMAP suspended. Ahead of the CPL
# test because ring 0 inherits AC just as readily: a fault or IRQ nested inside
# kernel code carries whatever AC the interrupted context had, and that context
# may itself be an entry that has not reached its own guard yet. Clearing first,
# unconditionally, means no path into the kernel is ever a path in with AC set.
#
# `clac` is #UD on a CPU without SMAP, so the image ships the 3-byte canonical NOP
# (`nopl (%rax)`) and per-cpu.zig overwrites it with `clac` (0f 01 ca) at boot,
# on the boot processor, only when CPUID says the instruction exists a one-time
# patch rather than a branch in the hottest path in the kernel. Neither encoding
# touches the flags the `testb` below sets, so the guard is invisible to the code
# that follows it either way.
.global isr_smap_patch
isr_common:
isr_smap_patch:
.byte 0x0f, 0x1f, 0x00 # nopl (%rax) -> patched to `clac` when the CPU has SMAP
testb $3, 24(%rsp)
jz 1f
swapgs
1: push %rax
push %rbx
push %rcx
push %rdx
push %rsi
push %rdi
push %rbp
push %r8
push %r9
push %r10
push %r11
push %r12
push %r13
push %r14
push %r15
mov %rsp, %rdi # first argument: pointer to the trap frame
# Save the interrupted SSE/x87 register file before any kernel code runs, and
# restore it on the way out the kernel and user both keep live values in XMM
# (a 16-byte struct copy is a movdqu), and the kernel never otherwise preserves
# them, so an interrupt handler (and whatever the scheduler runs in its place)
# would silently clobber the interrupted task's vector registers. rbx bridges the
# exact rsp across the call: it is callee-saved (interruptDispatch and every
# context switch preserve it), so it survives even a blocking dispatch, and the
# `and`/`sub` gives fxsave its required 16-byte-aligned scratch on the kernel stack.
mov %rsp, %rbx
and $-16, %rsp
sub $512, %rsp
fxsave (%rsp)
call interruptDispatch
fxrstor (%rsp)
mov %rbx, %rsp # back to the trap frame (undo the fxsave scratch)
pop %r15
pop %r14
pop %r13
pop %r12
pop %r11
pop %r10
pop %r9
pop %r8
pop %rbp
pop %rdi
pop %rsi
pop %rdx
pop %rcx
pop %rbx
pop %rax
add $16, %rsp # drop the vector and error code
# Symmetric to entry: if returning to ring 3, restore the user GS base. CS is
# now at offset 8 (RIP@0, CS@8).
testb $3, 8(%rsp)
jz 1f
swapgs
1: iretq