A Ryzen 3 3200G triple-faulted on its first timer tick after reaching init. Three defects in a chain, each hiding the one beneath it. STAR's SYSRET base was 0x10, so SS came back as base+8 = 0x18 with RPL 0 while CS carried RPL 3. Intel ORs RPL 3 into SS on SYSRET; AMD only does so for CS. Ring 3 ran fine — RPL is not checked on data access — and died the moment an interrupt tried to IRETQ back, where SS.RPL must equal CS.RPL. The base now carries the RPL (0x13), as Linux does. Two fixes below it, both of which made the first one unreadable: scheduler() read IA32_GS_BASE and dereferenced it without testing for zero, so every fault reporter faulted in turn — a panic inside a panic, and the machine reset before printing anything. Cast after the null test, plus a re-entrancy guard in the panic handler. NT is now masked in SFMASK alongside the rest, and isr.s exports isr_return_iretq at the faulting instruction so a frame dump can say which IRETQ died and print the CS/SS it was about to load. That dump is what identified the RPL mismatch.
531 lines
21 KiB
ArmAsm
531 lines
21 KiB
ArmAsm
# x86_64 low-level entry code: the CPU-exception stubs, plus the GDT/IDT load
|
|
# helpers. Kept in a dedicated assembly file rather than inline asm because these
|
|
# need real labels and cross-symbol jumps/calls (isr_common, interruptDispatch),
|
|
# and because `lgdt`/`lidt` memory operands aren't expressible in Zig inline asm.
|
|
#
|
|
# Each exception vector normalises the stack to a uniform trap frame — a dummy
|
|
# error code where the CPU pushes none, then the vector number — and jumps to the
|
|
# shared tail, which saves the general registers and calls the Zig handler with a
|
|
# pointer to the frame (matching src/arch/x86_64/idt.zig's CpuState).
|
|
|
|
.text
|
|
|
|
# _start: the kernel entry. The loader jumps here (higher-half address) with
|
|
# boot_info in RDI, still on the loader's low stack. Switch to a kernel-owned
|
|
# stack in .bss (the loader stack is a low address that goes away once the low
|
|
# half is dropped), keeping RDI, then call the Zig entry. kmainEntry never
|
|
# returns; the hlt loop is a belt-and-braces backstop.
|
|
.global _start
|
|
_start:
|
|
leaq bootstrap_stack_top(%rip), %rsp
|
|
call kmainEntry
|
|
1: hlt
|
|
jmp 1b
|
|
|
|
# The kernel's initial stack (used until the scheduler hands each task its own).
|
|
# 64 KiB: kmain's discovery path includes the recursive AML interpreter, so it
|
|
# needs more than a token stack. Lives in .bss (zeroed, higher-half).
|
|
.section .bss
|
|
.balign 16
|
|
bootstrap_stack:
|
|
.skip 65536
|
|
bootstrap_stack_top:
|
|
.text
|
|
|
|
# gdt_flush(rdi = *GDT descriptor): load the GDT, reload the data segment
|
|
# registers to the data selector, and reload CS to the code selector. CS can't be
|
|
# set with mov, so we far-return through the caller's own return address.
|
|
.global gdt_flush
|
|
gdt_flush:
|
|
lgdt (%rdi)
|
|
mov $0x10, %ax # kernel data selector
|
|
mov %ax, %ds
|
|
mov %ax, %es
|
|
mov %ax, %ss
|
|
mov %ax, %fs
|
|
mov %ax, %gs
|
|
pop %rax # caller's return address
|
|
push $0x08 # kernel code selector (new CS)
|
|
push %rax # return address (new RIP)
|
|
lretq
|
|
|
|
# idt_flush(rdi = *IDT descriptor): load the IDT.
|
|
.global idt_flush
|
|
idt_flush:
|
|
lidt (%rdi)
|
|
ret
|
|
|
|
# load_tr(di = TSS selector): load the task register.
|
|
.global load_tr
|
|
load_tr:
|
|
ltr %di
|
|
ret
|
|
|
|
# switch_context(rdi = &old_task.rsp, rsi = new_task.rsp)
|
|
# Cooperative context switch: save the callee-saved registers on the current
|
|
# stack, stash the stack pointer in the old task, load the new task's stack
|
|
# pointer, restore its callee-saved registers, and return into it. Caller-saved
|
|
# registers are the compiler's responsibility (this looks like a normal call).
|
|
.global switch_context
|
|
switch_context:
|
|
push %rbx
|
|
push %rbp
|
|
push %r12
|
|
push %r13
|
|
push %r14
|
|
push %r15
|
|
mov %rsp, (%rdi) # save old stack pointer into old_task.rsp
|
|
mov %rsi, %rsp # switch to the new task's stack
|
|
pop %r15
|
|
pop %r14
|
|
pop %r13
|
|
pop %r12
|
|
pop %rbp
|
|
pop %rbx
|
|
ret # return into the new task's saved instruction pointer
|
|
|
|
# task_trampoline: the first thing a freshly-spawned task runs. init_task_stack
|
|
# leaves its entry function in r15. A fresh task is switched to with the big kernel
|
|
# lock held (the hand-off rule in sync.zig) but has no enter/leave frame of its own,
|
|
# so it releases the lock here before running its body. r15 survives the call (it's
|
|
# callee-saved). New tasks then start with interrupts enabled.
|
|
.extern releaseForFreshTask
|
|
.global task_trampoline
|
|
task_trampoline:
|
|
call releaseForFreshTask # drop the kernel lock we inherited across the switch
|
|
sti
|
|
call *%r15 # call the task entry (fn() void)
|
|
1: hlt # if the entry returns, idle (still preemptible)
|
|
jmp 1b
|
|
|
|
# user_task_trampoline: the first thing a freshly-spawned *user* task runs.
|
|
# init_user_task_stack leaves the user entry in r15 and the user stack in r14
|
|
# (both callee-saved, so they survive the lock-release call). Like task_trampoline
|
|
# it drops the inherited kernel lock, then — instead of calling a kernel fn — it
|
|
# builds an iretq frame and drops to ring 3. The scheduler's switchTo already
|
|
# loaded this task's address space (CR3) and published its kernel stack
|
|
# (TSS.rsp0 + gs kernel_rsp) before switching here, so interrupts and syscalls
|
|
# from ring 3 land correctly. IF is set in the pushed RFLAGS (no sti needed).
|
|
# jump_to_user(rdi = user rip, rsi = user rsp): drop the current kernel context
|
|
# to ring 3, never returning. The scheduler calls this from a fresh user task's
|
|
# trampoline (after the lock is released and the entry/stack read from the Task).
|
|
# cli guards the swapgs..iretq window: an interrupt there would run in ring 0
|
|
# with the user GS base and mis-read per-CPU data. The pushed RFLAGS (IF set)
|
|
# re-enables interrupts on the drop to ring 3.
|
|
.global jump_to_user
|
|
jump_to_user:
|
|
cli
|
|
push $0x1B # user SS (0x18 | RPL 3)
|
|
push %rsi # user RSP
|
|
push $0x202 # RFLAGS: IF | reserved-1
|
|
push $0x23 # user CS (0x20 | RPL 3)
|
|
push %rdi # user RIP
|
|
swapgs # user GS base (isr_common/syscall swap back on entry)
|
|
iretq
|
|
|
|
# jump_to_user_arg(rdi = user rip, rsi = user rsp, rdx = user rdi/arg0): as
|
|
# jump_to_user, but delivers arg0 in the user's rdi — how a fresh **thread**
|
|
# receives its closure pointer (docs/threading.md). rdi carries the rip only until
|
|
# it is pushed into the iretq frame, after which we overwrite it with the arg.
|
|
.global jump_to_user_arg
|
|
jump_to_user_arg:
|
|
cli
|
|
push $0x1B # user SS (0x18 | RPL 3)
|
|
push %rsi # user RSP
|
|
push $0x202 # RFLAGS: IF | reserved-1
|
|
push $0x23 # user CS (0x20 | RPL 3)
|
|
push %rdi # user RIP (consumes rdi)
|
|
mov %rdx, %rdi # user rdi = arg0 (the thread's closure pointer)
|
|
swapgs # user GS base (isr_common/syscall swap back on entry)
|
|
iretq
|
|
|
|
# --- ring 3 entry/exit ------------------------------------------------------
|
|
|
|
# enter_user(rdi = user rip, rsi = user rsp, rdx = &TSS.rsp0)
|
|
# Drop to ring 3. Saves the callee-saved registers and the kernel stack pointer
|
|
# (so user_exit_to_kernel can unwind back here), publishes that stack pointer as
|
|
# this core's TSS.rsp0 — everything below it is dead, so ring-3 interrupt frames
|
|
# grow safely into it — then builds the 5-word iretq frame with the user
|
|
# selectors (RPL 3) and drops privilege. IF is set in the pushed RFLAGS so the
|
|
# timer keeps running in user mode.
|
|
.global enter_user
|
|
enter_user:
|
|
push %rbx
|
|
push %rbp
|
|
push %r12
|
|
push %r13
|
|
push %r14
|
|
push %r15
|
|
mov %rsp, user_saved_rsp(%rip) # where user_exit_to_kernel unwinds to
|
|
mov %rsp, (%rdx) # TSS.rsp0: ring-3 interrupts stack here
|
|
mov %rsp, %gs:0 # kernel_rsp: the syscall stub stacks here too
|
|
push $0x1B # user SS (0x18 | RPL 3)
|
|
push %rsi # user RSP
|
|
push $0x202 # RFLAGS: IF | reserved-1
|
|
push $0x23 # user CS (0x20 | RPL 3)
|
|
push %rdi # user RIP
|
|
swapgs # user GS base for ring 3 (isr_common swaps back)
|
|
iretq
|
|
|
|
# user_exit_to_kernel: abandon the in-flight ring-3 trap frame (it lives in the
|
|
# dead zone below user_saved_rsp) and return as if enter_user's call completed.
|
|
# Reached from the exit syscall's handler; interrupts are off (interrupt gate)
|
|
# and stay off — the Zig caller re-enables.
|
|
.global user_exit_to_kernel
|
|
user_exit_to_kernel:
|
|
mov user_saved_rsp(%rip), %rsp
|
|
pop %r15
|
|
pop %r14
|
|
pop %r13
|
|
pop %r12
|
|
pop %rbp
|
|
pop %rbx
|
|
ret
|
|
|
|
# syscall_entry: the target of the SYSCALL instruction (LSTAR). The CPU does NOT
|
|
# switch stacks — it puts the return RIP in RCX, the saved RFLAGS in R11, loads
|
|
# CS/SS from STAR, masks RFLAGS with SFMASK (so IF is already clear), and jumps
|
|
# here with RSP still the *user* stack. We swap in the kernel GS, switch to the
|
|
# task's kernel stack via the per-CPU block, build a CpuState frame identical to
|
|
# the interrupt path's, and reuse interruptDispatch (vector 128) — then SYSRET,
|
|
# unless the return RIP is non-canonical, in which case the canonical-RIP guard at
|
|
# the exit below returns through IRETQ instead (docs/os-development/smep-smap.md).
|
|
.global syscall_entry
|
|
syscall_entry:
|
|
swapgs # kernel GS base
|
|
movq %rsp, %gs:8 # stash user rsp in the scratch slot
|
|
movq %gs:0, %rsp # switch to this task's kernel stack
|
|
# Build the trap frame (same field order as isr_common), highest field first.
|
|
pushq $0x1B # ss (user data | 3)
|
|
pushq %gs:8 # rsp (user, from scratch)
|
|
pushq %r11 # rflags (saved by syscall)
|
|
pushq $0x23 # cs (user code | 3)
|
|
pushq %rcx # rip (saved by syscall)
|
|
pushq $0 # error_code (none for a syscall)
|
|
pushq $128 # vector (same as the int 0x80 gate)
|
|
push %rax
|
|
push %rbx
|
|
push %rcx
|
|
push %rdx
|
|
push %rsi
|
|
push %rdi
|
|
push %rbp
|
|
push %r8
|
|
push %r9
|
|
push %r10
|
|
push %r11
|
|
push %r12
|
|
push %r13
|
|
push %r14
|
|
push %r15
|
|
mov %rsp, %rdi # trap-frame pointer
|
|
# Preserve the caller's SSE/x87 register file across the syscall — see the same
|
|
# dance in isr_common. Without it a syscall (or a task the scheduler runs while
|
|
# this one blocks) clobbers the caller's live XMM values, which the compiler is
|
|
# free to hold across a syscall (its wrappers only clobber rcx/r11/memory).
|
|
mov %rsp, %rbx
|
|
and $-16, %rsp
|
|
sub $512, %rsp
|
|
fxsave (%rsp)
|
|
call interruptDispatch
|
|
fxrstor (%rsp)
|
|
mov %rbx, %rsp # back to the trap frame (undo the fxsave scratch)
|
|
pop %r15
|
|
pop %r14
|
|
pop %r13
|
|
pop %r12
|
|
pop %r11
|
|
pop %r10
|
|
pop %r9
|
|
pop %r8
|
|
pop %rbp
|
|
pop %rdi
|
|
pop %rsi
|
|
pop %rdx
|
|
pop %rcx
|
|
pop %rbx
|
|
pop %rax
|
|
add $16, %rsp # drop vector + error_code -> rsp at rip
|
|
popq %rcx # rip -> RCX (SYSRETQ restores RIP from RCX)
|
|
# --- canonical-RIP guard ---
|
|
# SYSRETQ with a non-canonical RCX raises #GP *in ring 0* on Intel: on the
|
|
# kernel stack, after the swapgs below has already installed the user's GS
|
|
# base — a fault in the trusted base, on user-influenced state, which is the
|
|
# classic escalation primitive (CVE-2012-0217). Ring 3 gets to choose that RIP
|
|
# without any kernel bug: the CPU saves the address of the instruction *after*
|
|
# `syscall`, so a program executing `syscall` as the last two bytes of the last
|
|
# canonical page returns to 0x0000_8000_0000_0000. Nothing between entry and
|
|
# here rewrites the frame's rip (no signal or context-restore path exists that
|
|
# could), so this is the whole attack surface — and one compare closes it.
|
|
#
|
|
# Canonical means bits 63:47 all equal bit 47, so sign-extending from bit 47
|
|
# and comparing is the complete test. Bit 47 is the right pivot because danos
|
|
# is 4-level only: paging.zig builds a PML4 and nothing anywhere sets CR4.LA57
|
|
# (bit 12), so a linear address is 48-bit on every core, on every machine we
|
|
# boot. A future 5-level port must pivot on bit 56 instead — and should patch
|
|
# this shift pair at boot rather than branch on a feature flag, to keep the
|
|
# return path of every syscall in the system free of loads.
|
|
#
|
|
# Cost: four register-only ALU ops and one forward branch the predictor sees
|
|
# taken exactly never (a correct program cannot have a non-canonical return
|
|
# address — it could not have executed there). R11 is free scratch: SYSCALL
|
|
# already destroyed the user's copy, and the fast path overwrites it with the
|
|
# saved RFLAGS two instructions further on.
|
|
movq %rcx, %r11
|
|
shlq $16, %r11
|
|
sarq $16, %r11 # sign-extend from bit 47
|
|
cmpq %rcx, %r11 # changed by the round trip => non-canonical
|
|
jne .Lnon_canonical_return
|
|
addq $8, %rsp # skip the cs slot (SYSRETQ loads CS from STAR)
|
|
popq %r11 # rflags -> R11 (SYSRETQ restores RFLAGS from R11)
|
|
popq %rsp # user rsp (the ss slot below is abandoned)
|
|
swapgs # user GS base
|
|
sysretq # -> ring 3: RIP=RCX, RFLAGS=R11, CS/SS from STAR
|
|
|
|
# The guard's cold path: return through IRETQ, which is safe where SYSRETQ is not.
|
|
# IRETQ loads CS — committing the privilege change to ring 3 — before the new RIP
|
|
# is fetched, so the #GP arrives *from ring 3*: through the IDT, onto this task's
|
|
# kernel stack, with a user CS in the frame, where isr_common swaps GS back and the
|
|
# kernel kills the process like any other user fault. That ordering is why IRETQ is
|
|
# the standard fallback for this exact case; the sysret-canonical test asserts it
|
|
# on the machine we run on (the process dies, the kernel does not).
|
|
#
|
|
# State handed to ring 3 is identical to what the fast path would have produced.
|
|
# IRETQ consumes the same 5-word frame the CPU pushes for an interrupt — rip, cs,
|
|
# rflags, rsp, ss — which is exactly the frame syscall_entry built and the fast
|
|
# path is part-way through dismantling, so un-popping the rip slot makes it whole:
|
|
# same user RIP, same user RSP, same RFLAGS, and CS/SS = 0x23/0x1B, the very
|
|
# selectors SYSRETQ would have loaded from STAR. R11 is reloaded from the frame's
|
|
# rflags slot so even the register SYSRET synthesizes matches. The swapgs sits in
|
|
# the same place relative to the ring change as the fast path's, so the swapgs
|
|
# discipline is untouched: kernel GS while we still touch kernel data, user GS for
|
|
# the instant before ring 3.
|
|
.Lnon_canonical_return:
|
|
# Cold-path diagnostic: how many hostile return addresses this boot refused.
|
|
# `lock` because every core shares the counter, and it costs nothing here — a
|
|
# process that reaches this line is about to die.
|
|
lock incq sysret_non_canonical_count(%rip)
|
|
movq 8(%rsp), %r11 # rflags -> R11, exactly as the fast path leaves it
|
|
subq $8, %rsp # un-pop the rip slot: rsp back at the iretq frame
|
|
swapgs # user GS base
|
|
iretq # -> ring 3, where the bad RIP faults harmlessly
|
|
|
|
.section .bss
|
|
.balign 8
|
|
user_saved_rsp:
|
|
.skip 8
|
|
# Times the canonical-RIP guard above refused a SYSRETQ this boot. Read through the
|
|
# architecture layer (cpu.zig nonCanonicalReturnCount); zero on any machine no
|
|
# process has attacked.
|
|
.global sysret_non_canonical_count
|
|
sysret_non_canonical_count:
|
|
.skip 8
|
|
.text
|
|
|
|
# --- user-mode test program --------------------------------------------------
|
|
# A hand-assembled ring-3 blob, copied by the kernel onto a user-mapped page and
|
|
# entered via enter_user. Position-independent (immediates and short jumps only).
|
|
# In .rodata: these bytes are data to the kernel — they only execute at CPL 3
|
|
# from the user mapping. (The old hello/ping blob was retired once /sbin/init
|
|
# became the real ring-3 exerciser; only the isolation proof remains.)
|
|
.section .rodata
|
|
|
|
# The isolation-proof program: read a kernel-only page from ring 3. The LAPIC
|
|
# lives in the kernel's physmap (physmap_base + 0xFEE00000) as a supervisor
|
|
# page, so this must take a #PF with error code 0x5 (present | user) before any
|
|
# access happens. movabs loads the full 64-bit higher-half address (a disp32
|
|
# would sign-extend and miss).
|
|
.global user_pf_start
|
|
.global user_pf_end
|
|
user_pf_start:
|
|
movabs $0xFFFF8800FEE00000, %rcx
|
|
mov (%rcx), %rax
|
|
1: jmp 1b
|
|
user_pf_end:
|
|
|
|
# The SYSRET-guard program: two harmless system calls, the second placed so that
|
|
# its *return address* is not canonical. The caller (tests.zig) copies these bytes
|
|
# to the very end of the last canonical user page, so the final `syscall` occupies
|
|
# the last two bytes of address space ring 3 can execute, and the RIP the CPU saves
|
|
# into RCX for it is 0x0000_8000_0000_0000 — the first non-canonical address.
|
|
# The first call proves the ordinary SYSRETQ path still works (the program only
|
|
# reaches the second instruction pair by returning correctly from the first).
|
|
# 39 is abi.SystemCall.current_core: no arguments, no side effects, always
|
|
# succeeds; the test asserts the immediate below still matches that enum.
|
|
.global user_sysret_start
|
|
.global user_sysret_end
|
|
user_sysret_start:
|
|
mov $39, %eax # current_core
|
|
syscall # canonical return address (mid-page): the fast path
|
|
mov $39, %eax # current_core
|
|
syscall # return address = the end of the page = non-canonical
|
|
user_sysret_end:
|
|
|
|
.text
|
|
|
|
# Stub for a vector the CPU does NOT push an error code for: push a dummy 0.
|
|
.macro STUB_NOERR vec
|
|
.global isr\vec
|
|
isr\vec:
|
|
pushq $0
|
|
pushq $\vec
|
|
jmp isr_common
|
|
.endm
|
|
|
|
# Stub for a vector the CPU DOES push an error code for: leave it in place.
|
|
.macro STUB_ERR vec
|
|
.global isr\vec
|
|
isr\vec:
|
|
pushq $\vec
|
|
jmp isr_common
|
|
.endm
|
|
|
|
STUB_NOERR 0
|
|
STUB_NOERR 1
|
|
STUB_NOERR 2
|
|
STUB_NOERR 3
|
|
STUB_NOERR 4
|
|
STUB_NOERR 5
|
|
STUB_NOERR 6
|
|
STUB_NOERR 7
|
|
STUB_ERR 8
|
|
STUB_NOERR 9
|
|
STUB_ERR 10
|
|
STUB_ERR 11
|
|
STUB_ERR 12
|
|
STUB_ERR 13
|
|
STUB_ERR 14
|
|
STUB_NOERR 15
|
|
STUB_NOERR 16
|
|
STUB_ERR 17
|
|
STUB_NOERR 18
|
|
STUB_NOERR 19
|
|
STUB_NOERR 20
|
|
STUB_ERR 21
|
|
STUB_NOERR 22
|
|
STUB_NOERR 23
|
|
STUB_NOERR 24
|
|
STUB_NOERR 25
|
|
STUB_NOERR 26
|
|
STUB_NOERR 27
|
|
STUB_NOERR 28
|
|
STUB_NOERR 29
|
|
STUB_NOERR 30
|
|
STUB_NOERR 31
|
|
|
|
# Device-interrupt vectors (timer, spurious, room for more). None push an error
|
|
# code, so they all use the dummy-zero form.
|
|
STUB_NOERR 32
|
|
STUB_NOERR 33
|
|
STUB_NOERR 34
|
|
STUB_NOERR 35
|
|
STUB_NOERR 36
|
|
STUB_NOERR 37
|
|
STUB_NOERR 38
|
|
STUB_NOERR 39
|
|
STUB_NOERR 40
|
|
STUB_NOERR 41
|
|
STUB_NOERR 42
|
|
STUB_NOERR 43
|
|
STUB_NOERR 44
|
|
STUB_NOERR 45
|
|
STUB_NOERR 46
|
|
STUB_NOERR 47
|
|
|
|
# The syscall gate (int $0x80 from ring 3). Same frame shape as every other
|
|
# vector; dispatched specially in interruptDispatch.
|
|
STUB_NOERR 128
|
|
|
|
.extern interruptDispatch
|
|
|
|
# Shared tail. Register push order here defines the CpuState field order.
|
|
# If the interrupt came from ring 3 the GS base holds the user's value, so swap
|
|
# in the kernel's before anything reads per-CPU data (swapgs discipline; see
|
|
# percpu.zig). CS sits at offset 24 here (vector@0, error@8, RIP@16, CS@24).
|
|
#
|
|
# The first three bytes are the SMAP guard, and they come before everything —
|
|
# before the CPL test, before the swapgs. Interrupt delivery does not clear
|
|
# EFLAGS.AC (SYSCALL does, through SFMASK; an IDT gate does not), and ring 3 sets
|
|
# AC freely with popfq, so without this a hostile process could take an interrupt
|
|
# with AC=1 and have the whole handler run with SMAP suspended. Ahead of the CPL
|
|
# test because ring 0 inherits AC just as readily: a fault or IRQ nested inside
|
|
# kernel code carries whatever AC the interrupted context had, and that context
|
|
# may itself be an entry that has not reached its own guard yet. Clearing first,
|
|
# unconditionally, means no path into the kernel is ever a path in with AC set.
|
|
#
|
|
# `clac` is #UD on a CPU without SMAP, so the image ships the 3-byte canonical NOP
|
|
# (`nopl (%rax)`) and per-cpu.zig overwrites it with `clac` (0f 01 ca) at boot,
|
|
# on the boot processor, only when CPUID says the instruction exists — a one-time
|
|
# patch rather than a branch in the hottest path in the kernel. Neither encoding
|
|
# touches the flags the `testb` below sets, so the guard is invisible to the code
|
|
# that follows it either way.
|
|
.global isr_smap_patch
|
|
isr_common:
|
|
isr_smap_patch:
|
|
.byte 0x0f, 0x1f, 0x00 # nopl (%rax) -> patched to `clac` when the CPU has SMAP
|
|
testb $3, 24(%rsp)
|
|
jz 1f
|
|
swapgs
|
|
1: push %rax
|
|
push %rbx
|
|
push %rcx
|
|
push %rdx
|
|
push %rsi
|
|
push %rdi
|
|
push %rbp
|
|
push %r8
|
|
push %r9
|
|
push %r10
|
|
push %r11
|
|
push %r12
|
|
push %r13
|
|
push %r14
|
|
push %r15
|
|
mov %rsp, %rdi # first argument: pointer to the trap frame
|
|
# Save the interrupted SSE/x87 register file before any kernel code runs, and
|
|
# restore it on the way out — the kernel and user both keep live values in XMM
|
|
# (a 16-byte struct copy is a movdqu), and the kernel never otherwise preserves
|
|
# them, so an interrupt handler (and whatever the scheduler runs in its place)
|
|
# would silently clobber the interrupted task's vector registers. rbx bridges the
|
|
# exact rsp across the call: it is callee-saved (interruptDispatch and every
|
|
# context switch preserve it), so it survives even a blocking dispatch, and the
|
|
# `and`/`sub` gives fxsave its required 16-byte-aligned scratch on the kernel stack.
|
|
mov %rsp, %rbx
|
|
and $-16, %rsp
|
|
sub $512, %rsp
|
|
fxsave (%rsp)
|
|
call interruptDispatch
|
|
fxrstor (%rsp)
|
|
mov %rbx, %rsp # back to the trap frame (undo the fxsave scratch)
|
|
pop %r15
|
|
pop %r14
|
|
pop %r13
|
|
pop %r12
|
|
pop %r11
|
|
pop %r10
|
|
pop %r9
|
|
pop %r8
|
|
pop %rbp
|
|
pop %rdi
|
|
pop %rsi
|
|
pop %rdx
|
|
pop %rcx
|
|
pop %rbx
|
|
pop %rax
|
|
add $16, %rsp # drop the vector and error code
|
|
# Symmetric to entry: if returning to ring 3, restore the user GS base. CS is
|
|
# now at offset 8 (RIP@0, CS@8).
|
|
testb $3, 8(%rsp)
|
|
jz 1f
|
|
swapgs
|
|
# Named so a fault reporter can recognise its own return instruction. A #GP
|
|
# here is the frame's fault, not this code's — the five words below RSP are
|
|
# what the CPU rejected, and they are the only evidence of why. Note that on
|
|
# the ring-3 path the swapgs above has already run, so a fault at this exact
|
|
# address re-enters the kernel with the *user's* GS base: per-CPU reads in
|
|
# that handler are reading user-controlled state and must not be trusted.
|
|
1:
|
|
.global isr_return_iretq
|
|
isr_return_iretq:
|
|
iretq
|