diff --git a/system/kernel/architecture/x86_64/isr.s b/system/kernel/architecture/x86_64/isr.s index 277fcb3..38f92af 100644 --- a/system/kernel/architecture/x86_64/isr.s +++ b/system/kernel/architecture/x86_64/isr.s @@ -518,4 +518,13 @@ isr_smap_patch: testb $3, 8(%rsp) jz 1f swapgs -1: iretq + # Named so a fault reporter can recognise its own return instruction. A #GP + # here is the frame's fault, not this code's — the five words below RSP are + # what the CPU rejected, and they are the only evidence of why. Note that on + # the ring-3 path the swapgs above has already run, so a fault at this exact + # address re-enters the kernel with the *user's* GS base: per-CPU reads in + # that handler are reading user-controlled state and must not be trusted. +1: +.global isr_return_iretq +isr_return_iretq: + iretq diff --git a/system/kernel/architecture/x86_64/per-cpu.zig b/system/kernel/architecture/x86_64/per-cpu.zig index 9cd2ace..6e84bdd 100644 --- a/system/kernel/architecture/x86_64/per-cpu.zig +++ b/system/kernel/architecture/x86_64/per-cpu.zig @@ -57,10 +57,21 @@ pub fn setLocal(index: usize, scheduler_ptr: usize) void { io.wrmsr(ia32_gs_base, @intFromPtr(&blocks[index])); } -/// The scheduler pointer for the running core (via the GS base). Valid in any -/// ring-0 context under the swapgs discipline. +/// The scheduler pointer for the running core (via the GS base), or 0 before this +/// core has one. Valid in any ring-0 context under the swapgs discipline. +/// +/// **Zero is an answer here, not a fault.** A core has no per-CPU block between +/// reset and its `setLocal`, and the fault reporters are documented to be callable +/// unconditionally: `currentIdSafe`, `currentNameSafe`, `currentCpuIndex` and +/// `sync.releaseIfHeldHere` all test this for 0 and mean it. Casting the base +/// before testing it put the null one instruction out of their reach — an early +/// fault reported itself by panicking on the cast, and the panic handler, calling +/// those same reporters on its next line, panicked again, so the machine reset +/// instead of halting with the message that would have said what went wrong. pub fn scheduler() usize { - return @as(*const ArchitecturePerCpu, @ptrFromInt(io.rdmsr(ia32_gs_base))).scheduler; + const base = io.rdmsr(ia32_gs_base); + if (base == 0) return 0; + return @as(*const ArchitecturePerCpu, @ptrFromInt(base)).scheduler; } /// Record core `index`'s kernel stack top, used by the system_call entry stub to @@ -76,16 +87,35 @@ const ia32_star = 0xC000_0081; const ia32_lstar = 0xC000_0082; const ia32_sfmask = 0xC000_0084; +/// The SYSRET half of STAR: `sysret` loads CS from base+16 and SS from base+8. +/// The base carries RPL 3 itself — 0x13, not the 0x10 the GDT layout suggests — +/// because **only CS is guaranteed to come out at ring 3**. CS must: `sysret` +/// sets CPL to 3, so the selector it loads is forced to match. SS is under no +/// such obligation, and the two vendors differ on it: Intel ORs 3 into the SS +/// selector as well, AMD hands it over exactly as the arithmetic produced it. +/// +/// With base 0x10 an AMD machine therefore enters ring 3 carrying SS 0x18 — RPL +/// 0 — and nothing complains, because a data access is checked against CPL and +/// the descriptor's DPL, never the selector's RPL. It runs perfectly until the +/// first interrupt: the CPU pushes that SS, and the `iretq` returning to ring 3 +/// requires SS.RPL to equal CS.RPL, refusing 0 against 3 with #GP(0x18). A +/// process that made system calls happily then dies on its first timer tick. +/// +/// Putting the 3 in the base makes both selectors right by construction on +/// either vendor: 0x13 + 8 = 0x1B, 0x13 + 16 = 0x23. Intel's OR is then a no-op +/// rather than the thing holding it together. +const star_sysret_base: u64 = 0x13; // user data 0x18 / user code 0x20, with RPL 3 +const star_syscall_base: u64 = 0x08; // kernel code 0x08 / kernel data 0x10 + /// Enable the `system_call`/`sysret` fast path on this core (BSP and each AP). EFER.SCE /// turns the instructions on; STAR sets the selectors system_call/sysret load; LSTAR /// is the entry stub (isr.s); SFMASK clears RFLAGS bits on entry (notably IF — /// the handler runs with interrupts off, like the int-gate path). The GDT is laid /// out (kernel code 0x08, then user data 0x18 / code 0x20) precisely so these line -/// up: system_call loads CS 0x08 / SS 0x10; sysret loads CS = base+16 and SS = base+8 -/// with RPL forced to 3, so base 0x10 gives CS 0x23 (user code|3) and SS 0x1B. +/// up: system_call loads CS 0x08 / SS 0x10; sysret loads CS 0x23 and SS 0x1B. pub fn initSystemCall() void { io.wrmsr(ia32_efer, io.rdmsr(ia32_efer) | 1); // SCE - io.wrmsr(ia32_star, (@as(u64, 0x08) << 32) | (@as(u64, 0x10) << 48)); + io.wrmsr(ia32_star, (star_syscall_base << 32) | (star_sysret_base << 48)); const entry = @extern(*const anyopaque, .{ .name = "syscall_entry" }); io.wrmsr(ia32_lstar, @intFromPtr(entry)); // Clear IF, TF, DF, AC and NT on entry. The first four are the usual diff --git a/system/kernel/kernel.zig b/system/kernel/kernel.zig index c7eee06..e6d55cc 100644 --- a/system/kernel/kernel.zig +++ b/system/kernel/kernel.zig @@ -525,6 +525,40 @@ fn exitReasonForVector(vector: u64) abi.ExitReason { }; } +/// A fault *at* the interrupt return itself has one piece of evidence worth +/// having: the five words the CPU refused to load. `iretq` reports a selector in +/// the error code but never says which of them carried it, and the frame is +/// otherwise gone the moment the machine halts — so print it while it is still +/// on the stack. Silent for every other fault, which is nearly all of them. +/// +/// Two honesty notes ride along. On the ring-3 path the entry stub has already +/// swapped in the user's GS base by the time this instruction runs, so the task +/// and core named above came from user-controlled state and mean nothing; the +/// frame's own CS says which case this is, so say so rather than let the reader +/// trust a number. And the stack is only read after checking it is a kernel +/// address, because a reporter that faults tells you nothing at all. +fn reportPendingReturnFrame(state: *const architecture.CpuState) void { + const site = @intFromPtr(@extern(*const anyopaque, .{ .name = "isr_return_iretq" })); + if (architecture.instructionPointer(state) != site) return; + + const sp = architecture.stackPointer(state); + if (sp < 0xFFFF_8000_0000_0000 or sp % 8 != 0) { + fatalPrint(" pending return frame: stack pointer unusable\n", .{}); + return; + } + const frame: [*]const u64 = @ptrFromInt(sp); + fatalPrint(" the return this refused (rip, cs, rflags, rsp, ss):\n", .{}); + fatalPrint(" rip : 0x{x:0>16}\n", .{frame[0]}); + fatalPrint(" cs : 0x{x:0>4} (rpl {d})\n", .{ frame[1], frame[1] & 3 }); + fatalPrint(" rflags : 0x{x:0>16}\n", .{frame[2]}); + fatalPrint(" rsp : 0x{x:0>16}\n", .{frame[3]}); + fatalPrint(" ss : 0x{x:0>4} (rpl {d})\n", .{ frame[4], frame[4] & 3 }); + if (frame[1] & 3 == 3) { + fatalPrint(" note: returning to ring 3, so the user GS base is installed —\n", .{}); + fatalPrint(" the task and core reported above are not to be believed.\n", .{}); + } +} + fn onException(state: *const architecture.CpuState) noreturn { if (architecture.fromUser(state) and scheduler.currentIsUserProcess() and recoverableFault(state.vector)) { statusPrint("\n/system/kernel: process {d} ({s}) killed by {s} (vector {d}) on core {d}\n", .{ scheduler.currentId(), scheduler.current().name(), architecture.exceptionName(state.vector), state.vector, scheduler.currentCpuIndex() }); @@ -548,6 +582,7 @@ fn onException(state: *const architecture.CpuState) noreturn { fatalPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)}); fatalPrint(" SP : 0x{x:0>16}\n", .{architecture.stackPointer(state)}); if (architecture.faultAddress(state)) |address| fatalPrint(" fault addr : 0x{x:0>16}\n", .{address}); + reportPendingReturnFrame(state); var buffer: [128]u8 = undefined; log.recordPanic(std.fmt.bufPrint(&buffer, "CPU exception {s} (vector {d}) on core {d} at IP 0x{x}", .{ architecture.exceptionName(state.vector), state.vector, core, architecture.instructionPointer(state) }) catch "cpu exception"); @@ -561,11 +596,23 @@ fn onException(state: *const architecture.CpuState) noreturn { /// Freestanding has no OS to receive a panic. Emit it to every output sink, drop a /// POST code + a persistent breadcrumb (so a post-mortem can recover it even with /// no live console), then halt. Assumes no console — the sinks self-guard. +/// Set on entry to the panic handler, and never cleared: a panic is terminal, so +/// the only reason to see this true is that reporting the first one faulted. +var panicking: bool = false; + pub const panic = std.debug.FullPanic(struct { fn panic(message: []const u8, first_trace_address: ?usize) noreturn { _ = first_trace_address; + // A panic while reporting a panic: stop where we are. Whatever the first + // one already printed is the diagnosis, and every line after this point + // risks faulting again — which on real hardware means a reset, taking the + // message off the screen of the person who needed to read it. Halting + // with a partial report beats rebooting with none. + if (panicking) architecture.halt(); + panicking = true; + log.checkpoint(cp_panic); - log.recordPanic(message); + log.recordPanic(message); // the breadcrumb first: it survives what follows fatal("\nKERNEL PANIC: "); // a panic outranks any display service holding the screen fatal(message); fatal("\n");