kernel: boot on AMD — SYSRET puts the RPL in the STAR base
A Ryzen 3 3200G triple-faulted on its first timer tick after reaching init. Three defects in a chain, each hiding the one beneath it. STAR's SYSRET base was 0x10, so SS came back as base+8 = 0x18 with RPL 0 while CS carried RPL 3. Intel ORs RPL 3 into SS on SYSRET; AMD only does so for CS. Ring 3 ran fine — RPL is not checked on data access — and died the moment an interrupt tried to IRETQ back, where SS.RPL must equal CS.RPL. The base now carries the RPL (0x13), as Linux does. Two fixes below it, both of which made the first one unreadable: scheduler() read IA32_GS_BASE and dereferenced it without testing for zero, so every fault reporter faulted in turn — a panic inside a panic, and the machine reset before printing anything. Cast after the null test, plus a re-entrancy guard in the panic handler. NT is now masked in SFMASK alongside the rest, and isr.s exports isr_return_iretq at the faulting instruction so a frame dump can say which IRETQ died and print the CS/SS it was about to load. That dump is what identified the RPL mismatch.
This commit is contained in:
@@ -525,6 +525,40 @@ fn exitReasonForVector(vector: u64) abi.ExitReason {
|
||||
};
|
||||
}
|
||||
|
||||
/// A fault *at* the interrupt return itself has one piece of evidence worth
|
||||
/// having: the five words the CPU refused to load. `iretq` reports a selector in
|
||||
/// the error code but never says which of them carried it, and the frame is
|
||||
/// otherwise gone the moment the machine halts — so print it while it is still
|
||||
/// on the stack. Silent for every other fault, which is nearly all of them.
|
||||
///
|
||||
/// Two honesty notes ride along. On the ring-3 path the entry stub has already
|
||||
/// swapped in the user's GS base by the time this instruction runs, so the task
|
||||
/// and core named above came from user-controlled state and mean nothing; the
|
||||
/// frame's own CS says which case this is, so say so rather than let the reader
|
||||
/// trust a number. And the stack is only read after checking it is a kernel
|
||||
/// address, because a reporter that faults tells you nothing at all.
|
||||
fn reportPendingReturnFrame(state: *const architecture.CpuState) void {
|
||||
const site = @intFromPtr(@extern(*const anyopaque, .{ .name = "isr_return_iretq" }));
|
||||
if (architecture.instructionPointer(state) != site) return;
|
||||
|
||||
const sp = architecture.stackPointer(state);
|
||||
if (sp < 0xFFFF_8000_0000_0000 or sp % 8 != 0) {
|
||||
fatalPrint(" pending return frame: stack pointer unusable\n", .{});
|
||||
return;
|
||||
}
|
||||
const frame: [*]const u64 = @ptrFromInt(sp);
|
||||
fatalPrint(" the return this refused (rip, cs, rflags, rsp, ss):\n", .{});
|
||||
fatalPrint(" rip : 0x{x:0>16}\n", .{frame[0]});
|
||||
fatalPrint(" cs : 0x{x:0>4} (rpl {d})\n", .{ frame[1], frame[1] & 3 });
|
||||
fatalPrint(" rflags : 0x{x:0>16}\n", .{frame[2]});
|
||||
fatalPrint(" rsp : 0x{x:0>16}\n", .{frame[3]});
|
||||
fatalPrint(" ss : 0x{x:0>4} (rpl {d})\n", .{ frame[4], frame[4] & 3 });
|
||||
if (frame[1] & 3 == 3) {
|
||||
fatalPrint(" note: returning to ring 3, so the user GS base is installed —\n", .{});
|
||||
fatalPrint(" the task and core reported above are not to be believed.\n", .{});
|
||||
}
|
||||
}
|
||||
|
||||
fn onException(state: *const architecture.CpuState) noreturn {
|
||||
if (architecture.fromUser(state) and scheduler.currentIsUserProcess() and recoverableFault(state.vector)) {
|
||||
statusPrint("\n/system/kernel: process {d} ({s}) killed by {s} (vector {d}) on core {d}\n", .{ scheduler.currentId(), scheduler.current().name(), architecture.exceptionName(state.vector), state.vector, scheduler.currentCpuIndex() });
|
||||
@@ -548,6 +582,7 @@ fn onException(state: *const architecture.CpuState) noreturn {
|
||||
fatalPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
|
||||
fatalPrint(" SP : 0x{x:0>16}\n", .{architecture.stackPointer(state)});
|
||||
if (architecture.faultAddress(state)) |address| fatalPrint(" fault addr : 0x{x:0>16}\n", .{address});
|
||||
reportPendingReturnFrame(state);
|
||||
|
||||
var buffer: [128]u8 = undefined;
|
||||
log.recordPanic(std.fmt.bufPrint(&buffer, "CPU exception {s} (vector {d}) on core {d} at IP 0x{x}", .{ architecture.exceptionName(state.vector), state.vector, core, architecture.instructionPointer(state) }) catch "cpu exception");
|
||||
@@ -561,11 +596,23 @@ fn onException(state: *const architecture.CpuState) noreturn {
|
||||
/// Freestanding has no OS to receive a panic. Emit it to every output sink, drop a
|
||||
/// POST code + a persistent breadcrumb (so a post-mortem can recover it even with
|
||||
/// no live console), then halt. Assumes no console — the sinks self-guard.
|
||||
/// Set on entry to the panic handler, and never cleared: a panic is terminal, so
|
||||
/// the only reason to see this true is that reporting the first one faulted.
|
||||
var panicking: bool = false;
|
||||
|
||||
pub const panic = std.debug.FullPanic(struct {
|
||||
fn panic(message: []const u8, first_trace_address: ?usize) noreturn {
|
||||
_ = first_trace_address;
|
||||
// A panic while reporting a panic: stop where we are. Whatever the first
|
||||
// one already printed is the diagnosis, and every line after this point
|
||||
// risks faulting again — which on real hardware means a reset, taking the
|
||||
// message off the screen of the person who needed to read it. Halting
|
||||
// with a partial report beats rebooting with none.
|
||||
if (panicking) architecture.halt();
|
||||
panicking = true;
|
||||
|
||||
log.checkpoint(cp_panic);
|
||||
log.recordPanic(message);
|
||||
log.recordPanic(message); // the breadcrumb first: it survives what follows
|
||||
fatal("\nKERNEL PANIC: "); // a panic outranks any display service holding the screen
|
||||
fatal(message);
|
||||
fatal("\n");
|
||||
|
||||
Reference in New Issue
Block a user