kernel: contain a fatal fault — name the task, release the BKL before halt
Two gaps a real-hardware crash exposed, both in onException's terminal path (and the panic path): - The report was anonymous. Add the faulting task's id + name and whether it trapped in ring 3 (a user process) or ring 0 (the trusted base) — so a fatal fault says WHAT crashed and WHERE, not just the vector. scheduler gains currentIdSafe/currentNameSafe (early-boot-guarded, like currentCpuIndex, so the reporter can't fault a second time). - halt() never released the big kernel lock, so a core that died holding it deadlocked every other core spinning in acquire() — the whole machine hangs, not just the one core the design promises. The BKL now records its owner (architecture.cpuLocal(), a unique per-core token); sync.releaseIfHeldHere() frees the lock only if this core holds it, called before halt on both fatal paths. Caveat: if we held it mid-mutation the shared state may be inconsistent, but letting the other cores + the supervisor keep running is strictly more recoverable than a guaranteed total hang.
This commit is contained in:
@@ -9,6 +9,7 @@ const wall_clock = @import("wall-clock.zig");
|
||||
const pmm = @import("pmm.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const scheduler = @import("scheduler.zig");
|
||||
const sync = @import("sync.zig");
|
||||
const process = @import("process.zig");
|
||||
const devices_broker = @import("devices-broker.zig");
|
||||
const irq = @import("irq.zig");
|
||||
@@ -508,6 +509,10 @@ fn onException(state: *const architecture.CpuState) noreturn {
|
||||
// console back on even if a display service was holding the framebuffer — on top of the
|
||||
// diagnostic log.
|
||||
fatalPrint("\nCPU EXCEPTION on core {d}: {s} (vector {d})\n", .{ core, architecture.exceptionName(state.vector), state.vector });
|
||||
// Name the culprit: which task, and whether it faulted in ring 3 (a process the
|
||||
// kernel would normally kill — landing here means it had no address space) or ring 0
|
||||
// (the trusted base itself). Without this the fatal report is anonymous.
|
||||
fatalPrint(" task : {d} ({s}), {s}\n", .{ scheduler.currentIdSafe(), scheduler.currentNameSafe(), if (architecture.fromUser(state)) "ring 3 (user)" else "ring 0 (kernel)" });
|
||||
fatalPrint(" error code : 0x{x}\n", .{state.error_code});
|
||||
fatalPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
|
||||
fatalPrint(" SP : 0x{x:0>16}\n", .{architecture.stackPointer(state)});
|
||||
@@ -515,6 +520,10 @@ fn onException(state: *const architecture.CpuState) noreturn {
|
||||
|
||||
var buffer: [128]u8 = undefined;
|
||||
log.recordPanic(std.fmt.bufPrint(&buffer, "CPU exception {s} (vector {d}) on core {d} at IP 0x{x}", .{ architecture.exceptionName(state.vector), state.vector, core, architecture.instructionPointer(state) }) catch "cpu exception");
|
||||
// Free the BKL if this core held it (a kernel-mode fault, or a nested fault in the
|
||||
// recovery teardown), so halting this one core doesn't deadlock every other core on
|
||||
// the lock. Only that core stops; the rest — and the supervisor — keep running.
|
||||
sync.releaseIfHeldHere();
|
||||
architecture.halt();
|
||||
}
|
||||
|
||||
@@ -529,6 +538,8 @@ pub const panic = std.debug.FullPanic(struct {
|
||||
fatal("\nKERNEL PANIC: "); // a panic outranks any display service holding the screen
|
||||
fatal(message);
|
||||
fatal("\n");
|
||||
fatalPrint(" task : {d} ({s})\n", .{ scheduler.currentIdSafe(), scheduler.currentNameSafe() });
|
||||
sync.releaseIfHeldHere(); // don't deadlock the other cores on the lock we may hold
|
||||
architecture.halt();
|
||||
}
|
||||
}.panic);
|
||||
|
||||
Reference in New Issue
Block a user