2 Commits
Author SHA1 Message Date
Daniel Samson a01a4f3b3d init: restart a crashed boot service
init supervises its boot services (spawned against an exit endpoint) but the
child-exit handler was `if (got.isNotification()) continue;` — it silently
dropped a dead service. Now init restarts it: on a child-exit notification it
finds the service, and unless it exited cleanly (chose to stop) or has hit the
crash-loop cap (maximum_restarts), respawns it and logs the death + reason. The
reincarnation half of resilience (docs/resilience.md) at the service level, the
counterpart to the device manager's driver restarts. A `shutting_down` flag
skips restarts during the orderly stop sequence, whose child deaths are expected.
2026-07-14 09:05:10 +01:00
Daniel Samson e1605e3235 kernel: contain a fatal fault — name the task, release the BKL before halt
Two gaps a real-hardware crash exposed, both in onException's terminal path (and
the panic path):

- The report was anonymous. Add the faulting task's id + name and whether it
  trapped in ring 3 (a user process) or ring 0 (the trusted base) — so a fatal
  fault says WHAT crashed and WHERE, not just the vector. scheduler gains
  currentIdSafe/currentNameSafe (early-boot-guarded, like currentCpuIndex, so the
  reporter can't fault a second time).

- halt() never released the big kernel lock, so a core that died holding it
  deadlocked every other core spinning in acquire() — the whole machine hangs,
  not just the one core the design promises. The BKL now records its owner
  (architecture.cpuLocal(), a unique per-core token); sync.releaseIfHeldHere()
  frees the lock only if this core holds it, called before halt on both fatal
  paths. Caveat: if we held it mid-mutation the shared state may be inconsistent,
  but letting the other cores + the supervisor keep running is strictly more
  recoverable than a guaranteed total hang.
2026-07-14 09:05:10 +01:00
4 changed files with 104 additions and 12 deletions
+11
View File
@@ -9,6 +9,7 @@ const wall_clock = @import("wall-clock.zig");
const pmm = @import("pmm.zig"); const pmm = @import("pmm.zig");
const heap = @import("heap.zig"); const heap = @import("heap.zig");
const scheduler = @import("scheduler.zig"); const scheduler = @import("scheduler.zig");
const sync = @import("sync.zig");
const process = @import("process.zig"); const process = @import("process.zig");
const devices_broker = @import("devices-broker.zig"); const devices_broker = @import("devices-broker.zig");
const irq = @import("irq.zig"); const irq = @import("irq.zig");
@@ -508,6 +509,10 @@ fn onException(state: *const architecture.CpuState) noreturn {
// console back on even if a display service was holding the framebuffer — on top of the // console back on even if a display service was holding the framebuffer — on top of the
// diagnostic log. // diagnostic log.
fatalPrint("\nCPU EXCEPTION on core {d}: {s} (vector {d})\n", .{ core, architecture.exceptionName(state.vector), state.vector }); fatalPrint("\nCPU EXCEPTION on core {d}: {s} (vector {d})\n", .{ core, architecture.exceptionName(state.vector), state.vector });
// Name the culprit: which task, and whether it faulted in ring 3 (a process the
// kernel would normally kill — landing here means it had no address space) or ring 0
// (the trusted base itself). Without this the fatal report is anonymous.
fatalPrint(" task : {d} ({s}), {s}\n", .{ scheduler.currentIdSafe(), scheduler.currentNameSafe(), if (architecture.fromUser(state)) "ring 3 (user)" else "ring 0 (kernel)" });
fatalPrint(" error code : 0x{x}\n", .{state.error_code}); fatalPrint(" error code : 0x{x}\n", .{state.error_code});
fatalPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)}); fatalPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
fatalPrint(" SP : 0x{x:0>16}\n", .{architecture.stackPointer(state)}); fatalPrint(" SP : 0x{x:0>16}\n", .{architecture.stackPointer(state)});
@@ -515,6 +520,10 @@ fn onException(state: *const architecture.CpuState) noreturn {
var buffer: [128]u8 = undefined; var buffer: [128]u8 = undefined;
log.recordPanic(std.fmt.bufPrint(&buffer, "CPU exception {s} (vector {d}) on core {d} at IP 0x{x}", .{ architecture.exceptionName(state.vector), state.vector, core, architecture.instructionPointer(state) }) catch "cpu exception"); log.recordPanic(std.fmt.bufPrint(&buffer, "CPU exception {s} (vector {d}) on core {d} at IP 0x{x}", .{ architecture.exceptionName(state.vector), state.vector, core, architecture.instructionPointer(state) }) catch "cpu exception");
// Free the BKL if this core held it (a kernel-mode fault, or a nested fault in the
// recovery teardown), so halting this one core doesn't deadlock every other core on
// the lock. Only that core stops; the rest — and the supervisor — keep running.
sync.releaseIfHeldHere();
architecture.halt(); architecture.halt();
} }
@@ -529,6 +538,8 @@ pub const panic = std.debug.FullPanic(struct {
fatal("\nKERNEL PANIC: "); // a panic outranks any display service holding the screen fatal("\nKERNEL PANIC: "); // a panic outranks any display service holding the screen
fatal(message); fatal(message);
fatal("\n"); fatal("\n");
fatalPrint(" task : {d} ({s})\n", .{ scheduler.currentIdSafe(), scheduler.currentNameSafe() });
sync.releaseIfHeldHere(); // don't deadlock the other cores on the lock we may hold
architecture.halt(); architecture.halt();
} }
}.panic); }.panic);
+15
View File
@@ -791,6 +791,21 @@ pub fn currentCpuIndex() u32 {
return thisCpu().index; return thisCpu().index;
} }
/// The running task's id, or 0 if this core's scheduler isn't up yet (early boot, no GS
/// base). Safe for a fault reporter to call unconditionally — like `currentCpuIndex`,
/// it never dereferences an unpublished per-CPU pointer and so can't fault a second time.
pub fn currentIdSafe() u32 {
if (architecture.cpuLocal() == 0) return 0;
return thisCpu().current.id;
}
/// The running task's name (argv[0]), or "" if this core's scheduler isn't up yet.
/// The companion to `currentIdSafe` for naming the culprit in a fatal fault report.
pub fn currentNameSafe() []const u8 {
if (architecture.cpuLocal() == 0) return "";
return thisCpu().current.name();
}
/// Change the running task's priority (takes effect next time it's enqueued). /// Change the running task's priority (takes effect next time it's enqueued).
pub fn setPriority(p: Priority) void { pub fn setPriority(p: Priority) void {
current().priority = p; current().priority = p;
+22
View File
@@ -33,6 +33,13 @@ const architecture = @import("architecture");
/// 0 = free, 1 = held. A single global lock for the whole kernel. /// 0 = free, 1 = held. A single global lock for the whole kernel.
var held = std.atomic.Value(u32).init(0); var held = std.atomic.Value(u32).init(0);
/// The per-CPU base pointer (`architecture.cpuLocal()`) of the core currently holding
/// the lock, or 0 when free. Metadata only — `held` is what enforces exclusion — read
/// solely by `releaseIfHeldHere` on the fatal-fault path. `cpuLocal()` is a unique,
/// architecture-level token per core (0 before this core's GS base is published, which
/// is fine: that window is single-core early boot, where no other core can deadlock).
var owner = std.atomic.Value(usize).init(0);
/// Enter the kernel: disable interrupts on this core, then spin until we own the /// Enter the kernel: disable interrupts on this core, then spin until we own the
/// lock. Returns the caller's prior interrupt flags for `leave` to restore. /// lock. Returns the caller's prior interrupt flags for `leave` to restore.
/// Interrupts stay off for the whole critical section so this core's timer tick /// Interrupts stay off for the whole critical section so this core's timer tick
@@ -67,14 +74,29 @@ export fn releaseForFreshTask() callconv(.c) void {
release(); release();
} }
/// Release the big kernel lock **only if this core is the one holding it** — a no-op
/// otherwise. For the fatal-fault path (a kernel-mode fault, or a nested fault inside the
/// recovery teardown, both of which run under the lock): a core that dies holding the BKL
/// must free it, or every other core spins forever in `acquire` and the whole machine
/// deadlocks instead of just that core stopping. It must NOT free a lock another core
/// owns, hence the owner check. Caveat: if we held it mid-mutation the shared state may be
/// inconsistent — but letting the other cores (and the supervisor) run on possibly-degraded
/// state is strictly more recoverable than a guaranteed total hang.
pub fn releaseIfHeldHere() void {
const me = architecture.cpuLocal();
if (me != 0 and owner.load(.monotonic) == me) release();
}
fn acquire() void { fn acquire() void {
// Test-and-test-and-set: try once, then spin read-only until the lock looks // Test-and-test-and-set: try once, then spin read-only until the lock looks
// free before retrying the (bus-locked) swap — cheaper on the coherency fabric. // free before retrying the (bus-locked) swap — cheaper on the coherency fabric.
while (held.swap(1, .acquire) != 0) { while (held.swap(1, .acquire) != 0) {
while (held.load(.monotonic) != 0) architecture.cpuRelax(); while (held.load(.monotonic) != 0) architecture.cpuRelax();
} }
owner.store(architecture.cpuLocal(), .monotonic);
} }
fn release() void { fn release() void {
owner.store(0, .monotonic);
held.store(0, .release); held.store(0, .release);
} }
+56 -12
View File
@@ -32,10 +32,20 @@ const log_path = "/mnt/usb/DANOS.LOG";
/// manifest under /system/services instead of a hardcoded list.) /// manifest under /system/services instead of a hardcoded list.)
const boot_services = [_][]const u8{ "vfs", "input", "device-manager", "fat", "display", "display-demo" }; const boot_services = [_][]const u8{ "vfs", "input", "device-manager", "fat", "display", "display-demo" };
var children: [boot_services.len]u32 = .{0} ** boot_services.len; /// The live process id of each boot service (0 = not running), indexed by its position
var child_count: usize = 0; /// in `boot_services`, plus how many times init has restarted it. init supervises these:
/// it spawns them against `supervision_endpoint` and, on a child's death, restarts it (up
/// to `maximum_restarts`) — the reincarnation half of resilience (docs/resilience.md), the
/// service-level counterpart to the device manager's driver restarts.
var child_ids: [boot_services.len]u32 = .{0} ** boot_services.len;
var restart_counts: [boot_services.len]u32 = .{0} ** boot_services.len;
var shutting_down = false;
var supervision_endpoint: runtime.ipc.Handle = 0; var supervision_endpoint: runtime.ipc.Handle = 0;
/// Give up restarting a service after this many crashes — a crash-loop cap, so a service
/// that faults immediately on every spawn doesn't respawn forever.
const maximum_restarts = 3;
pub fn main() void { pub fn main() void {
// Prove the heap end to end: allocate through the runtime allocator (which // Prove the heap end to end: allocate through the runtime allocator (which
// mmaps pages from the kernel and carves them with the free list), write into // mmaps pages from the kernel and carves them with the free list), write into
@@ -63,11 +73,8 @@ pub fn main() void {
// Bring up the boot services, supervised so init can stop them cleanly. // Bring up the boot services, supervised so init can stop them cleanly.
// Best-effort and silent: each service announces its own readiness, and in // Best-effort and silent: each service announces its own readiness, and in
// an isolation test with no initial-ramdisk the spawns simply no-op. // an isolation test with no initial-ramdisk the spawns simply no-op.
for (boot_services) |service| { for (boot_services, 0..) |service, i| {
if (runtime.system.spawnSupervised(service, &.{}, supervision_endpoint)) |id| { if (runtime.system.spawnSupervised(service, &.{}, supervision_endpoint)) |id| child_ids[i] = id;
children[child_count] = id;
child_count += 1;
}
} }
// Once the storage stack is up, a one-shot copies the boot log to the USB // Once the storage stack is up, a one-shot copies the boot log to the USB
@@ -105,11 +112,47 @@ pub fn main() void {
if (receive[1] == @intFromEnum(power.Event.power_button)) shutDown(); if (receive[1] == @intFromEnum(power.Event.power_button)) shutDown();
continue; continue;
} }
// Child-exit notifications and anything else: keep waiting. if (got.isChildExit()) {
restartChild(got.childProcessId());
continue;
}
// Anything else: keep waiting.
if (got.isNotification()) continue; if (got.isNotification()) continue;
} }
} }
/// A supervised boot service died. Find which one and restart it — unless it exited
/// cleanly (it chose to stop, e.g. a driver with no hardware) or has hit the crash-loop
/// cap. Reclaiming the dead process is already the kernel's job (docs/process-lifecycle.md
/// iron rule 1); init only decides whether to bring it back.
fn restartChild(id: u32) void {
if (shutting_down) return; // deaths during the stop sequence are expected, not crashes
for (boot_services, 0..) |service, i| {
if (child_ids[i] != id) continue;
child_ids[i] = 0;
// An unknown reason (the record aged out) is treated as a crash worth restarting.
const reason = runtime.process.exitReason(id) orelse .fault;
if (reason == .exited) {
logLine("/system/services/init: {s} exited cleanly; not restarting\n", .{service});
return;
}
restart_counts[i] += 1;
if (restart_counts[i] > maximum_restarts) {
logLine("/system/services/init: {s} keeps crashing; giving up after {d} restarts\n", .{ service, maximum_restarts });
return;
}
logLine("/system/services/init: {s} died ({s}); restarting ({d}/{d})\n", .{ service, @tagName(reason), restart_counts[i], maximum_restarts });
if (runtime.system.spawnSupervised(service, &.{}, supervision_endpoint)) |new_id| child_ids[i] = new_id;
return;
}
// An untracked child (e.g. the log-flush one-shot): nothing to restart.
}
fn logLine(comptime fmt: []const u8, args: anytype) void {
var line: [128]u8 = undefined;
_ = runtime.system.write(std.fmt.bufPrint(&line, fmt, args) catch return);
}
/// Look up the power service and subscribe our endpoint (handed over as the /// Look up the power service and subscribe our endpoint (handed over as the
/// call's capability) so events arrive as buffered messages here. /// call's capability) so events arrive as buffered messages here.
fn subscribePower() void { fn subscribePower() void {
@@ -153,15 +196,16 @@ fn flushKernelLog() void {
/// it), waiting up to a deadline for each to exit before killing it, then ask the /// it), waiting up to a deadline for each to exit before killing it, then ask the
/// power service to enter S5. /// power service to enter S5.
fn shutDown() void { fn shutDown() void {
shutting_down = true; // the stop loop below kills children — those deaths aren't crashes
_ = runtime.system.write("/system/services/init: shutting down\n"); _ = runtime.system.write("/system/services/init: shutting down\n");
// Persist the fullest log to the USB volume BEFORE tearing anything down: the // Persist the fullest log to the USB volume BEFORE tearing anything down: the
// reverse-order stop loop below kills the fat server (children[3]) first, so // reverse-order stop loop below kills the fat server first, so /mnt/usb must be
// /mnt/usb must be written while it is still mounted. // written while it is still mounted.
flushKernelLog(); flushKernelLog();
var i = child_count; var i = boot_services.len;
while (i > 0) { while (i > 0) {
i -= 1; i -= 1;
if (children[i] != 0) runtime.process.stop(children[i], 2000, supervision_endpoint); if (child_ids[i] != 0) runtime.process.stop(child_ids[i], 2000, supervision_endpoint);
} }
if (runtime.ipc.lookup(.power)) |h| { if (runtime.ipc.lookup(.power)) |h| {
const request = power.Shutdown{}; const request = power.Shutdown{};