Record and expose how every process ends (M17.2)

The kernel records an ExitReason at all three death sites — clean exit,
fault (classified by vector), and process_kill — into a bounded ring
before the exit notification posts, so a supervisor's query never races
the notice. process_exit_reason is gated by the same supervisor check as
kill; runtime.process.exitReason is the stable interface. This is the
input restart policy reads (docs/process-lifecycle.md iron rule 2).
This commit is contained in:
Daniel Samson
2026-07-12 23:34:09 +01:00
parent 888eaa74e1
commit 2ebfb0c3b0
10 changed files with 162 additions and 19 deletions
+33 -9
View File
@@ -1209,15 +1209,15 @@ fn userPfTest() void {
/// hand — address space, code page RO+X, stack page RW+NX — because the blob is a
/// raw code fragment, not an ELF `spawnProcess` could load. Returns false if any
/// allocation fails.
fn spawnFaultingProcess() bool {
fn spawnFaultingProcess() ?u32 {
const blob = process.pfBlob();
const flags = sync.enter();
defer sync.leave(flags);
const aspace = architecture.createAddressSpace() orelse return false;
const aspace = architecture.createAddressSpace() orelse return null;
const code_frame = pmm.alloc() orelse {
architecture.destroyAddressSpace(aspace);
return false;
return null;
};
// Fill through the physmap (the user mapping is read-only); pad with int3 so a
// stray jump traps instead of sliding.
@@ -1228,15 +1228,16 @@ fn spawnFaultingProcess() bool {
const stack_frame = pmm.alloc() orelse {
architecture.destroyAddressSpace(aspace); // frees code_frame too — it's mapped
return false;
return null;
};
architecture.mapUserPageInto(aspace, process.stack_base_virtual, stack_frame, true, false); // RW + NX
if (scheduler.spawnUserLocked(aspace, process.code_virtual, process.stack_base_virtual + abi.page_size, 4, "fault-probe", 0, null) == null) {
// Supervised by the calling test task, so exitReasonOf can read the verdict.
const id = scheduler.spawnUserLocked(aspace, process.code_virtual, process.stack_base_virtual + abi.page_size, 4, "fault-probe", scheduler.currentId(), null) orelse {
architecture.destroyAddressSpace(aspace);
return false;
}
return true;
return null;
};
return id;
}
/// Fault recovery (docs/resilience.md step 2): a scheduled ring-3 process that
@@ -1265,7 +1266,8 @@ fn faultRecoveryTest(boot_information: *const BootInformation) void {
scheduler.setPriority(4);
check("init heartbeat before the fault", process.write_count >= 1);
check("faulting process spawned", spawnFaultingProcess());
const probe = spawnFaultingProcess() orelse 0;
check("faulting process spawned", probe != 0);
// The kill: the faulting process #PFs on its first instruction and the kernel
// reaps it instead of halting.
@@ -1274,6 +1276,7 @@ fn faultRecoveryTest(boot_information: *const BootInformation) void {
while (process.fault_kill_count < 1 and architecture.millis() < deadline) scheduler.yield();
scheduler.setPriority(4);
check("faulting process was killed (not the machine)", process.fault_kill_count == 1);
check("the probe's reason reads segmentation_fault", process.exitReasonOf(scheduler.currentId(), probe) == @intFromEnum(abi.ExitReason.segmentation_fault));
// Life after the kill: init must keep beating on the same core.
const beats_at_kill = process.write_count;
@@ -1425,6 +1428,12 @@ fn processKillTest(boot_information: *const BootInformation) void {
check("the sleeper's exit notification arrived (length 0)", r == 0);
check("its badge carries the exit bit and the child id", badge == abi.notify_badge_bit | abi.notify_exit_bit | sleeper);
// M17.2: the recorded reason — the notification is the fence, so it is
// already readable, and gated by the same supervisor check as the kill.
check("the sleeper's reason reads killed", process.exitReasonOf(me, sleeper) == @intFromEnum(abi.ExitReason.killed));
check("a non-supervisor may not read the reason (-EPERM)", process.exitReasonOf(me + 12345, sleeper) == -ipcsync.EPERM);
check("an unknown id has no reason (-ESRCH)", process.exitReasonOf(me, 0xFFFF_FF00) == -ipcsync.ESRCH);
const beats_at_kill = process.write_count;
scheduler.sleep(1500); // more than one heartbeat period
check("the heartbeat stopped with the kill", process.write_count == beats_at_kill);
@@ -1445,6 +1454,21 @@ fn processKillTest(boot_information: *const BootInformation) void {
check("the spinner's exit notification arrived (length 0)", r == 0);
check("its badge carries the exit bit and the child id", badge == abi.notify_badge_bit | abi.notify_exit_bit | spinner);
// M17.2: a child that ends on its own must read exited, not killed —
// args-echo with arguments echoes once and returns from main.
var clean: u32 = 0;
i = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(item.name, "args-echo")) continue;
clean = process.spawnProcessSupervised(item.blob, 4, &.{ "args-echo", "clean-exit" }, me, endpoint) catch 0;
break;
}
check("args-echo spawned as the clean-exit child", clean != 0);
r = ipcsync.replyWait(endpoint, 0, 0, 0, 0, abi.no_cap, &badge, &received_cap);
check("the clean child's exit notification arrived", badge == abi.notify_badge_bit | abi.notify_exit_bit | clean);
check("the clean child's reason reads exited", process.exitReasonOf(me, clean) == @intFromEnum(abi.ExitReason.exited));
var table: [32]abi.ProcessDescriptor = undefined;
const total = scheduler.enumerate(&table);
var still_listed = false;