Kill a faulting user process instead of halting the machine

A CPU exception raised in ring 3 by a scheduled process now kills that
process - IRQ bindings, IPC handles, and address space reclaimed, a
client it owed a reply to failed with the new -EPEER instead of hung -
and the core reschedules (docs/resilience.md step 2). Kernel-mode
faults, NMI, double fault, and machine check stay terminal, as does the
borrowed-thread isolation probe. Proven by the new fault-recovery QEMU
test: init keeps heartbeating after a process page-faults to death.
This commit is contained in:
Daniel Samson
2026-07-11 04:57:02 +01:00
parent 59104dd988
commit 6b3ae0c997
8 changed files with 198 additions and 35 deletions
+83
View File
@@ -118,6 +118,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
userMemTest();
} else if (eql(case, "user-pf")) {
userPfTest();
} else if (eql(case, "fault-recovery")) {
faultRecoveryTest(boot_information);
} else if (eql(case, "init")) {
initTest(boot_information);
} else if (eql(case, "process")) {
@@ -1190,6 +1192,87 @@ fn userPfTest() void {
log("DANOS-TEST-RESULT: FAIL (user read of kernel memory did not fault)\n", .{});
}
/// Spawn a real scheduled ring-3 process whose body is the user-pf blob (a read of
/// a kernel-only page, then a spin), so its first instruction raises #PF. Built by
/// hand — address space, code page RO+X, stack page RW+NX — because the blob is a
/// raw code fragment, not an ELF `spawnProcess` could load. Returns false if any
/// allocation fails.
fn spawnFaultingProcess() bool {
const blob = process.pfBlob();
const flags = sync.enter();
defer sync.leave(flags);
const aspace = architecture.createAddressSpace() orelse return false;
const code_frame = pmm.alloc() orelse {
architecture.destroyAddressSpace(aspace);
return false;
};
// Fill through the physmap (the user mapping is read-only); pad with int3 so a
// stray jump traps instead of sliding.
const code: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(code_frame));
@memset(code[0..abi.page_size], 0xCC);
@memcpy(code[0..blob.len], blob);
architecture.mapUserPageInto(aspace, process.code_virtual, code_frame, false, true); // RO + X
const stack_frame = pmm.alloc() orelse {
architecture.destroyAddressSpace(aspace); // frees code_frame too — it's mapped
return false;
};
architecture.mapUserPageInto(aspace, process.stack_virtual, stack_frame, true, false); // RW + NX
if (!scheduler.spawnUserLocked(aspace, process.code_virtual, process.stack_virtual + abi.page_size, 4)) {
architecture.destroyAddressSpace(aspace);
return false;
}
return true;
}
/// Fault recovery (docs/resilience.md step 2): a scheduled ring-3 process that
/// faults must be killed — counted, resources reclaimed — while the rest of the
/// system keeps running. init heartbeats before and after the kill are the proof
/// the OS survived; the old behaviour (halt the core) would freeze the beat and
/// time the harness out.
fn faultRecoveryTest(boot_information: *const BootInformation) void {
log("DANOS-TEST-BEGIN: fault-recovery\n", .{});
check("bootloader handed over /system/services/init", boot_information.init_len != 0);
if (boot_information.init_len == 0) {
result();
return;
}
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
process.write_count = 0;
process.fault_kill_count = 0;
const spawned = if (process.spawnProcess(image, 4)) true else |_| false;
check("init spawned as the surviving process", spawned);
// A first heartbeat proves init runs before the fault.
scheduler.setPriority(1); // drop below the processes so they get the core
var deadline = architecture.millis() + 8000;
while (process.write_count < 1 and architecture.millis() < deadline) scheduler.yield();
scheduler.setPriority(4);
check("init heartbeat before the fault", process.write_count >= 1);
check("faulting process spawned", spawnFaultingProcess());
// The kill: the faulting process #PFs on its first instruction and the kernel
// reaps it instead of halting.
scheduler.setPriority(1);
deadline = architecture.millis() + 5000;
while (process.fault_kill_count < 1 and architecture.millis() < deadline) scheduler.yield();
scheduler.setPriority(4);
check("faulting process was killed (not the machine)", process.fault_kill_count == 1);
// Life after the kill: init must keep beating on the same core.
const beats_at_kill = process.write_count;
scheduler.setPriority(1);
deadline = architecture.millis() + 8000;
while (process.write_count < beats_at_kill + 2 and architecture.millis() < deadline) scheduler.yield();
scheduler.setPriority(4);
check("init kept heartbeating after the kill", process.write_count >= beats_at_kill + 2);
result();
}
/// The full PID-1 path: the bootloader read /system/services/init off the boot volume and
/// handed it over; load it as a user ELF and spawn it as a real ring-3 process
/// — the same call the normal boot path makes — then confirm it beats. init