Kill a faulting user process instead of halting the machine
A CPU exception raised in ring 3 by a scheduled process now kills that process - IRQ bindings, IPC handles, and address space reclaimed, a client it owed a reply to failed with the new -EPEER instead of hung - and the core reschedules (docs/resilience.md step 2). Kernel-mode faults, NMI, double fault, and machine check stay terminal, as does the borrowed-thread isolation probe. Proven by the new fault-recovery QEMU test: init keeps heartbeating after a process page-faults to death.
This commit is contained in:
@@ -118,6 +118,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
userMemTest();
|
||||
} else if (eql(case, "user-pf")) {
|
||||
userPfTest();
|
||||
} else if (eql(case, "fault-recovery")) {
|
||||
faultRecoveryTest(boot_information);
|
||||
} else if (eql(case, "init")) {
|
||||
initTest(boot_information);
|
||||
} else if (eql(case, "process")) {
|
||||
@@ -1190,6 +1192,87 @@ fn userPfTest() void {
|
||||
log("DANOS-TEST-RESULT: FAIL (user read of kernel memory did not fault)\n", .{});
|
||||
}
|
||||
|
||||
/// Spawn a real scheduled ring-3 process whose body is the user-pf blob (a read of
|
||||
/// a kernel-only page, then a spin), so its first instruction raises #PF. Built by
|
||||
/// hand — address space, code page RO+X, stack page RW+NX — because the blob is a
|
||||
/// raw code fragment, not an ELF `spawnProcess` could load. Returns false if any
|
||||
/// allocation fails.
|
||||
fn spawnFaultingProcess() bool {
|
||||
const blob = process.pfBlob();
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
|
||||
const aspace = architecture.createAddressSpace() orelse return false;
|
||||
const code_frame = pmm.alloc() orelse {
|
||||
architecture.destroyAddressSpace(aspace);
|
||||
return false;
|
||||
};
|
||||
// Fill through the physmap (the user mapping is read-only); pad with int3 so a
|
||||
// stray jump traps instead of sliding.
|
||||
const code: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(code_frame));
|
||||
@memset(code[0..abi.page_size], 0xCC);
|
||||
@memcpy(code[0..blob.len], blob);
|
||||
architecture.mapUserPageInto(aspace, process.code_virtual, code_frame, false, true); // RO + X
|
||||
|
||||
const stack_frame = pmm.alloc() orelse {
|
||||
architecture.destroyAddressSpace(aspace); // frees code_frame too — it's mapped
|
||||
return false;
|
||||
};
|
||||
architecture.mapUserPageInto(aspace, process.stack_virtual, stack_frame, true, false); // RW + NX
|
||||
|
||||
if (!scheduler.spawnUserLocked(aspace, process.code_virtual, process.stack_virtual + abi.page_size, 4)) {
|
||||
architecture.destroyAddressSpace(aspace);
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Fault recovery (docs/resilience.md step 2): a scheduled ring-3 process that
|
||||
/// faults must be killed — counted, resources reclaimed — while the rest of the
|
||||
/// system keeps running. init heartbeats before and after the kill are the proof
|
||||
/// the OS survived; the old behaviour (halt the core) would freeze the beat and
|
||||
/// time the harness out.
|
||||
fn faultRecoveryTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: fault-recovery\n", .{});
|
||||
check("bootloader handed over /system/services/init", boot_information.init_len != 0);
|
||||
if (boot_information.init_len == 0) {
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
||||
|
||||
process.write_count = 0;
|
||||
process.fault_kill_count = 0;
|
||||
const spawned = if (process.spawnProcess(image, 4)) true else |_| false;
|
||||
check("init spawned as the surviving process", spawned);
|
||||
|
||||
// A first heartbeat proves init runs before the fault.
|
||||
scheduler.setPriority(1); // drop below the processes so they get the core
|
||||
var deadline = architecture.millis() + 8000;
|
||||
while (process.write_count < 1 and architecture.millis() < deadline) scheduler.yield();
|
||||
scheduler.setPriority(4);
|
||||
check("init heartbeat before the fault", process.write_count >= 1);
|
||||
|
||||
check("faulting process spawned", spawnFaultingProcess());
|
||||
|
||||
// The kill: the faulting process #PFs on its first instruction and the kernel
|
||||
// reaps it instead of halting.
|
||||
scheduler.setPriority(1);
|
||||
deadline = architecture.millis() + 5000;
|
||||
while (process.fault_kill_count < 1 and architecture.millis() < deadline) scheduler.yield();
|
||||
scheduler.setPriority(4);
|
||||
check("faulting process was killed (not the machine)", process.fault_kill_count == 1);
|
||||
|
||||
// Life after the kill: init must keep beating on the same core.
|
||||
const beats_at_kill = process.write_count;
|
||||
scheduler.setPriority(1);
|
||||
deadline = architecture.millis() + 8000;
|
||||
while (process.write_count < beats_at_kill + 2 and architecture.millis() < deadline) scheduler.yield();
|
||||
scheduler.setPriority(4);
|
||||
check("init kept heartbeating after the kill", process.write_count >= beats_at_kill + 2);
|
||||
result();
|
||||
}
|
||||
|
||||
/// The full PID-1 path: the bootloader read /system/services/init off the boot volume and
|
||||
/// handed it over; load it as a user ELF and spawn it as a real ring-3 process
|
||||
/// — the same call the normal boot path makes — then confirm it beats. init
|
||||
|
||||
Reference in New Issue
Block a user