Record and expose how every process ends (M17.2)
The kernel records an ExitReason at all three death sites — clean exit, fault (classified by vector), and process_kill — into a bounded ring before the exit notification posts, so a supervisor's query never races the notice. process_exit_reason is gated by the same supervisor check as kill; runtime.process.exitReason is the stable interface. This is the input restart policy reads (docs/process-lifecycle.md iron rule 2).
This commit is contained in:
@@ -53,9 +53,27 @@ pub const SystemCall = enum(u64) {
|
||||
process_enumerate = 24, // process_enumerate(buffer, maximum) -> total: snapshot the task table
|
||||
process_kill = 25, // process_kill(id) -> 0/-errno: end a process this process spawned
|
||||
ipc_send = 26, // ipc_send(handle, message_ptr, message_len) -> 0/-errno: post a payload to an endpoint's async queue without blocking
|
||||
process_exit_reason = 27, // process_exit_reason(id) -> ExitReason/-errno: how a dead child ended (its supervisor only)
|
||||
_,
|
||||
};
|
||||
|
||||
/// How a process ended — recorded by the kernel at death, queried by the
|
||||
/// supervisor with `process_exit_reason`, and the input to its restart decision
|
||||
/// (docs/process-lifecycle.md): a clean exit meant to stop, a fault wants a
|
||||
/// restart with backoff, killed means the supervisor did it itself. The faults
|
||||
/// mirror the CPU exceptions a ring-3 process can die of; they are exit reasons,
|
||||
/// never delivered to the faulting process (recovery is restart, not a handler).
|
||||
pub const ExitReason = enum(u8) {
|
||||
exited = 0, // returned from main / called exit
|
||||
aborted = 1, // deliberate self-termination (reserved: no abort path yet)
|
||||
segmentation_fault = 2, // page fault
|
||||
illegal_instruction = 3, // invalid opcode
|
||||
arithmetic_fault = 4, // divide error, x87 or SIMD fault
|
||||
protection_fault = 5, // general protection fault
|
||||
fault = 6, // any other CPU exception
|
||||
killed = 7, // process_kill
|
||||
};
|
||||
|
||||
/// The x86 MSI message address base (`0xFEE0_0000`): a device raises an MSI by writing
|
||||
/// `data` to this address, which the Local APIC turns into an interrupt at the vector
|
||||
/// in `data`. The kernel returns the concrete (address, data) from `msi_bind`; this is
|
||||
|
||||
@@ -412,13 +412,26 @@ fn recoverableFault(vector: u64) bool {
|
||||
/// plus a POST code and a persistent breadcrumb. (A ring-3 fault on a *borrowed*
|
||||
/// kernel thread — process.run, the user-pf isolation probe — also lands here: there
|
||||
/// is no scheduled process to kill.)
|
||||
/// Classify a CPU exception vector as the ExitReason a supervisor reads — the
|
||||
/// fault classes of docs/process-lifecycle.md. Faults are exit reasons, never
|
||||
/// signals delivered to the faulting process: recovery is restart, not a handler.
|
||||
fn exitReasonForVector(vector: u64) abi.ExitReason {
|
||||
return switch (vector) {
|
||||
14 => .segmentation_fault, // page fault
|
||||
6 => .illegal_instruction, // invalid opcode
|
||||
0, 16, 19 => .arithmetic_fault, // divide error, x87, SIMD
|
||||
13 => .protection_fault, // general protection
|
||||
else => .fault,
|
||||
};
|
||||
}
|
||||
|
||||
fn onException(state: *const architecture.CpuState) noreturn {
|
||||
if (architecture.fromUser(state) and scheduler.currentIsUserProcess() and recoverableFault(state.vector)) {
|
||||
statusPrint("\ndanos: process {d} ({s}) killed by {s} (vector {d}) on core {d}\n", .{ scheduler.currentId(), scheduler.current().name(), architecture.exceptionName(state.vector), state.vector, scheduler.currentCpuIndex() });
|
||||
statusPrint(" error code : 0x{x}\n", .{state.error_code});
|
||||
statusPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)});
|
||||
if (architecture.faultAddress(state)) |address| statusPrint(" fault addr : 0x{x:0>16}\n", .{address});
|
||||
process.killCurrentProcess(); // reclaims everything, reschedules; never returns
|
||||
process.killCurrentProcess(exitReasonForVector(state.vector)); // reclaims everything, reschedules; never returns
|
||||
}
|
||||
|
||||
log.checkpoint(cp_exception);
|
||||
|
||||
@@ -164,6 +164,7 @@ fn system_call(state: *architecture.CpuState) void {
|
||||
// A scheduled process tears down fully (terminateCurrent); a borrowed
|
||||
// test thread unwinds back to the kernel that entered it.
|
||||
if (scheduler.currentIsUserProcess()) {
|
||||
scheduler.current().exit_reason = .exited;
|
||||
terminateCurrent();
|
||||
} else architecture.userExit();
|
||||
},
|
||||
@@ -199,6 +200,7 @@ fn system_call(state: *architecture.CpuState) void {
|
||||
.clock => systemClock(state),
|
||||
.process_enumerate => systemProcessEnumerate(state),
|
||||
.process_kill => systemProcessKill(state),
|
||||
.process_exit_reason => systemProcessExitReason(state),
|
||||
_ => fail(state),
|
||||
}
|
||||
}
|
||||
@@ -569,6 +571,7 @@ pub var fault_kill_count: u64 = 0;
|
||||
/// reference taken at spawn is dropped with it.
|
||||
/// Precondition: the big kernel lock is held.
|
||||
fn releaseTaskResourcesLocked(t: *scheduler.Task) void {
|
||||
recordExitLocked(t);
|
||||
irq.releaseOwner(t.id);
|
||||
devices_broker.releaseAllOwnedBy(t.id);
|
||||
if (t.ipc_client) |client| {
|
||||
@@ -633,6 +636,7 @@ pub fn killProcess(caller_id: u32, target_id: u32) i64 {
|
||||
const target = scheduler.taskByIdLocked(target_id) orelse return -ipc.ESRCH;
|
||||
if (target.aspace == 0) return -ipc.ESRCH; // kernel tasks are not processes
|
||||
if (target.supervisor != caller_id) return -ipc.EPERM;
|
||||
target.exit_reason = .killed;
|
||||
if (target.state == .running) {
|
||||
target.kill_pending = true;
|
||||
} else {
|
||||
@@ -645,12 +649,56 @@ pub fn killProcess(caller_id: u32, target_id: u32) i64 {
|
||||
/// The fault is confined to the process — the kernel trapped it on the task's own
|
||||
/// kernel stack and is intact — so everything the process held is reclaimed and the
|
||||
/// core reschedules. The system keeps running; only the faulting process dies
|
||||
/// (docs/resilience.md: fault -> kill -> continue).
|
||||
pub fn killCurrentProcess() noreturn {
|
||||
/// (docs/resilience.md: fault -> kill -> continue). `reason` is the fault class
|
||||
/// (from the vector), recorded for the supervisor's `process_exit_reason`.
|
||||
pub fn killCurrentProcess(reason: abi.ExitReason) noreturn {
|
||||
scheduler.current().exit_reason = reason;
|
||||
fault_kill_count += 1;
|
||||
terminateCurrent();
|
||||
}
|
||||
|
||||
/// The bounded record of recent deaths, for `process_exit_reason`: ids are never
|
||||
/// reused, so a ring keyed by id is enough — a record evicted by wraparound reads
|
||||
/// as -ESRCH, the same as an id that never lived, which a supervisor treats as
|
||||
/// "too late to ask". Written under the big kernel lock by the reap.
|
||||
const exit_record_capacity = 64;
|
||||
const ExitRecord = struct { id: u32 = 0, supervisor: u32 = 0, reason: abi.ExitReason = .exited, valid: bool = false };
|
||||
var exit_records: [exit_record_capacity]ExitRecord = .{ExitRecord{}} ** exit_record_capacity;
|
||||
var exit_record_next: usize = 0;
|
||||
|
||||
/// Record a dying task's (id, supervisor, reason) — called by the reap before the
|
||||
/// exit notification is posted, so a supervisor that hears the notification can
|
||||
/// always still query the reason. Precondition: the big kernel lock is held.
|
||||
fn recordExitLocked(t: *scheduler.Task) void {
|
||||
exit_records[exit_record_next] = .{ .id = t.id, .supervisor = t.supervisor, .reason = t.exit_reason, .valid = true };
|
||||
exit_record_next = (exit_record_next + 1) % exit_record_capacity;
|
||||
}
|
||||
|
||||
/// How dead process `id` ended, for `caller` — the kernel half of the
|
||||
/// process_exit_reason system call. Returns the ExitReason value, -ESRCH (never
|
||||
/// lived, still alive, or evicted from the ring), or -EPERM (the caller was not
|
||||
/// its supervisor — the same authority gate as process_kill).
|
||||
pub fn exitReasonOf(caller_id: u32, target_id: u32) i64 {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
for (&exit_records) |*record| {
|
||||
if (record.valid and record.id == target_id) {
|
||||
if (record.supervisor != caller_id) return -ipc.EPERM;
|
||||
return @intFromEnum(record.reason);
|
||||
}
|
||||
}
|
||||
return -ipc.ESRCH;
|
||||
}
|
||||
|
||||
fn systemProcessExitReason(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
const id = architecture.systemCallArg(state, 0);
|
||||
if (id > std.math.maxInt(u32)) return failErr(state, ipc.ESRCH);
|
||||
const r = exitReasonOf(t.id, @intCast(id));
|
||||
architecture.setSystemCallResult(state, @bitCast(r));
|
||||
}
|
||||
|
||||
/// Resolve `(device_id, resource_index)` to a GSI this process is entitled to bind, or null.
|
||||
/// The two checks are the whole security story: the device must be *claimed* by the
|
||||
/// caller, and the resource must be one of that device's `irq` resources as recorded
|
||||
|
||||
@@ -50,6 +50,10 @@ pub const Task = struct {
|
||||
// null. Holds its own reference, dropped when the notification is posted.
|
||||
// Opaque here for the same reason as `handles` below.
|
||||
exit_endpoint: ?*anyopaque = null,
|
||||
// How this process ended — set by the death paths (exit, fault, kill) just
|
||||
// before the reap records it for `process_exit_reason`. Meaningless while
|
||||
// the task lives.
|
||||
exit_reason: abi.ExitReason = .exited,
|
||||
// Set by process_kill on a task that is running on another core; the kernel
|
||||
// finishes the kill at that task's next system call or timer tick.
|
||||
kill_pending: bool = false,
|
||||
|
||||
+33
-9
@@ -1209,15 +1209,15 @@ fn userPfTest() void {
|
||||
/// hand — address space, code page RO+X, stack page RW+NX — because the blob is a
|
||||
/// raw code fragment, not an ELF `spawnProcess` could load. Returns false if any
|
||||
/// allocation fails.
|
||||
fn spawnFaultingProcess() bool {
|
||||
fn spawnFaultingProcess() ?u32 {
|
||||
const blob = process.pfBlob();
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
|
||||
const aspace = architecture.createAddressSpace() orelse return false;
|
||||
const aspace = architecture.createAddressSpace() orelse return null;
|
||||
const code_frame = pmm.alloc() orelse {
|
||||
architecture.destroyAddressSpace(aspace);
|
||||
return false;
|
||||
return null;
|
||||
};
|
||||
// Fill through the physmap (the user mapping is read-only); pad with int3 so a
|
||||
// stray jump traps instead of sliding.
|
||||
@@ -1228,15 +1228,16 @@ fn spawnFaultingProcess() bool {
|
||||
|
||||
const stack_frame = pmm.alloc() orelse {
|
||||
architecture.destroyAddressSpace(aspace); // frees code_frame too — it's mapped
|
||||
return false;
|
||||
return null;
|
||||
};
|
||||
architecture.mapUserPageInto(aspace, process.stack_base_virtual, stack_frame, true, false); // RW + NX
|
||||
|
||||
if (scheduler.spawnUserLocked(aspace, process.code_virtual, process.stack_base_virtual + abi.page_size, 4, "fault-probe", 0, null) == null) {
|
||||
// Supervised by the calling test task, so exitReasonOf can read the verdict.
|
||||
const id = scheduler.spawnUserLocked(aspace, process.code_virtual, process.stack_base_virtual + abi.page_size, 4, "fault-probe", scheduler.currentId(), null) orelse {
|
||||
architecture.destroyAddressSpace(aspace);
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
return null;
|
||||
};
|
||||
return id;
|
||||
}
|
||||
|
||||
/// Fault recovery (docs/resilience.md step 2): a scheduled ring-3 process that
|
||||
@@ -1265,7 +1266,8 @@ fn faultRecoveryTest(boot_information: *const BootInformation) void {
|
||||
scheduler.setPriority(4);
|
||||
check("init heartbeat before the fault", process.write_count >= 1);
|
||||
|
||||
check("faulting process spawned", spawnFaultingProcess());
|
||||
const probe = spawnFaultingProcess() orelse 0;
|
||||
check("faulting process spawned", probe != 0);
|
||||
|
||||
// The kill: the faulting process #PFs on its first instruction and the kernel
|
||||
// reaps it instead of halting.
|
||||
@@ -1274,6 +1276,7 @@ fn faultRecoveryTest(boot_information: *const BootInformation) void {
|
||||
while (process.fault_kill_count < 1 and architecture.millis() < deadline) scheduler.yield();
|
||||
scheduler.setPriority(4);
|
||||
check("faulting process was killed (not the machine)", process.fault_kill_count == 1);
|
||||
check("the probe's reason reads segmentation_fault", process.exitReasonOf(scheduler.currentId(), probe) == @intFromEnum(abi.ExitReason.segmentation_fault));
|
||||
|
||||
// Life after the kill: init must keep beating on the same core.
|
||||
const beats_at_kill = process.write_count;
|
||||
@@ -1425,6 +1428,12 @@ fn processKillTest(boot_information: *const BootInformation) void {
|
||||
check("the sleeper's exit notification arrived (length 0)", r == 0);
|
||||
check("its badge carries the exit bit and the child id", badge == abi.notify_badge_bit | abi.notify_exit_bit | sleeper);
|
||||
|
||||
// M17.2: the recorded reason — the notification is the fence, so it is
|
||||
// already readable, and gated by the same supervisor check as the kill.
|
||||
check("the sleeper's reason reads killed", process.exitReasonOf(me, sleeper) == @intFromEnum(abi.ExitReason.killed));
|
||||
check("a non-supervisor may not read the reason (-EPERM)", process.exitReasonOf(me + 12345, sleeper) == -ipcsync.EPERM);
|
||||
check("an unknown id has no reason (-ESRCH)", process.exitReasonOf(me, 0xFFFF_FF00) == -ipcsync.ESRCH);
|
||||
|
||||
const beats_at_kill = process.write_count;
|
||||
scheduler.sleep(1500); // more than one heartbeat period
|
||||
check("the heartbeat stopped with the kill", process.write_count == beats_at_kill);
|
||||
@@ -1445,6 +1454,21 @@ fn processKillTest(boot_information: *const BootInformation) void {
|
||||
check("the spinner's exit notification arrived (length 0)", r == 0);
|
||||
check("its badge carries the exit bit and the child id", badge == abi.notify_badge_bit | abi.notify_exit_bit | spinner);
|
||||
|
||||
// M17.2: a child that ends on its own must read exited, not killed —
|
||||
// args-echo with arguments echoes once and returns from main.
|
||||
var clean: u32 = 0;
|
||||
i = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(item.name, "args-echo")) continue;
|
||||
clean = process.spawnProcessSupervised(item.blob, 4, &.{ "args-echo", "clean-exit" }, me, endpoint) catch 0;
|
||||
break;
|
||||
}
|
||||
check("args-echo spawned as the clean-exit child", clean != 0);
|
||||
r = ipcsync.replyWait(endpoint, 0, 0, 0, 0, abi.no_cap, &badge, &received_cap);
|
||||
check("the clean child's exit notification arrived", badge == abi.notify_badge_bit | abi.notify_exit_bit | clean);
|
||||
check("the clean child's reason reads exited", process.exitReasonOf(me, clean) == @intFromEnum(abi.ExitReason.exited));
|
||||
|
||||
var table: [32]abi.ProcessDescriptor = undefined;
|
||||
const total = scheduler.enumerate(&table);
|
||||
var still_listed = false;
|
||||
|
||||
@@ -89,6 +89,12 @@ pub fn main(init: runtime.process.Init) void {
|
||||
if (listed(sleeper, "process-test")) fail("sleeper still listed after kill");
|
||||
if (listed(spinner, "process-test")) fail("spinner still listed after kill");
|
||||
|
||||
// M17.2: both children were killed by us, and the reason says so — the whole
|
||||
// restart-policy input, read through the runtime like a real supervisor would.
|
||||
if ((runtime.process.exitReason(sleeper) orelse .exited) != .killed) fail("sleeper reason not killed");
|
||||
if ((runtime.process.exitReason(spinner) orelse .exited) != .killed) fail("spinner reason not killed");
|
||||
if (runtime.process.exitReason(0xFFFF_FFF0) != null) fail("unknown id had a reason");
|
||||
|
||||
_ = runtime.system.write("process-test: ok\n");
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user