Record and expose how every process ends (M17.2)
The kernel records an ExitReason at all three death sites — clean exit, fault (classified by vector), and process_kill — into a bounded ring before the exit notification posts, so a supervisor's query never races the notice. process_exit_reason is gated by the same supervisor check as kill; runtime.process.exitReason is the stable interface. This is the input restart policy reads (docs/process-lifecycle.md iron rule 2).
This commit is contained in:
@@ -164,6 +164,7 @@ fn system_call(state: *architecture.CpuState) void {
|
||||
// A scheduled process tears down fully (terminateCurrent); a borrowed
|
||||
// test thread unwinds back to the kernel that entered it.
|
||||
if (scheduler.currentIsUserProcess()) {
|
||||
scheduler.current().exit_reason = .exited;
|
||||
terminateCurrent();
|
||||
} else architecture.userExit();
|
||||
},
|
||||
@@ -199,6 +200,7 @@ fn system_call(state: *architecture.CpuState) void {
|
||||
.clock => systemClock(state),
|
||||
.process_enumerate => systemProcessEnumerate(state),
|
||||
.process_kill => systemProcessKill(state),
|
||||
.process_exit_reason => systemProcessExitReason(state),
|
||||
_ => fail(state),
|
||||
}
|
||||
}
|
||||
@@ -569,6 +571,7 @@ pub var fault_kill_count: u64 = 0;
|
||||
/// reference taken at spawn is dropped with it.
|
||||
/// Precondition: the big kernel lock is held.
|
||||
fn releaseTaskResourcesLocked(t: *scheduler.Task) void {
|
||||
recordExitLocked(t);
|
||||
irq.releaseOwner(t.id);
|
||||
devices_broker.releaseAllOwnedBy(t.id);
|
||||
if (t.ipc_client) |client| {
|
||||
@@ -633,6 +636,7 @@ pub fn killProcess(caller_id: u32, target_id: u32) i64 {
|
||||
const target = scheduler.taskByIdLocked(target_id) orelse return -ipc.ESRCH;
|
||||
if (target.aspace == 0) return -ipc.ESRCH; // kernel tasks are not processes
|
||||
if (target.supervisor != caller_id) return -ipc.EPERM;
|
||||
target.exit_reason = .killed;
|
||||
if (target.state == .running) {
|
||||
target.kill_pending = true;
|
||||
} else {
|
||||
@@ -645,12 +649,56 @@ pub fn killProcess(caller_id: u32, target_id: u32) i64 {
|
||||
/// The fault is confined to the process — the kernel trapped it on the task's own
|
||||
/// kernel stack and is intact — so everything the process held is reclaimed and the
|
||||
/// core reschedules. The system keeps running; only the faulting process dies
|
||||
/// (docs/resilience.md: fault -> kill -> continue).
|
||||
pub fn killCurrentProcess() noreturn {
|
||||
/// (docs/resilience.md: fault -> kill -> continue). `reason` is the fault class
|
||||
/// (from the vector), recorded for the supervisor's `process_exit_reason`.
|
||||
pub fn killCurrentProcess(reason: abi.ExitReason) noreturn {
|
||||
scheduler.current().exit_reason = reason;
|
||||
fault_kill_count += 1;
|
||||
terminateCurrent();
|
||||
}
|
||||
|
||||
/// The bounded record of recent deaths, for `process_exit_reason`: ids are never
|
||||
/// reused, so a ring keyed by id is enough — a record evicted by wraparound reads
|
||||
/// as -ESRCH, the same as an id that never lived, which a supervisor treats as
|
||||
/// "too late to ask". Written under the big kernel lock by the reap.
|
||||
const exit_record_capacity = 64;
|
||||
const ExitRecord = struct { id: u32 = 0, supervisor: u32 = 0, reason: abi.ExitReason = .exited, valid: bool = false };
|
||||
var exit_records: [exit_record_capacity]ExitRecord = .{ExitRecord{}} ** exit_record_capacity;
|
||||
var exit_record_next: usize = 0;
|
||||
|
||||
/// Record a dying task's (id, supervisor, reason) — called by the reap before the
|
||||
/// exit notification is posted, so a supervisor that hears the notification can
|
||||
/// always still query the reason. Precondition: the big kernel lock is held.
|
||||
fn recordExitLocked(t: *scheduler.Task) void {
|
||||
exit_records[exit_record_next] = .{ .id = t.id, .supervisor = t.supervisor, .reason = t.exit_reason, .valid = true };
|
||||
exit_record_next = (exit_record_next + 1) % exit_record_capacity;
|
||||
}
|
||||
|
||||
/// How dead process `id` ended, for `caller` — the kernel half of the
|
||||
/// process_exit_reason system call. Returns the ExitReason value, -ESRCH (never
|
||||
/// lived, still alive, or evicted from the ring), or -EPERM (the caller was not
|
||||
/// its supervisor — the same authority gate as process_kill).
|
||||
pub fn exitReasonOf(caller_id: u32, target_id: u32) i64 {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
for (&exit_records) |*record| {
|
||||
if (record.valid and record.id == target_id) {
|
||||
if (record.supervisor != caller_id) return -ipc.EPERM;
|
||||
return @intFromEnum(record.reason);
|
||||
}
|
||||
}
|
||||
return -ipc.ESRCH;
|
||||
}
|
||||
|
||||
fn systemProcessExitReason(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
const id = architecture.systemCallArg(state, 0);
|
||||
if (id > std.math.maxInt(u32)) return failErr(state, ipc.ESRCH);
|
||||
const r = exitReasonOf(t.id, @intCast(id));
|
||||
architecture.setSystemCallResult(state, @bitCast(r));
|
||||
}
|
||||
|
||||
/// Resolve `(device_id, resource_index)` to a GSI this process is entitled to bind, or null.
|
||||
/// The two checks are the whole security story: the device must be *claimed* by the
|
||||
/// caller, and the resource must be one of that device's `irq` resources as recorded
|
||||
|
||||
Reference in New Issue
Block a user