From 2ebfb0c3b06e2ff19c16b39ff2a4cf013603ec08 Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Sun, 12 Jul 2026 23:34:09 +0100 Subject: [PATCH] Record and expose how every process ends (M17.2) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The kernel records an ExitReason at all three death sites — clean exit, fault (classified by vector), and process_kill — into a bounded ring before the exit notification posts, so a supervisor's query never races the notice. process_exit_reason is gated by the same supervisor check as kill; runtime.process.exitReason is the stable interface. This is the input restart policy reads (docs/process-lifecycle.md iron rule 2). --- docs/m17-m18-plan.md | 4 +- docs/process-lifecycle.md | 9 ++-- docs/process-management.md | 8 ++- library/runtime/process.zig | 23 +++++++- system/abi.zig | 18 +++++++ system/kernel/kernel.zig | 15 +++++- system/kernel/process.zig | 52 ++++++++++++++++++- system/kernel/scheduler.zig | 4 ++ system/kernel/tests.zig | 42 +++++++++++---- system/services/process-test/process-test.zig | 6 +++ 10 files changed, 162 insertions(+), 19 deletions(-) diff --git a/docs/m17-m18-plan.md b/docs/m17-m18-plan.md index 5d5b7dd..0d38428 100644 --- a/docs/m17-m18-plan.md +++ b/docs/m17-m18-plan.md @@ -32,7 +32,9 @@ only when its definition of green holds. - [x] **M17.1** — kernel releases claims/MSI on death (claims: `releaseAllOwnedBy` in the reap; MSI was already swept by `irq.releaseOwner`; `claim-release` test; suite 49/49) -- [ ] **M17.2** — exit reasons +- [x] **M17.2** — exit reasons (`ExitReason` recorded at exit/fault/kill before + the notification; `process_exit_reason` supervisor-gated; + `runtime.process.exitReason`; kernel + ring-3 assertions; suite 49/49) - [ ] **M17.3** — published exit events + VFS subscriber - [ ] **M17.4** — signals, timer notifications, `runtime.process`, the service harness - [ ] **merge** `feat/process-lifecycle` → main, push diff --git a/docs/process-lifecycle.md b/docs/process-lifecycle.md index 6944e8a..1b0dac8 100644 --- a/docs/process-lifecycle.md +++ b/docs/process-lifecycle.md @@ -245,13 +245,16 @@ pub fn stop(id: u32, deadline_ms: u64) void { ... } /// process_enumerate. pub fn subscribeExits(endpoint: usize) bool { ... } -/// How a process ended — from the exit notification. What restart policy reads. -pub const ExitReason = enum { +/// How a process ended — queried after the exit notification (the kernel records +/// it first, so the two never race). What restart policy reads. (Built in M17.2.) +pub const ExitReason = enum(u8) { exited, // returned from main / clean exit - aborted, // abort() — deliberate self-termination (SIGABRT's ghost) + aborted, // abort() — deliberate self-termination (SIGABRT's ghost; reserved) segmentation_fault, // SIGSEGV's ghost illegal_instruction, // SIGILL's ghost arithmetic_fault, // SIGFPE's ghost + protection_fault, // general protection fault + fault, // any other CPU exception killed, // process_kill }; ``` diff --git a/docs/process-management.md b/docs/process-management.md index dcc998b..3c31024 100644 --- a/docs/process-management.md +++ b/docs/process-management.md @@ -101,8 +101,12 @@ the architecture layer calls up into `tick`. again — the cleanup half of [process-lifecycle.md](process-lifecycle.md)'s iron rule 1. The `claim-release` test proves the kill → release → re-claim cycle. - Kernel stacks of dead tasks are leaked, as on every exit path (no reaper yet). -- There is no exit *status* in the notification, only the id; a supervisor that - needs the code can grow a wait-style call later. +- ~~There is no exit status in the notification~~ Closed (M17.2): the kernel + records how every process ends — exited, a fault class, or killed — before it + posts the exit notification, and the supervisor reads it with + `process_exit_reason` (`runtime.process.exitReason`). This is the input to + restart policy ([process-lifecycle.md](process-lifecycle.md)); an exit *code* + for the clean case can still ride alongside later. - Enumerate writes through the caller's raw pointer under the bring-up trust model, like `device_enumerate` (an unmapped page is a self-DoS, not an isolation break). diff --git a/library/runtime/process.zig b/library/runtime/process.zig index 9f41438..d0f66db 100644 --- a/library/runtime/process.zig +++ b/library/runtime/process.zig @@ -1,8 +1,13 @@ -//! Process-level runtime types: what a user program receives at entry. Mirrors +//! Process-level runtime types: what a user program receives at entry (`Init`, +//! the argv contract) and the process end of the lifecycle +//! (docs/process-lifecycle.md) — today the exit reason a supervisor reads to +//! decide restart; signals and the stop sequence land here with M17.4. Mirrors //! the spirit of `std.process.Init.Minimal` in danos terms — std's `Args` holds //! no data on freestanding targets, so the type is danos's own. const std = @import("std"); +const abi = @import("abi"); +const sc = @import("system-call.zig"); /// Everything a program receives at entry. Passed to /// `pub fn main(init: runtime.process.Init)`; programs that need nothing keep @@ -45,3 +50,19 @@ pub const Arguments = struct { } }; }; + +/// How a process ended — what a supervisor's restart policy reads: a clean exit +/// meant to stop, a fault wants a restart with backoff, killed means the +/// supervisor did it itself (docs/process-lifecycle.md). +pub const ExitReason = abi.ExitReason; + +/// How dead child `id` ended. Ask after the exit notification arrives — the +/// kernel records the reason before it posts the notification, so this never +/// races it. Returns null for an id that never lived, is still alive, was +/// evicted from the kernel's bounded record, or is not this process's child +/// (the same authority gate as `kill`). +pub fn exitReason(id: u32) ?ExitReason { + const r = sc.systemCall1(.process_exit_reason, id); + if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno + return @enumFromInt(r); +} diff --git a/system/abi.zig b/system/abi.zig index 69ba4df..c3c4d97 100644 --- a/system/abi.zig +++ b/system/abi.zig @@ -53,9 +53,27 @@ pub const SystemCall = enum(u64) { process_enumerate = 24, // process_enumerate(buffer, maximum) -> total: snapshot the task table process_kill = 25, // process_kill(id) -> 0/-errno: end a process this process spawned ipc_send = 26, // ipc_send(handle, message_ptr, message_len) -> 0/-errno: post a payload to an endpoint's async queue without blocking + process_exit_reason = 27, // process_exit_reason(id) -> ExitReason/-errno: how a dead child ended (its supervisor only) _, }; +/// How a process ended — recorded by the kernel at death, queried by the +/// supervisor with `process_exit_reason`, and the input to its restart decision +/// (docs/process-lifecycle.md): a clean exit meant to stop, a fault wants a +/// restart with backoff, killed means the supervisor did it itself. The faults +/// mirror the CPU exceptions a ring-3 process can die of; they are exit reasons, +/// never delivered to the faulting process (recovery is restart, not a handler). +pub const ExitReason = enum(u8) { + exited = 0, // returned from main / called exit + aborted = 1, // deliberate self-termination (reserved: no abort path yet) + segmentation_fault = 2, // page fault + illegal_instruction = 3, // invalid opcode + arithmetic_fault = 4, // divide error, x87 or SIMD fault + protection_fault = 5, // general protection fault + fault = 6, // any other CPU exception + killed = 7, // process_kill +}; + /// The x86 MSI message address base (`0xFEE0_0000`): a device raises an MSI by writing /// `data` to this address, which the Local APIC turns into an interrupt at the vector /// in `data`. The kernel returns the concrete (address, data) from `msi_bind`; this is diff --git a/system/kernel/kernel.zig b/system/kernel/kernel.zig index 900bf25..ce43ec0 100644 --- a/system/kernel/kernel.zig +++ b/system/kernel/kernel.zig @@ -412,13 +412,26 @@ fn recoverableFault(vector: u64) bool { /// plus a POST code and a persistent breadcrumb. (A ring-3 fault on a *borrowed* /// kernel thread — process.run, the user-pf isolation probe — also lands here: there /// is no scheduled process to kill.) +/// Classify a CPU exception vector as the ExitReason a supervisor reads — the +/// fault classes of docs/process-lifecycle.md. Faults are exit reasons, never +/// signals delivered to the faulting process: recovery is restart, not a handler. +fn exitReasonForVector(vector: u64) abi.ExitReason { + return switch (vector) { + 14 => .segmentation_fault, // page fault + 6 => .illegal_instruction, // invalid opcode + 0, 16, 19 => .arithmetic_fault, // divide error, x87, SIMD + 13 => .protection_fault, // general protection + else => .fault, + }; +} + fn onException(state: *const architecture.CpuState) noreturn { if (architecture.fromUser(state) and scheduler.currentIsUserProcess() and recoverableFault(state.vector)) { statusPrint("\ndanos: process {d} ({s}) killed by {s} (vector {d}) on core {d}\n", .{ scheduler.currentId(), scheduler.current().name(), architecture.exceptionName(state.vector), state.vector, scheduler.currentCpuIndex() }); statusPrint(" error code : 0x{x}\n", .{state.error_code}); statusPrint(" IP : 0x{x:0>16}\n", .{architecture.instructionPointer(state)}); if (architecture.faultAddress(state)) |address| statusPrint(" fault addr : 0x{x:0>16}\n", .{address}); - process.killCurrentProcess(); // reclaims everything, reschedules; never returns + process.killCurrentProcess(exitReasonForVector(state.vector)); // reclaims everything, reschedules; never returns } log.checkpoint(cp_exception); diff --git a/system/kernel/process.zig b/system/kernel/process.zig index c2de634..5896c74 100644 --- a/system/kernel/process.zig +++ b/system/kernel/process.zig @@ -164,6 +164,7 @@ fn system_call(state: *architecture.CpuState) void { // A scheduled process tears down fully (terminateCurrent); a borrowed // test thread unwinds back to the kernel that entered it. if (scheduler.currentIsUserProcess()) { + scheduler.current().exit_reason = .exited; terminateCurrent(); } else architecture.userExit(); }, @@ -199,6 +200,7 @@ fn system_call(state: *architecture.CpuState) void { .clock => systemClock(state), .process_enumerate => systemProcessEnumerate(state), .process_kill => systemProcessKill(state), + .process_exit_reason => systemProcessExitReason(state), _ => fail(state), } } @@ -569,6 +571,7 @@ pub var fault_kill_count: u64 = 0; /// reference taken at spawn is dropped with it. /// Precondition: the big kernel lock is held. fn releaseTaskResourcesLocked(t: *scheduler.Task) void { + recordExitLocked(t); irq.releaseOwner(t.id); devices_broker.releaseAllOwnedBy(t.id); if (t.ipc_client) |client| { @@ -633,6 +636,7 @@ pub fn killProcess(caller_id: u32, target_id: u32) i64 { const target = scheduler.taskByIdLocked(target_id) orelse return -ipc.ESRCH; if (target.aspace == 0) return -ipc.ESRCH; // kernel tasks are not processes if (target.supervisor != caller_id) return -ipc.EPERM; + target.exit_reason = .killed; if (target.state == .running) { target.kill_pending = true; } else { @@ -645,12 +649,56 @@ pub fn killProcess(caller_id: u32, target_id: u32) i64 { /// The fault is confined to the process — the kernel trapped it on the task's own /// kernel stack and is intact — so everything the process held is reclaimed and the /// core reschedules. The system keeps running; only the faulting process dies -/// (docs/resilience.md: fault -> kill -> continue). -pub fn killCurrentProcess() noreturn { +/// (docs/resilience.md: fault -> kill -> continue). `reason` is the fault class +/// (from the vector), recorded for the supervisor's `process_exit_reason`. +pub fn killCurrentProcess(reason: abi.ExitReason) noreturn { + scheduler.current().exit_reason = reason; fault_kill_count += 1; terminateCurrent(); } +/// The bounded record of recent deaths, for `process_exit_reason`: ids are never +/// reused, so a ring keyed by id is enough — a record evicted by wraparound reads +/// as -ESRCH, the same as an id that never lived, which a supervisor treats as +/// "too late to ask". Written under the big kernel lock by the reap. +const exit_record_capacity = 64; +const ExitRecord = struct { id: u32 = 0, supervisor: u32 = 0, reason: abi.ExitReason = .exited, valid: bool = false }; +var exit_records: [exit_record_capacity]ExitRecord = .{ExitRecord{}} ** exit_record_capacity; +var exit_record_next: usize = 0; + +/// Record a dying task's (id, supervisor, reason) — called by the reap before the +/// exit notification is posted, so a supervisor that hears the notification can +/// always still query the reason. Precondition: the big kernel lock is held. +fn recordExitLocked(t: *scheduler.Task) void { + exit_records[exit_record_next] = .{ .id = t.id, .supervisor = t.supervisor, .reason = t.exit_reason, .valid = true }; + exit_record_next = (exit_record_next + 1) % exit_record_capacity; +} + +/// How dead process `id` ended, for `caller` — the kernel half of the +/// process_exit_reason system call. Returns the ExitReason value, -ESRCH (never +/// lived, still alive, or evicted from the ring), or -EPERM (the caller was not +/// its supervisor — the same authority gate as process_kill). +pub fn exitReasonOf(caller_id: u32, target_id: u32) i64 { + const flags = sync.enter(); + defer sync.leave(flags); + for (&exit_records) |*record| { + if (record.valid and record.id == target_id) { + if (record.supervisor != caller_id) return -ipc.EPERM; + return @intFromEnum(record.reason); + } + } + return -ipc.ESRCH; +} + +fn systemProcessExitReason(state: *architecture.CpuState) void { + const t = scheduler.current(); + if (t.aspace == 0) return fail(state); + const id = architecture.systemCallArg(state, 0); + if (id > std.math.maxInt(u32)) return failErr(state, ipc.ESRCH); + const r = exitReasonOf(t.id, @intCast(id)); + architecture.setSystemCallResult(state, @bitCast(r)); +} + /// Resolve `(device_id, resource_index)` to a GSI this process is entitled to bind, or null. /// The two checks are the whole security story: the device must be *claimed* by the /// caller, and the resource must be one of that device's `irq` resources as recorded diff --git a/system/kernel/scheduler.zig b/system/kernel/scheduler.zig index e5e5b5c..6df7867 100644 --- a/system/kernel/scheduler.zig +++ b/system/kernel/scheduler.zig @@ -50,6 +50,10 @@ pub const Task = struct { // null. Holds its own reference, dropped when the notification is posted. // Opaque here for the same reason as `handles` below. exit_endpoint: ?*anyopaque = null, + // How this process ended — set by the death paths (exit, fault, kill) just + // before the reap records it for `process_exit_reason`. Meaningless while + // the task lives. + exit_reason: abi.ExitReason = .exited, // Set by process_kill on a task that is running on another core; the kernel // finishes the kill at that task's next system call or timer tick. kill_pending: bool = false, diff --git a/system/kernel/tests.zig b/system/kernel/tests.zig index 21acfe1..38bdf98 100644 --- a/system/kernel/tests.zig +++ b/system/kernel/tests.zig @@ -1209,15 +1209,15 @@ fn userPfTest() void { /// hand — address space, code page RO+X, stack page RW+NX — because the blob is a /// raw code fragment, not an ELF `spawnProcess` could load. Returns false if any /// allocation fails. -fn spawnFaultingProcess() bool { +fn spawnFaultingProcess() ?u32 { const blob = process.pfBlob(); const flags = sync.enter(); defer sync.leave(flags); - const aspace = architecture.createAddressSpace() orelse return false; + const aspace = architecture.createAddressSpace() orelse return null; const code_frame = pmm.alloc() orelse { architecture.destroyAddressSpace(aspace); - return false; + return null; }; // Fill through the physmap (the user mapping is read-only); pad with int3 so a // stray jump traps instead of sliding. @@ -1228,15 +1228,16 @@ fn spawnFaultingProcess() bool { const stack_frame = pmm.alloc() orelse { architecture.destroyAddressSpace(aspace); // frees code_frame too — it's mapped - return false; + return null; }; architecture.mapUserPageInto(aspace, process.stack_base_virtual, stack_frame, true, false); // RW + NX - if (scheduler.spawnUserLocked(aspace, process.code_virtual, process.stack_base_virtual + abi.page_size, 4, "fault-probe", 0, null) == null) { + // Supervised by the calling test task, so exitReasonOf can read the verdict. + const id = scheduler.spawnUserLocked(aspace, process.code_virtual, process.stack_base_virtual + abi.page_size, 4, "fault-probe", scheduler.currentId(), null) orelse { architecture.destroyAddressSpace(aspace); - return false; - } - return true; + return null; + }; + return id; } /// Fault recovery (docs/resilience.md step 2): a scheduled ring-3 process that @@ -1265,7 +1266,8 @@ fn faultRecoveryTest(boot_information: *const BootInformation) void { scheduler.setPriority(4); check("init heartbeat before the fault", process.write_count >= 1); - check("faulting process spawned", spawnFaultingProcess()); + const probe = spawnFaultingProcess() orelse 0; + check("faulting process spawned", probe != 0); // The kill: the faulting process #PFs on its first instruction and the kernel // reaps it instead of halting. @@ -1274,6 +1276,7 @@ fn faultRecoveryTest(boot_information: *const BootInformation) void { while (process.fault_kill_count < 1 and architecture.millis() < deadline) scheduler.yield(); scheduler.setPriority(4); check("faulting process was killed (not the machine)", process.fault_kill_count == 1); + check("the probe's reason reads segmentation_fault", process.exitReasonOf(scheduler.currentId(), probe) == @intFromEnum(abi.ExitReason.segmentation_fault)); // Life after the kill: init must keep beating on the same core. const beats_at_kill = process.write_count; @@ -1425,6 +1428,12 @@ fn processKillTest(boot_information: *const BootInformation) void { check("the sleeper's exit notification arrived (length 0)", r == 0); check("its badge carries the exit bit and the child id", badge == abi.notify_badge_bit | abi.notify_exit_bit | sleeper); + // M17.2: the recorded reason — the notification is the fence, so it is + // already readable, and gated by the same supervisor check as the kill. + check("the sleeper's reason reads killed", process.exitReasonOf(me, sleeper) == @intFromEnum(abi.ExitReason.killed)); + check("a non-supervisor may not read the reason (-EPERM)", process.exitReasonOf(me + 12345, sleeper) == -ipcsync.EPERM); + check("an unknown id has no reason (-ESRCH)", process.exitReasonOf(me, 0xFFFF_FF00) == -ipcsync.ESRCH); + const beats_at_kill = process.write_count; scheduler.sleep(1500); // more than one heartbeat period check("the heartbeat stopped with the kill", process.write_count == beats_at_kill); @@ -1445,6 +1454,21 @@ fn processKillTest(boot_information: *const BootInformation) void { check("the spinner's exit notification arrived (length 0)", r == 0); check("its badge carries the exit bit and the child id", badge == abi.notify_badge_bit | abi.notify_exit_bit | spinner); + // M17.2: a child that ends on its own must read exited, not killed — + // args-echo with arguments echoes once and returns from main. + var clean: u32 = 0; + i = 0; + while (i < rd.count) : (i += 1) { + const item = rd.entry(i) orelse continue; + if (!eql(item.name, "args-echo")) continue; + clean = process.spawnProcessSupervised(item.blob, 4, &.{ "args-echo", "clean-exit" }, me, endpoint) catch 0; + break; + } + check("args-echo spawned as the clean-exit child", clean != 0); + r = ipcsync.replyWait(endpoint, 0, 0, 0, 0, abi.no_cap, &badge, &received_cap); + check("the clean child's exit notification arrived", badge == abi.notify_badge_bit | abi.notify_exit_bit | clean); + check("the clean child's reason reads exited", process.exitReasonOf(me, clean) == @intFromEnum(abi.ExitReason.exited)); + var table: [32]abi.ProcessDescriptor = undefined; const total = scheduler.enumerate(&table); var still_listed = false; diff --git a/system/services/process-test/process-test.zig b/system/services/process-test/process-test.zig index ecf75bc..5536b96 100644 --- a/system/services/process-test/process-test.zig +++ b/system/services/process-test/process-test.zig @@ -89,6 +89,12 @@ pub fn main(init: runtime.process.Init) void { if (listed(sleeper, "process-test")) fail("sleeper still listed after kill"); if (listed(spinner, "process-test")) fail("spinner still listed after kill"); + // M17.2: both children were killed by us, and the reason says so — the whole + // restart-policy input, read through the runtime like a real supervisor would. + if ((runtime.process.exitReason(sleeper) orelse .exited) != .killed) fail("sleeper reason not killed"); + if ((runtime.process.exitReason(spinner) orelse .exited) != .killed) fail("spinner reason not killed"); + if (runtime.process.exitReason(0xFFFF_FFF0) != null) fail("unknown id had a reason"); + _ = runtime.system.write("process-test: ok\n"); }