kernel: M2 shared-fate — group fan-out, dying latch, deferred leader notification

All process deaths (exit from any thread, ring-3 fault, process_kill) now kill
the whole thread group via killGroupLocked: latch the AddressSpaceRef as dying
(refusing new members, closing the thread_spawn escape), stamp every member
(leader carries the group reason — the record the supervisor reads), reap
parked members to fixpoint, condemn running ones. The leader's exit
notification and subscriber broadcast move to the group-death moment — the
last address-space reference drop — via scheduler.group_exit_hook, which
re-stamps the leader's exit record first. Leader thread_exit is refused with
-EPERM. kill_pending is atomic; exit_reason and fault_kill_count writes moved
under the big kernel lock. (docs/shared-fate-plan.md M2)
This commit is contained in:
Daniel Samson
2026-07-22 10:23:41 +01:00
parent daca0d9216
commit b09a62bc36
2 changed files with 240 additions and 43 deletions
+154 -34
View File
@@ -167,6 +167,7 @@ pub fn init() void {
scheduler.terminate_current_hook = terminateCurrentLocked;
scheduler.reap_task_hook = reapTaskLocked;
scheduler.timer_tick_hook = timerSweepLocked;
scheduler.group_exit_hook = groupExitLocked;
}
/// Return -1 (as an unsigned bit pattern) in the system_call result register.
@@ -178,9 +179,9 @@ fn system_call(state: *architecture.CpuState) void {
const t = scheduler.current();
const user = t.address_space != 0;
if (user) {
// A condemned process (process_kill caught it running) dies at its next
// A condemned process (a kill caught it running) dies at its next
// kernel entry — before it can spawn, claim, or message anything else.
if (t.kill_pending) terminateCurrent();
if (t.kill_pending.load(.monotonic)) terminateCurrent();
// Mark the span of this call so the timer tick never tears the task down
// in the middle of a kernel operation (scheduler.reapKillPendingLocked).
t.in_system_call = true;
@@ -191,14 +192,15 @@ fn system_call(state: *architecture.CpuState) void {
switch (@as(SystemCall, @enumFromInt(architecture.systemCallNumber(state)))) {
.exit => {
exit_code = architecture.systemCallArg(state, 0);
// A scheduled process tears down fully (terminateCurrent); a borrowed
// test thread unwinds back to the kernel that entered it. A NONZERO
// code is a deliberate failure exit (`.aborted`): "the work exists
// but I could not do it" — supervisors restart those, unlike a clean
// `.exited` ("nothing for me here"), which they let lie.
// A scheduled process tears down fully — the WHOLE process: exit from
// any thread is group death, the exit_group lesson (docs/shared-fate-
// plan.md). A borrowed test thread unwinds back to the kernel that
// entered it. A NONZERO code is a deliberate failure exit
// (`.aborted`): "the work exists but I could not do it" — supervisors
// restart those, unlike a clean `.exited` ("nothing for me here"),
// which they let lie.
if (scheduler.currentIsUserProcess()) {
scheduler.current().exit_reason = if (exit_code == 0) .exited else .aborted;
terminateCurrent();
exitGroupCurrent(if (exit_code == 0) .exited else .aborted);
} else architecture.userExit();
},
.yield => {
@@ -256,12 +258,17 @@ fn system_call(state: *architecture.CpuState) void {
.futex_wait => systemFutexWait(state),
.futex_wake => systemFutexWake(state),
.thread_exit => {
// A thread ends like a process exit(0), but only this task: its
// resources are released and its address-space reference dropped (the
// space survives while sibling threads hold it). docs/threading.md.
// A WORKER thread ends like a process exit(0), but only this task:
// its resources are released and its address-space reference dropped
// (the space survives while sibling threads hold it). The LEADER may
// not thread_exit — the group ends only through exit, a fault, or
// process_kill (docs/shared-fate-plan.md, decided at sign-off).
if (scheduler.currentIsUserProcess()) {
scheduler.current().exit_reason = .exited;
terminateCurrent();
const dying = scheduler.current();
if (dying.id == dying.leader) return failErr(state, ipc.EPERM);
_ = sync.enter(); // handed off through the exit switch
dying.exit_reason = .exited;
terminateCurrentLocked();
} else architecture.userExit();
},
_ => fail(state),
@@ -916,9 +923,17 @@ fn releaseTaskResourcesLocked(t: *scheduler.Task) void {
ipc.closeHandles(t);
// Publish the exit to every subscriber (docs/process-lifecycle.md): the same
// badge encoding as the supervisor's notification, and equally late, so a
// subscriber also observes a fully-released child.
for (&exit_subscribers) |*slot| {
if (slot.*) |subscriber| ipc.notifyLocked(subscriber.endpoint, abi.notify_exit_bit | t.id);
// subscriber also observes a fully-released child. One exception: a dying
// group's LEADER defers its publication to the group_exit_hook — the group
// is only "fully released" when its LAST member is gone (docs/shared-fate-
// plan.md). Workers still publish per-task: subscribers like the FAT server
// sweep per-tid client state and need every id.
const defer_to_group_hook = t.address_space != 0 and t.id == t.leader and
scheduler.groupDyingLocked(t.address_space);
if (!defer_to_group_hook) {
for (&exit_subscribers) |*slot| {
if (slot.*) |subscriber| ipc.notifyLocked(subscriber.endpoint, abi.notify_exit_bit | t.id);
}
}
if (t.exit_endpoint) |raw| {
const endpoint: *ipc.Endpoint = @ptrCast(@alignCast(raw));
@@ -957,16 +972,19 @@ fn reapTaskLocked(t: *scheduler.Task) void {
scheduler.destroyTaskLocked(t);
}
/// Kill process `target_id` on behalf of `caller_id` — the kernel half of the
/// process_kill system call. Returns 0, -ESRCH (no such live process — kernel
/// tasks are not killable processes and stale ids miss, since ids are never
/// reused), or -EPERM (the caller is not the target's supervisor).
/// Kill the process containing `target_id` on behalf of `caller_id` — the kernel
/// half of the process_kill system call, and a WHOLE-GROUP kill: any member id
/// resolves to the leader, and every thread dies (docs/shared-fate-plan.md).
/// Returns 0, -ESRCH (no such live process — kernel tasks are not killable
/// processes and stale ids miss, since ids are never reused), or -EPERM (the
/// caller is not the leader's supervisor).
///
/// A target that is ready or blocked is reaped on the spot. One that is running
/// on another core cannot be torn down mid-instruction, so it is condemned
/// (`kill_pending`) and dies at its next system_call entry, block, or timer tick
/// — like a Unix signal, delivery is prompt but asynchronous. Either way the
/// call returns 0: the kill is accepted and irrevocable.
/// Members that are ready or blocked are reaped on the spot. Ones running on
/// another core cannot be torn down mid-instruction, so they are condemned
/// (`kill_pending`) and die at their next delivery point — like a Unix signal,
/// delivery is prompt but asynchronous. Either way the call returns 0: the kill
/// is accepted and irrevocable, and the supervisor's notification arrives only
/// once the last member is gone.
pub fn killProcess(caller_id: u32, target_id: u32) i64 {
const flags = sync.enter();
defer sync.leave(flags);
@@ -982,15 +1000,77 @@ pub fn killProcess(caller_id: u32, target_id: u32) i64 {
else
scheduler.taskByIdLocked(target.leader) orelse return -ipc.ESRCH;
if (leader.supervisor != caller_id) return -ipc.EPERM;
target.exit_reason = .killed;
if (target.state == .running) {
target.kill_pending = true;
} else {
reapTaskLocked(target);
}
// Whole-group kill (docs/shared-fate-plan.md). Already dying → the kill is
// already true: accepted and irrevocable either way, return 0. The caller is
// never a member (the leader's supervisor predates the group and cannot be
// inside it), so this always returns.
if (scheduler.groupDyingLocked(target.address_space)) return 0;
killGroupLocked(leader, .killed, null);
return 0;
}
/// Kill every member of `leader_task`'s group (docs/shared-fate-plan.md): latch
/// the address space as dying — closing the door to new members and moving the
/// leader's exit-endpoint reference into the latch's stash — stamp every
/// member's exit reason, reap the parked ones, and condemn the running ones
/// (there is no kill IPI: a condemned member dies at its next syscall entry, at
/// its own core's next tick in user mode, or — once parked — at any core's tick
/// reap). `trigger` is the current task when the kill came from inside (exit,
/// fault); it dies last and the call never returns. A supervisor's kill passes
/// null and returns. Caller holds the kernel lock and has already checked
/// `groupDyingLocked` — a second trigger must not re-stamp.
fn killGroupLocked(leader_task: *scheduler.Task, reason: abi.ExitReason, trigger: ?*scheduler.Task) void {
const leader_id = leader_task.id;
if (scheduler.markGroupDyingLocked(leader_task.address_space, leader_id, leader_task.supervisor, reason, leader_task.exit_endpoint)) {
// The stash now owns the leader's endpoint reference; the leader's own
// teardown sees null (no early notification, no double drop) and the
// group_exit_hook posts exactly once, at space destruction.
leader_task.exit_endpoint = null;
}
// Stamp before any teardown — recordExitLocked snapshots exit_reason as its
// first act. The LEADER carries the group reason: its record is the one the
// supervisor can read. The trigger keeps it too (its own record tells the
// truth); every other member died because the group died: .killed.
for (scheduler.allTasksLocked()) |*member| {
if (member.leader != leader_id) continue;
if (member.state == .free or member.state == .reaping) continue;
member.exit_reason = if (member.id == leader_id or member == trigger) reason else .killed;
member.kill_pending.store(true, .monotonic);
}
// Reap parked members, to fixpoint: one member's teardown can wake another
// (an -EPEER'd client, a joiner), flipping it .blocked -> .ready behind the
// scan. Terminates in at most one pass per member: the scrubs in
// releaseTaskResourcesLocked (abandonSenderLocked, removeFromWaitQueueLocked,
// forgetIpcClientLocked, killOwnedEndpointsLocked) run before destroy, so no
// wake path holds a pointer to a reaped member.
var progress = true;
while (progress) {
progress = false;
for (scheduler.allTasksLocked()) |*member| {
if (member.leader != leader_id) continue;
if (member.state != .ready and member.state != .blocked) continue;
reapTaskLocked(member);
progress = true;
}
}
// Members running on other cores stay condemned; the in-group trigger dies
// now — last, because terminateCurrentLocked switches away for good.
if (trigger != null) terminateCurrentLocked();
}
/// The group half of `exit`: end the CURRENT task's whole process with `reason`.
/// Takes the lock (handed off through the exit switch, like terminateCurrent)
/// and never returns. A second trigger — the group is already dying — keeps the
/// first trigger's stamp and just dies.
fn exitGroupCurrent(reason: abi.ExitReason) noreturn {
const t = scheduler.current();
_ = sync.enter();
if (scheduler.groupDyingLocked(t.address_space)) terminateCurrentLocked();
const leader_task = if (t.leader == t.id) t else scheduler.taskByIdLocked(t.leader) orelse t;
killGroupLocked(leader_task, reason, t);
unreachable; // killGroupLocked never returns for an in-group trigger
}
/// Kill the current user process in response to a CPU fault it raised in ring 3.
/// The fault is confined to the process — the kernel trapped it on the task's own
/// kernel stack and is intact — so everything the process held is reclaimed and the
@@ -998,9 +1078,16 @@ pub fn killProcess(caller_id: u32, target_id: u32) i64 {
/// (docs/resilience.md: fault -> kill -> continue). `reason` is the fault class
/// (from the vector), recorded for the supervisor's `process_exit_reason`.
pub fn killCurrentProcess(reason: abi.ExitReason) noreturn {
scheduler.current().exit_reason = reason;
const t = scheduler.current();
_ = sync.enter(); // handed off through the exit switch, released by the resumed task
// A second member faulting while the group already dies: no re-stamp, no
// second count bump — the first trigger owns the group's story
// (docs/shared-fate-plan.md). `fault_kill_count` is per faulting GROUP.
if (scheduler.groupDyingLocked(t.address_space)) terminateCurrentLocked();
fault_kill_count += 1;
terminateCurrent();
const leader_task = if (t.leader == t.id) t else scheduler.taskByIdLocked(t.leader) orelse t;
killGroupLocked(leader_task, reason, t);
unreachable; // killGroupLocked never returns for an in-group trigger
}
/// The bounded record of recent deaths, for `process_exit_reason`: ids are never
@@ -1020,6 +1107,39 @@ fn recordExitLocked(t: *scheduler.Task) void {
exit_record_next = (exit_record_next + 1) % exit_record_capacity;
}
/// The group-death moment (scheduler.group_exit_hook): the last member is gone
/// and the address space destroyed. Re-stamp the leader's exit record with the
/// group reason — the record must be present and current when the notification
/// lands, however many member deaths churned the ring in between — then publish
/// the leader's exit: the subscriber broadcast and the supervisor's endpoint
/// notification, whose stashed reference is dropped here, exactly once. Runs
/// under the big kernel lock, at both release sites (docs/shared-fate-plan.md).
fn groupExitLocked(leader: u32, supervisor: u32, reason: abi.ExitReason, exit_endpoint: ?*anyopaque) void {
restampExitRecordLocked(leader, supervisor, reason);
for (&exit_subscribers) |*slot| {
if (slot.*) |subscriber| ipc.notifyLocked(subscriber.endpoint, abi.notify_exit_bit | leader);
}
if (exit_endpoint) |raw| {
const endpoint: *ipc.Endpoint = @ptrCast(@alignCast(raw));
ipc.notifyLocked(endpoint, abi.notify_exit_bit | leader);
ipc.dropRef(endpoint);
}
}
/// Overwrite dead task `id`'s exit record with the group reason, or re-append it
/// if the group's death burst already evicted it — a supervisor must always be
/// able to read the reason for a notification it just received. Lock held.
fn restampExitRecordLocked(id: u32, supervisor: u32, reason: abi.ExitReason) void {
for (&exit_records) |*record| {
if (record.valid and record.id == id) {
record.reason = reason;
return;
}
}
exit_records[exit_record_next] = .{ .id = id, .supervisor = supervisor, .reason = reason, .valid = true };
exit_record_next = (exit_record_next + 1) % exit_record_capacity;
}
/// How dead process `id` ended, for `caller` — the kernel half of the
/// process_exit_reason system call. Returns the ExitReason value, -ESRCH (never
/// lived, still alive, or evicted from the ring), or -EPERM (the caller was not