kernel: M2 shared-fate — group fan-out, dying latch, deferred leader notification

All process deaths (exit from any thread, ring-3 fault, process_kill) now kill
the whole thread group via killGroupLocked: latch the AddressSpaceRef as dying
(refusing new members, closing the thread_spawn escape), stamp every member
(leader carries the group reason — the record the supervisor reads), reap
parked members to fixpoint, condemn running ones. The leader's exit
notification and subscriber broadcast move to the group-death moment — the
last address-space reference drop — via scheduler.group_exit_hook, which
re-stamps the leader's exit record first. Leader thread_exit is refused with
-EPERM. kill_pending is atomic; exit_reason and fault_kill_count writes moved
under the big kernel lock. (docs/shared-fate-plan.md M2)
This commit is contained in:
Daniel Samson
2026-07-22 10:23:41 +01:00
parent daca0d9216
commit b09a62bc36
2 changed files with 240 additions and 43 deletions
+86 -9
View File
@@ -70,9 +70,12 @@ pub const Task = struct {
// (docs/process-lifecycle.md). Bits are abi.Signal values. Signals pend here
// until an endpoint is bound; two pending terminates are one terminate.
pending_signals: u32 = 0,
// Set by process_kill on a task that is running on another core; the kernel
// finishes the kill at that task's next system call or timer tick.
kill_pending: bool = false,
// Set by process_kill (or a group fan-out) on a task that is running on
// another core; the kernel finishes the kill at that task's next system call
// or timer tick. Atomic because the syscall-entry check reads it without the
// lock while another core writes it under the lock — monotonic is enough: a
// missed read is caught at the next delivery point (x86-TSO or not).
kill_pending: std.atomic.Value(bool) = .init(false),
// True while this task executes its own system call — the timer tick must not
// tear a task down in the middle of a kernel operation, only while it runs
// user code (or sits at a block point, where teardown is safe).
@@ -165,7 +168,24 @@ var tasks = [_]Task{.{}} ** maximum_tasks;
// One live entry per address space; threads sharing an address space share this entry,
// so their mmap/mmio grants bump one cursor and never overlap (docs/threading-plan.md M7).
// `mmap_next`/`device_map_next` are 0 until process.zig seeds them to the arena base.
const AddressSpaceRef = struct { root: u64 = 0, count: u32 = 0, mmap_next: u64 = 0, device_map_next: u64 = 0 };
const AddressSpaceRef = struct {
root: u64 = 0,
count: u32 = 0,
mmap_next: u64 = 0,
device_map_next: u64 = 0,
// Group-death state (docs/shared-fate-plan.md), set once by the first kill
// trigger and never cleared while the entry lives. `dying` gates
// retainAddressSpace — no new member may join a dying group (closing the
// thread_spawn escape) — and the stash is what `group_exit_hook` posts when
// the last reference drops: the leader's identity, the group reason, and the
// leader's counted exit-endpoint reference (moved off its Task under the
// same lock hold that set the latch).
dying: bool = false,
group_leader: u32 = 0,
group_supervisor: u32 = 0,
group_reason: abi.ExitReason = .exited,
group_exit_endpoint: ?*anyopaque = null,
};
var address_space_refs = [_]AddressSpaceRef{.{}} ** maximum_tasks;
var address_space_destroy_count: u64 = 0;
@@ -213,6 +233,11 @@ fn retainAddressSpace(root: u64) bool {
var free: ?*AddressSpaceRef = null;
for (&address_space_refs) |*entry| {
if (entry.count != 0 and entry.root == root) {
// A dying group admits no new member: a thread_spawn that was already
// past its syscall-entry kill check when the group was condemned
// fails here, under the same lock that would have created the task
// (docs/shared-fate-plan.md).
if (entry.dying) return false;
entry.count += 1;
return true;
}
@@ -232,16 +257,68 @@ fn releaseAddressSpace(root: u64) void {
if (entry.count == 0 or entry.root != root) continue;
entry.count -= 1;
if (entry.count == 0) {
entry.root = 0;
// Copy the group-death stash out, then fully reset the slot before
// the hook runs — a stale stash must never survive into an unrelated
// process's reused entry (docs/shared-fate-plan.md).
const was_dying = entry.dying;
const leader = entry.group_leader;
const supervisor = entry.group_supervisor;
const reason = entry.group_reason;
const endpoint = entry.group_exit_endpoint;
entry.* = .{};
architecture.destroyAddressSpace(root);
address_space_destroy_count += 1;
// The group-death moment: the space is gone, every member is dead.
// process.zig posts the leader's deferred exit publication here.
if (was_dying) if (group_exit_hook) |hook| hook(leader, supervisor, reason, endpoint);
}
return;
}
// Never retained (a hand-built test space): destroyed directly, out of the
// group-death hook's scope, preserving the pre-refcount behaviour.
architecture.destroyAddressSpace(root);
address_space_destroy_count += 1;
}
/// Called (lock held) at the group-death moment — the last reference to a dying
/// group's address space dropped and the space was destroyed. Registered by
/// process.zig, which posts the leader's deferred exit notification and re-stamps
/// its exit record (docs/shared-fate-plan.md). Runs under the lock at both
/// release sites (`exitUserLocked`, `destroyTaskLocked`).
pub var group_exit_hook: ?*const fn (leader: u32, supervisor: u32, reason: abi.ExitReason, exit_endpoint: ?*anyopaque) void = null;
/// Latch `root`'s group as dying and stash the group-death payload for the hook.
/// Returns false — changing nothing — if the group is already dying (a concurrent
/// trigger lost the race) or `root` has no live entry. Caller holds the lock.
pub fn markGroupDyingLocked(root: u64, leader: u32, supervisor: u32, reason: abi.ExitReason, exit_endpoint: ?*anyopaque) bool {
for (&address_space_refs) |*entry| {
if (entry.count == 0 or entry.root != root) continue;
if (entry.dying) return false;
entry.dying = true;
entry.group_leader = leader;
entry.group_supervisor = supervisor;
entry.group_reason = reason;
entry.group_exit_endpoint = exit_endpoint;
return true;
}
return false;
}
/// Whether `root`'s group is already dying. Caller holds the lock.
pub fn groupDyingLocked(root: u64) bool {
for (&address_space_refs) |*entry| {
if (entry.count != 0 and entry.root == root) return entry.dying;
}
return false;
}
/// The whole static task pool, for process.zig's group fan-out — which must scan
/// members under the lock it already holds. Slots may be `.free`/`.reaping`;
/// callers filter by state and must not hold pointers past the lock.
pub fn allTasksLocked() []Task {
return tasks[0..];
}
/// Test-observable: how many address spaces are live (entries with a nonzero count).
pub fn liveAddressSpaceCount() u32 {
var live: u32 = 0;
@@ -916,12 +993,12 @@ fn reapKillPendingLocked() void {
// task_trampoline, not switchTo's tail), its stack is still queued here. The dying
// task switched away before this tick, so it is off its stack — drain now (M8).
drainReapListLocked(pc);
if (cur.kill_pending and cur.address_space != 0 and !cur.in_system_call) {
if (cur.kill_pending.load(.monotonic) and cur.address_space != 0 and !cur.in_system_call) {
if (terminate_current_hook) |hook| hook(); // noreturn
}
if (reap_task_hook) |hook| {
for (&tasks) |*t| {
if (!t.kill_pending) continue;
if (!t.kill_pending.load(.monotonic)) continue;
if (t.state == .ready or t.state == .blocked) hook(t);
}
}
@@ -997,7 +1074,7 @@ pub fn exitUserLocked() noreturn {
dying.state = .reaping; // dead but its slot stays reserved until the stack is freed
wakeJoinersLocked(dying.id); // let any thread_join(dying.id) return (M9)
dying.address_space = 0;
dying.kill_pending = false;
dying.kill_pending.store(false, .monotonic);
dying.in_system_call = false;
// Queue for reaping: the task we switch to (or the next tick) frees this stack (M8/M9).
dying.next = pc.reap_list;
@@ -1020,7 +1097,7 @@ pub fn destroyTaskLocked(t: *Task) void {
if (t.address_space != 0) releaseAddressSpace(t.address_space); // destroys only on the last reference
reapStackLocked(t); // safe to free now: `t` is not running on any core (M8)
t.address_space = 0;
t.kill_pending = false;
t.kill_pending.store(false, .monotonic);
t.in_system_call = false;
t.wake_at = 0;
t.state = .free;