kernel: M2 shared-fate — group fan-out, dying latch, deferred leader notification
All process deaths (exit from any thread, ring-3 fault, process_kill) now kill the whole thread group via killGroupLocked: latch the AddressSpaceRef as dying (refusing new members, closing the thread_spawn escape), stamp every member (leader carries the group reason — the record the supervisor reads), reap parked members to fixpoint, condemn running ones. The leader's exit notification and subscriber broadcast move to the group-death moment — the last address-space reference drop — via scheduler.group_exit_hook, which re-stamps the leader's exit record first. Leader thread_exit is refused with -EPERM. kill_pending is atomic; exit_reason and fault_kill_count writes moved under the big kernel lock. (docs/shared-fate-plan.md M2)
This commit is contained in:
@@ -70,9 +70,12 @@ pub const Task = struct {
|
||||
// (docs/process-lifecycle.md). Bits are abi.Signal values. Signals pend here
|
||||
// until an endpoint is bound; two pending terminates are one terminate.
|
||||
pending_signals: u32 = 0,
|
||||
// Set by process_kill on a task that is running on another core; the kernel
|
||||
// finishes the kill at that task's next system call or timer tick.
|
||||
kill_pending: bool = false,
|
||||
// Set by process_kill (or a group fan-out) on a task that is running on
|
||||
// another core; the kernel finishes the kill at that task's next system call
|
||||
// or timer tick. Atomic because the syscall-entry check reads it without the
|
||||
// lock while another core writes it under the lock — monotonic is enough: a
|
||||
// missed read is caught at the next delivery point (x86-TSO or not).
|
||||
kill_pending: std.atomic.Value(bool) = .init(false),
|
||||
// True while this task executes its own system call — the timer tick must not
|
||||
// tear a task down in the middle of a kernel operation, only while it runs
|
||||
// user code (or sits at a block point, where teardown is safe).
|
||||
@@ -165,7 +168,24 @@ var tasks = [_]Task{.{}} ** maximum_tasks;
|
||||
// One live entry per address space; threads sharing an address space share this entry,
|
||||
// so their mmap/mmio grants bump one cursor and never overlap (docs/threading-plan.md M7).
|
||||
// `mmap_next`/`device_map_next` are 0 until process.zig seeds them to the arena base.
|
||||
const AddressSpaceRef = struct { root: u64 = 0, count: u32 = 0, mmap_next: u64 = 0, device_map_next: u64 = 0 };
|
||||
const AddressSpaceRef = struct {
|
||||
root: u64 = 0,
|
||||
count: u32 = 0,
|
||||
mmap_next: u64 = 0,
|
||||
device_map_next: u64 = 0,
|
||||
// Group-death state (docs/shared-fate-plan.md), set once by the first kill
|
||||
// trigger and never cleared while the entry lives. `dying` gates
|
||||
// retainAddressSpace — no new member may join a dying group (closing the
|
||||
// thread_spawn escape) — and the stash is what `group_exit_hook` posts when
|
||||
// the last reference drops: the leader's identity, the group reason, and the
|
||||
// leader's counted exit-endpoint reference (moved off its Task under the
|
||||
// same lock hold that set the latch).
|
||||
dying: bool = false,
|
||||
group_leader: u32 = 0,
|
||||
group_supervisor: u32 = 0,
|
||||
group_reason: abi.ExitReason = .exited,
|
||||
group_exit_endpoint: ?*anyopaque = null,
|
||||
};
|
||||
var address_space_refs = [_]AddressSpaceRef{.{}} ** maximum_tasks;
|
||||
var address_space_destroy_count: u64 = 0;
|
||||
|
||||
@@ -213,6 +233,11 @@ fn retainAddressSpace(root: u64) bool {
|
||||
var free: ?*AddressSpaceRef = null;
|
||||
for (&address_space_refs) |*entry| {
|
||||
if (entry.count != 0 and entry.root == root) {
|
||||
// A dying group admits no new member: a thread_spawn that was already
|
||||
// past its syscall-entry kill check when the group was condemned
|
||||
// fails here, under the same lock that would have created the task
|
||||
// (docs/shared-fate-plan.md).
|
||||
if (entry.dying) return false;
|
||||
entry.count += 1;
|
||||
return true;
|
||||
}
|
||||
@@ -232,16 +257,68 @@ fn releaseAddressSpace(root: u64) void {
|
||||
if (entry.count == 0 or entry.root != root) continue;
|
||||
entry.count -= 1;
|
||||
if (entry.count == 0) {
|
||||
entry.root = 0;
|
||||
// Copy the group-death stash out, then fully reset the slot before
|
||||
// the hook runs — a stale stash must never survive into an unrelated
|
||||
// process's reused entry (docs/shared-fate-plan.md).
|
||||
const was_dying = entry.dying;
|
||||
const leader = entry.group_leader;
|
||||
const supervisor = entry.group_supervisor;
|
||||
const reason = entry.group_reason;
|
||||
const endpoint = entry.group_exit_endpoint;
|
||||
entry.* = .{};
|
||||
architecture.destroyAddressSpace(root);
|
||||
address_space_destroy_count += 1;
|
||||
// The group-death moment: the space is gone, every member is dead.
|
||||
// process.zig posts the leader's deferred exit publication here.
|
||||
if (was_dying) if (group_exit_hook) |hook| hook(leader, supervisor, reason, endpoint);
|
||||
}
|
||||
return;
|
||||
}
|
||||
// Never retained (a hand-built test space): destroyed directly, out of the
|
||||
// group-death hook's scope, preserving the pre-refcount behaviour.
|
||||
architecture.destroyAddressSpace(root);
|
||||
address_space_destroy_count += 1;
|
||||
}
|
||||
|
||||
/// Called (lock held) at the group-death moment — the last reference to a dying
|
||||
/// group's address space dropped and the space was destroyed. Registered by
|
||||
/// process.zig, which posts the leader's deferred exit notification and re-stamps
|
||||
/// its exit record (docs/shared-fate-plan.md). Runs under the lock at both
|
||||
/// release sites (`exitUserLocked`, `destroyTaskLocked`).
|
||||
pub var group_exit_hook: ?*const fn (leader: u32, supervisor: u32, reason: abi.ExitReason, exit_endpoint: ?*anyopaque) void = null;
|
||||
|
||||
/// Latch `root`'s group as dying and stash the group-death payload for the hook.
|
||||
/// Returns false — changing nothing — if the group is already dying (a concurrent
|
||||
/// trigger lost the race) or `root` has no live entry. Caller holds the lock.
|
||||
pub fn markGroupDyingLocked(root: u64, leader: u32, supervisor: u32, reason: abi.ExitReason, exit_endpoint: ?*anyopaque) bool {
|
||||
for (&address_space_refs) |*entry| {
|
||||
if (entry.count == 0 or entry.root != root) continue;
|
||||
if (entry.dying) return false;
|
||||
entry.dying = true;
|
||||
entry.group_leader = leader;
|
||||
entry.group_supervisor = supervisor;
|
||||
entry.group_reason = reason;
|
||||
entry.group_exit_endpoint = exit_endpoint;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Whether `root`'s group is already dying. Caller holds the lock.
|
||||
pub fn groupDyingLocked(root: u64) bool {
|
||||
for (&address_space_refs) |*entry| {
|
||||
if (entry.count != 0 and entry.root == root) return entry.dying;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// The whole static task pool, for process.zig's group fan-out — which must scan
|
||||
/// members under the lock it already holds. Slots may be `.free`/`.reaping`;
|
||||
/// callers filter by state and must not hold pointers past the lock.
|
||||
pub fn allTasksLocked() []Task {
|
||||
return tasks[0..];
|
||||
}
|
||||
|
||||
/// Test-observable: how many address spaces are live (entries with a nonzero count).
|
||||
pub fn liveAddressSpaceCount() u32 {
|
||||
var live: u32 = 0;
|
||||
@@ -916,12 +993,12 @@ fn reapKillPendingLocked() void {
|
||||
// task_trampoline, not switchTo's tail), its stack is still queued here. The dying
|
||||
// task switched away before this tick, so it is off its stack — drain now (M8).
|
||||
drainReapListLocked(pc);
|
||||
if (cur.kill_pending and cur.address_space != 0 and !cur.in_system_call) {
|
||||
if (cur.kill_pending.load(.monotonic) and cur.address_space != 0 and !cur.in_system_call) {
|
||||
if (terminate_current_hook) |hook| hook(); // noreturn
|
||||
}
|
||||
if (reap_task_hook) |hook| {
|
||||
for (&tasks) |*t| {
|
||||
if (!t.kill_pending) continue;
|
||||
if (!t.kill_pending.load(.monotonic)) continue;
|
||||
if (t.state == .ready or t.state == .blocked) hook(t);
|
||||
}
|
||||
}
|
||||
@@ -997,7 +1074,7 @@ pub fn exitUserLocked() noreturn {
|
||||
dying.state = .reaping; // dead but its slot stays reserved until the stack is freed
|
||||
wakeJoinersLocked(dying.id); // let any thread_join(dying.id) return (M9)
|
||||
dying.address_space = 0;
|
||||
dying.kill_pending = false;
|
||||
dying.kill_pending.store(false, .monotonic);
|
||||
dying.in_system_call = false;
|
||||
// Queue for reaping: the task we switch to (or the next tick) frees this stack (M8/M9).
|
||||
dying.next = pc.reap_list;
|
||||
@@ -1020,7 +1097,7 @@ pub fn destroyTaskLocked(t: *Task) void {
|
||||
if (t.address_space != 0) releaseAddressSpace(t.address_space); // destroys only on the last reference
|
||||
reapStackLocked(t); // safe to free now: `t` is not running on any core (M8)
|
||||
t.address_space = 0;
|
||||
t.kill_pending = false;
|
||||
t.kill_pending.store(false, .monotonic);
|
||||
t.in_system_call = false;
|
||||
t.wake_at = 0;
|
||||
t.state = .free;
|
||||
|
||||
Reference in New Issue
Block a user