threads(M8): the task reaper — reclaim dead tasks' kernel stacks

A dead task's kernel stack was leaked (no reaper), so every process/thread death
bled kernel memory. Now exit()/exitUserLocked record the dying task in a per-core
reap_after_switch slot and switch away; the task that resumes on that core frees
the stack in switchTo's tail (on its own stack, lock still held so the slot can't
be reused). reapKillPendingLocked drains the slot on the timer tick as a safety
net for the fresh-task case (a fresh task enters via the trampoline, bypassing
switchTo's tail). A task killed while not running is freed directly in
destroyTaskLocked. live_stack_bytes is the observable.

Fixed a migration bug this exposed: the post-switchContext reap read the pc
parameter, but a migrated task carries a stale pc in its saved switchTo frame ->
it freed the wrong core's pending stack (a #GP under SMP). Re-fetch thisCpu()
after the switch.

Gate task-reap PASS (5x isolated, 2x in the 24-case batch); full guardrail 24/24
incl. fault-recovery/supervision/process-kill/smp/affinity; build + host green.
This commit is contained in:
2026-07-20 23:18:30 +01:00
parent 8259678f0a
commit ed3b3f1c45
4 changed files with 124 additions and 14 deletions
+52
View File
@@ -152,6 +152,28 @@ const AspaceRef = struct { root: u64 = 0, count: u32 = 0, mmap_next: u64 = 0, de
var aspace_refs = [_]AspaceRef{.{}} ** maximum_tasks;
var aspace_destroy_count: u64 = 0;
/// Total bytes of task **kernel** stacks currently allocated from the kernel heap —
/// incremented when a task is created, decremented when the reaper frees a dead task's
/// stack. A test-observable proof that the reaper reclaims every stack (docs/threading-
/// plan.md M8): with no live tasks beyond the baseline, this returns to its baseline.
var live_stack_bytes: usize = 0;
/// Test-observable: bytes of task kernel stacks currently live (see `live_stack_bytes`).
pub fn liveStackBytes() usize {
return live_stack_bytes;
}
/// Free a dead task's kernel stack and drop it from `live_stack_bytes`. The task must be
/// off that stack already (killed while not running, or reaped after it switched away).
/// Caller holds the kernel lock.
fn reapStackLocked(t: *Task) void {
if (t.stack.len == 0) return; // boot/idle tasks run on a static stack — nothing to free
live_stack_bytes -= t.stack.len;
heap.allocator().free(t.stack);
t.stack = &.{};
t.kstack_top = 0;
}
/// Take a reference to address space `root` (0 = a kernel task, which owns none).
/// Returns false only if the ref table is full — bounded by `maximum_tasks`, so in
/// practice it never is. Caller holds the kernel lock.
@@ -246,6 +268,12 @@ pub const PerCpu = struct {
pinned_head: [number_priorities]?*Task = .{null} ** number_priorities,
pinned_tail: [number_priorities]?*Task = .{null} ** number_priorities,
pinned_bitmap: u8 = 0,
// A task that ended while running on THIS core: it could not free the kernel stack it
// was standing on, so it recorded itself here and switched away. The next task to run
// on this core frees that stack (from its own stack, safely) in `switchTo`. The big
// lock is held continuously across the switch, so the dead task's slot can't be reused
// before it is reaped (docs/threading-plan.md M8).
reap_after_switch: ?*Task = null,
};
const maximum_cpus = parameters.maximum_cpus;
@@ -422,6 +450,7 @@ pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, user_arg: u64, pri
heap.allocator().free(stack);
return null;
}
live_stack_bytes += stack.len; // the reaper drops this when the task dies (M8)
t.* = .{
.id = next_id,
.state = .ready,
@@ -465,6 +494,7 @@ fn startUserTask() void {
fn create(entry: *const fn () void, priority: Priority, affinity: ?u32) *Task {
const t = freeSlot() orelse @panic("sched: task table full");
const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack");
live_stack_bytes += stack.len; // the reaper drops this when the task dies (M8)
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack, .affinity = affinity };
next_id += 1;
const top = @intFromPtr(stack.ptr) + stack.len;
@@ -519,6 +549,17 @@ fn switchTo(pc: *PerCpu, save_sp: *usize, next: *Task) void {
pc.loaded_aspace = want;
}
architecture.switchContext(save_sp, next.sp);
// Resumed now (switchContext returned into our own switchTo frame). Re-fetch the core
// via thisCpu(): the `pc` parameter is from *our* earlier switchTo call, so it names
// the core we last ran on — stale if we migrated. switchContext only swaps stacks on
// the current core, so thisCpu() is the core the just-dead task died on. If a task
// died switching to us, free its kernel stack: we're on ours so it's safe, and the big
// lock is still held so its slot can't have been reused (docs/threading-plan.md M8).
const here = thisCpu();
if (here.reap_after_switch) |dead| {
here.reap_after_switch = null;
reapStackLocked(dead);
}
}
/// Voluntarily give up the CPU to the next ready task.
@@ -785,6 +826,14 @@ pub var reap_task_hook: ?*const fn (*Task) void = null;
fn reapKillPendingLocked() void {
const pc = thisCpu();
const cur = pc.current;
// Safety net for the reap-after-switch slot: if a dying task switched to a *fresh*
// task (which enters via task_trampoline, not switchTo's tail), its kernel stack is
// still pending here. The dying task switched away before this tick, so it is off its
// stack — reap it now (docs/threading-plan.md M8).
if (pc.reap_after_switch) |dead| {
pc.reap_after_switch = null;
reapStackLocked(dead);
}
if (cur.kill_pending and cur.aspace != 0 and !cur.in_system_call) {
if (terminate_current_hook) |hook| hook(); // noreturn
}
@@ -827,6 +876,7 @@ pub fn exit() noreturn {
_ = sync.enter();
const pc = thisCpu();
pc.current.state = .free;
pc.reap_after_switch = pc.current; // the task we switch to frees this stack (M8)
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
next.state = .running;
pc.current = next;
@@ -862,6 +912,7 @@ pub fn exitUserLocked() noreturn {
dying.aspace = 0;
dying.kill_pending = false;
dying.in_system_call = false;
pc.reap_after_switch = dying; // the task we switch to frees this stack (M8)
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
next.state = .running;
pc.current = next;
@@ -878,6 +929,7 @@ pub fn exitUserLocked() noreturn {
/// yet). Precondition: the big kernel lock is held.
pub fn destroyTaskLocked(t: *Task) void {
if (t.aspace != 0) releaseAspace(t.aspace); // destroys only on the last reference
reapStackLocked(t); // safe to free now: `t` is not running on any core (M8)
t.aspace = 0;
t.kill_pending = false;
t.in_system_call = false;
+38
View File
@@ -153,6 +153,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
threadIdTest(boot_information);
} else if (eql(case, "thread-alloc")) {
threadAllocTest(boot_information);
} else if (eql(case, "task-reap")) {
taskReapTest(boot_information);
} else if (eql(case, "args")) {
argsTest(boot_information);
} else if (eql(case, "init")) {
@@ -1738,6 +1740,42 @@ fn threadAllocTest(boot_information: *const BootInformation) void {
result();
}
/// The task reaper (docs/threading-plan.md M8): a dead task's kernel stack used to be
/// leaked ("no reaper yet"). Spawn and kill many ring-3 processes and confirm the total
/// kernel-stack bytes return to baseline — every stack reclaimed, no leak. (Threads exit
/// through the same exitUserLocked path, so this covers them too.)
fn taskReapTest(boot_information: *const BootInformation) void {
_ = boot_information;
log("DANOS-TEST-BEGIN: task-reap\n", .{});
const base = scheduler.liveStackBytes();
const rounds: u32 = 12;
var killed: u32 = 0;
var round: u32 = 0;
while (round < rounds) : (round += 1) {
process.fault_kill_count = 0;
const probe = spawnFaultingProcess() orelse break;
_ = probe;
scheduler.setPriority(1);
const deadline = architecture.millis() + 5000;
while (process.fault_kill_count < 1 and architecture.millis() < deadline) scheduler.yield();
scheduler.setPriority(4);
if (process.fault_kill_count >= 1) killed += 1;
}
// Reaping is asynchronous — a dead task's stack is freed when its core next switches
// or ticks. Poll (bounded) until the live bytes return to baseline: a correct reaper
// gets there in a few ms; a genuine leak never does and this times out.
scheduler.setPriority(1);
const settle_deadline = architecture.millis() + 3000;
while (scheduler.liveStackBytes() != base and architecture.millis() < settle_deadline) scheduler.yield();
scheduler.setPriority(4);
check("all probes spawned and were killed", killed == rounds);
check("kernel stacks reclaimed to baseline (no leak)", scheduler.liveStackBytes() == base);
if (killed == rounds and scheduler.liveStackBytes() == base)
log("task-reap: kernel stacks reclaimed to baseline ok\n", .{});
result();
}
/// The full PID-1 path: the bootloader read /system/services/init off the boot volume and
/// handed it over; load it as a user ELF and spawn it as a real ring-3 process
/// — the same call the normal boot path makes — then confirm it beats. init