From e78810195dbd7c2a244701b2c19d30abb619cddd Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Wed, 22 Jul 2026 10:12:42 +0100 Subject: [PATCH 1/7] =?UTF-8?q?docs:=20shared-fate=20plan=20=E2=80=94=20ap?= =?UTF-8?q?proved=20design=20for=20whole-process=20death?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Red-teamed draft: leader id (tgid-style), AddressSpaceRef dying latch, killGroupLocked fan-out, group notification at address-space destruction, shared-memory mapping refs. Leader thread_exit refused with -EPERM (decided at sign-off). Milestones M1-M4. --- docs/shared-fate-plan.md | 291 +++++++++++++++++++++++++++++++++++++++ 1 file changed, 291 insertions(+) create mode 100644 docs/shared-fate-plan.md diff --git a/docs/shared-fate-plan.md b/docs/shared-fate-plan.md new file mode 100644 index 0000000..f03a002 --- /dev/null +++ b/docs/shared-fate-plan.md @@ -0,0 +1,291 @@ +# Shared fate: whole-process death (plan) + +**Status: approved 2026-07-22; leader `thread_exit` → `-EPERM`. Implementation in +progress (M1–M4 below).** + +[threading.md](threading.md) promises that a process dies *whole* — a fault in any +thread, or a kill, takes down every thread. The kernel doesn't do that yet: every +death path (`exit`, `thread_exit`, a ring-3 fault, `process_kill`) tears down +exactly one `Task`, and the address-space refcount keeps the space alive for the +siblings — so a faulting worker orphans its threads, which keep running in the +possibly-corrupted address space (`process.zig` `killCurrentProcess`; +`scheduler.zig` `exitUserLocked`). This plan closes that gap the way Linux, Windows, +and Fuchsia all did: **the process is the unit of fate; only a voluntary +`thread_exit` is per-thread.** The supervisor contract — one exit notification, +then `process_exit_reason` — is deliberately unchanged. + +## The contract + +| Event | Who dies | Reason the supervisor reads (leader's record) | +|---|---|---| +| CPU fault in **any** thread (recoverable vector) | the whole group | the fault class (`segmentation_fault`, …) | +| `process_kill` on **any member id** | the whole group | `killed` | +| `exit(code)` from **any** thread | the whole group | `exited` (0) / `aborted` (≠0) | +| `thread_exit` from a worker | that worker only | — (per-task record: `exited`) | +| `thread_exit` from the **leader** | nobody — refused, `-EPERM` (decided, below) | — | +| NMI, double fault, machine check | the core halts (unchanged) | — | + +`exit` gaining group semantics is the `exit_group` lesson from Linux: the runtime's +main-return path calls `exit`, and a process whose main returned must not leave +workers running. `thread_exit` (what the worker trampoline calls) keeps today's +per-thread behavior, refcount and all. + +**The leader-`thread_exit` rule (decided: refuse).** The syscall is reachable from +the leader even though the runtime never does it. Options weighed: (i) **refuse +with `-EPERM`** — cheapest and honest; the group ends only through +`exit`/fault/kill; (ii) escalate to `exit(0)` — Linux-flavored, but silently turns +a (buggy) library call into process death; (iii) a Linux-style zombie leader whose +slot survives until the group ends — the most faithful, and by far the most +machinery. **(i) chosen at sign-off**; rows and tests below follow it. + +## Group identity: a leader id + +Nothing on `Task` names a process today — threads are tied to their process only by +an equal `address_space`, their name is `"thread"`, and their `supervisor` is +whichever *task* spawned them (possibly another worker), so supervision links form a +chain, not a group. Scanning by `address_space` is also fragile during teardown, +because both death paths zero it. + +So: **add `leader: u32` to `Task`** — the Linux tgid, in danos clothes. +`spawnProcessSupervised` sets `leader = own id`; `spawnThreadSupervised` copies the +*caller's* leader; kernel tasks keep `leader = 0`, which is never followed. The +leader id is exactly the id `system_spawn` returned to the supervisor, so the +outside world already speaks it. Group membership = equal `leader`, where a *live +member* means `state` ∈ {`.ready`, `.blocked`, `.running`} — the same filter +`taskByIdLocked` applies; `.reaping` corpses are excluded. While the field is being +introduced, add `leader` to `ProcessDescriptor` too (the ABI is private, so this is +cheap now and lets `process_enumerate` consumers group threads). + +`process_kill` re-derives its authority through the leader — with the existing +guard order preserved: a kernel task (`address_space == 0`) is `-ESRCH` *before* +any leader resolution (the kernel test asserts exactly that). Then: resolve the +target, follow `target.leader`, require `leader.supervisor == caller`. The kill +capability becomes per-*process*, aimed at any member id, and the odd +today-behavior where a worker can be individually killed by its spawning thread +disappears. (Audited: nothing in-tree kills a worker tid or relies on +thread-supervisor kill semantics.) + +## The group-dying latch + +`AddressSpaceRef` — one per space, refcounted by its member tasks, recycled with a +full struct re-init — gains the group-death state: + +```zig +dying: bool = false, // set by the first trigger; never cleared +group_reason: abi.ExitReason, // what the leader's record will say +exit_endpoint: ?*ipc.Endpoint, // the leader's counted ref, moved here +leader: u32, +``` + +The latch answers three attacks the red team confirmed against a latch-free +design: + +- **The spawn gate.** A member already *inside* `thread_spawn` on another core + when the fan-out runs (it passed the syscall-entry `kill_pending` check, then + spun on the BKL) would otherwise complete the spawn after the fan-out's lock + hold ends — a fresh, uncondemned member that escapes the kill and, worse, holds + a space reference that keeps the group-death hook from ever firing. Fix: + `retainAddressSpace` (equivalently `spawnUserLocked`) **refuses a dying + space**; the in-flight `thread_spawn` fails with `-ESRCH` under the same lock + that would have created the member. +- **Concurrent triggers.** A second member faulting (or exiting) on another core + while the first fan-out runs must not re-run the fan-out, double-bump + `fault_kill_count`, or re-stamp reasons. Every kill path checks the latch first: + already dying → skip straight to `terminateCurrentLocked`, no stamp, no count. + First trigger wins, deterministically. `exit_reason` and `fault_kill_count` + writes move under the BKL as part of this. +- **Notification ownership.** The leader's `exit_endpoint` is a counted + birth-to-death reference dropped at notify time. The stamp pass **moves** that + reference onto the `AddressSpaceRef` and nulls `Task.exit_endpoint` in the same + hold, so the leader's own `releaseTaskResourcesLocked` sees null (no early + notify, no double drop); the group-death hook notifies and drops exactly once. + +## The fan-out: `killGroupLocked` + +One new function in `process.zig`, running under a **single BKL hold** (built from +the `*Locked` primitives — the lock is non-recursive, and `terminateCurrentLocked` +never returns, which forces the shape): + +``` +killGroupLocked(leader: u32, reason: ExitReason, trigger: ?*Task) + 0. Latch: AddressSpaceRef.dying = true, stash {reason, leader, + leader's exit_endpoint (moved)}. + 1. Stamp pass: the LEADER's exit_reason = reason — the leader's + record is the one the supervisor can read, so it carries the + group reason even when the trigger is a worker. The trigger + also keeps `reason` (its own record tells the truth); every + other live member gets .killed. All members get kill_pending. + Stamping precedes any teardown, because recordExitLocked + snapshots the reason first thing. + 2. Reap pass, to fixpoint: reap every member in .ready or .blocked + via reapTaskLocked, re-reading Task.state each iteration — a + member's teardown can wake another member (-EPEER wakes, joiner + wakes), flipping it .blocked → .ready behind the scan cursor. + Terminates in ≤ one pass per member: the scrub calls in + releaseTaskResourcesLocked (abandonSenderLocked, + removeFromWaitQueueLocked, forgetIpcClientLocked, + killOwnedEndpointsLocked) run before destroy, so no wake path + holds a pointer to a reaped member. + 3. Members .running on other cores stay condemned (kill_pending); + a condemned member dies at its next syscall entry, at its own + core's next tick while in user mode, or — once it blocks or is + preempted — at any core's next tick reap. There is no kill IPI. + (The entry check reads kill_pending unlocked; benign on + x86-TSO — a missed read is caught by the next delivery point — + but make the field atomic when touching it.) + 4. If the current task is a member (fault, exit, in-group kill): + terminateCurrentLocked, last, because it switches away and the + reap paths free the kernel stack being stood on. + If the caller is outside the group (supervisor kill): return. +``` + +The invariants this preserves, each load-bearing today: + +- **Only `.ready`/`.blocked` tasks are reaped synchronously.** A member running on + another core can only be condemned — it tears itself down after switching CR3 + off the dying page tables (the stack it stands on is freed later by the reap + list), and its address-space reference protects the page tables its CR3 still + points at. Force-destroying the space under a running sibling is the one + unrecoverable mistake available here. +- **The refcount decides when the space dies.** Reaping N members drops N + references; the last drop — possibly on a condemned sibling's core, a tick + later — destroys the space. No path forces it. +- **`fault_kill_count` bumps once per group**, not per member (`fault-recovery` + asserts `== 1` exactly); the latch is what enforces this under racing faults. +- **Per-tid resource sweeps stay per-tid.** Each member's + `releaseTaskResourcesLocked` releases what *that tid* owns — claims, GSI/MSI + bindings, registered endpoints, handles. That keying is correct under shared + fate (and is today's hazard: a lone worker death already yanks its claims out + from under live siblings). A worker that *does* carry an `exit_endpoint` (the + ABI allows it; the runtime passes `no_cap`) keeps today's per-task posting at + its own teardown — only the leader's notification moves. + +## When is the group dead? The notification + +Today each task posts its own exit notification as the *last* step of its release, +so a supervisor observes a fully-released child. For a group that guarantee must +hold for the **whole group**: if the leader's notification fires while a condemned +sibling still runs on another core, the device manager can respawn the driver into +a claim conflict with a not-yet-dead sibling. + +The clean fix falls out of the refcount: **the group is dead exactly when the +address space is destroyed.** `releaseAddressSpace`'s last-drop path calls a new +`group_exit_hook` (the scheduler already calls up through hooks — +`terminate_current_hook` — precisely to keep this layering), which: + +1. **re-stamps the leader's exit record** with the stashed `group_reason` — the + record is written (again) at group-death time, so "the reason is recorded + before the notification posts" stays true and a supervisor can never be + notified and then read `-ESRCH` because the burst evicted an old record; +2. posts the leader's exit notification (and subscriber broadcast) from the + stashed endpoint, and drops that reference — exactly once. + +Both `releaseAddressSpace` call sites (`exitUserLocked`, `destroyTaskLocked`) run +under the BKL, so the hook does too; its wakes are safe at both (verified). For a +single-threaded process the behavior is *externally indistinguishable* from +today — the order of notify vs. destroy inverts, but both sit inside one lock +hold, so no other core can observe the space destroyed but the notification +unposted, or vice versa. That sentence is the correctness argument; it is also the +first invariant to re-examine if the BKL is ever split, along with +`killGroupLocked`'s single-hold atomicity. (Hand-built spaces that were never +retained take the immediate-destroy path and are out of the hook's scope.) + +Workers' `exit_subscribers` broadcasts still fire per task — the FAT server's +dead-client sweep is keyed by tid and needs those. + +**Signals.** `signal_bind` is per-task and the service harness binds on the main +thread, so signals address the leader in practice; that stays. During a group +death, `process_signal` may return `0` (accepted by a condemned member — never +delivered, every delivery point kills first) or `-ESRCH` (member already reaped); +init's stop sequence already tolerates both, and its timer escalation to +`process_kill` covers the gap. `process_signal` follows `process_kill`'s +leader re-key for consistency. + +## The shared-memory frame hazard + +`dropSharedMemoryReference` frees a region's physical frames when the last *handle* +reference drops, but mappings die only with the address space. If the last handle +lived in a torn-down member while any task still has the region mapped, that task +holds a live mapping onto freed frames — and the red team showed this is **not** +group-specific: a plain `thread_exit` of the handle-holding thread, or a last-ref +drop by a task *outside* the dying group during the condemned window, hits the same +use-after-free. + +So the fix is a property of the **object**, not the dropper: give +`SharedMemoryObject` a per-*mapping* reference — `shared_memory_map` (and create's +self-map) retains; each space's destruction releases. "Last reference" then means +*no handles and no mappings*, both hazard paths collapse into the existing +refcount, and no group-kill special case is needed at all. + +## Deliberately unchanged + +- Worker `thread_exit`: per-thread, full per-tid resource sweep, refcount drop. +- The condemned-but-running window: a member on another core can finish its + in-flight syscall and run user code for up to a tick before dying — identical to + today's single-task `process_kill` semantics ("prompt but asynchronous, like a + Unix signal"). A kill IPI would shrink it; it is not part of this plan. +- `thread_join` returns 0 for a killed thread; joiners inside a dying group are + woken and then reaped like any member. +- The `.reaping` state, reap lists, and stack reaper. + +## Accepted limits (documented, not fixed here) + +- **Notify-ring overflow**: a group death posts one subscriber badge per member + into 8-slot rings; a >8-member group can drop badges. Group size is bounded by + the 48-task table; today's largest production group is 2 (display) and + thread-test already reaches 5. +- **Exit-record ring pressure**: one 64-entry ring, one record per member — made + harmless for the supervisor by the hook's group-death re-stamp. +- **Enumerate shows a partial group** mid-death: reaped members vanish at once, + condemned members linger up to a tick (audited: no in-tree consumer + misbehaves; the `leader` field in `ProcessDescriptor` lets future consumers + group correctly). +- **Pre-existing reap race, not widened**: a task preempted *mid-syscall* is + `.ready` with `in_system_call = true`, and the tick's reap loop will reap it — + an existing hazard the fan-out inherits but must not add new instances of. + Filed to investigate separately. +- **Per-task DMA/shm cursors** (`dma_map_next`, `shared_memory_map_next`): two + sibling threads allocating overlap the same arena — pre-existing thread bug, + adjacent to but not part of this plan (the mmap/MMIO cursors already moved + per-space for exactly this reason). + +## Milestones + +- **M1 — the leader id.** `Task.leader` (kernel tasks: 0, never followed), set on + both spawn paths; `leader` added to `ProcessDescriptor`; `process_kill` and + `process_signal` re-keyed (kernel-task `-ESRCH` guard *before* leader + resolution). No fan-out yet. Existing tests must pass untouched. +- **M2 — the latch + fan-out.** `AddressSpaceRef.dying` + stash; + `retainAddressSpace` refuses dying spaces; `killGroupLocked`; wire the fault + path, `exit`, and `process_kill` into it; leader `thread_exit` → `-EPERM`; + `exit_reason`/`fault_kill_count` writes under the BKL; `kill_pending` made + atomic. Group notification via the `group_exit_hook` re-stamp + post. +- **M3 — shared-memory mapping refs.** `SharedMemoryObject` counts mappings; + space destruction releases them; frames free only at zero handles *and* zero + mappings. +- **M4 — tests + docs.** New `-Dtest-case`s (all `smp: 4` where cross-core + matters), driving `thread-test` with new argv modes: + - `thread-fault-group`: a worker faults; assert both tasks gone from + `enumerate`, `fault_kill_count == 1`, `process_exit_reason(leader) == + segmentation_fault`, address-space and stack-bytes counters return to base. + - `kill-threaded-group`: `process_kill(leader)` with a worker spinning on + another core; assert the worker dies by the deferred path, exactly one exit + badge, delivered only after both members are dead, and + `process_exit_reason(leader) == .killed`. + - `kill-via-worker-tid`: `process_kill(worker)` kills the whole group; + `-EPERM` for a non-supervisor aiming at the worker. + - `racing-triggers`: two members fault/exit simultaneously on different cores; + assert a deterministic leader reason and `fault_kill_count == 1`. + - `exit-group`: a *worker* calls `exit(3)`; group dies, leader reason + `.aborted`. + - `leader-thread-exit`: asserts the chosen rule (`-EPERM`, workers unaffected). + - `thread-exit-solo`: regression — worker `thread_exit` still leaves siblings + running. + - `group-claim-release`: a member claims a device; assert the claim is free and + the leader notification arrives only after every member is dead. + - `shm-mapping-ref`: last handle dropped by a dying thread; sibling's mapping + stays valid until space death (M3 regression). + Then update [threading.md](threading.md) (the shared-fate gap note), + [process-lifecycle.md](process-lifecycle.md), + [process-management.md](process-management.md), and + [ipc.md](ipc.md)/[drivers.md](drivers.md) mentions. From daca0d9216166a158574760410e3ce9a55b4a6f8 Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Wed, 22 Jul 2026 10:17:13 +0100 Subject: [PATCH 2/7] =?UTF-8?q?kernel:=20M1=20shared-fate=20=E2=80=94=20Ta?= =?UTF-8?q?sk.leader=20id,=20kill/signal=20re-keyed=20to=20the=20leader?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Every task carries its process leader's id (main task: own id; threads: copied from the spawner; kernel tasks: 0, never followed). process_kill and process_signal resolve any member id to the leader and authorize against the leader's supervisor, making both capabilities per-process. ProcessDescriptor gains the leader field. No fan-out yet (docs/shared-fate-plan.md M1). --- system/abi.zig | 1 + system/kernel/process.zig | 33 +++++++++++++++++++++++++-------- system/kernel/scheduler.zig | 14 ++++++++++++-- system/kernel/tests.zig | 2 +- 4 files changed, 39 insertions(+), 11 deletions(-) diff --git a/system/abi.zig b/system/abi.zig index d005fbc..18adda2 100644 --- a/system/abi.zig +++ b/system/abi.zig @@ -186,6 +186,7 @@ pub const ProcessState = enum(u32) { pub const ProcessDescriptor = extern struct { id: u32, // kernel-assigned process id; never reused (monotonic) supervisor: u32, // id of the process that spawned it (0 = the kernel) + leader: u32, // process-leader id: == id for a main task, the main task's id for a thread state: u32, // a ProcessState value priority: u32, name_length: u32, diff --git a/system/kernel/process.zig b/system/kernel/process.zig index da1546a..2fb4bba 100644 --- a/system/kernel/process.zig +++ b/system/kernel/process.zig @@ -715,17 +715,17 @@ fn systemThreadSpawn(state: *architecture.CpuState) void { null else ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF); - const tid = spawnThreadSupervised(t.address_space, entry, stack_top, arg, t.priority, t.id, exit_endpoint) orelse return fail(state); + const tid = spawnThreadSupervised(t.address_space, entry, stack_top, arg, t.priority, t.id, exit_endpoint, t.leader) orelse return fail(state); architecture.setSystemCallResult(state, tid); } /// Spawn a thread sharing `address_space`, taking the exit-endpoint reference under the **same** /// lock as the spawn (as `spawnProcessSupervised` does), so the thread cannot die before /// its reference exists. Returns the new thread id, or null on resource exhaustion. -fn spawnThreadSupervised(address_space: u64, entry: u64, stack_top: u64, arg: u64, priority: scheduler.Priority, supervisor: u32, exit_endpoint: ?*ipc.Endpoint) ?u32 { +fn spawnThreadSupervised(address_space: u64, entry: u64, stack_top: u64, arg: u64, priority: scheduler.Priority, supervisor: u32, exit_endpoint: ?*ipc.Endpoint, leader: u32) ?u32 { const flags = sync.enter(); defer sync.leave(flags); - const tid = scheduler.spawnUserLocked(address_space, entry, stack_top, arg, priority, "thread", supervisor, if (exit_endpoint) |e| @ptrCast(e) else null) orelse return null; + const tid = scheduler.spawnUserLocked(address_space, entry, stack_top, arg, priority, "thread", supervisor, if (exit_endpoint) |e| @ptrCast(e) else null, leader) orelse return null; if (exit_endpoint) |endpoint| endpoint.refcount += 1; // the thread holds it birth-to-death return tid; } @@ -972,7 +972,16 @@ pub fn killProcess(caller_id: u32, target_id: u32) i64 { defer sync.leave(flags); const target = scheduler.taskByIdLocked(target_id) orelse return -ipc.ESRCH; if (target.address_space == 0) return -ipc.ESRCH; // kernel tasks are not processes - if (target.supervisor != caller_id) return -ipc.EPERM; + // The kill authority is the supervision link of the process's LEADER, so the + // capability is per-process, aimed at any member id — a thread's own + // `supervisor` (the task that spawned it) grants nothing here + // (docs/shared-fate-plan.md M1). Kernel tasks were -ESRCH'd above, so a + // leader of 0 is unreachable. + const leader = if (target.leader == target.id) + target + else + scheduler.taskByIdLocked(target.leader) orelse return -ipc.ESRCH; + if (leader.supervisor != caller_id) return -ipc.EPERM; target.exit_reason = .killed; if (target.state == .running) { target.kill_pending = true; @@ -1088,9 +1097,17 @@ fn systemProcessSignal(state: *architecture.CpuState) void { if (signal > 31) return failErr(state, ipc.EBADF); // not a Signal bit position const flags = sync.enter(); defer sync.leave(flags); - const target = scheduler.taskByIdLocked(@intCast(id)) orelse return failErr(state, ipc.ESRCH); - if (target.address_space == 0) return failErr(state, ipc.ESRCH); - if (target.supervisor != t.id and target.id != t.id) return failErr(state, ipc.EPERM); + const member = scheduler.taskByIdLocked(@intCast(id)) orelse return failErr(state, ipc.ESRCH); + if (member.address_space == 0) return failErr(state, ipc.ESRCH); + // Signals address the process: any member id resolves to the LEADER, whose + // endpoint the service harness binds. Authority mirrors process_kill (the + // leader's supervisor), plus the group may signal itself + // (docs/shared-fate-plan.md M1). + const target = if (member.leader == member.id) + member + else + scheduler.taskByIdLocked(member.leader) orelse return failErr(state, ipc.ESRCH); + if (target.supervisor != t.id and target.id != t.leader) return failErr(state, ipc.EPERM); target.pending_signals |= @as(u32, 1) << @intCast(signal); if (target.signal_endpoint) |raw| { const endpoint: *ipc.Endpoint = @ptrCast(@alignCast(raw)); @@ -1765,7 +1782,7 @@ pub fn spawnProcessSupervised(image: []const u8, priority: u3, argv: []const []c architecture.mapUserPageInto(address_space, page_virtual, stack_frame, true, false); // RW + NX } - const child = scheduler.spawnUserLocked(address_space, parsed.entry, user_sp, 0, priority, argv[0], supervisor, if (exit_endpoint) |endpoint| @ptrCast(endpoint) else null) orelse + const child = scheduler.spawnUserLocked(address_space, parsed.entry, user_sp, 0, priority, argv[0], supervisor, if (exit_endpoint) |endpoint| @ptrCast(endpoint) else null, 0) orelse return error.OutOfMemory; // The child holds a reference to its exit endpoint from birth to death. Taken // only now, after nothing can fail; the lock is still held, so the child diff --git a/system/kernel/scheduler.zig b/system/kernel/scheduler.zig index 7fef809..738e313 100644 --- a/system/kernel/scheduler.zig +++ b/system/kernel/scheduler.zig @@ -49,6 +49,12 @@ pub const Task = struct { // Id of the process that spawned this one (0 = the kernel). The supervision // link is the kill authority: only the supervisor may process_kill a child. supervisor: u32 = 0, + // Id of this task's process leader — the main task's own id, copied to every + // thread it (transitively) spawns; 0 for kernel tasks and never followed. + // Equal `leader` is what makes two tasks one process; the leader's id is the + // id `system_spawn` returned, so it is the process id the supervisor speaks + // (docs/shared-fate-plan.md). + leader: u32 = 0, // Endpoint to notify when this process ends (any way: exit, fault, kill), or // null. Holds its own reference, dropped when the notification is posted. // Opaque here for the same reason as `handles` below. @@ -460,14 +466,16 @@ pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool { /// in user mode at `entry` on `user_sp`, recorded under `name` (its argv[0]). /// `supervisor` is the id of the spawning process (0 = the kernel) — the kill /// authority — and `exit_endpoint` (an *ipc.Endpoint whose reference the caller -/// has already taken, or null) is notified when this process ends. +/// has already taken, or null) is notified when this process ends. `leader` is +/// the process leader's id for a thread, or 0 to make the new task its own +/// leader (a process spawn). /// It gets a fresh kernel stack for syscalls/interrupts, and its first switch-in /// lands in `user_task_trampoline`. /// Returns the new process id, or null (creating nothing) if the table is full or /// out of memory. /// **Caller must hold the kernel lock** (the loader that builds `address_space` holds it /// across the whole spawn, so the address space and the task appear atomically). -pub fn spawnUserLocked(address_space: u64, entry: u64, user_sp: u64, user_arg: u64, priority: Priority, task_name: []const u8, supervisor: u32, exit_endpoint: ?*anyopaque) ?u32 { +pub fn spawnUserLocked(address_space: u64, entry: u64, user_sp: u64, user_arg: u64, priority: Priority, task_name: []const u8, supervisor: u32, exit_endpoint: ?*anyopaque, leader: u32) ?u32 { const t = freeSlot() orelse return null; const stack = heap.allocator().alloc(u8, stack_size) catch return null; // Take this task's reference to the address space before we commit the slot, so a @@ -488,6 +496,7 @@ pub fn spawnUserLocked(address_space: u64, entry: u64, user_sp: u64, user_arg: u .user_arg = user_arg, .supervisor = supervisor, .exit_endpoint = exit_endpoint, + .leader = if (leader == 0) next_id else leader, }; const name_length = @min(task_name.len, maximum_task_name); @memcpy(t.name_buffer[0..name_length], task_name[0..name_length]); @@ -1035,6 +1044,7 @@ pub fn enumerate(out: []abi.ProcessDescriptor) u64 { d.* = .{ .id = t.id, .supervisor = t.supervisor, + .leader = t.leader, .state = @intFromEnum(@as(abi.ProcessState, switch (t.state) { .ready => .ready, .running => .running, diff --git a/system/kernel/tests.zig b/system/kernel/tests.zig index 50e0648..476ff03 100644 --- a/system/kernel/tests.zig +++ b/system/kernel/tests.zig @@ -1430,7 +1430,7 @@ fn spawnFaultingProcess() ?u32 { architecture.mapUserPageInto(address_space, process.stack_base_virtual, stack_frame, true, false); // RW + NX // Supervised by the calling test task, so exitReasonOf can read the verdict. - const id = scheduler.spawnUserLocked(address_space, process.code_virtual, process.stack_base_virtual + abi.page_size, 0, 4, "fault-probe", scheduler.currentId(), null) orelse { + const id = scheduler.spawnUserLocked(address_space, process.code_virtual, process.stack_base_virtual + abi.page_size, 0, 4, "fault-probe", scheduler.currentId(), null, 0) orelse { architecture.destroyAddressSpace(address_space); return null; }; From b09a62bc360148393709fc60ccaf5d6c5d3c5d8a Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Wed, 22 Jul 2026 10:23:41 +0100 Subject: [PATCH 3/7] =?UTF-8?q?kernel:=20M2=20shared-fate=20=E2=80=94=20gr?= =?UTF-8?q?oup=20fan-out,=20dying=20latch,=20deferred=20leader=20notificat?= =?UTF-8?q?ion?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit All process deaths (exit from any thread, ring-3 fault, process_kill) now kill the whole thread group via killGroupLocked: latch the AddressSpaceRef as dying (refusing new members, closing the thread_spawn escape), stamp every member (leader carries the group reason — the record the supervisor reads), reap parked members to fixpoint, condemn running ones. The leader's exit notification and subscriber broadcast move to the group-death moment — the last address-space reference drop — via scheduler.group_exit_hook, which re-stamps the leader's exit record first. Leader thread_exit is refused with -EPERM. kill_pending is atomic; exit_reason and fault_kill_count writes moved under the big kernel lock. (docs/shared-fate-plan.md M2) --- system/kernel/process.zig | 188 +++++++++++++++++++++++++++++------- system/kernel/scheduler.zig | 95 ++++++++++++++++-- 2 files changed, 240 insertions(+), 43 deletions(-) diff --git a/system/kernel/process.zig b/system/kernel/process.zig index 2fb4bba..314e029 100644 --- a/system/kernel/process.zig +++ b/system/kernel/process.zig @@ -167,6 +167,7 @@ pub fn init() void { scheduler.terminate_current_hook = terminateCurrentLocked; scheduler.reap_task_hook = reapTaskLocked; scheduler.timer_tick_hook = timerSweepLocked; + scheduler.group_exit_hook = groupExitLocked; } /// Return -1 (as an unsigned bit pattern) in the system_call result register. @@ -178,9 +179,9 @@ fn system_call(state: *architecture.CpuState) void { const t = scheduler.current(); const user = t.address_space != 0; if (user) { - // A condemned process (process_kill caught it running) dies at its next + // A condemned process (a kill caught it running) dies at its next // kernel entry — before it can spawn, claim, or message anything else. - if (t.kill_pending) terminateCurrent(); + if (t.kill_pending.load(.monotonic)) terminateCurrent(); // Mark the span of this call so the timer tick never tears the task down // in the middle of a kernel operation (scheduler.reapKillPendingLocked). t.in_system_call = true; @@ -191,14 +192,15 @@ fn system_call(state: *architecture.CpuState) void { switch (@as(SystemCall, @enumFromInt(architecture.systemCallNumber(state)))) { .exit => { exit_code = architecture.systemCallArg(state, 0); - // A scheduled process tears down fully (terminateCurrent); a borrowed - // test thread unwinds back to the kernel that entered it. A NONZERO - // code is a deliberate failure exit (`.aborted`): "the work exists - // but I could not do it" — supervisors restart those, unlike a clean - // `.exited` ("nothing for me here"), which they let lie. + // A scheduled process tears down fully — the WHOLE process: exit from + // any thread is group death, the exit_group lesson (docs/shared-fate- + // plan.md). A borrowed test thread unwinds back to the kernel that + // entered it. A NONZERO code is a deliberate failure exit + // (`.aborted`): "the work exists but I could not do it" — supervisors + // restart those, unlike a clean `.exited` ("nothing for me here"), + // which they let lie. if (scheduler.currentIsUserProcess()) { - scheduler.current().exit_reason = if (exit_code == 0) .exited else .aborted; - terminateCurrent(); + exitGroupCurrent(if (exit_code == 0) .exited else .aborted); } else architecture.userExit(); }, .yield => { @@ -256,12 +258,17 @@ fn system_call(state: *architecture.CpuState) void { .futex_wait => systemFutexWait(state), .futex_wake => systemFutexWake(state), .thread_exit => { - // A thread ends like a process exit(0), but only this task: its - // resources are released and its address-space reference dropped (the - // space survives while sibling threads hold it). docs/threading.md. + // A WORKER thread ends like a process exit(0), but only this task: + // its resources are released and its address-space reference dropped + // (the space survives while sibling threads hold it). The LEADER may + // not thread_exit — the group ends only through exit, a fault, or + // process_kill (docs/shared-fate-plan.md, decided at sign-off). if (scheduler.currentIsUserProcess()) { - scheduler.current().exit_reason = .exited; - terminateCurrent(); + const dying = scheduler.current(); + if (dying.id == dying.leader) return failErr(state, ipc.EPERM); + _ = sync.enter(); // handed off through the exit switch + dying.exit_reason = .exited; + terminateCurrentLocked(); } else architecture.userExit(); }, _ => fail(state), @@ -916,9 +923,17 @@ fn releaseTaskResourcesLocked(t: *scheduler.Task) void { ipc.closeHandles(t); // Publish the exit to every subscriber (docs/process-lifecycle.md): the same // badge encoding as the supervisor's notification, and equally late, so a - // subscriber also observes a fully-released child. - for (&exit_subscribers) |*slot| { - if (slot.*) |subscriber| ipc.notifyLocked(subscriber.endpoint, abi.notify_exit_bit | t.id); + // subscriber also observes a fully-released child. One exception: a dying + // group's LEADER defers its publication to the group_exit_hook — the group + // is only "fully released" when its LAST member is gone (docs/shared-fate- + // plan.md). Workers still publish per-task: subscribers like the FAT server + // sweep per-tid client state and need every id. + const defer_to_group_hook = t.address_space != 0 and t.id == t.leader and + scheduler.groupDyingLocked(t.address_space); + if (!defer_to_group_hook) { + for (&exit_subscribers) |*slot| { + if (slot.*) |subscriber| ipc.notifyLocked(subscriber.endpoint, abi.notify_exit_bit | t.id); + } } if (t.exit_endpoint) |raw| { const endpoint: *ipc.Endpoint = @ptrCast(@alignCast(raw)); @@ -957,16 +972,19 @@ fn reapTaskLocked(t: *scheduler.Task) void { scheduler.destroyTaskLocked(t); } -/// Kill process `target_id` on behalf of `caller_id` — the kernel half of the -/// process_kill system call. Returns 0, -ESRCH (no such live process — kernel -/// tasks are not killable processes and stale ids miss, since ids are never -/// reused), or -EPERM (the caller is not the target's supervisor). +/// Kill the process containing `target_id` on behalf of `caller_id` — the kernel +/// half of the process_kill system call, and a WHOLE-GROUP kill: any member id +/// resolves to the leader, and every thread dies (docs/shared-fate-plan.md). +/// Returns 0, -ESRCH (no such live process — kernel tasks are not killable +/// processes and stale ids miss, since ids are never reused), or -EPERM (the +/// caller is not the leader's supervisor). /// -/// A target that is ready or blocked is reaped on the spot. One that is running -/// on another core cannot be torn down mid-instruction, so it is condemned -/// (`kill_pending`) and dies at its next system_call entry, block, or timer tick -/// — like a Unix signal, delivery is prompt but asynchronous. Either way the -/// call returns 0: the kill is accepted and irrevocable. +/// Members that are ready or blocked are reaped on the spot. Ones running on +/// another core cannot be torn down mid-instruction, so they are condemned +/// (`kill_pending`) and die at their next delivery point — like a Unix signal, +/// delivery is prompt but asynchronous. Either way the call returns 0: the kill +/// is accepted and irrevocable, and the supervisor's notification arrives only +/// once the last member is gone. pub fn killProcess(caller_id: u32, target_id: u32) i64 { const flags = sync.enter(); defer sync.leave(flags); @@ -982,15 +1000,77 @@ pub fn killProcess(caller_id: u32, target_id: u32) i64 { else scheduler.taskByIdLocked(target.leader) orelse return -ipc.ESRCH; if (leader.supervisor != caller_id) return -ipc.EPERM; - target.exit_reason = .killed; - if (target.state == .running) { - target.kill_pending = true; - } else { - reapTaskLocked(target); - } + // Whole-group kill (docs/shared-fate-plan.md). Already dying → the kill is + // already true: accepted and irrevocable either way, return 0. The caller is + // never a member (the leader's supervisor predates the group and cannot be + // inside it), so this always returns. + if (scheduler.groupDyingLocked(target.address_space)) return 0; + killGroupLocked(leader, .killed, null); return 0; } +/// Kill every member of `leader_task`'s group (docs/shared-fate-plan.md): latch +/// the address space as dying — closing the door to new members and moving the +/// leader's exit-endpoint reference into the latch's stash — stamp every +/// member's exit reason, reap the parked ones, and condemn the running ones +/// (there is no kill IPI: a condemned member dies at its next syscall entry, at +/// its own core's next tick in user mode, or — once parked — at any core's tick +/// reap). `trigger` is the current task when the kill came from inside (exit, +/// fault); it dies last and the call never returns. A supervisor's kill passes +/// null and returns. Caller holds the kernel lock and has already checked +/// `groupDyingLocked` — a second trigger must not re-stamp. +fn killGroupLocked(leader_task: *scheduler.Task, reason: abi.ExitReason, trigger: ?*scheduler.Task) void { + const leader_id = leader_task.id; + if (scheduler.markGroupDyingLocked(leader_task.address_space, leader_id, leader_task.supervisor, reason, leader_task.exit_endpoint)) { + // The stash now owns the leader's endpoint reference; the leader's own + // teardown sees null (no early notification, no double drop) and the + // group_exit_hook posts exactly once, at space destruction. + leader_task.exit_endpoint = null; + } + // Stamp before any teardown — recordExitLocked snapshots exit_reason as its + // first act. The LEADER carries the group reason: its record is the one the + // supervisor can read. The trigger keeps it too (its own record tells the + // truth); every other member died because the group died: .killed. + for (scheduler.allTasksLocked()) |*member| { + if (member.leader != leader_id) continue; + if (member.state == .free or member.state == .reaping) continue; + member.exit_reason = if (member.id == leader_id or member == trigger) reason else .killed; + member.kill_pending.store(true, .monotonic); + } + // Reap parked members, to fixpoint: one member's teardown can wake another + // (an -EPEER'd client, a joiner), flipping it .blocked -> .ready behind the + // scan. Terminates in at most one pass per member: the scrubs in + // releaseTaskResourcesLocked (abandonSenderLocked, removeFromWaitQueueLocked, + // forgetIpcClientLocked, killOwnedEndpointsLocked) run before destroy, so no + // wake path holds a pointer to a reaped member. + var progress = true; + while (progress) { + progress = false; + for (scheduler.allTasksLocked()) |*member| { + if (member.leader != leader_id) continue; + if (member.state != .ready and member.state != .blocked) continue; + reapTaskLocked(member); + progress = true; + } + } + // Members running on other cores stay condemned; the in-group trigger dies + // now — last, because terminateCurrentLocked switches away for good. + if (trigger != null) terminateCurrentLocked(); +} + +/// The group half of `exit`: end the CURRENT task's whole process with `reason`. +/// Takes the lock (handed off through the exit switch, like terminateCurrent) +/// and never returns. A second trigger — the group is already dying — keeps the +/// first trigger's stamp and just dies. +fn exitGroupCurrent(reason: abi.ExitReason) noreturn { + const t = scheduler.current(); + _ = sync.enter(); + if (scheduler.groupDyingLocked(t.address_space)) terminateCurrentLocked(); + const leader_task = if (t.leader == t.id) t else scheduler.taskByIdLocked(t.leader) orelse t; + killGroupLocked(leader_task, reason, t); + unreachable; // killGroupLocked never returns for an in-group trigger +} + /// Kill the current user process in response to a CPU fault it raised in ring 3. /// The fault is confined to the process — the kernel trapped it on the task's own /// kernel stack and is intact — so everything the process held is reclaimed and the @@ -998,9 +1078,16 @@ pub fn killProcess(caller_id: u32, target_id: u32) i64 { /// (docs/resilience.md: fault -> kill -> continue). `reason` is the fault class /// (from the vector), recorded for the supervisor's `process_exit_reason`. pub fn killCurrentProcess(reason: abi.ExitReason) noreturn { - scheduler.current().exit_reason = reason; + const t = scheduler.current(); + _ = sync.enter(); // handed off through the exit switch, released by the resumed task + // A second member faulting while the group already dies: no re-stamp, no + // second count bump — the first trigger owns the group's story + // (docs/shared-fate-plan.md). `fault_kill_count` is per faulting GROUP. + if (scheduler.groupDyingLocked(t.address_space)) terminateCurrentLocked(); fault_kill_count += 1; - terminateCurrent(); + const leader_task = if (t.leader == t.id) t else scheduler.taskByIdLocked(t.leader) orelse t; + killGroupLocked(leader_task, reason, t); + unreachable; // killGroupLocked never returns for an in-group trigger } /// The bounded record of recent deaths, for `process_exit_reason`: ids are never @@ -1020,6 +1107,39 @@ fn recordExitLocked(t: *scheduler.Task) void { exit_record_next = (exit_record_next + 1) % exit_record_capacity; } +/// The group-death moment (scheduler.group_exit_hook): the last member is gone +/// and the address space destroyed. Re-stamp the leader's exit record with the +/// group reason — the record must be present and current when the notification +/// lands, however many member deaths churned the ring in between — then publish +/// the leader's exit: the subscriber broadcast and the supervisor's endpoint +/// notification, whose stashed reference is dropped here, exactly once. Runs +/// under the big kernel lock, at both release sites (docs/shared-fate-plan.md). +fn groupExitLocked(leader: u32, supervisor: u32, reason: abi.ExitReason, exit_endpoint: ?*anyopaque) void { + restampExitRecordLocked(leader, supervisor, reason); + for (&exit_subscribers) |*slot| { + if (slot.*) |subscriber| ipc.notifyLocked(subscriber.endpoint, abi.notify_exit_bit | leader); + } + if (exit_endpoint) |raw| { + const endpoint: *ipc.Endpoint = @ptrCast(@alignCast(raw)); + ipc.notifyLocked(endpoint, abi.notify_exit_bit | leader); + ipc.dropRef(endpoint); + } +} + +/// Overwrite dead task `id`'s exit record with the group reason, or re-append it +/// if the group's death burst already evicted it — a supervisor must always be +/// able to read the reason for a notification it just received. Lock held. +fn restampExitRecordLocked(id: u32, supervisor: u32, reason: abi.ExitReason) void { + for (&exit_records) |*record| { + if (record.valid and record.id == id) { + record.reason = reason; + return; + } + } + exit_records[exit_record_next] = .{ .id = id, .supervisor = supervisor, .reason = reason, .valid = true }; + exit_record_next = (exit_record_next + 1) % exit_record_capacity; +} + /// How dead process `id` ended, for `caller` — the kernel half of the /// process_exit_reason system call. Returns the ExitReason value, -ESRCH (never /// lived, still alive, or evicted from the ring), or -EPERM (the caller was not diff --git a/system/kernel/scheduler.zig b/system/kernel/scheduler.zig index 738e313..313e5aa 100644 --- a/system/kernel/scheduler.zig +++ b/system/kernel/scheduler.zig @@ -70,9 +70,12 @@ pub const Task = struct { // (docs/process-lifecycle.md). Bits are abi.Signal values. Signals pend here // until an endpoint is bound; two pending terminates are one terminate. pending_signals: u32 = 0, - // Set by process_kill on a task that is running on another core; the kernel - // finishes the kill at that task's next system call or timer tick. - kill_pending: bool = false, + // Set by process_kill (or a group fan-out) on a task that is running on + // another core; the kernel finishes the kill at that task's next system call + // or timer tick. Atomic because the syscall-entry check reads it without the + // lock while another core writes it under the lock — monotonic is enough: a + // missed read is caught at the next delivery point (x86-TSO or not). + kill_pending: std.atomic.Value(bool) = .init(false), // True while this task executes its own system call — the timer tick must not // tear a task down in the middle of a kernel operation, only while it runs // user code (or sits at a block point, where teardown is safe). @@ -165,7 +168,24 @@ var tasks = [_]Task{.{}} ** maximum_tasks; // One live entry per address space; threads sharing an address space share this entry, // so their mmap/mmio grants bump one cursor and never overlap (docs/threading-plan.md M7). // `mmap_next`/`device_map_next` are 0 until process.zig seeds them to the arena base. -const AddressSpaceRef = struct { root: u64 = 0, count: u32 = 0, mmap_next: u64 = 0, device_map_next: u64 = 0 }; +const AddressSpaceRef = struct { + root: u64 = 0, + count: u32 = 0, + mmap_next: u64 = 0, + device_map_next: u64 = 0, + // Group-death state (docs/shared-fate-plan.md), set once by the first kill + // trigger and never cleared while the entry lives. `dying` gates + // retainAddressSpace — no new member may join a dying group (closing the + // thread_spawn escape) — and the stash is what `group_exit_hook` posts when + // the last reference drops: the leader's identity, the group reason, and the + // leader's counted exit-endpoint reference (moved off its Task under the + // same lock hold that set the latch). + dying: bool = false, + group_leader: u32 = 0, + group_supervisor: u32 = 0, + group_reason: abi.ExitReason = .exited, + group_exit_endpoint: ?*anyopaque = null, +}; var address_space_refs = [_]AddressSpaceRef{.{}} ** maximum_tasks; var address_space_destroy_count: u64 = 0; @@ -213,6 +233,11 @@ fn retainAddressSpace(root: u64) bool { var free: ?*AddressSpaceRef = null; for (&address_space_refs) |*entry| { if (entry.count != 0 and entry.root == root) { + // A dying group admits no new member: a thread_spawn that was already + // past its syscall-entry kill check when the group was condemned + // fails here, under the same lock that would have created the task + // (docs/shared-fate-plan.md). + if (entry.dying) return false; entry.count += 1; return true; } @@ -232,16 +257,68 @@ fn releaseAddressSpace(root: u64) void { if (entry.count == 0 or entry.root != root) continue; entry.count -= 1; if (entry.count == 0) { - entry.root = 0; + // Copy the group-death stash out, then fully reset the slot before + // the hook runs — a stale stash must never survive into an unrelated + // process's reused entry (docs/shared-fate-plan.md). + const was_dying = entry.dying; + const leader = entry.group_leader; + const supervisor = entry.group_supervisor; + const reason = entry.group_reason; + const endpoint = entry.group_exit_endpoint; + entry.* = .{}; architecture.destroyAddressSpace(root); address_space_destroy_count += 1; + // The group-death moment: the space is gone, every member is dead. + // process.zig posts the leader's deferred exit publication here. + if (was_dying) if (group_exit_hook) |hook| hook(leader, supervisor, reason, endpoint); } return; } + // Never retained (a hand-built test space): destroyed directly, out of the + // group-death hook's scope, preserving the pre-refcount behaviour. architecture.destroyAddressSpace(root); address_space_destroy_count += 1; } +/// Called (lock held) at the group-death moment — the last reference to a dying +/// group's address space dropped and the space was destroyed. Registered by +/// process.zig, which posts the leader's deferred exit notification and re-stamps +/// its exit record (docs/shared-fate-plan.md). Runs under the lock at both +/// release sites (`exitUserLocked`, `destroyTaskLocked`). +pub var group_exit_hook: ?*const fn (leader: u32, supervisor: u32, reason: abi.ExitReason, exit_endpoint: ?*anyopaque) void = null; + +/// Latch `root`'s group as dying and stash the group-death payload for the hook. +/// Returns false — changing nothing — if the group is already dying (a concurrent +/// trigger lost the race) or `root` has no live entry. Caller holds the lock. +pub fn markGroupDyingLocked(root: u64, leader: u32, supervisor: u32, reason: abi.ExitReason, exit_endpoint: ?*anyopaque) bool { + for (&address_space_refs) |*entry| { + if (entry.count == 0 or entry.root != root) continue; + if (entry.dying) return false; + entry.dying = true; + entry.group_leader = leader; + entry.group_supervisor = supervisor; + entry.group_reason = reason; + entry.group_exit_endpoint = exit_endpoint; + return true; + } + return false; +} + +/// Whether `root`'s group is already dying. Caller holds the lock. +pub fn groupDyingLocked(root: u64) bool { + for (&address_space_refs) |*entry| { + if (entry.count != 0 and entry.root == root) return entry.dying; + } + return false; +} + +/// The whole static task pool, for process.zig's group fan-out — which must scan +/// members under the lock it already holds. Slots may be `.free`/`.reaping`; +/// callers filter by state and must not hold pointers past the lock. +pub fn allTasksLocked() []Task { + return tasks[0..]; +} + /// Test-observable: how many address spaces are live (entries with a nonzero count). pub fn liveAddressSpaceCount() u32 { var live: u32 = 0; @@ -916,12 +993,12 @@ fn reapKillPendingLocked() void { // task_trampoline, not switchTo's tail), its stack is still queued here. The dying // task switched away before this tick, so it is off its stack — drain now (M8). drainReapListLocked(pc); - if (cur.kill_pending and cur.address_space != 0 and !cur.in_system_call) { + if (cur.kill_pending.load(.monotonic) and cur.address_space != 0 and !cur.in_system_call) { if (terminate_current_hook) |hook| hook(); // noreturn } if (reap_task_hook) |hook| { for (&tasks) |*t| { - if (!t.kill_pending) continue; + if (!t.kill_pending.load(.monotonic)) continue; if (t.state == .ready or t.state == .blocked) hook(t); } } @@ -997,7 +1074,7 @@ pub fn exitUserLocked() noreturn { dying.state = .reaping; // dead but its slot stays reserved until the stack is freed wakeJoinersLocked(dying.id); // let any thread_join(dying.id) return (M9) dying.address_space = 0; - dying.kill_pending = false; + dying.kill_pending.store(false, .monotonic); dying.in_system_call = false; // Queue for reaping: the task we switch to (or the next tick) frees this stack (M8/M9). dying.next = pc.reap_list; @@ -1020,7 +1097,7 @@ pub fn destroyTaskLocked(t: *Task) void { if (t.address_space != 0) releaseAddressSpace(t.address_space); // destroys only on the last reference reapStackLocked(t); // safe to free now: `t` is not running on any core (M8) t.address_space = 0; - t.kill_pending = false; + t.kill_pending.store(false, .monotonic); t.in_system_call = false; t.wake_at = 0; t.state = .free; From 08e139ebbade4f17b6d8ff14eb688a45d749fe71 Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Wed, 22 Jul 2026 10:34:03 +0100 Subject: [PATCH 4/7] =?UTF-8?q?kernel:=20M3=20shared-fate=20=E2=80=94=20sh?= =?UTF-8?q?ared-memory=20frames=20live=20while=20any=20mapping=20does?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Each address space that maps a shared-memory region now holds its own reference, recorded on the AddressSpaceRef and dropped when the space is destroyed — so 'last reference' means no handles AND no mappings, and a region's frames can no longer be freed out from under a sibling thread (or any other live mapper) when the handle-holding task dies. The group-death notification still posts after every mapping release. (docs/shared-fate-plan.md M3) --- system/kernel/ipc-synchronous.zig | 15 ++++++++--- system/kernel/process.zig | 39 ++++++++++++++++++++++++++- system/kernel/scheduler.zig | 45 ++++++++++++++++++++++++++++++- 3 files changed, 94 insertions(+), 5 deletions(-) diff --git a/system/kernel/ipc-synchronous.zig b/system/kernel/ipc-synchronous.zig index dfbfab8..b6f3a27 100644 --- a/system/kernel/ipc-synchronous.zig +++ b/system/kernel/ipc-synchronous.zig @@ -173,9 +173,18 @@ pub fn createSharedMemory(phys: u64, pages: usize) ?*SharedMemoryObject { return shared_memory; } -/// Drop a shared-memory reference; when the last one goes, return its frames to the -/// allocator and free the object. (The mappings themselves are torn down with each -/// sharer's address space; `device_grant` keeps that from freeing the frames early.) +/// Take a shared-memory reference — a MAPPING's reference (docs/shared-fate-plan.md +/// M3): each address space that maps the region holds one, recorded on the space +/// and dropped at its destruction. Caller holds the big kernel lock. +pub fn retainSharedMemory(shared_memory: *SharedMemoryObject) void { + shared_memory.refcount += 1; +} + +/// Drop a shared-memory reference; when the last one goes — no handles AND no +/// mappings left — return its frames to the allocator and free the object. +/// (The mappings themselves are torn down with each sharer's address space; +/// `device_grant` keeps that sweep from freeing the frames, and the space's +/// recorded mapping references keep this drop from freeing them early.) pub fn dropSharedMemoryReference(shared_memory: *SharedMemoryObject) void { if (shared_memory.refcount > 1) { shared_memory.refcount -= 1; diff --git a/system/kernel/process.zig b/system/kernel/process.zig index 314e029..30e6bd6 100644 --- a/system/kernel/process.zig +++ b/system/kernel/process.zig @@ -168,6 +168,7 @@ pub fn init() void { scheduler.reap_task_hook = reapTaskLocked; scheduler.timer_tick_hook = timerSweepLocked; scheduler.group_exit_hook = groupExitLocked; + scheduler.space_mapping_release_hook = dropSpaceMappingHook; } /// Return -1 (as an unsigned bit pattern) in the system_call result register. @@ -570,9 +571,26 @@ fn systemSharedMemoryCreate(state: *architecture.CpuState) void { for (0..pages) |i| pmm.free(phys + i * page_size); return fail(state); }; + // The creator's mapping holds its own reference, recorded on the space + // (docs/shared-fate-plan.md M3): frames must outlive every MAPPING, not just + // every handle — a sibling thread keeps using the region after the + // handle-holding thread dies. + { + const flags = sync.enter(); + if (!scheduler.recordSpaceMappingLocked(t.address_space, @ptrCast(shared_memory))) { + sync.leave(flags); + ipc.dropSharedMemoryReference(shared_memory); // last ref: frees the object AND its frames + return fail(state); + } + ipc.retainSharedMemory(shared_memory); + sync.leave(flags); + } const handle = ipc.installSharedMemoryHandle(t, shared_memory); if (handle < 0) { - ipc.dropSharedMemoryReference(shared_memory); // last ref: frees the object and its frames + // The mapping record keeps one reference; drop only the creator's. The + // object (and frames) now live until this space is destroyed — the + // region was never named, so nothing else can reach it. + ipc.dropSharedMemoryReference(shared_memory); return fail(state); } @@ -597,6 +615,18 @@ fn systemSharedMemoryMap(state: *architecture.CpuState) void { const size = shared_memory.pages * page_size; if (base_v + size > shared_memory_arena_end) return fail(state); + // This mapping holds its own reference, recorded on the space and dropped at + // its destruction (docs/shared-fate-plan.md M3) — the handle's reference is + // separate and may be closed while the mapping lives on. + { + const flags = sync.enter(); + if (!scheduler.recordSpaceMappingLocked(t.address_space, @ptrCast(shared_memory))) { + sync.leave(flags); + return fail(state); // mapping table full: refuse rather than map unrecorded + } + ipc.retainSharedMemory(shared_memory); + sync.leave(flags); + } architecture.mapUserSharedInto(t.address_space, base_v, shared_memory.phys, size); t.shared_memory_map_next = base_v + size; architecture.setSystemCallResult(state, base_v); @@ -1126,6 +1156,13 @@ fn groupExitLocked(leader: u32, supervisor: u32, reason: abi.ExitReason, exit_en } } +/// scheduler.space_mapping_release_hook: drop one shared-memory MAPPING +/// reference when the space that held it is destroyed (docs/shared-fate-plan.md +/// M3). Lock held by the release path. +fn dropSpaceMappingHook(object: *anyopaque) void { + ipc.dropSharedMemoryReference(@ptrCast(@alignCast(object))); +} + /// Overwrite dead task `id`'s exit record with the group reason, or re-append it /// if the group's death burst already evicted it — a supervisor must always be /// able to read the reason for a notification it just received. Lock held. diff --git a/system/kernel/scheduler.zig b/system/kernel/scheduler.zig index 313e5aa..34dcef8 100644 --- a/system/kernel/scheduler.zig +++ b/system/kernel/scheduler.zig @@ -185,7 +185,18 @@ const AddressSpaceRef = struct { group_supervisor: u32 = 0, group_reason: abi.ExitReason = .exited, group_exit_endpoint: ?*anyopaque = null, + // Shared-memory objects mapped into this space (docs/shared-fate-plan.md M3). + // Each mapping holds one reference to its object, dropped through + // `space_mapping_release_hook` when the space is destroyed — so "last + // reference" means no handles AND no mappings, and frames can never be freed + // while a live space still maps them. Opaque: the object type is the IPC + // layer's. + mappings: [maximum_space_mappings]?*anyopaque = .{null} ** maximum_space_mappings, }; + +/// Shared-memory mappings one address space can hold — matches the per-task +/// handle table's order of magnitude; `shared_memory_map` fails when full. +const maximum_space_mappings = 16; var address_space_refs = [_]AddressSpaceRef{.{}} ** maximum_tasks; var address_space_destroy_count: u64 = 0; @@ -265,11 +276,19 @@ fn releaseAddressSpace(root: u64) void { const supervisor = entry.group_supervisor; const reason = entry.group_reason; const endpoint = entry.group_exit_endpoint; + const mappings = entry.mappings; entry.* = .{}; architecture.destroyAddressSpace(root); address_space_destroy_count += 1; + // Release the mapping references now that no mapping exists — + // `device_grant`-tagged leaves kept destroyAddressSpace's sweep off + // the frames, so this drop is what may actually free them (M3). + if (space_mapping_release_hook) |release| { + for (mappings) |slot| if (slot) |object| release(object); + } // The group-death moment: the space is gone, every member is dead. - // process.zig posts the leader's deferred exit publication here. + // process.zig posts the leader's deferred exit publication here — + // last, so the supervisor's notification postdates every release. if (was_dying) if (group_exit_hook) |hook| hook(leader, supervisor, reason, endpoint); } return; @@ -304,6 +323,30 @@ pub fn markGroupDyingLocked(root: u64, leader: u32, supervisor: u32, reason: abi return false; } +/// Called (lock held) once per recorded shared-memory mapping when an address +/// space is destroyed — drops the mapping's object reference. Registered by +/// process.zig (the object type lives in the IPC layer). +pub var space_mapping_release_hook: ?*const fn (*anyopaque) void = null; + +/// Record a shared-memory mapping on `root`'s space; its reference is dropped +/// via `space_mapping_release_hook` at space destruction. Returns false — +/// recording nothing — if the space has no live entry or its mapping table is +/// full. On true, the caller has transferred one object reference to the space. +/// Caller holds the lock. +pub fn recordSpaceMappingLocked(root: u64, object: *anyopaque) bool { + for (&address_space_refs) |*entry| { + if (entry.count == 0 or entry.root != root) continue; + for (&entry.mappings) |*slot| { + if (slot.* == null) { + slot.* = object; + return true; + } + } + return false; + } + return false; +} + /// Whether `root`'s group is already dying. Caller holds the lock. pub fn groupDyingLocked(root: u64) bool { for (&address_space_refs) |*entry| { From 1882161cb437accaa0d0884f0bed18ba8e29419c Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Wed, 22 Jul 2026 10:37:55 +0100 Subject: [PATCH 5/7] =?UTF-8?q?docs:=20system-image.md=20=E2=80=94=20the?= =?UTF-8?q?=20boot=20capsule=20documented=20in=20full?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit system.img was only mentioned in passing (efi.md, system-requirements.md); its format, builder, fallback chain, and kernel-side life were spread across pack-system-image.py, build.zig, efi.zig, initial-ramdisk.zig, and vfs.zig. New dedicated page covers all of it, cross-linked from every prior mention and added to the docs index. --- docs/README.md | 49 ++++++++------ docs/efi.md | 6 +- docs/system-image.md | 128 ++++++++++++++++++++++++++++++++++++ docs/system-requirements.md | 3 +- 4 files changed, 162 insertions(+), 24 deletions(-) create mode 100644 docs/system-image.md diff --git a/docs/README.md b/docs/README.md index 11d0362..84b0be5 100644 --- a/docs/README.md +++ b/docs/README.md @@ -7,79 +7,86 @@ rather than restate it. Roughly in the order things happen at runtime: runs the bootloader, what the loader gathers before `ExitBootServices`, how it loads the kernel ELF, and the ABI contract for the jump into the kernel. Start here. -2. **[gop.md](gop.md) — the Graphics Output Protocol.** How UEFI exposes graphics +2. **[system-image.md](system-image.md) — system.img, the boot capsule.** The + bundled user binaries packed into one file in the initial-ramdisk wire + format, because one open + one sequential read is the only file I/O shape + firmware is fast at. The trivial container format, the three artifacts one + build list derives (tree, manifest, capsule), the loader's three-strategy + fallback chain, and the capsule's kernel-side life as both the spawn table + and the read-only `/system` mount. +3. **[gop.md](gop.md) — the Graphics Output Protocol.** How UEFI exposes graphics modes (unlike fixed VGA modes), how we detect the monitor's native resolution from EDID and switch to it, and the pixel formats we accept or reject. -3. **[framebuffer.md](framebuffer.md) — the framebuffer.** What the linear +4. **[framebuffer.md](framebuffer.md) — the framebuffer.** What the linear framebuffer the loader hands over actually is, and what **pitch** (stride) means versus width — the detail you have to get right to avoid a skewed image. -4. **[memory-map.md](memory-map.md) — the memory map.** How the loader learns what +5. **[memory-map.md](memory-map.md) — the memory map.** How the loader learns what physical RAM exists and hands it to the kernel in danos's own neutral format, rather than leaking UEFI's memory descriptors across the boundary. -5. **[frame-allocator.md](frame-allocator.md) — the physical frame allocator.** The +6. **[frame-allocator.md](frame-allocator.md) — the physical frame allocator.** The bitmap allocator that hands out and reclaims 4 KiB physical frames from that map — the primitive page tables and the heap are built on. -6. **[interrupts.md](interrupts.md) — interrupts and exceptions.** The GDT, IDT and +7. **[interrupts.md](interrupts.md) — interrupts and exceptions.** The GDT, IDT and TSS, the exception stubs, and the handler that reports a CPU fault in red instead of letting it triple-fault into a silent reset. -7. **[paging.md](paging.md) — the kernel's page tables.** Building our own 4-level +8. **[paging.md](paging.md) — the kernel's page tables.** Building our own 4-level page tables, identity-mapping the low 4 GiB, and switching CR3 off the firmware's tables onto ours. -8. **[device-interrupts.md](device-interrupts.md) — device interrupts.** The Local +9. **[device-interrupts.md](device-interrupts.md) — device interrupts.** The Local APIC and its timer — the kernel's first interrupt that is *handled and returned from*, giving it a heartbeat. -9. **[heap.md](heap.md) — the kernel heap.** A growable free-list allocator built on +10. **[heap.md](heap.md) — the kernel heap.** A growable free-list allocator built on the VMM, exposed as a `std.mem.Allocator` so std containers work — dynamic allocation for the kernel. -10. **[scheduling.md](scheduling.md) — the scheduler.** Fixed-priority preemptive +11. **[scheduling.md](scheduling.md) — the scheduler.** Fixed-priority preemptive multitasking: kernel threads, the context switch, O(1) priority selection, and blocking (sleep, wait queues) — the leap to a running system. -11. **[ipc.md](ipc.md) — inter-process communication.** Bounded blocking +12. **[ipc.md](ipc.md) — inter-process communication.** Bounded blocking message-passing channels, then synchronous call/reply between *processes* over endpoints — the backbone the microkernel's isolated servers talk over. -12. **[syscall.md](syscall.md) — system calls.** How ring 3 asks the kernel for +13. **[syscall.md](syscall.md) — system calls.** How ring 3 asks the kernel for something: the `syscall`/`sysret` fast path, the trap frame, and why the table is deliberately tiny. The numbers are a **private** ABI — [vdso.md](vdso.md) designs the public boundary that will hide them. -13. **[vfs-protocol.md](vfs-protocol.md) — the VFS wire protocol.** The language-neutral +14. **[vfs-protocol.md](vfs-protocol.md) — the VFS wire protocol.** The language-neutral byte-level spec of the file protocol spoken over IPC: request/reply headers, the operation table, mount routing, and the append-only evolution rules — the first IPC protocol documented as public ABI. -14. **[drivers.md](drivers.md) — writing a driver.** The payoff: a driver is an +15. **[drivers.md](drivers.md) — writing a driver.** The payoff: a driver is an ordinary ring-3 process that claims a device, maps its registers, and **sleeps until its hardware interrupts it**. The claim is the capability; `irq_ack` is the unmask. -15. **[driver-model.md](driver-model.md) — buses, classes and host controllers.** How +16. **[driver-model.md](driver-model.md) — buses, classes and host controllers.** How real driver stacks factor into three shapes and how families share code. The three primitives it proposed are long since built (M13 capability passing, M14 DMA + barriers, M15 MSI), and the driver *contract* on top of them — hello, supervision, restart — is built too (device-manager.md, M18). -16. **[usb-hub.md](usb-hub.md) — USB hubs.** Built (M22): why hub topology is handled +17. **[usb-hub.md](usb-hub.md) — USB hubs.** Built (M22): why hub topology is handled *inside* the `usb-xhci-bus` driver rather than a separate hub class driver — a device behind a hub is reached by the **controller**, programmed with a route string in its slot context — plus the compound-hub reality (a USB 3.0 hub is physically two hubs) and detection via the hub's status-change interrupt endpoint. -17. **[process-management.md](process-management.md) — process management.** The +18. **[process-management.md](process-management.md) — process management.** The microkernel's `ps`/`kill`/SIGCHLD: enumerate as a table snapshot, the supervision link as the kill authority, and child-exit notifications over the same endpoints IRQs arrive on. -18. **[process-lifecycle.md](process-lifecycle.md) — the process lifecycle.** Built +19. **[process-lifecycle.md](process-lifecycle.md) — the process lifecycle.** Built (M17): signals over IPC as the one lifecycle vocabulary every process speaks — the POSIX.1-1990 words with message delivery instead of stack hijack, the stable `runtime.process` interface, exit reasons, published exit events any stateful service can subscribe to (the VFS releasing dead clients' handles), and the two iron rules (cleanup is the kernel's job; kill is not a signal). -19. **[device-manager.md](device-manager.md) — the device manager.** Built (M18, +20. **[device-manager.md](device-manager.md) — the device manager.** Built (M18, through the app surface): the tree, the matcher, and the supervisor. Tree structure lives in the manager, authority stays in the kernel; bus drivers report what they see; drivers are restarted through the lifecycle vocabulary — the plan that turns [resilience.md](resilience.md)'s restart goal into increments. -20. **[input.md](input.md) — the input module.** Broadcasting input events (keyboard, +21. **[input.md](input.md) — the input module.** Broadcasting input events (keyboard, mouse, joystick): why a synchronous rendezvous can't fan out to many listeners, the asynchronous `ipc_send` primitive built to fix it, and the per-device subscribe/publish service layered on top. -21. **[display.md](display.md) — the display service.** The display half of the GUI +22. **[display.md](display.md) — the display service.** The display half of the GUI track: a user-space compositor that owns the framebuffer, composes a layer stack into a double buffer, and presents it. Why GOP and the PCI display device are two views of one controller, the device-node + write-combining handoff, and what flicker-free buys @@ -90,7 +97,7 @@ rather than restate it. Roughly in the order things happen at runtime: further out, two research snapshots survey what a *native* driver for real GPU silicon would take as another `.scanout` backend: [nvidia-gpus.md](nvidia-gpus.md) (RTX 3060 / Ampere) and [intel-igpu.md](intel-igpu.md) (Intel iGPU). -22. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and +23. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and how `while (true) hlt` parks the CPU safely once there's nothing left to do. Start with the north star: diff --git a/docs/efi.md b/docs/efi.md index eb3206b..15ce503 100644 --- a/docs/efi.md +++ b/docs/efi.md @@ -27,7 +27,8 @@ The boot volume is **FHS-shaped** (see the repository-layout note in [README.md](README.md)): `build.zig` installs `boot/efi.zig` (built for the `uefi` target) at `EFI/BOOT/BOOTX64.efi` — the one path UEFI firmware fixes — and lays the rest out by FHS path: the kernel at `system/kernel`, init at -`system/services/init`, the pre-packed boot capsule at `boot/system.img`. +`system/services/init`, the pre-packed boot capsule at `boot/system.img` +([system-image.md](system-image.md)). `zig-out` mirrors that tree, but what a machine actually boots is the self-contained FAT32 image `tools/make-fat-image.py` builds from the same files (`danos-usb.img`). The `run-x86-64` step points QEMU at OVMF (UEFI firmware for @@ -67,7 +68,8 @@ services are still up), loads the system binaries into an in-RAM **initial ramdisk** (`loadSystemTree` — normally a single read of the pre-packed `boot\system.img` capsule, which already *is* the ramdisk wire format; it falls back to opening each manifest-listed path, and walks the `/system` tree only as -a last resort for hand-assembled sticks. Best-effort either way — a kernel-only +a last resort for hand-assembled sticks — the capsule's format, builder, and +fallback chain are documented in [system-image.md](system-image.md). Best-effort either way — a kernel-only volume still boots), and builds the **bootstrap page tables** the kernel starts life on (`buildBootstrapTables`), all before the jump: diff --git a/docs/system-image.md b/docs/system-image.md new file mode 100644 index 0000000..ddc7615 --- /dev/null +++ b/docs/system-image.md @@ -0,0 +1,128 @@ +# system.img — the boot capsule + +## What it is + +`boot/system.img` is the **boot capsule**: every bundled user binary — init, the +services, the drivers, the test programs — packed into **one file** on the boot +volume. It is not a filesystem image and it is not compressed; it is exactly the +kernel's **initial-ramdisk wire format** (`system/initial-ramdisk.zig`, format +v2), written to disk ahead of time. The EFI loader reads it in a single +sequential pass and hands the bytes to the kernel unmodified. + +The capsule is a *performance artifact*, not a source of truth. The boot +volume's `/system` file tree remains the canonical layout (see +[danos-file-system-hierarchy-FSH.md](danos-file-system-hierarchy-FSH.md)); +the capsule is a pre-baked snapshot of the same binaries, derived from the same +build graph, so the running system is identical whether the loader read the +capsule or walked the tree. + +## Why it exists + +Firmware file I/O has exactly one fast shape: **one open + one sequential +read**. Everything else is a lottery. Loading the system per-file — dozens of +opens, seeks, and short reads through the firmware's FAT driver — measured +**minutes** on real hardware, against milliseconds in QEMU/OVMF. Packing the +binaries into a single file turns the whole of user space into the shape +firmware is good at. + +Because the capsule already *is* the ramdisk wire format, the loader doesn't +even repack it: `loadCapsule` (`boot/efi.zig`) validates the magic and passes +the buffer straight through as `BootInformation.initial_ramdisk_base`/`len`. + +## The format + +The container is deliberately trivial — danos owns both producer and consumer, +so it need be no fancier. Little-endian throughout: + +``` +Header magic: u32 = "DNR2" (0x32524E44), count: u32 +Entry × count name: [64]u8 (NUL-padded FHS path), offset: u64, len: u64 +blobs... each entry's file bytes, at its offset within the image +``` + +- **Names are full FHS paths** (`/system/services/init`), not basenames — that + is what "v2" means. The 64-byte capacity matches `abi.maximum_process_name`, + so a task named after its binary path is never truncated. Paths longer than + 63 bytes are a build error (`pack-system-image.py` rejects them). +- **The v1 magic (`"DNRD"`, basename entries) is rejected**, not tolerated: a + stale image should fail loudly at `Reader.init`, not misparse names. +- `initial_ramdisk.Reader` is the one validated view over the bytes — magic + check, table bounds, per-blob bounds — used by the kernel and shared with the + loader. `Reader.find` resolves a binary by exact path first, then by unique + basename, ASCII case-insensitively (the entries come from a FAT volume, whose + name lookups are case-insensitive by definition). + +## How it is built + +`build.zig` maintains one `bundled` list — every user binary and its FHS home. +Three artifacts are derived from that same list, in the same build graph, so +they cannot drift apart: + +1. **The tree**: each binary installed at its FHS path (`zig-out/system/...`, + mirrored onto the FAT boot volume by `tools/make-fat-image.py`). +2. **The manifest** (`system/manifest`): the FHS path of every bundled binary, + one per line — the loader's per-file fallback input. +3. **The capsule**: `tools/pack-system-image.py` packs the same binaries into + the v2 container, installed at `zig-out/boot/system.img` and placed on the + boot volume at `boot/system.img`. + +Note what the capsule does *not* contain: the kernel (`system/kernel` is loaded +separately by `loadKernel`, as an ELF) and the EFI loader itself. It is user +space only. + +## How it is loaded + +`loadSystemTree` (`boot/efi.zig`) tries three strategies, most portable first — +the running system cannot tell which one ran, because all three produce the +same in-RAM ramdisk image: + +1. **The capsule** — open `boot\system.img`, read it whole, check the magic, + hand it over as-is. The normal path on any build-produced volume. +2. **The manifest** — read `system\manifest` and open each listed path *by + name*. FAT name lookup is case-insensitive and firmware-portable, unlike + directory enumeration. The loader assembles the v2 image in RAM itself. +3. **The tree walk** — enumerate `/system` recursively. Last resort for + hand-assembled sticks with neither file: some firmware FAT drivers return + bare 8.3 names uppercase from enumeration, which is why this is the + fallback and not the primary path. + +All three are best-effort: a **kernel-only volume still boots** — the kernel +just has no user binaries to spawn and reports the absence. + +One operational consequence of the ordering: the capsule *shadows* the tree. +If you hand-edit binaries on a stick that also carries a `boot/system.img`, +your edits are invisible — the loader boots the capsule's snapshot. Delete +`boot/system.img` from the volume to force the manifest/tree path. + +## What the kernel does with it + +The loader records the image's physical base and length in `BootInformation`; +the kernel (`kernel.zig`) then publishes the same bytes twice, to two +consumers: + +- **The process layer** (`process.zig`): `system_spawn` looks binaries up in + the ramdisk via `Reader.find` — exact FHS path, or unique basename for + pre-path callers — and loads them as fresh ring-3 processes. The stored path + becomes the task's name. +- **The VFS root** (`vfs.zig`, `setInitialRamdisk`): the image is mounted as + the kernel-backed, read-only `/system` mount. Directory nodes are derived + from the entry paths (the unique parents), so `/system` is listable and its + files readable over the normal VFS protocol — the FHS boot tree every + process sees comes straight out of the capsule bytes. + +The image is never copied after the handoff and never mutated: the initrd is +immutable, which is what makes the VFS's node serving lock-free. + +## What it is not + +- **Not `danos-usb.img`.** That is the 64 MiB FAT32 *boot volume* built by + `tools/make-fat-image.py` — the thing a machine actually boots, which + *contains* `boot/system.img` alongside the loader, kernel, manifest, and + tree. See [efi.md](efi.md) and [release-iso.md](release-iso.md). +- **Not a mountable filesystem.** No FAT, no block device, no driver — just a + header, a table, and concatenated blobs, parsed by ~90 lines of + `initial-ramdisk.zig`. +- **Not required.** It is the fast path, with two slower equivalents behind + it. +- **Not a place where state lives.** It is regenerated on every build from the + bundled binaries; nothing writes to it, at build time or runtime. diff --git a/docs/system-requirements.md b/docs/system-requirements.md index 034e649..e9efb16 100644 --- a/docs/system-requirements.md +++ b/docs/system-requirements.md @@ -117,7 +117,8 @@ hypervisor configured for UEFI firmware and an xHCI USB controller. FADT (power / PM timer). Optionally consumed: HPET, DMAR, SPCR. (`system/devices/acpi.zig:3`) - The loader reads `/system/kernel` off the FAT boot volume, then loads user - space: a prebuilt `boot\system.img` capsule when present, otherwise it walks + space: a prebuilt `boot\system.img` capsule + ([system-image.md](system-image.md)) when present, otherwise it walks the volume's `/system` tree (init included) into the initial ramdisk. The kernel can boot "kernel-only" without either. (`efi.zig:16`, `efi.zig:68`) From 2bc2a0d70d0eac159bca4fb9cfc4e7ff5f50ef85 Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Wed, 22 Jul 2026 10:49:46 +0100 Subject: [PATCH 6/7] =?UTF-8?q?kernel+tests:=20M4=20shared-fate=20?= =?UTF-8?q?=E2=80=94=20nine=20group-death=20test=20cases,=20per-space=20ar?= =?UTF-8?q?ena=20cursors?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Eight new QEMU cases (thread-fault-group, kill-threaded-group, kill-via-worker-tid, racing-triggers, exit-group, leader-thread-exit, thread-exit-solo, shm-mapping-ref) driving seven new thread-test modes; a shared checkGroupDead asserts the contract everywhere: one notification, badged with the leader, reason on the leader's record, no member listed, claims released first. The shm-mapping-ref case flushed out the per-task DMA/shared-memory arena cursor bug directly (a sibling's regions mapped over the worker's), so both cursors moved to the AddressSpaceRef like the mmap/MMIO cursors before them (threading M7 pattern). Runtime gains Thread.tryExitCurrent for the leader -EPERM refusal path. Docs updated: threading.md's shared-fate gap is closed, process-management.md and process-lifecycle.md describe the leader re-key, plan status = implemented. Full suite: 100/100. --- docs/process-lifecycle.md | 3 + docs/process-management.md | 10 +- docs/shared-fate-plan.md | 16 +- docs/threading.md | 20 +- library/runtime/thread.zig | 9 + system/kernel/process.zig | 76 +++++-- system/kernel/scheduler.zig | 26 ++- system/kernel/tests.zig | 218 ++++++++++++++++++++ system/services/thread-test/thread-test.zig | 165 +++++++++++++++ test/qemu_test.py | 65 ++++++ 10 files changed, 565 insertions(+), 43 deletions(-) diff --git a/docs/process-lifecycle.md b/docs/process-lifecycle.md index 9190a51..b4095e4 100644 --- a/docs/process-lifecycle.md +++ b/docs/process-lifecycle.md @@ -55,6 +55,9 @@ pattern reused. Signals are the same pattern reused a third time. - **`process_signal(id, signal)`** — posts the signal as an asynchronous notification to the target's bound endpoint: badge = `notify_badge_bit | notify_signal_bit | pending signals`. Non-blocking for the sender, always. + Signals address the *process*: `id` may name any member of a threaded process + and resolves to its leader — whose endpoint the harness binds — with authority + mirroring `process_kill` ([shared-fate-plan.md](shared-fate-plan.md)). - **Pending signals coalesce** in a per-process bitmask while the target has no signal endpoint bound, and the whole mask arrives as one notification at bind — POSIX's own semantics for non-realtime signals (two pending SIGTERMs are one diff --git a/docs/process-management.md b/docs/process-management.md index e52c07f..d0d5fec 100644 --- a/docs/process-management.md +++ b/docs/process-management.md @@ -57,9 +57,13 @@ dangle even if the supervisor dies first. ### `process_kill(id) -> 0 / -ESRCH / -EPERM` -Only the supervisor may kill; kernel tasks are not killable processes. Like a -signal, delivery is prompt but asynchronous — 0 means the kill is accepted and -irrevocable; the exit notification confirms completion. +Only the supervisor may kill; kernel tasks are not killable processes. The kill +is a **whole-process** kill ([shared-fate-plan.md](shared-fate-plan.md)): `id` +may name any member of a threaded process — it resolves to the group's leader, +authorization is checked against the *leader's* supervisor, and every thread +dies. Like a signal, delivery is prompt but asynchronous — 0 means the kill is +accepted and irrevocable; the exit notification (badged with the leader, posted +once the last member is gone) confirms completion. ## How a kill lands (the kernel mechanics) diff --git a/docs/shared-fate-plan.md b/docs/shared-fate-plan.md index f03a002..564e1c1 100644 --- a/docs/shared-fate-plan.md +++ b/docs/shared-fate-plan.md @@ -1,7 +1,11 @@ # Shared fate: whole-process death (plan) -**Status: approved 2026-07-22; leader `thread_exit` → `-EPERM`. Implementation in -progress (M1–M4 below).** +**Status: implemented 2026-07-22 (branch shared-fate), M1–M4 all landed; leader +`thread_exit` → `-EPERM` as decided. One scope addition forced by M4: the +per-task DMA/shared-memory arena cursors moved to the per-space object (the +`shm-mapping-ref` test could not distinguish corruption-by-remap from +corruption-by-free while sibling threads overlapped the arena) — the same move +the mmap/MMIO cursors made in threading M7.** [threading.md](threading.md) promises that a process dies *whole* — a fault in any thread, or a kill, takes down every thread. The kernel doesn't do that yet: every @@ -244,10 +248,10 @@ refcount, and no group-kill special case is needed at all. `.ready` with `in_system_call = true`, and the tick's reap loop will reap it — an existing hazard the fan-out inherits but must not add new instances of. Filed to investigate separately. -- **Per-task DMA/shm cursors** (`dma_map_next`, `shared_memory_map_next`): two - sibling threads allocating overlap the same arena — pre-existing thread bug, - adjacent to but not part of this plan (the mmap/MMIO cursors already moved - per-space for exactly this reason). +- **Per-task DMA/shm cursors** — *fixed during M4 after all*: the + `shm-mapping-ref` test tripped the overlap (the sibling's churn regions mapped + over the worker's region), so both cursors moved to the `AddressSpaceRef` + like the mmap/MMIO cursors before them. ## Milestones diff --git a/docs/threading.md b/docs/threading.md index 0a8a43b..d8580dd 100644 --- a/docs/threading.md +++ b/docs/threading.md @@ -252,20 +252,18 @@ stays single-threaded and lean. halts" property intact under lock contention — no busy-wait. - **Lifecycle** ([process-lifecycle.md](process-lifecycle.md)): the contract is that killing a process kills *all* its threads and only then drops the last address-space - ref. **The kernel does not implement that fan-out yet**: `process_kill` reaps only - the one task it resolves, and no death path loops over the tasks sharing an address - space — the refcount keeps the space (and the sibling threads) alive and running. - The gap is hit in practice: of the only threaded binaries (the `display` service and - the `thread-test` harness), `display` is a boot service that handles no `.terminate` - signal, so init's stop sequence escalates to `process_kill` on every orderly - shutdown — benign only because poweroff follows. Whether to implement the fan-out or - amend the contract is a decision still to be made. + ref — and the kernel now implements exactly that + ([shared-fate-plan.md](shared-fate-plan.md)): every death path (`exit` from any + thread, a fault, `process_kill` aimed at any member id) fans out through the whole + group via a `dying` latch on the address space; the supervisor's one exit + notification — badged with the leader — fires only when the last member is gone. + A worker's voluntary `thread_exit` stays per-thread; the leader's is refused + (`-EPERM`). - **Resilience** ([resilience.md](resilience.md)): by the same contract, a faulting thread kills its whole process (shared fate); the supervisor restarts the **process**, which respawns its threads from a known-good state — restart - granularity stays the process. Today a CPU fault kills only the faulting task - (`killCurrentProcess` tears down a single task), so sibling threads keep running — - the same implementation gap as above. + granularity stays the process. The leader's recorded exit reason carries the fault + class even when a worker faulted, so restart policy is unchanged. - **IPC — two consequences threads forced ([ipc.md](ipc.md)):** - *Handles do not cross threads.* The handle table lives on the `Task` ([scheduler.zig](../system/kernel/scheduler.zig)), so a handle number is meaningful diff --git a/library/runtime/thread.zig b/library/runtime/thread.zig index 0a59df0..5d2da4a 100644 --- a/library/runtime/thread.zig +++ b/library/runtime/thread.zig @@ -120,6 +120,15 @@ pub const Thread = struct { return @intCast(sc.systemCall0(.current_core)); } + /// Ask the kernel to end the calling thread. A WORKER never returns from this; + /// the process's MAIN thread gets the kernel's refusal (-EPERM — the group ends + /// only through exit, a fault, or process_kill, docs/shared-fate-plan.md) and + /// the call returns. Exists for exactly that refusal path; workers end through + /// the spawn trampoline, and a process ends through `system.exit`. + pub fn tryExitCurrent() void { + _ = sc.systemCall0(.thread_exit); + } + /// `std.Thread.Futex`-shaped block/wake on a `u32` atomic — the primitive the /// blocking `Mutex`/`Condition`/`Semaphore` are built on. Waiters park in the /// kernel (no busy-wait), so an idle core still halts (docs/halting.md). diff --git a/system/kernel/process.zig b/system/kernel/process.zig index 30e6bd6..d627e6b 100644 --- a/system/kernel/process.zig +++ b/system/kernel/process.zig @@ -78,15 +78,15 @@ pub const device_arena_end: u64 = device_arena_base + (4 << 30); /// The DMA arena: where `dma_alloc` places coherent DMA buffers, in PML4[228] — a /// user-exclusive region distinct from the MMIO arena. Unlike MMIO grants these back -/// real RAM (contiguous frames), so they are reclaimed on teardown. Per-process cursor -/// in `Task.dma_map_next`. +/// real RAM (contiguous frames), so they are reclaimed on teardown. Per-SPACE cursor +/// on the AddressSpaceRef (sibling threads share the arena). pub const dma_arena_base: u64 = 0x0000_7200_0000_0000; pub const dma_arena_end: u64 = dma_arena_base + (256 << 20); // 256 MiB per process /// The shared-memory arena: where `shared_memory_create`/`shared_memory_map` place shared cacheable regions, in /// PML4[230] — a user-exclusive region distinct from the DMA arena. The frames are owned by /// a refcounted shared-memory object and freed when its last capability drops, not on teardown, so the -/// mapping carries `device_grant`. Per-process cursor in `Task.shared_memory_map_next` (docs/display-v2.md). +/// mapping carries `device_grant`. Per-SPACE cursor on the AddressSpaceRef (docs/display-v2.md). pub const shared_memory_arena_base: u64 = 0x0000_7300_0000_0000; pub const shared_memory_arena_end: u64 = shared_memory_arena_base + (256 << 20); // 256 MiB per process @@ -501,19 +501,31 @@ fn systemDmaAlloc(state: *architecture.CpuState) void { const max_phys: u64 = if (flags & abi.dma_below_4g != 0) (@as(u64, 4) << 30) else ~@as(u64, 0); const phys = pmm.allocContiguous(pages, max_phys) orelse return fail(state); - if (t.dma_map_next == 0) t.dma_map_next = dma_arena_base; - const base_v = t.dma_map_next; - if (base_v + pages * page_size > dma_arena_end) { - for (0..pages) |i| pmm.free(phys + i * page_size); // arena exhausted; give the frames back - return fail(state); + // Reserve arena virtual space from the per-SPACE cursor, under the lock — + // sibling threads must hand out disjoint windows of the one shared arena. + var base_v: u64 = 0; + { + const lock_flags = sync.enter(); + const cursor = scheduler.addressSpaceDmaNextPtr(t.address_space) orelse { + sync.leave(lock_flags); + for (0..pages) |i| pmm.free(phys + i * page_size); + return fail(state); + }; + if (cursor.* == 0) cursor.* = dma_arena_base; + base_v = cursor.*; + if (base_v + pages * page_size > dma_arena_end) { + sync.leave(lock_flags); + for (0..pages) |i| pmm.free(phys + i * page_size); // arena exhausted; give the frames back + return fail(state); + } + cursor.* = base_v + pages * page_size; + sync.leave(lock_flags); } // Zero through the physmap (the frames aren't mapped in the caller yet), then map. const kernel_view: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(phys)); @memset(kernel_view[0 .. pages * page_size], 0); architecture.mapUserDmaInto(t.address_space, base_v, phys, pages * page_size); - - t.dma_map_next = base_v + pages * page_size; architecture.setSystemCallResult(state, base_v); // virtual address for the CPU architecture.setSystemCallResult2(state, phys); // physical address for the device } @@ -557,10 +569,24 @@ fn systemSharedMemoryCreate(state: *architecture.CpuState) void { const pages: usize = @intCast((len + page_size - 1) / page_size); if (pages == 0 or pages > maximum_shared_memory_pages) return fail(state); - // Reserve arena virtual space up front, so a mapping failure needs no rollback. - if (t.shared_memory_map_next == 0) t.shared_memory_map_next = shared_memory_arena_base; - const base_v = t.shared_memory_map_next; - if (base_v + pages * page_size > shared_memory_arena_end) return fail(state); // arena exhausted + // Reserve arena virtual space up front (per-SPACE cursor: sibling threads + // hand out disjoint windows), so a mapping failure needs no rollback. + var base_v: u64 = 0; + { + const lock_flags = sync.enter(); + const cursor = scheduler.addressSpaceSharedMemoryNextPtr(t.address_space) orelse { + sync.leave(lock_flags); + return fail(state); + }; + if (cursor.* == 0) cursor.* = shared_memory_arena_base; + base_v = cursor.*; + if (base_v + pages * page_size > shared_memory_arena_end) { + sync.leave(lock_flags); + return fail(state); // arena exhausted + } + cursor.* = base_v + pages * page_size; + sync.leave(lock_flags); + } const phys = pmm.allocContiguous(pages, ~@as(u64, 0)) orelse return fail(state); // Zero through the physmap (the frames aren't mapped in the caller yet). @@ -595,7 +621,6 @@ fn systemSharedMemoryCreate(state: *architecture.CpuState) void { } architecture.mapUserSharedInto(t.address_space, base_v, phys, pages * page_size); - t.shared_memory_map_next = base_v + pages * page_size; architecture.setSystemCallResult(state, base_v); // virtual_address for the CPU architecture.setSystemCallResult2(state, @intCast(handle)); // capability handle to pass on } @@ -610,25 +635,34 @@ fn systemSharedMemoryMap(state: *architecture.CpuState) void { if (t.address_space == 0) return fail(state); const shared_memory = ipc.resolveSharedMemory(t, cap) orelse return fail(state); // not a shared-memory handle we hold - if (t.shared_memory_map_next == 0) t.shared_memory_map_next = shared_memory_arena_base; - const base_v = t.shared_memory_map_next; const size = shared_memory.pages * page_size; - if (base_v + size > shared_memory_arena_end) return fail(state); - // This mapping holds its own reference, recorded on the space and dropped at - // its destruction (docs/shared-fate-plan.md M3) — the handle's reference is + // Reserve arena space (per-SPACE cursor) and record the mapping's own + // reference in one locked section — the reference is dropped at space + // destruction (docs/shared-fate-plan.md M3); the handle's reference is // separate and may be closed while the mapping lives on. + var base_v: u64 = 0; { const flags = sync.enter(); + const cursor = scheduler.addressSpaceSharedMemoryNextPtr(t.address_space) orelse { + sync.leave(flags); + return fail(state); + }; + if (cursor.* == 0) cursor.* = shared_memory_arena_base; + base_v = cursor.*; + if (base_v + size > shared_memory_arena_end) { + sync.leave(flags); + return fail(state); // arena exhausted + } if (!scheduler.recordSpaceMappingLocked(t.address_space, @ptrCast(shared_memory))) { sync.leave(flags); return fail(state); // mapping table full: refuse rather than map unrecorded } ipc.retainSharedMemory(shared_memory); + cursor.* = base_v + size; sync.leave(flags); } architecture.mapUserSharedInto(t.address_space, base_v, shared_memory.phys, size); - t.shared_memory_map_next = base_v + size; architecture.setSystemCallResult(state, base_v); } diff --git a/system/kernel/scheduler.zig b/system/kernel/scheduler.zig index 34dcef8..27b99c0 100644 --- a/system/kernel/scheduler.zig +++ b/system/kernel/scheduler.zig @@ -121,8 +121,10 @@ pub const Task = struct { ipc_reply_ptr: u64 = 0, // client: reply buffer (virtual_address) ipc_reply_cap: u64 = 0, ipc_status: i64 = 0, // client: reply length / -errno, written by the replier - dma_map_next: u64 = 0, // bump pointer into this task's DMA arena (0 = unseeded) - shared_memory_map_next: u64 = 0, // bump pointer into this task's shared-memory arena (0 = unseeded) + // The DMA and shared-memory arena cursors moved to the per-address-space object + // (`AddressSpaceRef`) like the mmap/MMIO cursors before them — per-TASK cursors made + // sibling threads hand out overlapping windows of the one shared arena + // (docs/shared-fate-plan.md M4 tripped exactly that). ipc_send_cap: u64 = ~@as(u64, 0), // handle to transfer with this message (abi.no_cap = none) ipc_received_cap: u64 = ~@as(u64, 0), // client: handle the reply's transferred cap landed at (abi.no_cap = none) next: ?*Task = null, // ready-queue link (also the endpoint sender-FIFO link) @@ -173,6 +175,8 @@ const AddressSpaceRef = struct { count: u32 = 0, mmap_next: u64 = 0, device_map_next: u64 = 0, + dma_next: u64 = 0, // bump pointer into this space's DMA arena (0 = unseeded) + shared_memory_next: u64 = 0, // bump pointer into this space's shared-memory arena (0 = unseeded) // Group-death state (docs/shared-fate-plan.md), set once by the first kill // trigger and never cleared while the entry lives. `dying` gates // retainAddressSpace — no new member may join a dying group (closing the @@ -395,6 +399,24 @@ pub fn addressSpaceDeviceMapNextPtr(root: u64) ?*u64 { } return null; } + +/// Pointer to the DMA arena cursor for address space `root` (see +/// `addressSpaceMmapNextPtr`). Caller holds the kernel lock. +pub fn addressSpaceDmaNextPtr(root: u64) ?*u64 { + for (&address_space_refs) |*entry| { + if (entry.count != 0 and entry.root == root) return &entry.dma_next; + } + return null; +} + +/// Pointer to the shared-memory arena cursor for address space `root` (see +/// `addressSpaceMmapNextPtr`). Caller holds the kernel lock. +pub fn addressSpaceSharedMemoryNextPtr(root: u64) ?*u64 { + for (&address_space_refs) |*entry| { + if (entry.count != 0 and entry.root == root) return &entry.shared_memory_next; + } + return null; +} var next_id: u32 = 1; /// Per-CPU scheduler state: the task each core is running, its own idle task, and a diff --git a/system/kernel/tests.zig b/system/kernel/tests.zig index 476ff03..25488cc 100644 --- a/system/kernel/tests.zig +++ b/system/kernel/tests.zig @@ -146,6 +146,22 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void { faultRecoveryTest(boot_information); } else if (eql(case, "address-space-refcount")) { addressSpaceRefcountTest(boot_information); + } else if (eql(case, "thread-fault-group")) { + threadFaultGroupTest(boot_information); + } else if (eql(case, "kill-threaded-group")) { + killThreadedGroupTest(boot_information); + } else if (eql(case, "kill-via-worker-tid")) { + killViaWorkerTidTest(boot_information); + } else if (eql(case, "racing-triggers")) { + racingTriggersTest(boot_information); + } else if (eql(case, "exit-group")) { + exitGroupTest(boot_information); + } else if (eql(case, "leader-thread-exit")) { + threadTestMarkerCase(boot_information, "leader-thread-exit", "leader-exit"); + } else if (eql(case, "thread-exit-solo")) { + threadTestMarkerCase(boot_information, "thread-exit-solo", "solo"); + } else if (eql(case, "shm-mapping-ref")) { + threadTestMarkerCase(boot_information, "shm-mapping-ref", "shm-worker"); } else if (eql(case, "thread-spawn")) { threadSpawnTest(boot_information); } else if (eql(case, "thread-join")) { @@ -3167,6 +3183,208 @@ fn kernelVfsTest(boot_information: *const BootInformation) void { result(); } +// --- shared-fate helpers (docs/shared-fate-plan.md M4) ----------------------- + +/// Spawn the ramdisk's thread-test with `mode` as argv[1], supervised by the +/// calling test task on `endpoint`. Returns the child (leader) id, or 0. +fn spawnThreadTestSupervised(boot_information: *const BootInformation, mode: []const u8, endpoint: *ipcsync.Endpoint) u32 { + const ramdisk = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len]; + const rd = initial_ramdisk.Reader.init(ramdisk) orelse return 0; + var i: u32 = 0; + while (i < rd.count) : (i += 1) { + const item = rd.entry(i) orelse continue; + if (!eql(initial_ramdisk.basename(item.name), "thread-test")) continue; + return process.spawnProcessSupervised(item.blob, 4, &.{ "thread-test", mode }, scheduler.currentId(), endpoint) catch 0; + } + return 0; +} + +/// Whether any live task still belongs to `leader`'s group. +fn groupListed(leader: u32) bool { + var table: [32]abi.ProcessDescriptor = undefined; + const total = scheduler.enumerate(&table); + for (table[0..@min(total, table.len)]) |descriptor| { + if (descriptor.id == leader or descriptor.leader == leader) return true; + } + return false; +} + +/// Poll enumerate for a WORKER of `leader` (same leader, different id) until +/// `deadline` (millis); returns its id, or 0. +fn findWorkerOf(leader: u32, deadline: u64) u32 { + while (architecture.millis() < deadline) { + var table: [32]abi.ProcessDescriptor = undefined; + const total = scheduler.enumerate(&table); + for (table[0..@min(total, table.len)]) |descriptor| { + if (descriptor.leader == leader and descriptor.id != leader) return descriptor.id; + } + scheduler.yield(); + } + return 0; +} + +/// Block on `endpoint` for the next notification and return its badge. +fn awaitExitBadge(endpoint: *ipcsync.Endpoint) u64 { + var badge: u64 = 0; + var received_cap: u64 = 0; + _ = ipcsync.replyWait(endpoint, 0, 0, 0, 0, abi.no_cap, &badge, &received_cap); + return badge; +} + +/// A group death is one notification, badged with the LEADER, arriving only +/// after every member (and the address space) is gone — asserted by every +/// shared-fate case below. +fn checkGroupDead(me: u32, leader: u32, badge: u64, reason: abi.ExitReason) void { + check("one exit notification, badged with the leader", badge == abi.notify_badge_bit | abi.notify_exit_bit | leader); + check("the leader's recorded reason is the group reason", process.exitReasonOf(me, leader) == @intFromEnum(reason)); + check("no group member is listed after the death", !groupListed(leader)); +} + +fn threadFaultGroupTest(boot_information: *const BootInformation) void { + log("DANOS-TEST-BEGIN: thread-fault-group\n", .{}); + const me = scheduler.currentId(); + const endpoint = ipcsync.createIpcEndpoint() orelse { + check("exit endpoint allocated", false); + result(); + return; + }; + const stacks_base = scheduler.liveStackBytes(); + const spaces_base = scheduler.liveAddressSpaceCount(); + process.fault_kill_count = 0; + const child = spawnThreadTestSupervised(boot_information, "fault-worker", endpoint); + check("thread-test spawned (fault-worker)", child != 0); + if (child == 0) { + result(); + return; + } + const badge = awaitExitBadge(endpoint); + checkGroupDead(me, child, badge, .segmentation_fault); + check("one fault kill for the whole group", process.fault_kill_count == 1); + const deadline = architecture.millis() + 5000; + while (scheduler.liveStackBytes() > stacks_base and architecture.millis() < deadline) scheduler.yield(); + check("address spaces returned to base", scheduler.liveAddressSpaceCount() == spaces_base); + check("kernel stacks returned to base", scheduler.liveStackBytes() == stacks_base); + result(); +} + +fn killThreadedGroupTest(boot_information: *const BootInformation) void { + log("DANOS-TEST-BEGIN: kill-threaded-group\n", .{}); + const me = scheduler.currentId(); + var buffer: [2]device_abi.DeviceDescriptor = undefined; + check("the device tree is seeded (>= 2 devices)", devices_broker.enumerate(&buffer) >= 2); + const endpoint = ipcsync.createIpcEndpoint() orelse { + check("exit endpoint allocated", false); + result(); + return; + }; + const child = spawnThreadTestSupervised(boot_information, "spin-forever", endpoint); + check("thread-test spawned (spin-forever)", child != 0); + if (child == 0) { + result(); + return; + } + const worker = findWorkerOf(child, architecture.millis() + 8000); + check("the spinning worker is enumerable with leader = the child", worker != 0); + // Claims for BOTH members: group death must release every member's claims + // before the supervisor hears anything — the worker's by the deferred + // (condemned) path. + check("device 0 claimed for the leader", devices_broker.claim(0, child)); + check("device 1 claimed for the worker", devices_broker.claim(1, worker)); + scheduler.sleep(100); // let the worker really be running on another core + check("the supervisor's kill is accepted", process.killProcess(me, child) == 0); + const badge = awaitExitBadge(endpoint); + checkGroupDead(me, child, badge, .killed); + check("the leader's claim was released before the notification", devices_broker.ownerOf(0) == null); + check("the worker's claim was released before the notification", devices_broker.ownerOf(1) == null); + check("a dead group stays dead (-ESRCH)", process.killProcess(me, child) == -ipcsync.ESRCH); + result(); +} + +fn killViaWorkerTidTest(boot_information: *const BootInformation) void { + log("DANOS-TEST-BEGIN: kill-via-worker-tid\n", .{}); + const me = scheduler.currentId(); + const endpoint = ipcsync.createIpcEndpoint() orelse { + check("exit endpoint allocated", false); + result(); + return; + }; + const child = spawnThreadTestSupervised(boot_information, "spin-forever", endpoint); + check("thread-test spawned (spin-forever)", child != 0); + if (child == 0) { + result(); + return; + } + const worker = findWorkerOf(child, architecture.millis() + 8000); + check("the spinning worker is enumerable", worker != 0); + check("a non-supervisor aiming at the worker is refused (-EPERM)", process.killProcess(me + 12345, worker) == -ipcsync.EPERM); + check("the supervisor's kill aimed at the WORKER id is accepted", process.killProcess(me, worker) == 0); + const badge = awaitExitBadge(endpoint); + checkGroupDead(me, child, badge, .killed); + result(); +} + +fn racingTriggersTest(boot_information: *const BootInformation) void { + log("DANOS-TEST-BEGIN: racing-triggers\n", .{}); + const me = scheduler.currentId(); + const endpoint = ipcsync.createIpcEndpoint() orelse { + check("exit endpoint allocated", false); + result(); + return; + }; + process.fault_kill_count = 0; + const child = spawnThreadTestSupervised(boot_information, "race", endpoint); + check("thread-test spawned (race)", child != 0); + if (child == 0) { + result(); + return; + } + const badge = awaitExitBadge(endpoint); + checkGroupDead(me, child, badge, .segmentation_fault); + check("two racing faults counted as ONE group kill", process.fault_kill_count == 1); + result(); +} + +fn exitGroupTest(boot_information: *const BootInformation) void { + log("DANOS-TEST-BEGIN: exit-group\n", .{}); + const me = scheduler.currentId(); + const endpoint = ipcsync.createIpcEndpoint() orelse { + check("exit endpoint allocated", false); + result(); + return; + }; + const child = spawnThreadTestSupervised(boot_information, "exit-worker", endpoint); + check("thread-test spawned (exit-worker)", child != 0); + if (child == 0) { + result(); + return; + } + const badge = awaitExitBadge(endpoint); + checkGroupDead(me, child, badge, .aborted); + result(); +} + +/// leader-thread-exit, thread-exit-solo, and shm-mapping-ref share one shape: +/// the child asserts its own property, prints a marker the harness matches, and +/// exits clean — the kernel side asserts the clean group death. +fn threadTestMarkerCase(boot_information: *const BootInformation, case_name: []const u8, mode: []const u8) void { + log("DANOS-TEST-BEGIN: {s}\n", .{case_name}); + const me = scheduler.currentId(); + const endpoint = ipcsync.createIpcEndpoint() orelse { + check("exit endpoint allocated", false); + result(); + return; + }; + const child = spawnThreadTestSupervised(boot_information, mode, endpoint); + check("thread-test spawned", child != 0); + if (child == 0) { + result(); + return; + } + const badge = awaitExitBadge(endpoint); + checkGroupDead(me, child, badge, .exited); + result(); +} + fn spawnNamed(rd: initial_ramdisk.Reader, name: []const u8) bool { var i: u32 = 0; while (i < rd.count) : (i += 1) { diff --git a/system/services/thread-test/thread-test.zig b/system/services/thread-test/thread-test.zig index 50a72c4..cc5c263 100644 --- a/system/services/thread-test/thread-test.zig +++ b/system/services/thread-test/thread-test.zig @@ -469,6 +469,157 @@ fn runRwlockMode() void { } } +// --- shared-fate modes (docs/shared-fate-plan.md M4) ------------------------- + +fn faultNullPage() void { + @as(*volatile u32, @ptrFromInt(0x10)).* = 1; // unmapped null page: #PF, group death +} + +fn faultingWorker() void { + write("thread-test: worker faulting\n"); + faultNullPage(); +} + +/// fault-worker: a worker faults; shared fate must end the whole group, so the +/// main thread parks forever and never prints anything more. +fn runFaultWorkerMode() void { + write("thread-test: spawning faulting worker\n"); + _ = runtime.Thread.spawn(.{}, faultingWorker, .{}) catch { + write("thread-test: FAIL spawn refused\n"); + return; + }; + while (true) runtime.system.yield(); +} + +fn spinningWorker() void { + while (true) {} // no system calls: only a tick can deliver a deferred kill +} + +/// spin-forever: a kill target. The worker spins without syscalls (the condemned +/// path); the main thread yields (the parked path). +fn runSpinForeverMode() void { + _ = runtime.Thread.spawn(.{}, spinningWorker, .{}) catch { + write("thread-test: FAIL spawn refused\n"); + return; + }; + write("thread-test: spinning\n"); + while (true) runtime.system.yield(); +} + +fn exitingWorker() void { + write("thread-test: worker exiting the process\n"); + runtime.system.exit(3); // exit from ANY thread is group death (.aborted) +} + +/// exit-worker: a WORKER calls exit(3); the group must die with the leader's +/// reason reading .aborted. +fn runExitWorkerMode() void { + _ = runtime.Thread.spawn(.{}, exitingWorker, .{}) catch { + write("thread-test: FAIL spawn refused\n"); + return; + }; + while (true) runtime.system.yield(); +} + +var race_go = std.atomic.Value(u32).init(0); + +fn racingWorker() void { + while (race_go.load(.acquire) == 0) {} + faultNullPage(); +} + +/// race: two members fault as near-simultaneously as user space can arrange — +/// the group-dying latch must make the two triggers count as one death. +fn runRaceMode() void { + _ = runtime.Thread.spawn(.{}, racingWorker, .{}) catch { + write("thread-test: FAIL spawn refused\n"); + return; + }; + write("thread-test: racing\n"); + race_go.store(1, .release); + faultNullPage(); +} + +var leader_exit_done = std.atomic.Value(u32).init(0); + +fn patientWorker() void { + while (leader_exit_done.load(.acquire) == 0) runtime.system.yield(); +} + +/// leader-exit: the MAIN thread asks for thread_exit; the kernel must refuse +/// (-EPERM) and the worker must be entirely unaffected. +fn runLeaderExitMode() void { + const worker = runtime.Thread.spawn(.{}, patientWorker, .{}) catch { + write("thread-test: FAIL spawn refused\n"); + return; + }; + runtime.Thread.tryExitCurrent(); // refused: we are the leader + write("thread-test: leader thread_exit refused ok\n"); + leader_exit_done.store(1, .release); + worker.join(); +} + +fn promptWorker() void {} + +/// solo: regression — a WORKER's thread_exit stays per-thread; the sibling +/// (main) survives it. +fn runSoloMode() void { + const worker = runtime.Thread.spawn(.{}, promptWorker, .{}) catch { + write("thread-test: FAIL spawn refused\n"); + return; + }; + worker.join(); + write("thread-test: solo sibling survived ok\n"); +} + +const shm_pattern_length: usize = 4096; +var shm_region_base = std.atomic.Value(usize).init(0); + +fn patternByte(i: usize) u8 { + return @truncate(i *% 31 +% 7); +} + +fn shmWorker() void { + const region = runtime.shared_memory.create(shm_pattern_length) orelse { + write("thread-shm: FAIL create refused\n"); + return; + }; + for (0..shm_pattern_length) |i| region.ptr[i] = patternByte(i); + shm_region_base.store(@intFromPtr(region.ptr), .release); + // Returning thread_exits this worker: its handle table — holding the + // region's ONLY handle — closes. The sibling's view must survive on the + // mapping reference (docs/shared-fate-plan.md M3). +} + +/// shm-worker: the M3 use-after-free regression. The worker creates and fills a +/// shared-memory region and dies; the main thread churns the frame allocator and +/// then checks the mapping is intact — freed frames would have been reused and +/// scribbled on. +fn runShmWorkerMode() void { + const worker = runtime.Thread.spawn(.{}, shmWorker, .{}) catch { + write("thread-test: FAIL spawn refused\n"); + return; + }; + worker.join(); + const base = shm_region_base.load(.acquire); + if (base == 0) return; // the worker already printed the failure + var churn: usize = 0; + while (churn < 8) : (churn += 1) { + const noise = runtime.shared_memory.create(shm_pattern_length) orelse break; + @memset(noise.ptr[0..shm_pattern_length], 0xFF); + } + const view: [*]const u8 = @ptrFromInt(base); + var intact = true; + for (0..shm_pattern_length) |i| { + if (view[i] != patternByte(i)) intact = false; + } + if (intact) { + write("thread-shm: mapping survives creator ok\n"); + } else { + write("thread-shm: FAIL mapping corrupted after creator death\n"); + } +} + pub fn main(init: runtime.process.Init) void { const mode = init.arguments.get(1) orelse "spawn"; if (std.mem.eql(u8, mode, "join")) { @@ -485,6 +636,20 @@ pub fn main(init: runtime.process.Init) void { runTlsMode(); } else if (std.mem.eql(u8, mode, "rwlock")) { runRwlockMode(); + } else if (std.mem.eql(u8, mode, "fault-worker")) { + runFaultWorkerMode(); + } else if (std.mem.eql(u8, mode, "spin-forever")) { + runSpinForeverMode(); + } else if (std.mem.eql(u8, mode, "exit-worker")) { + runExitWorkerMode(); + } else if (std.mem.eql(u8, mode, "race")) { + runRaceMode(); + } else if (std.mem.eql(u8, mode, "leader-exit")) { + runLeaderExitMode(); + } else if (std.mem.eql(u8, mode, "solo")) { + runSoloMode(); + } else if (std.mem.eql(u8, mode, "shm-worker")) { + runShmWorkerMode(); } else { runSpawnMode(); } diff --git a/test/qemu_test.py b/test/qemu_test.py index 5fb6f9f..b8e6291 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -327,6 +327,71 @@ CASES = [ "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, + # docs/shared-fate-plan.md M4: a worker's CPU fault kills the whole group — + # one notification (badged with the leader), the leader's reason is the fault + # class, and every counter returns to base. + {"name": "thread-fault-group", + "smp": 4, + "timeout": 60, + "expect": r"thread-test: worker faulting[\s\S]*DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, + + # docs/shared-fate-plan.md M4: process_kill on a threaded group — the + # spinning worker dies by the deferred (condemned) path, both members' + # device claims are free before the single leader-badged notification. + {"name": "kill-threaded-group", + "smp": 4, + "timeout": 60, + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, + + # docs/shared-fate-plan.md M4: process_kill aimed at a WORKER tid kills the + # whole group (the capability is per-process), still badged with the leader. + {"name": "kill-via-worker-tid", + "smp": 4, + "timeout": 60, + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, + + # docs/shared-fate-plan.md M4: two members fault near-simultaneously on + # different cores; the dying latch makes them one group death + # (fault_kill_count == 1) with a deterministic leader reason. + {"name": "racing-triggers", + "smp": 4, + "timeout": 60, + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, + + # docs/shared-fate-plan.md M4: exit(3) from a WORKER is group death with the + # leader's reason reading .aborted (the exit_group lesson). + {"name": "exit-group", + "smp": 4, + "timeout": 60, + "expect": r"thread-test: worker exiting the process[\s\S]*DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, + + # docs/shared-fate-plan.md M4: the LEADER's thread_exit is refused (-EPERM); + # the worker is unaffected and the process then exits clean. + {"name": "leader-thread-exit", + "timeout": 60, + "expect": r"thread-test: leader thread_exit refused ok[\s\S]*DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, + + # docs/shared-fate-plan.md M4: regression — a WORKER's thread_exit stays + # per-thread; the sibling survives it. + {"name": "thread-exit-solo", + "timeout": 60, + "expect": r"thread-test: solo sibling survived ok[\s\S]*DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, + + # docs/shared-fate-plan.md M3/M4: a shared-memory region whose only HANDLE + # died with its creator thread survives on the MAPPING reference — the + # sibling's view stays intact through allocator churn. + {"name": "shm-mapping-ref", + "timeout": 60, + "expect": r"thread-shm: mapping survives creator ok[\s\S]*DANOS-TEST-RESULT: PASS", + "fail": r"thread-shm: FAIL|DANOS-TEST-RESULT: FAIL"}, + # docs/threading-plan.md M4: futex — a thread parks in futex_wait and is woken by # futex_wake (serial order waiting/waking/woke), and timedWait reports a timeout. {"name": "thread-futex", From c8191570e18c65269f8b6b2b196ac6af74bdd391 Mon Sep 17 00:00:00 2001 From: Daniel Samson <12231216+daniel-samson@users.noreply.github.com> Date: Wed, 22 Jul 2026 11:23:32 +0100 Subject: [PATCH 7/7] =?UTF-8?q?kernel:=20shared-fate=20review=20fixes=20?= =?UTF-8?q?=E2=80=94=20lock=20the=20shm/dma=20walks,=20contract=20edges?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The adversarial review of the branch confirmed the big one: the shm/DMA page-table walks and their pmm/heap calls ran outside the big kernel lock — pre-existing, but fatal once the per-space cursors invited sibling threads to race them (two concurrent creates could orphan a page table: one thread's region silently unmapped, the frame leaked — a plausible root for the long-standing intermittent AP ring-3 fault at the shm base). All three paths now follow the mmap discipline: allocation, object build, record, and handle under one lock hold with full rollback; the map itself per-page under brief holds; dma_free's translate/unmap/free per-page likewise. Contract edges from the same review: thread_spawn into a dying group returns -ESRCH (was generic -1); process_kill during the condemned window answers from the latch's stashed supervisor (0 or -EPERM, was -ESRCH once the leader slot was reaped); exit derives the group reason from its own argument rather than the racy exit_code global; a worker's thread_exit no longer overwrites a concurrent group-kill stamp; checkGroupDead now asserts exactly-one notification via the drained ring. Full suite: 100/100. --- docs/shared-fate-plan.md | 17 +++- system/kernel/process.zig | 157 ++++++++++++++++++++++++------------ system/kernel/scheduler.zig | 28 +++++++ system/kernel/tests.zig | 19 +++-- 4 files changed, 160 insertions(+), 61 deletions(-) diff --git a/docs/shared-fate-plan.md b/docs/shared-fate-plan.md index 564e1c1..c8631f7 100644 --- a/docs/shared-fate-plan.md +++ b/docs/shared-fate-plan.md @@ -251,7 +251,22 @@ refcount, and no group-kill special case is needed at all. - **Per-task DMA/shm cursors** — *fixed during M4 after all*: the `shm-mapping-ref` test tripped the overlap (the sibling's churn regions mapped over the worker's region), so both cursors moved to the `AddressSpaceRef` - like the mmap/MMIO cursors before them. + like the mmap/MMIO cursors before them. The post-implementation review then + found the other half: the shm/DMA page-table walks and their pmm/heap calls + ran *outside* the big kernel lock — pre-existing, but fatal once siblings + were invited to race them (and a plausible root for the long-standing + intermittent AP ring-3 fault at the shm base). All three paths now follow + the mmap discipline: metadata and allocation under one hold, the map itself + per-page under brief holds. +- **Mapping-record slots are never recycled**: 16 per space, one per + `shared_memory_create`/`map`, freed only at space destruction (there is no + shm unmap). A long-lived compositor that churns surfaces will hit the cap; + the failure is a clean refused create, and slot recycling can ride whatever + adds `shared_memory_unmap`. +- **Two properties lack direct tests**: the spawn gate (an in-flight + `thread_spawn` racing the fan-out — inherently nondeterministic to arrange; + covered by code inspection and the `-ESRCH` path) and the `process_signal` + leader re-key (exercised only implicitly by the signals case). ## Milestones diff --git a/system/kernel/process.zig b/system/kernel/process.zig index d627e6b..bff36a2 100644 --- a/system/kernel/process.zig +++ b/system/kernel/process.zig @@ -201,7 +201,10 @@ fn system_call(state: *architecture.CpuState) void { // restart those, unlike a clean `.exited` ("nothing for me here"), // which they let lie. if (scheduler.currentIsUserProcess()) { - exitGroupCurrent(if (exit_code == 0) .exited else .aborted); + // The reason derives from this call's own argument (a local), not + // the shared exit_code global — two racing exits must not decide + // each other's group reason. + exitGroupCurrent(if (architecture.systemCallArg(state, 0) == 0) .exited else .aborted); } else architecture.userExit(); }, .yield => { @@ -268,7 +271,10 @@ fn system_call(state: *architecture.CpuState) void { const dying = scheduler.current(); if (dying.id == dying.leader) return failErr(state, ipc.EPERM); _ = sync.enter(); // handed off through the exit switch - dying.exit_reason = .exited; + // A group kill may have stamped us .killed while we raced to + // this dispatch (past the entry check, spinning on the lock) — + // keep that stamp; the fan-out's story wins. + if (!dying.kill_pending.load(.monotonic)) dying.exit_reason = .exited; terminateCurrentLocked(); } else architecture.userExit(); }, @@ -499,33 +505,44 @@ fn systemDmaAlloc(state: *architecture.CpuState) void { const pages: usize = @intCast((len + page_size - 1) / page_size); const max_phys: u64 = if (flags & abi.dma_below_4g != 0) (@as(u64, 4) << 30) else ~@as(u64, 0); - const phys = pmm.allocContiguous(pages, max_phys) orelse return fail(state); - // Reserve arena virtual space from the per-SPACE cursor, under the lock — - // sibling threads must hand out disjoint windows of the one shared arena. + // Frame allocation and cursor reservation under the lock — the pmm has no + // lock of its own, and sibling threads must hand out disjoint windows of + // the one shared arena. var base_v: u64 = 0; + var phys: u64 = 0; { const lock_flags = sync.enter(); const cursor = scheduler.addressSpaceDmaNextPtr(t.address_space) orelse { sync.leave(lock_flags); - for (0..pages) |i| pmm.free(phys + i * page_size); return fail(state); }; if (cursor.* == 0) cursor.* = dma_arena_base; base_v = cursor.*; if (base_v + pages * page_size > dma_arena_end) { sync.leave(lock_flags); - for (0..pages) |i| pmm.free(phys + i * page_size); // arena exhausted; give the frames back - return fail(state); + return fail(state); // arena exhausted } + phys = pmm.allocContiguous(pages, max_phys) orelse { + sync.leave(lock_flags); + return fail(state); + }; cursor.* = base_v + pages * page_size; sync.leave(lock_flags); } - // Zero through the physmap (the frames aren't mapped in the caller yet), then map. + // Zero through the physmap OUTSIDE the lock (the frames are still private to + // this call), then map page-by-page, each under a brief lock hold — the lock + // serializes the shared page-table walk (siblings on other cores walk the + // same tables), and per-page holds keep the mmap discipline: never pin every + // core's ticks across a multi-MiB operation. const kernel_view: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(phys)); @memset(kernel_view[0 .. pages * page_size], 0); - architecture.mapUserDmaInto(t.address_space, base_v, phys, pages * page_size); + for (0..pages) |i| { + const lock_flags = sync.enter(); + architecture.mapUserDmaInto(t.address_space, base_v + i * page_size, phys + i * page_size, page_size); + sync.leave(lock_flags); + } architecture.setSystemCallResult(state, base_v); // virtual address for the CPU architecture.setSystemCallResult2(state, phys); // physical address for the device } @@ -545,10 +562,14 @@ fn systemDmaFree(state: *architecture.CpuState) void { for (0..pages) |i| { const va = base_v + i * page_size; + // Per-page lock hold: the translate/unmap walks the shared page tables + // and pmm.free mutates the unlocked frame bitmap. + const lock_flags = sync.enter(); if (architecture.translate(t.address_space, va)) |phys| { architecture.unmapUserPageInto(t.address_space, va); pmm.free(phys); } + sync.leave(lock_flags); } architecture.setSystemCallResult(state, 0); } @@ -569,9 +590,15 @@ fn systemSharedMemoryCreate(state: *architecture.CpuState) void { const pages: usize = @intCast((len + page_size - 1) / page_size); if (pages == 0 or pages > maximum_shared_memory_pages) return fail(state); - // Reserve arena virtual space up front (per-SPACE cursor: sibling threads - // hand out disjoint windows), so a mapping failure needs no rollback. + // One locked section builds the whole named object — cursor reservation, + // frames, the refcounted object, the space's mapping record (M3: frames must + // outlive every MAPPING, not just every handle), and the handle — with full + // rollback, so no failure path ever touches the unlocked pmm/heap and no + // half-built region is ever reachable. var base_v: u64 = 0; + var phys: u64 = 0; + var handle: i64 = -1; + var shared_memory: *ipc.SharedMemoryObject = undefined; { const lock_flags = sync.enter(); const cursor = scheduler.addressSpaceSharedMemoryNextPtr(t.address_space) orelse { @@ -584,43 +611,46 @@ fn systemSharedMemoryCreate(state: *architecture.CpuState) void { sync.leave(lock_flags); return fail(state); // arena exhausted } + phys = pmm.allocContiguous(pages, ~@as(u64, 0)) orelse { + sync.leave(lock_flags); + return fail(state); + }; + shared_memory = ipc.createSharedMemory(phys, pages) orelse { + for (0..pages) |i| pmm.free(phys + i * page_size); + sync.leave(lock_flags); + return fail(state); + }; + if (!scheduler.recordSpaceMappingLocked(t.address_space, @ptrCast(shared_memory))) { + ipc.dropSharedMemoryReference(shared_memory); // last ref: frees the object AND its frames + sync.leave(lock_flags); + return fail(state); + } + ipc.retainSharedMemory(shared_memory); // the mapping's own reference + handle = ipc.installSharedMemoryHandle(t, shared_memory); + if (handle < 0) { + // Full rollback: unrecord the mapping, then drop both references — + // the second drop frees the object and its frames, all under the lock. + scheduler.removeSpaceMappingLocked(t.address_space, @ptrCast(shared_memory)); + ipc.dropSharedMemoryReference(shared_memory); + ipc.dropSharedMemoryReference(shared_memory); + sync.leave(lock_flags); + return fail(state); + } cursor.* = base_v + pages * page_size; sync.leave(lock_flags); } - const phys = pmm.allocContiguous(pages, ~@as(u64, 0)) orelse return fail(state); - // Zero through the physmap (the frames aren't mapped in the caller yet). + // Zero through the physmap OUTSIDE the lock (the frames are private until + // the handle is shared and the pages mapped), then map page-by-page under + // brief lock holds — the lock serializes the shared page-table walk against + // sibling threads, per-page so no core's tick starves (the mmap discipline). const kernel_view: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(phys)); @memset(kernel_view[0 .. pages * page_size], 0); - - const shared_memory = ipc.createSharedMemory(phys, pages) orelse { - for (0..pages) |i| pmm.free(phys + i * page_size); - return fail(state); - }; - // The creator's mapping holds its own reference, recorded on the space - // (docs/shared-fate-plan.md M3): frames must outlive every MAPPING, not just - // every handle — a sibling thread keeps using the region after the - // handle-holding thread dies. - { - const flags = sync.enter(); - if (!scheduler.recordSpaceMappingLocked(t.address_space, @ptrCast(shared_memory))) { - sync.leave(flags); - ipc.dropSharedMemoryReference(shared_memory); // last ref: frees the object AND its frames - return fail(state); - } - ipc.retainSharedMemory(shared_memory); - sync.leave(flags); + for (0..pages) |i| { + const lock_flags = sync.enter(); + architecture.mapUserSharedInto(t.address_space, base_v + i * page_size, phys + i * page_size, page_size); + sync.leave(lock_flags); } - const handle = ipc.installSharedMemoryHandle(t, shared_memory); - if (handle < 0) { - // The mapping record keeps one reference; drop only the creator's. The - // object (and frames) now live until this space is destroyed — the - // region was never named, so nothing else can reach it. - ipc.dropSharedMemoryReference(shared_memory); - return fail(state); - } - - architecture.mapUserSharedInto(t.address_space, base_v, phys, pages * page_size); architecture.setSystemCallResult(state, base_v); // virtual_address for the CPU architecture.setSystemCallResult2(state, @intCast(handle)); // capability handle to pass on } @@ -662,7 +692,14 @@ fn systemSharedMemoryMap(state: *architecture.CpuState) void { cursor.* = base_v + size; sync.leave(flags); } - architecture.mapUserSharedInto(t.address_space, base_v, shared_memory.phys, size); + // Map page-by-page under brief lock holds — the shared page-table walk must + // be serialized against sibling threads (the mmap discipline). + var page_index: usize = 0; + while (page_index * page_size < size) : (page_index += 1) { + const lock_flags = sync.enter(); + architecture.mapUserSharedInto(t.address_space, base_v + page_index * page_size, shared_memory.phys + page_index * page_size, page_size); + sync.leave(lock_flags); + } architecture.setSystemCallResult(state, base_v); } @@ -786,17 +823,24 @@ fn systemThreadSpawn(state: *architecture.CpuState) void { null else ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF); - const tid = spawnThreadSupervised(t.address_space, entry, stack_top, arg, t.priority, t.id, exit_endpoint, t.leader) orelse return fail(state); - architecture.setSystemCallResult(state, tid); + const tid = spawnThreadSupervised(t.address_space, entry, stack_top, arg, t.priority, t.id, exit_endpoint, t.leader); + if (tid == -ipc.ESRCH) return failErr(state, ipc.ESRCH); // dying group admits no member + if (tid < 0) return fail(state); + architecture.setSystemCallResult(state, @intCast(tid)); } /// Spawn a thread sharing `address_space`, taking the exit-endpoint reference under the **same** /// lock as the spawn (as `spawnProcessSupervised` does), so the thread cannot die before /// its reference exists. Returns the new thread id, or null on resource exhaustion. -fn spawnThreadSupervised(address_space: u64, entry: u64, stack_top: u64, arg: u64, priority: scheduler.Priority, supervisor: u32, exit_endpoint: ?*ipc.Endpoint, leader: u32) ?u32 { +fn spawnThreadSupervised(address_space: u64, entry: u64, stack_top: u64, arg: u64, priority: scheduler.Priority, supervisor: u32, exit_endpoint: ?*ipc.Endpoint, leader: u32) i64 { const flags = sync.enter(); defer sync.leave(flags); - const tid = scheduler.spawnUserLocked(address_space, entry, stack_top, arg, priority, "thread", supervisor, if (exit_endpoint) |e| @ptrCast(e) else null, leader) orelse return null; + // The spawn gate (docs/shared-fate-plan.md): a dying group admits no new + // member — checked under the same lock that would create it, and reported + // as -ESRCH (the process is as good as gone). retainAddressSpace inside + // spawnUserLocked backstops the same refusal. + if (scheduler.groupDyingLocked(address_space)) return -ipc.ESRCH; + const tid = scheduler.spawnUserLocked(address_space, entry, stack_top, arg, priority, "thread", supervisor, if (exit_endpoint) |e| @ptrCast(e) else null, leader) orelse return -1; if (exit_endpoint) |endpoint| endpoint.refcount += 1; // the thread holds it birth-to-death return tid; } @@ -1059,16 +1103,21 @@ pub fn killProcess(caller_id: u32, target_id: u32) i64 { // `supervisor` (the task that spawned it) grants nothing here // (docs/shared-fate-plan.md M1). Kernel tasks were -ESRCH'd above, so a // leader of 0 is unreachable. + // A group already dying answers from the latch's stash — the leader's slot + // may already be reaped while a condemned member is still enumerable. Same + // authority gate, then 0: the kill is already true, accepted and + // irrevocable (docs/shared-fate-plan.md). + if (scheduler.groupSupervisorLocked(target.address_space)) |group_supervisor| { + return if (group_supervisor == caller_id) 0 else -ipc.EPERM; + } const leader = if (target.leader == target.id) target else scheduler.taskByIdLocked(target.leader) orelse return -ipc.ESRCH; if (leader.supervisor != caller_id) return -ipc.EPERM; - // Whole-group kill (docs/shared-fate-plan.md). Already dying → the kill is - // already true: accepted and irrevocable either way, return 0. The caller is - // never a member (the leader's supervisor predates the group and cannot be - // inside it), so this always returns. - if (scheduler.groupDyingLocked(target.address_space)) return 0; + // Whole-group kill (docs/shared-fate-plan.md). The caller is never a member + // (the leader's supervisor predates the group and cannot be inside it), so + // this always returns. killGroupLocked(leader, .killed, null); return 0; } @@ -1130,6 +1179,9 @@ fn exitGroupCurrent(reason: abi.ExitReason) noreturn { const t = scheduler.current(); _ = sync.enter(); if (scheduler.groupDyingLocked(t.address_space)) terminateCurrentLocked(); + // The orelse fallback is unreachable today — a leader outlives its members + // (its thread_exit is refused; any leader death IS group death). If a future + // change ever made it reachable, it degrades to single-task semantics. const leader_task = if (t.leader == t.id) t else scheduler.taskByIdLocked(t.leader) orelse t; killGroupLocked(leader_task, reason, t); unreachable; // killGroupLocked never returns for an in-group trigger @@ -1149,6 +1201,7 @@ pub fn killCurrentProcess(reason: abi.ExitReason) noreturn { // (docs/shared-fate-plan.md). `fault_kill_count` is per faulting GROUP. if (scheduler.groupDyingLocked(t.address_space)) terminateCurrentLocked(); fault_kill_count += 1; + // orelse fallback unreachable today: a leader outlives its members (see exitGroupCurrent). const leader_task = if (t.leader == t.id) t else scheduler.taskByIdLocked(t.leader) orelse t; killGroupLocked(leader_task, reason, t); unreachable; // killGroupLocked never returns for an in-group trigger diff --git a/system/kernel/scheduler.zig b/system/kernel/scheduler.zig index 27b99c0..6be85f9 100644 --- a/system/kernel/scheduler.zig +++ b/system/kernel/scheduler.zig @@ -359,6 +359,34 @@ pub fn groupDyingLocked(root: u64) bool { return false; } +/// The stashed supervisor of `root`'s DYING group, or null if the group is not +/// dying. Answers the process_kill authority question during the condemned +/// window, when the leader's task slot may already be reaped. Lock held. +pub fn groupSupervisorLocked(root: u64) ?u32 { + for (&address_space_refs) |*entry| { + if (entry.count != 0 and entry.root == root) { + return if (entry.dying) entry.group_supervisor else null; + } + } + return null; +} + +/// Undo a `recordSpaceMappingLocked` — the rollback half for a caller whose +/// later step failed. Removes one matching slot; the caller drops the reference +/// it had transferred. Lock held. +pub fn removeSpaceMappingLocked(root: u64, object: *anyopaque) void { + for (&address_space_refs) |*entry| { + if (entry.count == 0 or entry.root != root) continue; + for (&entry.mappings) |*slot| { + if (slot.* == object) { + slot.* = null; + return; + } + } + return; + } +} + /// The whole static task pool, for process.zig's group fan-out — which must scan /// members under the lock it already holds. Slots may be `.free`/`.reaping`; /// callers filter by state and must not hold pointers past the lock. diff --git a/system/kernel/tests.zig b/system/kernel/tests.zig index 25488cc..9ef99ae 100644 --- a/system/kernel/tests.zig +++ b/system/kernel/tests.zig @@ -3233,11 +3233,14 @@ fn awaitExitBadge(endpoint: *ipcsync.Endpoint) u64 { /// A group death is one notification, badged with the LEADER, arriving only /// after every member (and the address space) is gone — asserted by every -/// shared-fate case below. -fn checkGroupDead(me: u32, leader: u32, badge: u64, reason: abi.ExitReason) void { +/// shared-fate case below. "Exactly one": after a settling sleep, a second +/// (buggy, double-posted) badge would still sit in the endpoint's notify ring. +fn checkGroupDead(me: u32, leader: u32, badge: u64, reason: abi.ExitReason, endpoint: *ipcsync.Endpoint) void { check("one exit notification, badged with the leader", badge == abi.notify_badge_bit | abi.notify_exit_bit | leader); check("the leader's recorded reason is the group reason", process.exitReasonOf(me, leader) == @intFromEnum(reason)); check("no group member is listed after the death", !groupListed(leader)); + scheduler.sleep(200); + check("exactly one exit notification (ring drained)", endpoint.notify_head == endpoint.notify_tail); } fn threadFaultGroupTest(boot_information: *const BootInformation) void { @@ -3258,7 +3261,7 @@ fn threadFaultGroupTest(boot_information: *const BootInformation) void { return; } const badge = awaitExitBadge(endpoint); - checkGroupDead(me, child, badge, .segmentation_fault); + checkGroupDead(me, child, badge, .segmentation_fault, endpoint); check("one fault kill for the whole group", process.fault_kill_count == 1); const deadline = architecture.millis() + 5000; while (scheduler.liveStackBytes() > stacks_base and architecture.millis() < deadline) scheduler.yield(); @@ -3293,7 +3296,7 @@ fn killThreadedGroupTest(boot_information: *const BootInformation) void { scheduler.sleep(100); // let the worker really be running on another core check("the supervisor's kill is accepted", process.killProcess(me, child) == 0); const badge = awaitExitBadge(endpoint); - checkGroupDead(me, child, badge, .killed); + checkGroupDead(me, child, badge, .killed, endpoint); check("the leader's claim was released before the notification", devices_broker.ownerOf(0) == null); check("the worker's claim was released before the notification", devices_broker.ownerOf(1) == null); check("a dead group stays dead (-ESRCH)", process.killProcess(me, child) == -ipcsync.ESRCH); @@ -3319,7 +3322,7 @@ fn killViaWorkerTidTest(boot_information: *const BootInformation) void { check("a non-supervisor aiming at the worker is refused (-EPERM)", process.killProcess(me + 12345, worker) == -ipcsync.EPERM); check("the supervisor's kill aimed at the WORKER id is accepted", process.killProcess(me, worker) == 0); const badge = awaitExitBadge(endpoint); - checkGroupDead(me, child, badge, .killed); + checkGroupDead(me, child, badge, .killed, endpoint); result(); } @@ -3339,7 +3342,7 @@ fn racingTriggersTest(boot_information: *const BootInformation) void { return; } const badge = awaitExitBadge(endpoint); - checkGroupDead(me, child, badge, .segmentation_fault); + checkGroupDead(me, child, badge, .segmentation_fault, endpoint); check("two racing faults counted as ONE group kill", process.fault_kill_count == 1); result(); } @@ -3359,7 +3362,7 @@ fn exitGroupTest(boot_information: *const BootInformation) void { return; } const badge = awaitExitBadge(endpoint); - checkGroupDead(me, child, badge, .aborted); + checkGroupDead(me, child, badge, .aborted, endpoint); result(); } @@ -3381,7 +3384,7 @@ fn threadTestMarkerCase(boot_information: *const BootInformation, case_name: []c return; } const badge = awaitExitBadge(endpoint); - checkGroupDead(me, child, badge, .exited); + checkGroupDead(me, child, badge, .exited, endpoint); result(); }