diff --git a/build.zig b/build.zig index b6e6dea..ae18855 100644 --- a/build.zig +++ b/build.zig @@ -926,6 +926,22 @@ pub fn build(b: *std.Build) void { }); test_step.dependOn(&b.addRunArtifact(time_tests).step); + // runtime.Thread's lock/condvar state machines (Mutex/Condition/RwLock/WaitGroup). Its + // Futex seam falls back to std.Thread.Futex off the danos target, so the tests exercise + // them with real host threads (docs/threading-plan.md M11). Like time.zig it pulls in + // system.zig (syscall wrappers), which needs the `abi` module. + const thread_tests = b.addTest(.{ + .root_module = b.createModule(.{ + .root_source_file = b.path("library/runtime/thread.zig"), + .target = target, + .optimize = optimize, + .imports = &.{ + .{ .name = "abi", .module = abi_module }, + }, + }), + }); + test_step.dependOn(&b.addRunArtifact(thread_tests).step); + // Convenience: `zig build gen-xkeyboard-config` regenerates the layout tables from the // vendored data (offline). `fetch` (the network step) stays a manual script run. const gen_xkb = b.addSystemCommand(&.{ "python3", "tools/make-xkeyboard-config.py", "generate" }); diff --git a/docs/README.md b/docs/README.md index 5778206..96663c8 100644 --- a/docs/README.md +++ b/docs/README.md @@ -107,7 +107,7 @@ Start with the north star: - **[threading.md](threading.md) — threads, the std-shaped way.** **Built** (M1–M6): `runtime.Thread` mirrors `std.Thread`'s API (spawn/join/detach, Mutex/Condition/ Semaphore) over a **private** thread ABI — several tasks sharing one address space via - a `thread_spawn` syscall, futex-backed blocking, aspace refcounting. Why it's the + a `thread_spawn` syscall, futex-backed blocking, address-space refcounting. Why it's the native type and not literal `std.Thread` (the [private ABI](syscall.md)), and why threads stay a narrow opt-in against the [resilience](resilience.md) default. Build plan + gates: [threading-plan.md](threading-plan.md). diff --git a/docs/display-v2-plan.md b/docs/display-v2-plan.md index f08d75d..ca97ce4 100644 --- a/docs/display-v2-plan.md +++ b/docs/display-v2-plan.md @@ -62,7 +62,7 @@ is the only backend), and `zig build test` stays green. - [x] [abi.zig](../system/abi.zig): `shm_create` (34) / `shm_map` (35) syscalls + a `shm_test` service id. Handlers in process.zig: `shm_create(len)` allocates contiguous, zeroed, **cacheable** frames, wraps them in a refcounted object, installs a capability - handle, maps them into the caller's shm arena → returns vaddr + handle; `shm_map(cap)` + handle, maps them into the caller's shm arena → returns virtual_address + handle; `shm_map(cap)` maps the same physical pages into the receiver. Reclaimed on death (see below). - [x] The capability core (ipc-synchronous.zig) is now **kind-tagged**: `scheduler.Task`'s handle table holds `HandleObject{kind, ptr}`; `closeHandles` and `shareCapability` diff --git a/docs/display-v2.md b/docs/display-v2.md index 42dedad..ada8c3e 100644 --- a/docs/display-v2.md +++ b/docs/display-v2.md @@ -77,9 +77,9 @@ deferred (docs/display.md, "What v1 does not do"). v2 builds it: the natural gen of M13 capability-passing from *endpoints* to *memory objects* — ``` -shm_create(len) -> {handle, vaddr} // a shareable, page-aligned RAM region +shm_create(len) -> {handle, virtual_address} // a shareable, page-aligned RAM region … pass `handle` as the send_cap on an ipc_call … -shm_map(cap) -> vaddr // the receiver maps the same physical pages +shm_map(cap) -> virtual_address // the receiver maps the same physical pages ``` The payoff is leverage: the **same** primitive unlocks **both** native GPU drivers *and* diff --git a/docs/display.md b/docs/display.md index 3f4fd1b..b4c6dad 100644 --- a/docs/display.md +++ b/docs/display.md @@ -220,8 +220,8 @@ both are clean additions behind the interfaces v1 establishes. to render into its *own* buffer and hand the compositor a *reference*, not a stream of commands. That needs the missing cross-process shared-memory primitive — best built as the natural generalization of the existing M13 [capability passing](driver-model.md) - from *endpoints* to *memory objects* (`shm_create(len) → {cap, vaddr}`, pass `cap` on - an `ipc_call`, receiver `shm_map(cap) → vaddr`). v1 avoids it because server-owned + from *endpoints* to *memory objects* (`shm_create(len) → {cap, virtual_address}`, pass `cap` on + an `ipc_call`, receiver `shm_map(cap) → virtual_address`). v1 avoids it because server-owned surfaces already prove the whole pipeline. - **Runtime mode-setting (a native backend).** Detecting the EDID mode list and changing diff --git a/docs/driver-model.md b/docs/driver-model.md index 9016e1b..23e1946 100644 --- a/docs/driver-model.md +++ b/docs/driver-model.md @@ -240,8 +240,8 @@ once per page, maps writeback-cached, and never reveals a physical address. **The fix.** ``` -dma_alloc(len, flags) -> vaddr (rax), paddr (rdx) -dma_free(vaddr, len) -> 0 +dma_alloc(len, flags) -> virtual_address (rax), physical_address (rdx) +dma_free(virtual_address, len) -> 0 flags: dma_coherent (1) uncacheable; the default and the only one that's portable dma_wc (2) write-combining — needs PAT programmed; for framebuffers diff --git a/docs/drivers.md b/docs/drivers.md index 74ea9bd..44749d2 100644 --- a/docs/drivers.md +++ b/docs/drivers.md @@ -74,7 +74,7 @@ The driver syscall numbers (`system/abi.zig`) with the device types they carry |---|------|---------| | 11 | `device_enumerate(buf, max) -> total` | Snapshot the device table | | 12 | `device_claim(id) -> ok` | Take **exclusive** ownership | -| 13 | `mmio_map(id, res_idx) -> vaddr` | Map a claimed device's register window | +| 13 | `mmio_map(id, res_idx) -> virtual_address` | Map a claimed device's register window | | 14 | `irq_bind(id, res_idx, endpoint)` | Deliver that device's IRQ as a notification | | 15 | `irq_ack(id, res_idx)` | Re-arm the IRQ after servicing the device | | 16 | `device_register(parent_id, desc) -> id` | Publish a child of a device you claimed | diff --git a/docs/threading-plan.md b/docs/threading-plan.md index 7b06344..3f245c8 100644 --- a/docs/threading-plan.md +++ b/docs/threading-plan.md @@ -105,28 +105,28 @@ first unchecked box. ## M1 — Address-space refcount (kernel foundation, no API, no behaviour change) ✅ The one invariant change threads require, landed and proven **before** anything shares -an address space. Today aspace is 1:1 with a task and teardown destroys it on any user +an address space. Today address space is 1:1 with a task and teardown destroys it on any user task's exit; make destruction happen on the **last** exit. - [x] A refcount keyed by the address-space root, held in `scheduler.zig` - (`aspace_refs`): `retainAspace` takes a reference in `spawnUserLocked` (on the + (`address_space_refs`): `retainAddressSpace` takes a reference in `spawnUserLocked` (on the success path, after the slot + stack are secured), all under the big kernel lock. - [x] Both task-teardown paths ([scheduler.zig](../system/kernel/scheduler.zig): `exitUserLocked` and `destroyTaskLocked`) call `releaseAspace`, which decrements and only `destroyAddressSpace`s at **zero**; an unretained space (hand-built test spaces) is destroyed directly, preserving prior behaviour. -- [x] `-Dtest-case=aspace-refcount`: spawn and reap several ring-3 processes in sequence - and assert (via test-observable `liveAspaceCount`/`aspaceDestroyCount`) that the +- [x] `-Dtest-case=address-space-refcount`: spawn and reap several ring-3 processes in sequence + and assert (via test-observable `liveAddressSpaceCount`/`addressSpaceDestroyCount`) that the live-space count returns to **baseline** and destructions advance by exactly that many — each space destroyed exactly once, no leak, no double-free. (Refcount observables, not raw frame counts, since kernel stacks are still leaked on exit.) -**Gate (met):** `python3 test/qemu_test.py aspace-refcount` passes -(`aspace-refcount: spaces released to baseline ok` → `DANOS-TEST-RESULT: PASS`), and the +**Gate (met):** `python3 test/qemu_test.py address-space-refcount` passes +(`address-space-refcount: spaces released to baseline ok` → `DANOS-TEST-RESULT: PASS`), and the full guardrail set passes unchanged — 13/13 (`smoke`, `sched`, `priority`, `smp`, `affinity`, `process`, `process-kill`, `supervision`, `fault-recovery`, `vfs-client-death`, `ipc`, `ipc-cap`, `display-service`); default `zig build` clean, -`zig build test` green. The reframing is invisible until an aspace is actually shared. +`zig build test` green. The reframing is invisible until an address space is actually shared. ## M2 — `thread_spawn` + `thread_exit`: a thread runs in the shared address space ✅ @@ -135,7 +135,7 @@ space and exits cleanly. - [x] [abi.zig](../system/abi.zig): `thread_spawn = 37`, `thread_exit = 38`. Handlers in process.zig; `thread_spawn` calls `scheduler.spawnThread` (shares the caller's - aspace, `retainAspace`); `thread_exit` ends the task like a process `exit(0)` + address space, `retainAddressSpace`); `thread_exit` ends the task like a process `exit(0)` (`terminateCurrent` → `releaseAspace`). The closure pointer is delivered in the new thread's **rdi** via a new `jump_to_user_arg` asm path (`t.user_arg`, 0 for a process) — no naked runtime asm. @@ -152,14 +152,14 @@ space and exits cleanly. address space. **Gate (met):** `python3 test/qemu_test.py thread-spawn` passes -(`thread-test: child ran in shared aspace ok` → `DANOS-TEST-RESULT: PASS`); guardrail set +(`thread-test: child ran in shared address space ok` → `DANOS-TEST-RESULT: PASS`); guardrail set 16/16 green (incl. `args`/`init`/`process`, which exercise the new `jump_to_user_arg` -process path with arg 0) plus `aspace-refcount`; `zig build` clean, `zig build test` +process path with arg 0) plus `address-space-refcount`; `zig build` clean, `zig build test` green. > **Note (deferred to M3+):** the mmap arena is per-*task* (`heap_next`), so two threads -> in one aspace that both `mmap` would collide. Fine for M2 (only the parent maps, for the -> child's stack); make the arena per-aspace and the runtime heap thread-safe alongside the +> in one address space that both `mmap` would collide. Fine for M2 (only the parent maps, for the +> child's stack); make the arena per-address-space and the runtime heap thread-safe alongside the > `Mutex` work (M5). ## M3 — `join` + `detach` + real parallelism ✅ @@ -184,7 +184,7 @@ green. **Gate (met):** `python3 test/qemu_test.py thread-join` passes (`thread-test: join ok` → `DANOS-TEST-RESULT: PASS`), robust across 4 runs; guardrail 17/17 green (incl. `smp`, `affinity`, `process-kill`, and `args`/`init`/`process` on the exit-endpoint spawn path) -plus `aspace-refcount`/`thread-spawn`; `zig build` clean, `zig build test` green. +plus `address-space-refcount`/`thread-spawn`; `zig build` clean, `zig build test` green. > **Note (deferred):** a detached thread's stack is freed only at process exit (not by the > reaper on thread exit) — kernel user-stack tracking + reclaim is a later refinement. And @@ -212,7 +212,7 @@ plus `aspace-refcount`/`thread-spawn`; `zig build` clean, `zig build test` green **Gate (met):** `python3 test/qemu_test.py thread-futex` passes, robust across 3 runs — the case's **ordered** regex asserts `waiting → waking → woke → PASS` on the serial stream (the handoff proof), and `thread-futex: timeout ok` confirms the timeout. -Guardrail 18/18 green (incl. `sleep`/`event`/`ipc` blocking paths) + `aspace-refcount`, +Guardrail 18/18 green (incl. `sleep`/`event`/`ipc` blocking paths) + `address-space-refcount`, `thread-spawn`, `thread-join`; `zig build` clean, `zig build test` green. > **Note:** the kernel test checks only the freshest verdict marker via `bufferHas` (the @@ -252,11 +252,11 @@ green. - [x] `getCurrentId` via a small `thread_self = 42` syscall (`runtime.Thread.getCurrentId` returns the kernel task id). **Per-thread `threadlocal` TLS is deferred** — no - consumer needs it, and it would require context-switching `fs.base` per task (real - kernel + per-switch cost) for an unused feature; threaded binaries have run fine + consumer needs it, and it would require context-switching the thread pointer per task + (real kernel + per-switch cost) for an unused feature; threaded binaries have run fine without it through M2–M5. threading.md's TLS reasoning already scoped it as deferred-unless-needed. When a consumer appears, the shape is: `thread_spawn` - allocates a per-thread TLS block, sets `fs.base`, and the context switch saves/ + allocates a per-thread TLS block, sets the thread pointer, and the context switch saves/ restores it. - [x] `RwLock` / `WaitGroup` deferred (no consumer yet); they slot onto the same `Futex`/`Mutex`/`Condition` when wanted. @@ -279,8 +279,11 @@ green. cross-core parallelism, futex, and `Mutex`/`Condition`/`Semaphore`, all over a private thread ABI behind the runtime. -**Phase 2 (M7–M11): planned below** — hardening the deferred parts so threads are safe -for real workloads and reclaimed like everything else danos owns. +**Phase 2 (M7–M11): built.** Thread-safe allocation (M7), a task reaper that reclaims dead +tasks' kernel stacks (M8), endpoint-free `thread_join` (M9), the per-thread thread pointer (M10), +and `RwLock`/`WaitGroup` + host-testable sync (M11). Two things stay deferred by design +(no consumer): the Zig `threadlocal` *compiler* layer (M10) and detached-thread user-stack +reclaim (M9) — both noted in place. --- @@ -290,112 +293,170 @@ The organising principle, so Phase 2 reinforces danos's goals rather than erodin - **Everything a thread owns is reclaimed on process death.** Thread stacks, TLS blocks, and futex words live in the process's **address space**, and the kernel's per-process - state is keyed by the aspace root — so the M1 refcount + `destroyAddressSpace` already + state is keyed by the address-space root — so the M1 refcount + `destroyAddressSpace` already free all of it when the last thread exits. A crashed or killed threaded process leaves - **nothing** behind. Phase 2 closes the one thing that is *not* aspace-owned — the + **nothing** behind. Phase 2 closes the one thing that is *not* address-space-owned — the per-task **kernel** stack (kernel heap) — with a reaper (M8). This is the [resilience](resilience.md) restart guarantee, extended to threads. - **Kernel owns mechanism; the runtime owns policy.** The kernel maps pages, saves/ - restores `fs.base`, and reaps dead tasks; the runtime decides allocation, TLS layout, + restores the thread pointer, and reaps dead tasks; the runtime decides allocation, TLS layout, and lock algorithms. Every new kernel entry stays a private syscall behind the runtime ([syscall.md](syscall.md)) — the ABI stays renumberable. - **The process is still the isolation and restart boundary.** Threads share fate within one process; Phase 2 never adds a way for one process to reach into another (the cross-process futex stays explicitly out of scope, below). -### M7 — Thread-safe allocation (the correctness gap) +### M7 — Thread-safe allocation (the correctness gap) ✅ Today the mmap arena cursor is per-*task* and the runtime heap is unlocked, so two threads in one process that both allocate corrupt each other. The thread *machinery* avoids this (closure on the stack, stacks mmap'd only by the spawner), but real -multi-threaded code would hit it. Close it: +multi-threaded code would hit it. Closed it: -- [ ] **Kernel — per-address-space mmap arena.** Grow M1's `aspace_refs` entry into a - small address-space object holding the `mmap`/`mmio` arena cursors (moved off - `Task`); `systemMmap`/`mmio_map` bump the *aspace's* cursor under the big lock, so - sibling threads get disjoint, serialized grants. Freed at refcount zero, so the - cursors vanish with the process. -- [ ] **Runtime — thread-safe heap.** Guard the allocator with a `Thread.Mutex`, gated on +- [x] **Kernel — per-address-space mmap arena.** Grew M1's `address_space_refs` entry into the + per-address-space object holding the `mmap`/`mmio` arena cursors (moved off `Task`); + `scheduler.addressSpaceMmapNextPtr`/`addressSpaceDeviceMapNextPtr` expose them. `systemMmap` + reserves a disjoint range under a *brief* lock, then maps **per page** under a + short-held lock — not the whole grant — because the big lock is held with interrupts + disabled, so pinning it across a multi-MiB memset+map froze other cores (it timed + the `affinity` scenario out mid-bring-up). Freed at refcount zero, so the cursors + vanish with the process. +- [x] **Runtime — thread-safe heap.** The allocator's two free-list mutators + (`rawAlloc`/`rawFree`) take a `Thread.Mutex`, gated on `!@import("builtin").single_threaded` so single-threaded binaries compile it out and - pay nothing. (The heap grows via mmap, now safe per above.) -- [ ] `-Dtest-case=thread-alloc` (`smp: 4`): N threads each do many `alloc`/`free` of - varied sizes, write a per-thread pattern, verify it, and free; assert every block - round-trips intact and all memory returns — no corruption under concurrent - allocation. A direct check confirms two threads' concurrent `mmap`s are disjoint. + pay nothing. Uncontended acquisition is a single CAS (no syscall). +- [x] `-Dtest-case=thread-alloc` (`smp: 4`): 4 threads each do 500 `alloc`/fill/verify/ + `free` cycles of varied sizes; each block is filled with a per-thread pattern and + verified before free, so any overlap between concurrent allocations is caught. -**Gate:** `thread-alloc` passes; guardrail + all `thread-*` cases green. +**Gate (met):** `thread-alloc` passes (3× non-flaky); full guardrail 23/23 green, +`zig build`/`zig build test` clean. -### M8 — The task reaper (cleanup + resilience) +> **Also fixed here:** the `affinity` guardrail's fixed-count busy-loop (`while (spins < +> 3e9)`) had codegen-dependent wall-time — adding a function to `tests.zig` flipped how +> the optimiser compiled it, swinging affinity from ~4 s to ~63 s and timing it out. +> Reworked it (and the settle loop) to wait on the wall clock instead, so its duration is +> independent of unrelated code changes. -A dead task's **kernel** stack is currently leaked ("no reaper yet") — every process -*and* thread death loses one, so a crash loop bleeds kernel memory. A reaper fixes it and -serves the [resilience](resilience.md) restart goal directly: +### M8 — The task reaper (cleanup + resilience) ✅ -- [ ] A dying task cannot free the kernel stack it runs on, so it hands itself to a - **reap list** and switches away; the kernel stack (and, for a detached thread, its - user stack) is reclaimed from another context — a low-priority reaper step drained - on the scheduler tick and when a core goes idle. Extends the existing - `reap_task_hook`/`destroyTaskLocked` path rather than inventing a parallel one. -- [ ] `-Dtest-case=task-reap`: spawn and exit many threads and processes; assert the - kernel-heap free bytes (a new test observable) return to **baseline** — kernel - stacks reclaimed, no leak — and that the `fault-recovery`/kill paths reclaim too. +A dead task's **kernel** stack was leaked ("no reaper yet") — every process *and* thread +death lost one, so a crash loop bled kernel memory. The reaper fixes it and serves the +[resilience](resilience.md) restart goal directly: -**Gate:** `task-reap` passes; `fault-recovery`, `supervision`, `process-kill`, -`aspace-refcount` still green. +- [x] A dying task cannot free the kernel stack it runs on, so `exit()`/`exitUserLocked` + record it in a **per-core `reap_after_switch` slot** and switch away; the task that + resumes on that core frees the stack in `switchTo`'s tail (it's on its own stack, the + big lock is still held so the slot can't have been reused). A **tick-time drain** + (`reapKillPendingLocked`) is the safety net for the case where the next task is + *fresh* (enters via the trampoline, bypassing `switchTo`'s tail). A task killed while + *not* running is freed immediately in `destroyTaskLocked`. A `live_stack_bytes` + counter is the observable. *(Detached-thread user-stack reclaim moves to M9, which + adds the joinable/detached flag.)* +- [x] `-Dtest-case=task-reap` (`smp: 4`): spawn and kill 12 processes; poll the + test-observable `scheduler.liveStackBytes()` until it returns to **baseline** (a + correct reaper gets there in a few ms; a genuine leak times out) — every kernel + stack reclaimed, no leak. Threads exit through the same `exitUserLocked`, so covered. + +**Gate (met):** `task-reap` passes (5× isolated + 2× in the full batch); `fault-recovery`, +`supervision`, `process-kill`, `address-space-refcount`, `smp`, `affinity` all still green (24/24 +full guardrail); `zig build`/`zig build test` clean. + +> **Bug found + fixed here (touches every context switch):** the post-`switchContext` reap +> first read the `pc` **parameter**, but a task that migrated cores carries a *stale* `pc` +> in its saved `switchTo` frame — so it read the wrong core's slot and freed a live stack +> (a #GP under SMP). Fixed to re-fetch `thisCpu()` after the switch (the switch only swaps +> stacks on the current core). ### M9 — Futex-completion join (retire the per-thread endpoint) With the reaper (M8) able to act *after* a thread is fully off its stack, migrate `join` to the std shape and drop M3's per-thread exit endpoint: -- [ ] `thread_spawn` takes a user **completion word** (in the `Thread` handle's memory) - and a joinable/detached flag. On reap the kernel writes 0 to that word and - `futex_wake`s it (a `CLONE_CHILD_CLEARTID` equivalent — safe now the thread is off - its stack). `join` = `futex_wait` on the word, then `munmap` the stack; a - **detached** thread's stack is `munmap`ped by the reaper instead. No IPC endpoint - per thread. -- [ ] `thread-join` passes on the new path; a check confirms joining N threads creates no - per-thread endpoints (handle count stable). +- [x] A **`thread_join(tid)` syscall** (not a user futex word): it blocks the caller until + the task with id `tid` exits, and the exit paths call `wakeJoinersLocked`. `join` + only reclaims the joined thread's **user** stack, which the thread vacates the moment + it enters the kernel to exit — so waking at *exit* time (not reap time) is safe, and + no reaper/address-space juggling or user-memory write is needed. This is equally + std-shaped (like `pthread_join`) and much simpler/safer than the planned + reaper-written completion word. `thread_spawn` no longer takes an exit endpoint (the + runtime passes `no_cap`); the per-thread IPC endpoint is gone. +- [x] `thread-join` passes on the new path, and its join mode now runs **40 spawn+join + cycles** — under the old per-thread-endpoint scheme those leaked handles would + exhaust the 16-slot handle table; here they all succeed, proving join is endpoint-free. -**Gate:** `thread-join`/`thread-mutex` green on futex-completion join; guardrail green. +**Gate (met):** `thread-join` passes (3× isolated) on the `thread_join` path; full +guardrail 26/26 (incl. `process-kill`, `supervision`, `fault-recovery`, `task-reap`); +`zig build`/`zig build test` clean. -### M10 — Per-thread TLS (`threadlocal`) +> **Reaper hardened here (fixes an M8 flake).** M8's single per-core reap slot could be +> *overwritten* by a second death on that core before the first drained (a fresh-task/SMP +> timing window) — an intermittent one-stack leak (`task-reap` flaked ~20%). Replaced it +> with a per-core reap **list** plus a `.reaping` task state so a pending slot can't be +> reused before its stack is freed. `task-reap` now 11/11 isolated + 2× in the batch. -Give each thread its own `threadlocal` storage — the piece self-hosting Zig -([zig-self-hosting.md](zig-self-hosting.md)) will force: +> **Deferred:** detached-thread **user-stack** reclaim (still freed at process exit, as in +> M3). Doing it in the reaper needs the saved address space + stack range and a +> translate/unmap in a not-currently-loaded address space — real complexity for a bounded leak. +> A follow-up when a consumer needs it. -- [ ] **Runtime** allocates a per-thread TLS block from the binary's `PT_TLS` template - (linker symbols: copy `.tdata`, zero `.tbss`, variant-II TCB self-pointer) and hands - its thread pointer to `thread_spawn`; the main thread sets its own via a new - `set_thread_pointer` syscall in `_start`. The block is aspace memory → reclaimed on - teardown. -- [ ] **Kernel** stores `fs_base` on `Task`, loads it at first entry and restores it on - context switch only when it changes (the same conditional-load pattern as CR3). - `getCurrentId` can then read a TLS self-slot instead of a syscall. -- [ ] `-Dtest-case=thread-tls`: two threads each write and read their own `threadlocal` - slot with no cross-talk, and observe distinct `getCurrentId`. +### M10 — Per-thread TLS: the thread-pointer mechanism ✅ + +Give each thread its own thread pointer and private TLS storage — the foundation +self-hosting Zig ([zig-self-hosting.md](zig-self-hosting.md)) will build `threadlocal` on. + +- [x] **Kernel** stores `thread_pointer` on `Task` and restores it on every context switch + **only when it changes** (the same conditional-load discipline as CR3; + `architecture.setThreadPointer` → `wrmsr IA32_FS_BASE` on x86_64). A + `set_thread_pointer(addr)` = 44 syscall sets the caller's `thread_pointer` and loads it + now. The kernel never touches FS, so there is no swapgs complication. +- [x] **Runtime** lays a small per-thread TLS block at the top of each thread's stack + (self-pointer at `%fs:0` + scratch slots) and the thread trampoline calls + `set_thread_pointer` before any user code — so every spawned thread has a private, + switch-stable thread pointer. Reclaimed with the stack. +- [x] `-Dtest-case=thread-tls` (`smp: 4`): two threads each write a unique marker to their + own `%fs:8` slot and — after both have written — read it back; a shared (non-per-thread) + FS base would clobber one and cause cross-talk. Both read their own marker → pass. + +**Gate (met):** `thread-tls` passes (3×); full guardrail 25/25 (the switch-time thread-pointer +restore touches every context switch); `zig build`/`zig build test` clean. + +> **Deferred: the Zig `threadlocal` *compiler* layer.** Real `threadlocal` variables need +> the ELF **variant-II TLS** surface — `.tdata`/`.tbss` sections + a `PT_TLS` program header +> in `user.ld`, a runtime that copies the template with exact negative-offset layout, and +> the `.large`-code-model TLS section names — a high-uncertainty lift for a feature with +> **no consumer today** (threading.md scopes it "only if a consumer needs it"). What lands +> here is the load-bearing piece — the per-thread thread pointer, context-switched — so adding the +> compiler layer later is purely runtime+linker work on top, no kernel change. `getCurrentId` +> stays the `thread_self` syscall (M6) rather than an fs self-slot (which would need the +> main thread's TLS set up in `_start` too). **Gate:** `thread-tls` passes; full `thread-*` suite + guardrail green. -### M11 — `RwLock`, `WaitGroup`, and host-testable sync +### M11 — `RwLock`, `WaitGroup`, and host-testable sync ✅ -- [ ] `runtime.Thread.RwLock` and `WaitGroup` on the existing `Futex`/`Mutex`/ - `Condition`. -- [ ] A compile-time `Futex` seam: syscalls on the danos target, a host-backed impl under - `zig build test`, so the `Mutex`/`Condition`/`RwLock` state machines run as host - unit tests (fast iteration, no QEMU). -- [ ] `-Dtest-case=thread-rwlock` (`smp: 4`): many readers + writers over an `RwLock` keep - an invariant (a reader never observes a half-written value); host tests cover the - lock transitions. +- [x] `runtime.Thread.RwLock` (reader-preferring: `>0` readers / `-1` writer / `0` free, + with `lock`/`tryLock`/`unlock` + `lockShared`/`tryLockShared`/`unlockShared`) and + `WaitGroup` (`start`/`finish`/`wait`), both on the existing `Mutex`/`Condition`. +- [x] A compile-time `Futex` seam gated on `builtin.os.tag == .freestanding`: the futex + syscalls on danos, a spin+yield mock off-target (Zig 0.16 has no `std.Thread.Futex`; + `wake` is a no-op since the state machines re-check). `thread.zig` is wired into + `zig build test`, so `Mutex`/`RwLock`/`WaitGroup` run as **host unit tests** with real + `std.Thread` threads (`test` blocks only compile under test). +- [x] `-Dtest-case=thread-rwlock` (`smp: 4`): 2 writers set both halves of a value under + the exclusive lock while 3 readers check the halves match under the shared lock — + zero half-write observations across ~150k reads. Host tests cover the Mutex, + RwLock, and WaitGroup state machines. -**Gate:** host `zig build test` covers the sync primitives; `thread-rwlock` passes; -guardrail green. +**Gate (met):** `zig build test` covers the sync primitives (host threads); `thread-rwlock` +passes (3×); full Done gate **26/26** (whole `thread-*` suite + guardrail); `zig build` +clean. --- ## Deferred (explicitly not in this plan) -- **Cross-process shared-memory futex** — the `(aspace, vaddr)` key can become a +- **Cross-process shared-memory futex** — the `(address_space, virtual_address)` key can become a physical-address key so two processes share a futex through an [shm](display-v2.md) region. Not needed for intra-process threads. - **Per-thread priorities / affinity distinct from the process** — threads inherit the diff --git a/docs/threading.md b/docs/threading.md index abcd350..35ba304 100644 --- a/docs/threading.md +++ b/docs/threading.md @@ -2,12 +2,13 @@ A note on danos **threads** — several tasks sharing one address space — provided by a `runtime.Thread` type that mirrors the shape of Zig's `std.Thread` while keeping every -kernel entry behind the [runtime](../library/runtime). **Built** (M1–M6, see -[threading-plan.md](threading-plan.md)): `spawn`/`join`/`detach`, cross-core -parallelism, a futex (`futex_wait`/`futex_wake`), and a futex-backed -`Mutex`/`Condition`/`Semaphore`, plus `getCurrentId`/`currentCore`. Deferred by design -(no consumer yet): per-thread `threadlocal` TLS, `RwLock`/`WaitGroup`, and migrating -`join` to a futex completion word — see the plan's M5/M6 notes. The analysis is against +kernel entry behind the [runtime](../library/runtime). **Built** (M1–M11, see +[threading-plan.md](threading-plan.md)): `spawn`/`join`/`detach`, cross-core parallelism, +a futex, `Mutex`/`Condition`/`Semaphore`/`RwLock`/`WaitGroup`, `getCurrentId`/`currentCore`, +per-thread thread-pointer TLS, thread-safe allocation, and a task reaper that reclaims dead +tasks' kernel stacks. Deferred by design (no consumer yet): the Zig `threadlocal` +*compiler* layer (the per-thread thread pointer is in place, so it's runtime+linker work on top) and +detached-thread user-stack reclaim — see the plan's M9/M10 notes. The analysis is against **Zig 0.16** (the pinned toolchain); `std.Thread`'s internals move between releases, so treat upstream shapes as "0.16.x." @@ -147,10 +148,10 @@ Plus one invariant change with no new syscall: **address-space reference countin ### Address-space reference counting -Today an address space is 1:1 with a task: `spawnUserLocked` records `aspace` on the -Task, and teardown does `destroyAddressSpace(t.aspace)` when **any** user task exits +Today an address space is 1:1 with a task: `spawnUserLocked` records `address_space` on the +Task, and teardown does `destroyAddressSpace(t.address_space)` when **any** user task exits ([scheduler.zig](../system/kernel/scheduler.zig)). With threads, several tasks share -one `aspace`, so the first to exit would rip the address space out from under its +one `address_space`, so the first to exit would rip the address space out from under its siblings. Fix: a small refcount keyed by the address-space root (`createAddressSpace` in @@ -161,7 +162,7 @@ that must land and be proven before anything shares an address space. ### `thread_spawn` and the trampoline -The scheduler already accepts an arbitrary `aspace` and does **not** smuggle values +The scheduler already accepts an arbitrary `address_space` and does **not** smuggle values through registers — `startUserTask` reads the entry/stack from the Task and `jumpToUser`s ([scheduler.zig](../system/kernel/scheduler.zig)). That makes the thread path clean: @@ -170,7 +171,7 @@ path clean: `{ fn_ptr, args_tuple, completion }`, the std "Instance" pattern — and writes the closure pointer to the **top word of the new stack**. 2. It calls `thread_spawn(entry = &threadTrampoline, stack_top, arg = closure_ptr)`. - The kernel calls the same `spawnUserLocked` path with the **caller's aspace** + The kernel calls the same `spawnUserLocked` path with the **caller's address space** (refcount++), `entry`, and `user_sp = stack_top`. 3. `threadTrampoline` (a small runtime shim) reads the closure off its stack, calls the user function, then calls `thread_exit`. No new register ABI — the closure @@ -184,7 +185,7 @@ Unlike a process start, there is **no** System V argc/argv/auxv block - **`thread_exit`** marks the task dead and hands the kernel the thread's user-stack range. The kernel reaps the task on the scheduler (already running on a *kernel* - stack, so it can safely unmap the user stack), decrements the aspace refcount, and + stack, so it can safely unmap the user stack), decrements the address-space refcount, and frees the task slot. - **`join` — Stage 1** reuses the existing exit-notification machinery ([process-lifecycle.md](process-lifecycle.md)): `spawn` passes a per-thread @@ -205,12 +206,12 @@ Unlike a process start, there is **no** System V argc/argv/auxv block call the futex wrappers on the slow path — the same construction `std.Thread` uses, so the algorithms port directly. -Keying: threads share an address space, so a **virtual address within that aspace** -identifies a futex uniquely; the kernel keys its wait queue by `(aspace_root, vaddr)`. -Keying by the **physical** address instead (translate `vaddr -> paddr` on entry) is a +Keying: threads share an address space, so a **virtual address within that address space** +identifies a futex uniquely; the kernel keys its wait queue by `(address_space_root, virtual_address)`. +Keying by the **physical** address instead (translate `virtual_address -> physical_address` on entry) is a deliberate forward door: it lets two *processes* share a futex through an [shm](display-v2.md) region later, without changing the API. We start with the -private-per-aspace key and note the physical-key upgrade. +private-per-address-space key and note the physical-key upgrade. No spinning: a contended lock parks the task in the kernel and the core is free to run other work or `hlt` ([halting.md](halting.md)). This is why futex is a locked @@ -218,12 +219,12 @@ decision, not a "maybe later." ### TLS and `getCurrentId` -danos sets up no `fs.base` TLS today (fine under `single_threaded`). Two scoped needs: +danos sets up no thread-pointer TLS today (fine under `single_threaded`). Two scoped needs: - **`getCurrentId`** returns the kernel task id — either a trivial syscall or, better, a value the runtime stashes in a per-thread control block. -- **`threadlocal` variables** need a real per-thread TLS block and `fs.base` set per - thread. `thread_spawn` sets `fs.base` to a runtime-allocated per-thread block; full +- **`threadlocal` variables** need a real per-thread TLS block and the thread pointer set per + thread. `thread_spawn` sets the thread pointer to a runtime-allocated per-thread block; full `threadlocal` support is Stage 3, only if a consumer needs it. Nothing in the core spawn/join/mutex path requires `threadlocal`. @@ -237,14 +238,14 @@ it may call `runtime.Thread.spawn`. Everyone else stays single-threaded and lean ## Interaction with the rest of the kernel - **Scheduler / SMP** ([scheduling.md](scheduling.md), [smp.md](smp.md)): a thread is - just another `Task` with an `aspace` shared with its siblings; the existing + just another `Task` with an `address_space` shared with its siblings; the existing per-core ready queues, priorities, and affinity apply unchanged. Threads of one process can run on different cores simultaneously — that is the point. - **Halting** ([halting.md](halting.md)): futex-parked waiters keep the "idle core halts" property intact under lock contention — no busy-wait. - **Lifecycle** ([process-lifecycle.md](process-lifecycle.md)): killing a process - must kill *all* its threads and only then drop the last aspace ref. The kill path - already targets a process; it fans out to every task on that aspace. + must kill *all* its threads and only then drop the last address-space ref. The kill path + already targets a process; it fans out to every task on that address space. - **Resilience** ([resilience.md](resilience.md)): a faulting thread kills its whole process (shared fate). The supervisor restarts the **process**, which respawns its threads from a known-good state — restart granularity stays the process. @@ -257,8 +258,8 @@ The ordered, `/loop`-runnable milestones live in a verifiable gate (`python3 test/qemu_test.py `, asserting serial markers; `zig build test` for host unit tests). The stages below are the shape it expands. -- **Stage 0 — address-space refcount.** Refcount on the aspace root; teardown destroys - at zero. No API yet; nothing shares an aspace, so refcount is 1 everywhere. +- **Stage 0 — address-space refcount.** Refcount on the address-space root; teardown destroys + at zero. No API yet; nothing shares an address space, so refcount is 1 everywhere. *Gate:* the full QEMU suite stays green (no regression) — proves the reframing is invisible until used. - **Stage 1 — spawn / join / detach.** `thread_spawn` + `thread_exit`, the trampoline, @@ -272,7 +273,7 @@ a verifiable gate (`python3 test/qemu_test.py `, asserting serial markers; word. *Gate:* `-Dtest-case=thread-mutex` — a bounded producer/consumer over a `Mutex` + `Condition` moves K items with no lost wakeups and no busy-wait (assert the consumer blocked, e.g. via a low idle tick count). -- **Stage 3 — polish.** Per-thread TLS / `fs.base` and `threadlocal` (only if a +- **Stage 3 — polish.** Per-thread TLS / thread pointer and `threadlocal` (only if a consumer needs it), `RwLock`/`WaitGroup` as demanded, and this doc's cases wired into [test/qemu_test.py](../test/qemu_test.py). @@ -292,7 +293,7 @@ are — user code never names a syscall. - **No thread priorities distinct from the process.** Threads inherit the process priority; per-thread priority is a later question if it ever earns its keep. - **No cross-process shared-memory futex yet** — the physical-address key leaves the - door open, but the first cut is private-per-aspace. + door open, but the first cut is private-per-address-space. - **No `pthread`/POSIX surface.** The API is `std.Thread`-shaped Zig, nothing more. ## The self-hosting endgame diff --git a/docs/vdso.md b/docs/vdso.md index 34203cd..3c2d89c 100644 --- a/docs/vdso.md +++ b/docs/vdso.md @@ -119,7 +119,7 @@ One table entry per kernel call, C ABI (System V AMD64), names prefixed returns are `u64`, errors return as negative values exactly as today. The calls that return two values in `rax:rdx` today — `dma_alloc` -(vaddr + paddr), `msi_bind` (address + data), `shm_create` (vaddr + handle) — +(virtual_address + physical_address), `msi_bind` (address + data), `shm_create` (virtual_address + handle) — become functions returning a two-`u64` struct. The System V ABI returns a 16-byte struct in `rax:rdx`, so the stub is a plain `syscall; ret` — the C-ABI spelling of the existing convention, at zero cost. diff --git a/library/runtime/heap.zig b/library/runtime/heap.zig index 0ff4fd7..0e525b7 100644 --- a/library/runtime/heap.zig +++ b/library/runtime/heap.zig @@ -9,15 +9,32 @@ //! — `grow` asks the kernel for pages via `mmap` instead of mapping frames //! itself, and the kernel picks the base address. //! -//! Single-threaded and 16-byte maximum alignment, exactly like the kernel heap; a -//! lock and larger alignments come when user programs gain threads. +//! 16-byte maximum alignment, exactly like the kernel heap. The free list is guarded by +//! a `Thread.Mutex` **only in multi-threaded binaries** (`addThreadedUserBinary`): the +//! guard is gated on `builtin.single_threaded`, so an ordinary single-threaded binary +//! compiles it out and pays nothing, while a threaded one can allocate safely from +//! several threads at once (docs/threading-plan.md M7). The lock lives at the two +//! free-list mutators — `rawAlloc`/`rawFree` — which every entry point funnels through. const std = @import("std"); +const builtin = @import("builtin"); const abi = @import("abi"); const system_calls = @import("system.zig"); +const Mutex = @import("thread.zig").Thread.Mutex; const page_size = abi.page_size; +/// Guards `free_list`. A no-op in single-threaded builds (compiled out); a real futex +/// mutex in threaded ones. Uncontended acquisition is a single CAS — no syscall. +var heap_mutex: Mutex = .{}; + +inline fn lockHeap() void { + if (comptime !builtin.single_threaded) heap_mutex.lock(); +} +inline fn unlockHeap() void { + if (comptime !builtin.single_threaded) heap_mutex.unlock(); +} + /// A block header, at the start of every block; while free it also links the /// free list via `next`. const Block = extern struct { @@ -84,8 +101,11 @@ fn insertFree(block: *Block) void { } } -/// Allocate `len` bytes (16-byte aligned), or null if out of memory. +/// Allocate `len` bytes (16-byte aligned), or null if out of memory. Holds the heap lock +/// across the free-list search and any `grow` (which also touches the free list). fn rawAlloc(len: usize) ?[*]u8 { + lockHeap(); + defer unlockHeap(); const need = alignUp(header_size + len, 16); var attempts: u32 = 0; @@ -119,6 +139,8 @@ fn rawAlloc(len: usize) ?[*]u8 { } fn rawFree(ptr: [*]u8) void { + lockHeap(); + defer unlockHeap(); const block: *Block = @ptrFromInt(@intFromPtr(ptr) - header_size); insertFree(block); } diff --git a/library/runtime/shm.zig b/library/runtime/shm.zig index 1a5ec5b..a75056c 100644 --- a/library/runtime/shm.zig +++ b/library/runtime/shm.zig @@ -22,7 +22,7 @@ pub const Region = struct { }; /// Grant `len` bytes (rounded up to whole pages) of shareable, zeroed, cacheable RAM. -/// Returns the region or null on failure. Two return values — vaddr in rax, handle in rdx — +/// Returns the region or null on failure. Two return values — virtual_address in rax, handle in rdx — /// so this is a hand-written stub like `dma.alloc`. pub fn create(len: usize) ?Region { var rax: usize = undefined; diff --git a/library/runtime/thread.zig b/library/runtime/thread.zig index 2f84ebb..0a59df0 100644 --- a/library/runtime/thread.zig +++ b/library/runtime/thread.zig @@ -11,19 +11,27 @@ //! A binary must be built multi-threaded (`addThreadedUserBinary`) before it may spawn. const std = @import("std"); +const builtin = @import("builtin"); const abi = @import("abi"); const sc = @import("system-call.zig"); const system = @import("system.zig"); -const ipc = @import("ipc.zig"); + +/// True in a real danos binary; false when this module is compiled for host unit tests. +/// The `Futex` seam and the test blocks below branch on it so the lock/condvar state +/// machines can be exercised on the host against `std.Thread.Futex` (docs/threading-plan.md +/// M11), while the danos build uses the futex syscalls. +const on_danos = builtin.os.tag == .freestanding; /// A thread stack, if the caller does not override it. 64 KiB of mmap'd, zeroed pages. pub const default_stack_size: usize = 64 * 1024; +/// Bytes reserved at the top of each thread's stack for its per-thread TLS block (the +/// self-pointer plus scratch slots reachable via `%fs`). docs/threading-plan.md M10. +const tls_block_size: usize = 64; + pub const Thread = struct { - /// The kernel task id of the spawned thread. + /// The kernel task id of the spawned thread — what `join` waits on. tid: u32, - /// The endpoint the kernel notifies when this thread ends — what `join` blocks on. - exit_endpoint: ipc.Handle, /// The mmap'd stack, reclaimed by `join` (or at process exit after `detach`). stack_base: usize, stack_size: usize, @@ -46,51 +54,52 @@ pub const Thread = struct { pub fn spawn(config: SpawnConfig, comptime function: anytype, args: anytype) SpawnError!Thread { const Args = @TypeOf(args); const Closure = struct { + tls_base: usize, args: Args, - /// Entered directly by the kernel with `self` in rdi (C ABI). Runs the user - /// function, then ends the thread — never returns. + /// Entered directly by the kernel with `self` in rdi (C ABI). Establishes this + /// thread's TLS pointer, runs the user function, then ends the thread. fn entry(self_addr: usize) callconv(.c) noreturn { const self: *@This() = @ptrFromInt(self_addr); + setThreadPointer(self.tls_base); // per-thread thread pointer before any user code @call(.auto, function, self.args); exitThread(); } }; - // The endpoint the kernel posts this thread's exit notification to. - const endpoint = ipc.createIpcEndpoint() orelse return error.SystemResources; - const base = system.mmap(config.stack_size, system.PROT_READ | system.PROT_WRITE); if (system.mmapFailed(base)) return error.SystemResources; - // Lay the closure at the very top of the thread's own stack, then start the - // thread's rsp just below it (16-aligned minus 8, the alignment a `call` leaves - // for a C-ABI entry) so the growing stack never overwrites the args. + // Top of the thread's own stack, downward: the closure, then a small per-thread TLS + // block (the thread pointer points here; slot 0 is the variant-II self-pointer, the rest is + // scratch for user TLS), then the stack proper (rsp starts below the TLS block, so + // the growing stack never overwrites either). var closure_addr = (base + config.stack_size) - @sizeOf(Closure); closure_addr &= ~@as(usize, @alignOf(Closure) - 1); // align the closure down - const closure: *Closure = @ptrFromInt(closure_addr); - closure.* = .{ .args = args }; - var stack_top = closure_addr & ~@as(usize, 15); // 16-align below the closure + const tls_base = (closure_addr - tls_block_size) & ~@as(usize, 15); + const tls: [*]usize = @ptrFromInt(tls_base); + tls[0] = tls_base; // self-pointer (fs:0), as the x86_64 TLS ABI expects + + const closure: *Closure = @ptrFromInt(closure_addr); + closure.* = .{ .tls_base = tls_base, .args = args }; + + var stack_top = tls_base & ~@as(usize, 15); // 16-align below the TLS block stack_top -= 8; // ...then rsp % 16 == 8 at the C entry - const tid = threadSpawn(@intFromPtr(&Closure.entry), stack_top, closure_addr, endpoint); + const tid = threadSpawn(@intFromPtr(&Closure.entry), stack_top, closure_addr); if (threadSpawnFailed(tid)) { _ = system.munmap(base, config.stack_size); return error.SystemResources; } - return .{ .tid = @intCast(tid), .exit_endpoint = endpoint, .stack_base = base, .stack_size = config.stack_size }; + return .{ .tid = @intCast(tid), .stack_base = base, .stack_size = config.stack_size }; } /// Block until this thread finishes, then reclaim its stack. Mirrors /// `std.Thread.join`. The exit endpoint is private to this thread, so the first /// child-exit notification on it is this thread's. pub fn join(self: Thread) void { - var receive: [0]u8 = undefined; - while (true) { - const got = ipc.replyWait(self.exit_endpoint, &.{}, &receive, null); - if (got.isChildExit() and got.childProcessId() == self.tid) break; - } - _ = system.munmap(self.stack_base, self.stack_size); + _ = sc.systemCall1(.thread_join, self.tid); // block until the thread has exited + _ = system.munmap(self.stack_base, self.stack_size); // reclaim its (now-vacated) stack } /// Relinquish the right to join: never wait for or reclaim this thread. Its stack is @@ -119,17 +128,37 @@ pub const Thread = struct { /// the value already differs (safe against spurious returns, as in std): the /// caller re-checks its condition in a loop. pub fn wait(ptr: *const std.atomic.Value(u32), expect: u32) void { - _ = futexWait(@intFromPtr(ptr), expect, 0); + if (comptime on_danos) { + _ = futexWait(@intFromPtr(ptr), expect, 0); + } else { + // Host unit-test mock: spin+yield until the value changes (`wake` is a + // no-op — the callers re-check their condition in a loop anyway). Correct, + // if busy; fine for the state-machine tests. + while (ptr.load(.acquire) == expect) std.Thread.yield() catch {}; + } } /// As `wait`, but returns `error.Timeout` if `timeout_ns` elapses first. pub fn timedWait(ptr: *const std.atomic.Value(u32), expect: u32, timeout_ns: u64) error{Timeout}!void { - if (futexWait(@intFromPtr(ptr), expect, timeout_ns) == abi.futex_timed_out) return error.Timeout; + if (comptime on_danos) { + if (futexWait(@intFromPtr(ptr), expect, timeout_ns) == abi.futex_timed_out) return error.Timeout; + } else { + var spins: u64 = 0; + const limit = timeout_ns / 1000 + 1; + while (ptr.load(.acquire) == expect) : (spins += 1) { + if (spins >= limit) return error.Timeout; + std.Thread.yield() catch {}; + } + } } /// Wake up to `max_waiters` threads blocked on `ptr`. pub fn wake(ptr: *const std.atomic.Value(u32), max_waiters: u32) void { - _ = futexWake(@intFromPtr(ptr), max_waiters); + if (comptime on_danos) { + _ = futexWake(@intFromPtr(ptr), max_waiters); + } else { + // host mock: spin-waiters re-check their condition, so no wake is needed. + } } }; @@ -230,10 +259,101 @@ pub const Thread = struct { s.cond.signal(); } }; + + /// A reader/writer lock, `std.Thread.RwLock`-shaped: many concurrent readers OR one + /// exclusive writer. Reader-preferring (a steady stream of readers can delay a writer), + /// built on `Mutex` + `Condition` over a signed state: `>0` = that many readers hold + /// it, `-1` = a writer holds it, `0` = free. + pub const RwLock = struct { + mutex: Mutex = .{}, + cond: Condition = .{}, + state: i64 = 0, + + /// Acquire shared (read) access, blocking while a writer holds the lock. + pub fn lockShared(rw: *RwLock) void { + rw.mutex.lock(); + defer rw.mutex.unlock(); + while (rw.state < 0) rw.cond.wait(&rw.mutex); + rw.state += 1; + } + + /// Try to acquire shared access without blocking. + pub fn tryLockShared(rw: *RwLock) bool { + rw.mutex.lock(); + defer rw.mutex.unlock(); + if (rw.state < 0) return false; + rw.state += 1; + return true; + } + + /// Release shared access; wake a waiting writer once the last reader leaves. + pub fn unlockShared(rw: *RwLock) void { + rw.mutex.lock(); + defer rw.mutex.unlock(); + rw.state -= 1; + if (rw.state == 0) rw.cond.broadcast(); + } + + /// Acquire exclusive (write) access, blocking until no readers or writer remain. + pub fn lock(rw: *RwLock) void { + rw.mutex.lock(); + defer rw.mutex.unlock(); + while (rw.state != 0) rw.cond.wait(&rw.mutex); + rw.state = -1; + } + + /// Try to acquire exclusive access without blocking. + pub fn tryLock(rw: *RwLock) bool { + rw.mutex.lock(); + defer rw.mutex.unlock(); + if (rw.state != 0) return false; + rw.state = -1; + return true; + } + + /// Release exclusive access; wake all waiters (they re-check their condition). + pub fn unlock(rw: *RwLock) void { + rw.mutex.lock(); + defer rw.mutex.unlock(); + rw.state = 0; + rw.cond.broadcast(); + } + }; + + /// A `std.Thread.WaitGroup`-shaped counter: `start` before spawning work, `finish` as + /// each unit completes, `wait` blocks until the count returns to zero. + pub const WaitGroup = struct { + mutex: Mutex = .{}, + cond: Condition = .{}, + counter: usize = 0, + + /// Register one pending unit of work. + pub fn start(wg: *WaitGroup) void { + wg.mutex.lock(); + defer wg.mutex.unlock(); + wg.counter += 1; + } + + /// Mark one unit done; wake waiters if that was the last. + pub fn finish(wg: *WaitGroup) void { + wg.mutex.lock(); + defer wg.mutex.unlock(); + wg.counter -= 1; + if (wg.counter == 0) wg.cond.broadcast(); + } + + /// Block until every started unit has finished. + pub fn wait(wg: *WaitGroup) void { + wg.mutex.lock(); + defer wg.mutex.unlock(); + while (wg.counter != 0) wg.cond.wait(&wg.mutex); + } + }; }; /// thread_spawn(entry, stack_top, arg, exit_endpoint) -> tid, or a wrapped error. -fn threadSpawn(entry: usize, stack_top: usize, arg: usize, exit_endpoint: ipc.Handle) usize { +fn threadSpawn(entry: usize, stack_top: usize, arg: usize) usize { + const exit_endpoint: usize = @intCast(abi.no_cap); // join uses thread_join, not an endpoint return sc.systemCall4(.thread_spawn, entry, stack_top, arg, exit_endpoint); } @@ -249,6 +369,11 @@ fn exitThread() noreturn { unreachable; } +/// Set the calling thread's FS base (its user TLS thread pointer). +fn setThreadPointer(addr: usize) void { + _ = sc.systemCall1(.set_thread_pointer, addr); +} + /// futex_wait(addr, expect, timeout_ns) -> status (abi.futex_*). fn futexWait(addr: usize, expect: u32, timeout_ns: u64) usize { return sc.systemCall3(.futex_wait, addr, expect, timeout_ns); @@ -258,3 +383,102 @@ fn futexWait(addr: usize, expect: u32, timeout_ns: u64) usize { fn futexWake(addr: usize, count: u32) usize { return sc.systemCall2(.futex_wake, addr, count); } + +// --- host unit tests (docs/threading-plan.md M11) --------------------------- +// +// These run under `zig build test` on the host: the `Futex` seam above uses +// `std.Thread.Futex` off-danos, so the lock/condvar state machines can be exercised by +// real host threads. They are never compiled into a danos binary (test blocks only build +// under test), so their `std.Thread` use is fine even though `std.Thread` is unavailable +// on the freestanding target. + +test "Mutex serialises concurrent increments across host threads" { + var m: Thread.Mutex = .{}; + var counter: u64 = 0; + const workers = 8; + const per = 20_000; + const Ctx = struct { + m: *Thread.Mutex, + c: *u64, + fn run(ctx: @This()) void { + var i: usize = 0; + while (i < per) : (i += 1) { + ctx.m.lock(); + ctx.c.* += 1; + ctx.m.unlock(); + } + } + }; + var handles: [workers]std.Thread = undefined; + for (&handles) |*h| h.* = try std.Thread.spawn(.{}, Ctx.run, .{Ctx{ .m = &m, .c = &counter }}); + for (handles) |h| h.join(); + try std.testing.expectEqual(@as(u64, workers * per), counter); +} + +test "RwLock never lets a reader observe a half-written pair" { + var rw: Thread.RwLock = .{}; + var a: u64 = 0; + var b: u64 = 0; // invariant while a lock is held: a == b + var stop = std.atomic.Value(bool).init(false); + var ok = std.atomic.Value(bool).init(true); + + const Writer = struct { + rw: *Thread.RwLock, + a: *u64, + b: *u64, + stop: *std.atomic.Value(bool), + fn run(w: @This()) void { + var v: u64 = 1; + while (!w.stop.load(.acquire)) : (v +%= 1) { + w.rw.lock(); + w.a.* = v; // update both halves under the exclusive lock... + w.b.* = v; + w.rw.unlock(); + } + } + }; + const Reader = struct { + rw: *Thread.RwLock, + a: *u64, + b: *u64, + ok: *std.atomic.Value(bool), + fn run(r: @This()) void { + var i: usize = 0; + while (i < 200_000) : (i += 1) { + r.rw.lockShared(); + if (r.a.* != r.b.*) r.ok.store(false, .release); // ...so a reader must never see them differ + r.rw.unlockShared(); + } + } + }; + + var writers: [2]std.Thread = undefined; + for (&writers) |*w| w.* = try std.Thread.spawn(.{}, Writer.run, .{Writer{ .rw = &rw, .a = &a, .b = &b, .stop = &stop }}); + var readers: [4]std.Thread = undefined; + for (&readers) |*rd| rd.* = try std.Thread.spawn(.{}, Reader.run, .{Reader{ .rw = &rw, .a = &a, .b = &b, .ok = &ok }}); + for (readers) |rd| rd.join(); + stop.store(true, .release); + for (writers) |w| w.join(); + try std.testing.expect(ok.load(.acquire)); +} + +test "WaitGroup blocks until every started unit finishes" { + var wg: Thread.WaitGroup = .{}; + var done = std.atomic.Value(u32).init(0); + const n = 6; + const Ctx = struct { + wg: *Thread.WaitGroup, + done: *std.atomic.Value(u32), + fn run(c: @This()) void { + _ = c.done.fetchAdd(1, .monotonic); + c.wg.finish(); + } + }; + var i: usize = 0; + while (i < n) : (i += 1) wg.start(); + var handles: [n]std.Thread = undefined; + for (&handles) |*h| h.* = try std.Thread.spawn(.{}, Ctx.run, .{Ctx{ .wg = &wg, .done = &done }}); + wg.wait(); // must not return until all n finished + try std.testing.expectEqual(@as(u32, n), done.load(.acquire)); + for (handles) |h| h.join(); +} diff --git a/system/abi.zig b/system/abi.zig index d536237..7ab350b 100644 --- a/system/abi.zig +++ b/system/abi.zig @@ -39,13 +39,13 @@ pub const SystemCall = enum(u64) { ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, receive, cap) -> receive_len (+badge in rdx) device_enumerate = 11, // device_enumerate(buffer, maximum) -> count: snapshot the device table device_claim = 12, // device_claim(id) -> ok: take exclusive ownership of a device - mmio_map = 13, // mmio_map(id, resource_index) -> vaddr: map a claimed device's MMIO into this AS + mmio_map = 13, // mmio_map(id, resource_index) -> virtual_address: map a claimed device's MMIO into this address space irq_bind = 14, // irq_bind(id, resource_index, endpoint): deliver a device IRQ as an IPC notification irq_ack = 15, // irq_ack(id, resource_index): re-arm a bound IRQ after servicing it device_register = 16, // device_register(parent_id, descriptor) -> id: publish a child of a device you claimed system_spawn = 17, // system_spawn(name_ptr, name_len, arguments_ptr, arguments_len, exit_endpoint) -> child process id: start a named initial-ramdisk binary as a new ring-3 process - dma_alloc = 18, // dma_alloc(len, flags) -> vaddr (rax), paddr (rdx): contiguous, pinned, uncacheable DMA memory - dma_free = 19, // dma_free(vaddr, len) -> 0: release a prior dma_alloc + dma_alloc = 18, // dma_alloc(len, flags) -> virtual_address (rax), physical_address (rdx): contiguous, pinned, uncacheable DMA memory + dma_free = 19, // dma_free(virtual_address, len) -> 0: release a prior dma_alloc msi_bind = 20, // msi_bind(device_id, endpoint) -> address (rax), data (rdx): a per-device MSI vector for a claimed device io_read = 21, // io_read(device_id, resource_index, offset, width) -> value: read a port in a claimed device's io_port resource io_write = 22, // io_write(device_id, resource_index, offset, width, value) -> 0: write a port in a claimed device's io_port resource @@ -60,15 +60,17 @@ pub const SystemCall = enum(u64) { timer_bind = 31, // timer_bind(endpoint, ms) -> 0/-errno: one-shot timer — posts a notification when ms elapse klog_read = 32, // klog_read(offset, ptr, len) -> bytes copied: copy the kernel RAM log buffer out to a user buffer (for persisting the boot log to disk) wall_clock = 33, // wall_clock() -> Unix epoch seconds (UTC): the RTC wall-clock time, for filesystem timestamps (mtime). Monotonic time is `clock`. - shm_create = 34, // shm_create(len) -> vaddr (rax), handle (rdx): a shareable, zeroed, cacheable RAM region mapped into this AS; the handle is a capability passed to another process as an ipc_call send_cap (docs/display-v2.md) - shm_map = 35, // shm_map(cap) -> vaddr: map the shared region named by a received capability into this AS (the same physical pages the creator sees) - shm_physical = 36, // shm_physical(cap) -> paddr: the guest-physical base of a shared region held by capability, so a driver can program it into a device (e.g. virtio-gpu attach_backing); the pages are contiguous (docs/display-v2.md) + shm_create = 34, // shm_create(len) -> virtual_address (rax), handle (rdx): a shareable, zeroed, cacheable RAM region mapped into this AS; the handle is a capability passed to another process as an ipc_call send_cap (docs/display-v2.md) + shm_map = 35, // shm_map(cap) -> virtual_address: map the shared region named by a received capability into this address space (the same physical pages the creator sees) + shm_physical = 36, // shm_physical(cap) -> physical_address: the guest-physical base of a shared region held by capability, so a driver can program it into a device (e.g. virtio-gpu attach_backing); the pages are contiguous (docs/display-v2.md) thread_spawn = 37, // thread_spawn(entry, stack_top, arg, exit_endpoint) -> tid: start a task sharing the caller's address space at `entry` on `stack_top`, `arg` in rdi; exit_endpoint (a handle, or no_cap) is notified when it ends — how join waits (docs/threading.md) thread_exit = 38, // thread_exit(): end the calling thread, dropping one reference to its address space (destroyed on the last) current_core = 39, // current_core() -> index: the dense 0-based index of the core the caller is running on (for parallelism/affinity introspection) futex_wait = 40, // futex_wait(addr, expected, timeout_ns) -> status: if *addr == expected, block until woken or the timeout; returns futex_woken/mismatch/timed_out (docs/threading.md) futex_wake = 41, // futex_wake(addr, count) -> woken: wake up to `count` tasks blocked in futex_wait on `addr` in this address space thread_self = 42, // thread_self() -> tid: the calling thread's kernel task id (runtime.Thread.getCurrentId) + thread_join = 43, // thread_join(tid) -> 0: block until the thread with id `tid` has exited (runtime.Thread.join; no per-thread IPC endpoint) (docs/threading.md) + set_thread_pointer = 44, // set_thread_pointer(addr) -> 0: set the caller's thread pointer (user-space TLS base; x86_64 IA32_FS_BASE, aarch64 TPIDR_EL0); restored per task across context switches (docs/threading-plan.md M10) _, }; diff --git a/system/kernel/architecture/x86_64/cpu.zig b/system/kernel/architecture/x86_64/cpu.zig index 3625c2a..9bbd7a9 100644 --- a/system/kernel/architecture/x86_64/cpu.zig +++ b/system/kernel/architecture/x86_64/cpu.zig @@ -304,6 +304,17 @@ pub fn cpuLocal() usize { return pcpu.scheduler(); } +const ia32_fs_base = 0xC000_0100; + +/// Set the user-space TLS **thread pointer** — the arch-neutral name the generic scheduler +/// calls (`architecture.setThreadPointer`). On x86_64 that is the FS-segment base +/// (`IA32_FS_BASE`); an aarch64 port implements the same call against `TPIDR_EL0`. The +/// kernel never touches FS, so this only affects the user task that runs next, which the +/// scheduler restores per task across context switches (docs/threading-plan.md M10). +pub fn setThreadPointer(base: u64) void { + io.wrmsr(ia32_fs_base, base); +} + // --- SMP: application-processor bring-up ---------------------------------- /// Record the low (<1 MiB) frame reserved for the AP trampoline. Run once at boot. diff --git a/system/kernel/architecture/x86_64/paging.zig b/system/kernel/architecture/x86_64/paging.zig index b4b2213..5838185 100644 --- a/system/kernel/architecture/x86_64/paging.zig +++ b/system/kernel/architecture/x86_64/paging.zig @@ -509,7 +509,7 @@ pub fn unmapInto(pml4: u64, virtual: u64) void { /// any address space, not just the live one). Returns null if `virtual` is not /// mapped at any level. Stops at a 2 MiB huge-page leaf (the physmap uses them), /// resolving the offset within it. The foundation for cross-address-space copies -/// and for munmap (which needs the frame behind a user vaddr to free it). +/// and for munmap (which needs the frame behind a user virtual_address to free it). pub fn translateIn(pml4: u64, virtual: u64) ?u64 { const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF]; if (pml4e & present == 0) return null; diff --git a/system/kernel/ipc-synchronous.zig b/system/kernel/ipc-synchronous.zig index 70edee7..34476bf 100644 --- a/system/kernel/ipc-synchronous.zig +++ b/system/kernel/ipc-synchronous.zig @@ -355,7 +355,7 @@ pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_pt me.ipc_client = null; const n = @min(reply_len, client.ipc_reply_cap); client.ipc_received_cap = abi.no_cap; - if (!copyAcross(me.aspace, reply_ptr, client.aspace, client.ipc_reply_ptr, n)) { + if (!copyAcross(me.address_space, reply_ptr, client.address_space, client.ipc_reply_ptr, n)) { client.ipc_status = -EFAULT; } else if (send_cap != abi.no_cap) { // Transfer the reply's capability into the client. A failure fails the @@ -383,8 +383,8 @@ pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_pt } if (popPost(endpoint)) |slot| { const n = @min(@as(usize, slot.length), receive_cap); - // Copy from the kernel-resident ring slot (source aspace 0) into the receiver. - if (!copyAcross(0, @intFromPtr(&slot.bytes), me.aspace, receive_ptr, n)) { + // Copy from the kernel-resident ring slot (source address_space 0) into the receiver. + if (!copyAcross(0, @intFromPtr(&slot.bytes), me.address_space, receive_ptr, n)) { continue; // bad receive buffer: drop this message, keep serving } out_badge.* = slot.sender_id | notify_badge_bit | notify_message_bit; @@ -392,7 +392,7 @@ pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_pt } if (dequeueSender(endpoint)) |caller| { const n = @min(caller.ipc_send_len, receive_cap); - if (!copyAcross(caller.aspace, caller.ipc_send_ptr, me.aspace, receive_ptr, n)) { + if (!copyAcross(caller.address_space, caller.ipc_send_ptr, me.address_space, receive_ptr, n)) { caller.ipc_status = -EFAULT; // bad sender buffer: fail it, keep serving scheduler.readyLocked(caller); continue; diff --git a/system/kernel/process.zig b/system/kernel/process.zig index 4fbe9c0..266bdfe 100644 --- a/system/kernel/process.zig +++ b/system/kernel/process.zig @@ -58,8 +58,9 @@ pub const stack_top_virtual: u64 = stack_base_virtual + parameters.user_stack_pa /// The mmap grant arena: where `mmap` hands out fresh user pages, above the image /// and stack but still inside PML4[224] (so no kernel mapping is widened). Each -/// process bump-allocates from `heap_arena_base` upward via `Task.heap_next`; a -/// 1 GiB window is far more than any user heap needs today. +/// process bump-allocates from `heap_arena_base` upward via a per-address-space cursor +/// (`scheduler.addressSpaceMmapNextPtr`, shared by its threads); a 1 GiB window is far more +/// than any user heap needs today. pub const heap_arena_base: u64 = 0x0000_7000_1000_0000; pub const heap_arena_end: u64 = heap_arena_base + (1 << 30); @@ -69,8 +70,8 @@ pub const user_half_end: u64 = 0x0000_8000_0000_0000; /// The MMIO-grant arena: where `mmio_map` places device windows, in PML4[226] — /// a user-exclusive region distinct from code/stack/heap (PML4[224]), so mapping -/// device pages user-accessible widens no kernel mapping. Per-process cursor in -/// `Task.device_map_next`. +/// device pages user-accessible widens no kernel mapping. Per-address-space cursor +/// (`scheduler.addressSpaceDeviceMapNextPtr`). pub const device_arena_base: u64 = 0x0000_7100_0000_0000; pub const device_arena_end: u64 = device_arena_base + (4 << 30); @@ -163,7 +164,7 @@ fn fail(state: *architecture.CpuState) void { fn system_call(state: *architecture.CpuState) void { const t = scheduler.current(); - const user = t.aspace != 0; + const user = t.address_space != 0; if (user) { // A condemned process (process_kill caught it running) dies at its next // kernel entry — before it can spawn, claim, or message anything else. @@ -230,6 +231,8 @@ fn system_call(state: *architecture.CpuState) void { .thread_spawn => systemThreadSpawn(state), .current_core => systemCurrentCore(state), .thread_self => systemThreadSelf(state), + .thread_join => systemThreadJoin(state), + .set_thread_pointer => systemSetThreadPointer(state), .futex_wait => systemFutexWait(state), .futex_wake => systemFutexWake(state), .thread_exit => { @@ -314,7 +317,7 @@ fn systemIpcReplyWait(state: *architecture.CpuState) void { fn systemIpcSend(state: *architecture.CpuState) void { const me = scheduler.current(); const endpoint = ipc.resolveHandle(me, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF); - const r = ipc.send(endpoint, me.aspace, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), me.id); + const r = ipc.send(endpoint, me.address_space, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), me.id); architecture.setSystemCallResult(state, @bitCast(r)); } @@ -324,7 +327,7 @@ fn systemDeviceEnumerate(state: *architecture.CpuState) void { const buffer_ptr = architecture.systemCallArg(state, 0); const maximum = architecture.systemCallArg(state, 1); const t = scheduler.current(); - if (t.aspace == 0 or buffer_ptr >= user_half_end) return fail(state); + if (t.address_space == 0 or buffer_ptr >= user_half_end) return fail(state); const sz = @sizeOf(device_abi.DeviceDescriptor); const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half const out: [*]device_abi.DeviceDescriptor = @ptrFromInt(buffer_ptr); @@ -348,14 +351,14 @@ fn systemDeviceClaim(state: *architecture.CpuState) void { } else fail(state); } -/// mmio_map(device_id, resource_index) -> vaddr: map a claimed device's MMIO window into +/// mmio_map(device_id, resource_index) -> virtual_address: map a claimed device's MMIO window into /// this address space (strong-uncacheable) and return the register base address. /// The claim is the capability — a process can only map hardware it owns. fn systemMmioMap(state: *architecture.CpuState) void { const device_id = architecture.systemCallArg(state, 0); const resource_index = architecture.systemCallArg(state, 1); const t = scheduler.current(); - if (t.aspace == 0) return fail(state); + if (t.address_space == 0) return fail(state); // Read the broker table under the lock: ring-3 device_register (M19) now // mutates it concurrently on other cores, so a lock-free read here could // see a torn resource (and a torn length used to panic the arithmetic @@ -373,18 +376,23 @@ fn systemMmioMap(state: *architecture.CpuState) void { if (r.len == 0) return fail(state); if (@addWithOverflow(r.start, r.len)[1] != 0) return fail(state); - if (t.device_map_next == 0) t.device_map_next = device_arena_base; const first = r.start & ~@as(u64, page_size - 1); const last = (r.start + r.len - 1) & ~@as(u64, page_size - 1); const pages = (last - first) / page_size + 1; - const base_v = t.device_map_next; - if (base_v + pages * page_size > device_arena_end) return fail(state); - // A framebuffer resource asks (via its flag) to be mapped write-combining rather // than the strong-uncacheable default that register MMIO needs. const write_combining = (r.flags & device_abi.resource_flag_write_combining) != 0; - architecture.mapUserDeviceInto(t.aspace, base_v, r.start, r.len, write_combining); - t.device_map_next = base_v + pages * page_size; + + // Per-address-space cursor + shared page tables → serialize under the big lock, + // same as mmap (docs/threading-plan.md M7). + const flags = sync.enter(); + defer sync.leave(flags); + const cursor = scheduler.addressSpaceDeviceMapNextPtr(t.address_space) orelse return fail(state); + if (cursor.* == 0) cursor.* = device_arena_base; // seed the arena lazily + const base_v = cursor.*; + if (base_v + pages * page_size > device_arena_end) return fail(state); + architecture.mapUserDeviceInto(t.address_space, base_v, r.start, r.len, write_combining); + cursor.* = base_v + pages * page_size; architecture.setSystemCallResult(state, base_v + (r.start & (page_size - 1))); // register base } @@ -413,7 +421,7 @@ pub fn resolveIoPort(t: *scheduler.Task, device_id: u64, resource_index: u64, of /// is fine. See docs/drivers.md. fn systemIoRead(state: *architecture.CpuState) void { const t = scheduler.current(); - if (t.aspace == 0) return fail(state); + if (t.address_space == 0) return fail(state); const width = architecture.systemCallArg(state, 3); const port = resolveIoPort(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), width) orelse return fail(state); architecture.setSystemCallResult(state, architecture.pioRead(@intCast(width), port)); @@ -424,14 +432,14 @@ fn systemIoRead(state: *architecture.CpuState) void { /// gate as `io_read`. fn systemIoWrite(state: *architecture.CpuState) void { const t = scheduler.current(); - if (t.aspace == 0) return fail(state); + if (t.address_space == 0) return fail(state); const width = architecture.systemCallArg(state, 3); const port = resolveIoPort(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), width) orelse return fail(state); architecture.pioWrite(@intCast(width), port, @intCast(architecture.systemCallArg(state, 4))); architecture.setSystemCallResult(state, 0); } -/// dma_alloc(len, flags) -> vaddr (rax), paddr (rdx): grant `len` bytes (rounded up to +/// dma_alloc(len, flags) -> virtual_address (rax), physical_address (rdx): grant `len` bytes (rounded up to /// whole pages) of DMA-capable memory — physically contiguous, zeroed, pinned, and /// strong-uncacheable (coherent) — mapping it into the caller's DMA arena and handing /// back both the virtual address to touch and the physical address to program into the @@ -443,7 +451,7 @@ fn systemDmaAlloc(state: *architecture.CpuState) void { const len = architecture.systemCallArg(state, 0); const flags = architecture.systemCallArg(state, 1); const t = scheduler.current(); - if (t.aspace == 0 or len == 0) return fail(state); + if (t.address_space == 0 or len == 0) return fail(state); const pages: usize = @intCast((len + page_size - 1) / page_size); const max_phys: u64 = if (flags & abi.dma_below_4g != 0) (@as(u64, 4) << 30) else ~@as(u64, 0); @@ -459,14 +467,14 @@ fn systemDmaAlloc(state: *architecture.CpuState) void { // Zero through the physmap (the frames aren't mapped in the caller yet), then map. const kernel_view: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(phys)); @memset(kernel_view[0 .. pages * page_size], 0); - architecture.mapUserDmaInto(t.aspace, base_v, phys, pages * page_size); + architecture.mapUserDmaInto(t.address_space, base_v, phys, pages * page_size); t.dma_map_next = base_v + pages * page_size; architecture.setSystemCallResult(state, base_v); // virtual address for the CPU architecture.setSystemCallResult2(state, phys); // physical address for the device } -/// dma_free(vaddr, len) -> 0: release a prior `dma_alloc`. Bounded to the DMA arena so +/// dma_free(virtual_address, len) -> 0: release a prior `dma_alloc`. Bounded to the DMA arena so /// it can never unmap-and-free the caller's stack, heap, or an MMIO grant; only pages /// actually mapped are freed (an unmapped hole is skipped). Teardown also reclaims any /// DMA pages left mapped at exit (they carry no `device_grant`, so `freeSubtree` frees @@ -475,21 +483,21 @@ fn systemDmaFree(state: *architecture.CpuState) void { const base_v = architecture.systemCallArg(state, 0); const len = architecture.systemCallArg(state, 1); const t = scheduler.current(); - if (t.aspace == 0) return fail(state); + if (t.address_space == 0) return fail(state); const pages: usize = @intCast((len + page_size - 1) / page_size); if (base_v < dma_arena_base or base_v + pages * page_size > dma_arena_end) return fail(state); for (0..pages) |i| { const va = base_v + i * page_size; - if (architecture.translate(t.aspace, va)) |phys| { - architecture.unmapUserPageInto(t.aspace, va); + if (architecture.translate(t.address_space, va)) |phys| { + architecture.unmapUserPageInto(t.address_space, va); pmm.free(phys); } } architecture.setSystemCallResult(state, 0); } -/// shm_create(len) -> vaddr (rax), handle (rdx): grant `len` bytes (rounded up to whole +/// shm_create(len) -> virtual_address (rax), handle (rdx): grant `len` bytes (rounded up to whole /// pages) of **shareable, zeroed, cacheable** RAM — contiguous frames mapped into the /// caller's shm arena — and hand back the virtual address plus a capability handle. Unlike /// `dma_alloc` the memory is write-back cacheable (for CPU compositing, not device DMA) and @@ -500,7 +508,7 @@ fn systemDmaFree(state: *architecture.CpuState) void { fn systemShmCreate(state: *architecture.CpuState) void { const len = architecture.systemCallArg(state, 0); const t = scheduler.current(); - if (t.aspace == 0 or len == 0) return fail(state); + if (t.address_space == 0 or len == 0) return fail(state); const pages: usize = @intCast((len + page_size - 1) / page_size); if (pages == 0 or pages > maximum_shm_pages) return fail(state); @@ -525,20 +533,20 @@ fn systemShmCreate(state: *architecture.CpuState) void { return fail(state); } - architecture.mapUserSharedInto(t.aspace, base_v, phys, pages * page_size); + architecture.mapUserSharedInto(t.address_space, base_v, phys, pages * page_size); t.shm_map_next = base_v + pages * page_size; - architecture.setSystemCallResult(state, base_v); // vaddr for the CPU + architecture.setSystemCallResult(state, base_v); // virtual_address for the CPU architecture.setSystemCallResult2(state, @intCast(handle)); // capability handle to pass on } -/// shm_map(cap) -> vaddr: map the shared region named by a capability handle the caller +/// shm_map(cap) -> virtual_address: map the shared region named by a capability handle the caller /// received (via an `ipc_call` send_cap) into its shm arena — the same physical frames the /// creator sees — returning the virtual address. The handle already holds a reference (taken /// when the capability was shared), so this only adds a mapping; it never bumps the refcount. fn systemShmMap(state: *architecture.CpuState) void { const cap = architecture.systemCallArg(state, 0); const t = scheduler.current(); - if (t.aspace == 0) return fail(state); + if (t.address_space == 0) return fail(state); const shm = ipc.resolveShm(t, cap) orelse return fail(state); // not an shm handle we hold if (t.shm_map_next == 0) t.shm_map_next = shm_arena_base; @@ -546,12 +554,12 @@ fn systemShmMap(state: *architecture.CpuState) void { const size = shm.pages * page_size; if (base_v + size > shm_arena_end) return fail(state); - architecture.mapUserSharedInto(t.aspace, base_v, shm.phys, size); + architecture.mapUserSharedInto(t.address_space, base_v, shm.phys, size); t.shm_map_next = base_v + size; architecture.setSystemCallResult(state, base_v); } -/// shm_physical(cap) -> paddr: the guest-physical base of a shared region the caller holds a +/// shm_physical(cap) -> physical_address: the guest-physical base of a shared region the caller holds a /// capability for. The frames are contiguous (allocated by `allocContiguous`), so a single /// physical base + length describes the whole region — which is exactly what a driver needs /// to hand a shm surface to a device (virtio-gpu `attach_backing`). Only a holder of the @@ -559,7 +567,7 @@ fn systemShmMap(state: *architecture.CpuState) void { fn systemShmPhysical(state: *architecture.CpuState) void { const cap = architecture.systemCallArg(state, 0); const t = scheduler.current(); - if (t.aspace == 0) return fail(state); + if (t.address_space == 0) return fail(state); const shm = ipc.resolveShm(t, cap) orelse return fail(state); // not an shm handle we hold architecture.setSystemCallResult(state, shm.phys); } @@ -580,10 +588,10 @@ fn systemDeviceRegister(state: *architecture.CpuState) void { const parent_id = architecture.systemCallArg(state, 0); const descriptor_ptr = architecture.systemCallArg(state, 1); const t = scheduler.current(); - if (t.aspace == 0) return fail(state); + if (t.address_space == 0) return fail(state); var descriptor: device_abi.DeviceDescriptor = undefined; - if (!ipc.copyFromUser(t.aspace, descriptor_ptr, std.mem.asBytes(&descriptor))) return fail(state); + if (!ipc.copyFromUser(t.address_space, descriptor_ptr, std.mem.asBytes(&descriptor))) return fail(state); // Under the big kernel lock: the broker's table is also mutated by the // death sweep (releaseAllOwnedBy) and read by enumerate on other cores — @@ -667,7 +675,7 @@ fn systemThreadSpawn(state: *architecture.CpuState) void { const arg = architecture.systemCallArg(state, 2); const exit_handle = architecture.systemCallArg(state, 3); const t = scheduler.current(); - if (t.aspace == 0) return fail(state); // kernel tasks own no address space to share + if (t.address_space == 0) return fail(state); // kernel tasks own no address space to share if (entry == 0 or entry >= user_half_end) return fail(state); if (stack_top == 0 or stack_top > user_half_end) return fail(state); // The endpoint the thread notifies on exit (how join waits), or none. @@ -675,17 +683,17 @@ fn systemThreadSpawn(state: *architecture.CpuState) void { null else ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF); - const tid = spawnThreadSupervised(t.aspace, entry, stack_top, arg, t.priority, t.id, exit_endpoint) orelse return fail(state); + const tid = spawnThreadSupervised(t.address_space, entry, stack_top, arg, t.priority, t.id, exit_endpoint) orelse return fail(state); architecture.setSystemCallResult(state, tid); } -/// Spawn a thread sharing `aspace`, taking the exit-endpoint reference under the **same** +/// Spawn a thread sharing `address_space`, taking the exit-endpoint reference under the **same** /// lock as the spawn (as `spawnProcessSupervised` does), so the thread cannot die before /// its reference exists. Returns the new thread id, or null on resource exhaustion. -fn spawnThreadSupervised(aspace: u64, entry: u64, stack_top: u64, arg: u64, priority: scheduler.Priority, supervisor: u32, exit_endpoint: ?*ipc.Endpoint) ?u32 { +fn spawnThreadSupervised(address_space: u64, entry: u64, stack_top: u64, arg: u64, priority: scheduler.Priority, supervisor: u32, exit_endpoint: ?*ipc.Endpoint) ?u32 { const flags = sync.enter(); defer sync.leave(flags); - const tid = scheduler.spawnUserLocked(aspace, entry, stack_top, arg, priority, "thread", supervisor, if (exit_endpoint) |e| @ptrCast(e) else null) orelse return null; + const tid = scheduler.spawnUserLocked(address_space, entry, stack_top, arg, priority, "thread", supervisor, if (exit_endpoint) |e| @ptrCast(e) else null) orelse return null; if (exit_endpoint) |endpoint| endpoint.refcount += 1; // the thread holds it birth-to-death return tid; } @@ -700,6 +708,34 @@ fn systemThreadSelf(state: *architecture.CpuState) void { architecture.setSystemCallResult(state, scheduler.currentId()); } +/// set_thread_pointer(addr) -> 0: set the caller's user-space TLS thread pointer. The +/// arch layer maps it to IA32_FS_BASE on x86_64, `TPIDR_EL0` on aarch64; the kernel +/// never reads it, and the scheduler restores it per task across context switches +/// (docs/threading-plan.md M10). `addr` must be a user-half address. +fn systemSetThreadPointer(state: *architecture.CpuState) void { + const addr = architecture.systemCallArg(state, 0); + const t = scheduler.current(); + if (t.address_space == 0) return fail(state); // kernel tasks have no user TLS + if (addr >= user_half_end) return fail(state); + const flags = sync.enter(); + scheduler.setThreadPointerLocked(addr); + sync.leave(flags); + architecture.setSystemCallResult(state, 0); +} + +/// thread_join(tid) -> 0: block until the thread with id `tid` has exited (docs/threading- +/// plan.md M9). Needs no per-thread IPC endpoint. The compare-and-block is one critical +/// section, so an exit cannot slip between "is it alive?" and the block. +fn systemThreadJoin(state: *architecture.CpuState) void { + const tid: u32 = @truncate(architecture.systemCallArg(state, 0)); + const t = scheduler.current(); + if (t.address_space == 0) return fail(state); // kernel tasks don't join + const flags = sync.enter(); + scheduler.joinThreadLocked(tid); + sync.leave(flags); + architecture.setSystemCallResult(state, 0); +} + /// futex_wait(addr, expected, timeout_ns) -> status (docs/threading.md): if the 4-byte /// user word at `addr` still equals `expected`, block until a futex_wake on `addr` or /// (if timeout_ns > 0) the deadline. The compare and the block are one critical section, @@ -710,12 +746,12 @@ fn systemFutexWait(state: *architecture.CpuState) void { const expected: u32 = @truncate(architecture.systemCallArg(state, 1)); const timeout_ns = architecture.systemCallArg(state, 2); const t = scheduler.current(); - if (t.aspace == 0) return fail(state); + if (t.address_space == 0) return fail(state); if (addr == 0 or (addr & 3) != 0 or addr + 4 > user_half_end) return fail(state); const flags = sync.enter(); var word_bytes: [4]u8 = undefined; - if (!ipc.copyFromUser(t.aspace, addr, &word_bytes)) { + if (!ipc.copyFromUser(t.address_space, addr, &word_bytes)) { sync.leave(flags); return fail(state); } @@ -739,10 +775,10 @@ fn systemFutexWake(state: *architecture.CpuState) void { const addr = architecture.systemCallArg(state, 0); const count: u32 = @truncate(architecture.systemCallArg(state, 1)); const t = scheduler.current(); - if (t.aspace == 0) return fail(state); + if (t.address_space == 0) return fail(state); if (addr == 0 or (addr & 3) != 0 or addr + 4 > user_half_end) return fail(state); const flags = sync.enter(); - const woken = scheduler.futexWakeLocked(t.aspace, addr, count); + const woken = scheduler.futexWakeLocked(t.address_space, addr, count); sync.leave(flags); architecture.setSystemCallResult(state, woken); } @@ -757,7 +793,7 @@ fn systemProcessEnumerate(state: *architecture.CpuState) void { const buffer_ptr = architecture.systemCallArg(state, 0); const maximum = architecture.systemCallArg(state, 1); const t = scheduler.current(); - if (t.aspace == 0 or buffer_ptr >= user_half_end) return fail(state); + if (t.address_space == 0 or buffer_ptr >= user_half_end) return fail(state); const sz = @sizeOf(abi.ProcessDescriptor); const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half const out: [*]abi.ProcessDescriptor = @ptrFromInt(buffer_ptr); @@ -770,7 +806,7 @@ fn systemProcessEnumerate(state: *architecture.CpuState) void { /// cannot be a weapon (ids are never reused, so a stale one just misses). fn systemProcessKill(state: *architecture.CpuState) void { const t = scheduler.current(); - if (t.aspace == 0) return fail(state); + if (t.address_space == 0) return fail(state); const id = architecture.systemCallArg(state, 0); if (id > std.math.maxInt(u32)) return failErr(state, ipc.ESRCH); const r = killProcess(t.id, @intCast(id)); @@ -903,7 +939,7 @@ pub fn killProcess(caller_id: u32, target_id: u32) i64 { const flags = sync.enter(); defer sync.leave(flags); const target = scheduler.taskByIdLocked(target_id) orelse return -ipc.ESRCH; - if (target.aspace == 0) return -ipc.ESRCH; // kernel tasks are not processes + if (target.address_space == 0) return -ipc.ESRCH; // kernel tasks are not processes if (target.supervisor != caller_id) return -ipc.EPERM; target.exit_reason = .killed; if (target.state == .running) { @@ -973,7 +1009,7 @@ var exit_subscribers: [exit_subscriber_capacity]?ExitSubscriber = .{null} ** exi /// secret between cooperating processes. -ENOSPC when the table is full. fn systemProcessSubscribe(state: *architecture.CpuState) void { const t = scheduler.current(); - if (t.aspace == 0) return fail(state); + if (t.address_space == 0) return fail(state); const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF); const flags = sync.enter(); defer sync.leave(flags); @@ -993,7 +1029,7 @@ fn systemProcessSubscribe(state: *architecture.CpuState) void { /// delivered immediately on bind, coalesced into one notification. fn systemSignalBind(state: *architecture.CpuState) void { const t = scheduler.current(); - if (t.aspace == 0) return fail(state); + if (t.address_space == 0) return fail(state); const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF); const flags = sync.enter(); defer sync.leave(flags); @@ -1013,7 +1049,7 @@ fn systemSignalBind(state: *architecture.CpuState) void { /// targets accumulate the signal in their pending mask. fn systemProcessSignal(state: *architecture.CpuState) void { const t = scheduler.current(); - if (t.aspace == 0) return fail(state); + if (t.address_space == 0) return fail(state); const id = architecture.systemCallArg(state, 0); const signal = architecture.systemCallArg(state, 1); if (id > std.math.maxInt(u32)) return failErr(state, ipc.ESRCH); @@ -1021,7 +1057,7 @@ fn systemProcessSignal(state: *architecture.CpuState) void { const flags = sync.enter(); defer sync.leave(flags); const target = scheduler.taskByIdLocked(@intCast(id)) orelse return failErr(state, ipc.ESRCH); - if (target.aspace == 0) return failErr(state, ipc.ESRCH); + if (target.address_space == 0) return failErr(state, ipc.ESRCH); if (target.supervisor != t.id and target.id != t.id) return failErr(state, ipc.EPERM); target.pending_signals |= @as(u32, 1) << @intCast(signal); if (target.signal_endpoint) |raw| { @@ -1058,7 +1094,7 @@ fn timerSweepLocked() void { /// timer_bind(endpoint, ms): arm a one-shot timer. -ENOSPC when the table is full. fn systemTimerBind(state: *architecture.CpuState) void { const t = scheduler.current(); - if (t.aspace == 0) return fail(state); + if (t.address_space == 0) return fail(state); const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF); const ms = architecture.systemCallArg(state, 1); const flags = sync.enter(); @@ -1075,7 +1111,7 @@ fn systemTimerBind(state: *architecture.CpuState) void { fn systemProcessExitReason(state: *architecture.CpuState) void { const t = scheduler.current(); - if (t.aspace == 0) return fail(state); + if (t.address_space == 0) return fail(state); const id = architecture.systemCallArg(state, 0); if (id > std.math.maxInt(u32)) return failErr(state, ipc.ESRCH); const r = exitReasonOf(t.id, @intCast(id)); @@ -1101,7 +1137,7 @@ fn ownedGsi(t: *scheduler.Task, device_id: u64, resource_index: u64) ?u32 { /// IPC_ReplyWait and is woken by the ISR; see system/kernel/irq.zig for the cycle. fn systemIrqBind(state: *architecture.CpuState) void { const t = scheduler.current(); - if (t.aspace == 0) return fail(state); + if (t.address_space == 0) return fail(state); const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse return fail(state); const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 2)) orelse return fail(state); @@ -1121,7 +1157,7 @@ fn systemIrqBind(state: *architecture.CpuState) void { fn systemMsiBind(state: *architecture.CpuState) void { const device_id = architecture.systemCallArg(state, 0); const t = scheduler.current(); - if (t.aspace == 0) return fail(state); + if (t.address_space == 0) return fail(state); const owner = devices_broker.ownerOf(device_id) orelse return fail(state); if (owner != t.id) return fail(state); // not claimed by this process const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 1)) orelse return failErr(state, ipc.EBADF); @@ -1140,7 +1176,7 @@ fn systemMsiBind(state: *architecture.CpuState) void { /// more arrives until the driver says it has serviced the hardware. fn systemIrqAck(state: *architecture.CpuState) void { const t = scheduler.current(); - if (t.aspace == 0) return fail(state); + if (t.address_space == 0) return fail(state); const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse return fail(state); @@ -1228,37 +1264,52 @@ fn systemKlogRead(state: *architecture.CpuState) void { fn systemMmap(state: *architecture.CpuState) void { const len = architecture.systemCallArg(state, 0); const t = scheduler.current(); - if (t.aspace == 0) return fail(state); // not a user process — nothing to map into + if (t.address_space == 0) return fail(state); // not a user process — nothing to map into const pages = (len + page_size - 1) / page_size; if (pages == 0 or pages > maximum_mmap_pages) return fail(state); - if (t.heap_next == 0) t.heap_next = heap_arena_base; // seed the arena lazily - const base = t.heap_next; - if (base + pages * page_size > heap_arena_end) return fail(state); // arena exhausted + // Reserve a disjoint range under a *brief* lock (the cursor is shared by every thread + // in this address space). The mapping below then takes the lock **per page**, not for + // the whole grant: the big lock is held with interrupts disabled, so pinning it across + // a multi-MiB memset+map would freeze every other core on its next tick — which timed + // the `affinity` scenario out (docs/threading-plan.md M7). + const base = reserve: { + const flags = sync.enter(); + defer sync.leave(flags); + const cursor = scheduler.addressSpaceMmapNextPtr(t.address_space) orelse return fail(state); + if (cursor.* == 0) cursor.* = heap_arena_base; // seed the arena lazily + const b = cursor.*; + if (b + pages * page_size > heap_arena_end) return fail(state); // arena exhausted + cursor.* = b + pages * page_size; // reserve now, so concurrent grants can't overlap + break :reserve b; + }; - // Map page by page. On mid-way frame exhaustion, roll back the pages already mapped - // (unmap + free) so no partial grant leaks into the address space — the same - // all-or-nothing guarantee as before, but without a fixed scratch array, so the - // per-call size can be a multi-MiB framebuffer. + // Map the reserved range page by page, each page under a short-held lock (the range is + // already reserved, so pages can't overlap another thread's; the lock only serializes + // the shared page-table walk). On mid-way frame exhaustion, roll back the mapped pages + // so no partial grant leaks — the reserved-but-unmapped tail of the arena is left + // fallow (a rare, bounded address-space leak, not a memory leak). var mapped: usize = 0; while (mapped < pages) : (mapped += 1) { + const flags = sync.enter(); const frame = pmm.alloc() orelse { var i: usize = 0; while (i < mapped) : (i += 1) { const va = base + i * page_size; - if (architecture.translate(t.aspace, va)) |physical| { - architecture.unmapUserPageInto(t.aspace, va); + if (architecture.translate(t.address_space, va)) |physical| { + architecture.unmapUserPageInto(t.address_space, va); pmm.free(physical); } } + sync.leave(flags); return fail(state); }; const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(frame)); @memset(destination[0..page_size], 0); // hand out zeroed memory - architecture.mapUserPageInto(t.aspace, base + mapped * page_size, frame, true, false); // RW + NX + architecture.mapUserPageInto(t.address_space, base + mapped * page_size, frame, true, false); // RW + NX + sync.leave(flags); } - t.heap_next = base + pages * page_size; - architecture.setSystemCallResult(state, base); + architecture.setSystemCallResult(state, base); // the cursor was already advanced at reserve } /// munmap(base, len): release a range previously handed out by `mmap`. Unmaps @@ -1270,14 +1321,14 @@ fn systemMunmap(state: *architecture.CpuState) void { const base = architecture.systemCallArg(state, 0); const len = architecture.systemCallArg(state, 1); const t = scheduler.current(); - if (t.aspace == 0 or base % page_size != 0) return fail(state); + if (t.address_space == 0 or base % page_size != 0) return fail(state); const pages = (len + page_size - 1) / page_size; if (base < heap_arena_base or base + pages * page_size > heap_arena_end) return fail(state); for (0..pages) |i| { const va = base + i * page_size; - if (architecture.translate(t.aspace, va)) |physical| { - architecture.unmapUserPageInto(t.aspace, va); + if (architecture.translate(t.address_space, va)) |physical| { + architecture.unmapUserPageInto(t.address_space, va); pmm.free(physical); } } @@ -1341,7 +1392,7 @@ const maximum_segments = 16; const maximum_pages = 256; // 1 MiB loader budget; the user region caps at 2 MiB anyway const Segment = struct { - vaddr: u64, + virtual_address: u64, memsz: u64, filesz: u64, off: u64, @@ -1389,7 +1440,7 @@ fn parseSegments(image: []const u8, segs: *[maximum_segments]Segment) InitError! if (w and x) return error.BadSegment; // W^X, even for init const seg = Segment{ - .vaddr = phdr.p_vaddr, + .virtual_address = phdr.p_vaddr, .memsz = phdr.p_memsz, .filesz = phdr.p_filesz, .off = phdr.p_offset, @@ -1398,9 +1449,9 @@ fn parseSegments(image: []const u8, segs: *[maximum_segments]Segment) InitError! }; // No overlap with any earlier segment (page-granular, since mapping is). for (segs[0..count]) |other| { - const a_end = seg.vaddr + seg.pages() * page_size; - const b_end = other.vaddr + other.pages() * page_size; - if (seg.vaddr < b_end and other.vaddr < a_end) return error.BadSegment; + const a_end = seg.virtual_address + seg.pages() * page_size; + const b_end = other.virtual_address + other.pages() * page_size; + if (seg.virtual_address < b_end and other.virtual_address < a_end) return error.BadSegment; } total_pages += seg.pages(); if (total_pages > maximum_pages) return error.ProgramTooBig; @@ -1411,17 +1462,17 @@ fn parseSegments(image: []const u8, segs: *[maximum_segments]Segment) InitError! // The entry point must land inside an executable segment. for (segs[0..count]) |seg| { - if (seg.executable and ehdr.e_entry >= seg.vaddr and ehdr.e_entry < seg.vaddr + seg.memsz) + if (seg.executable and ehdr.e_entry >= seg.virtual_address and ehdr.e_entry < seg.virtual_address + seg.memsz) return .{ .count = count, .entry = ehdr.e_entry }; } return error.BadEntry; } -/// Load one page of a segment into address space `aspace`: a fresh frame, zeroed +/// Load one page of a segment into address space `address_space`: a fresh frame, zeroed /// and filled through the physmap, mapped user-accessible with the segment's W^X. /// On a later failure the whole address space is torn down, which frees every /// frame mapped into it — so no per-page rollback list is needed here. -fn loadPageInto(aspace: u64, image: []const u8, seg: Segment, page_index: u64) InitError!void { +fn loadPageInto(address_space: u64, image: []const u8, seg: Segment, page_index: u64) InitError!void { const frame = pmm.alloc() orelse return error.OutOfMemory; const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(frame)); @memset(destination[0..page_size], 0); @@ -1430,7 +1481,7 @@ fn loadPageInto(aspace: u64, image: []const u8, seg: Segment, page_index: u64) I const n = @min(page_size, seg.filesz - page_off); @memcpy(destination[0..n], image[seg.off + page_off ..][0..n]); } - architecture.mapUserPageInto(aspace, seg.vaddr + page_off, frame, seg.writable, seg.executable); + architecture.mapUserPageInto(address_space, seg.virtual_address + page_off, frame, seg.writable, seg.executable); } /// Build the System V AMD64 process-entry block at the top of a process's stack @@ -1517,11 +1568,11 @@ pub fn spawnProcessSupervised(image: []const u8, priority: u3, argv: []const []c const flags = sync.enter(); defer sync.leave(flags); - const aspace = architecture.createAddressSpace() orelse return error.OutOfMemory; - errdefer architecture.destroyAddressSpace(aspace); + const address_space = architecture.createAddressSpace() orelse return error.OutOfMemory; + errdefer architecture.destroyAddressSpace(address_space); for (segs[0..parsed.count]) |seg| { - for (0..seg.pages()) |i| try loadPageInto(aspace, image, seg, i); + for (0..seg.pages()) |i| try loadPageInto(address_space, image, seg, i); } // The stack: `user_stack_pages` zeroed pages below stack_top_virtual, RW + NX. @@ -1535,10 +1586,10 @@ pub fn spawnProcessSupervised(image: []const u8, priority: u3, argv: []const []c const page_virtual = stack_base_virtual + i * page_size; if (i == parameters.user_stack_pages - 1) user_sp = buildEntryStack(stack_page, page_virtual, argv); - architecture.mapUserPageInto(aspace, page_virtual, stack_frame, true, false); // RW + NX + architecture.mapUserPageInto(address_space, page_virtual, stack_frame, true, false); // RW + NX } - const child = scheduler.spawnUserLocked(aspace, parsed.entry, user_sp, 0, priority, argv[0], supervisor, if (exit_endpoint) |endpoint| @ptrCast(endpoint) else null) orelse + const child = scheduler.spawnUserLocked(address_space, parsed.entry, user_sp, 0, priority, argv[0], supervisor, if (exit_endpoint) |endpoint| @ptrCast(endpoint) else null) orelse return error.OutOfMemory; // The child holds a reference to its exit endpoint from birth to death. Taken // only now, after nothing can fail; the lock is still held, so the child diff --git a/system/kernel/scheduler.zig b/system/kernel/scheduler.zig index 4a91ef8..b6ede31 100644 --- a/system/kernel/scheduler.zig +++ b/system/kernel/scheduler.zig @@ -31,7 +31,10 @@ const number_priorities = 8; const stack_size = parameters.kernel_stack_size; // each task's kernel stack const maximum_tasks = parameters.maximum_tasks; // maximum tasks alive at once (static pool) -const State = enum { free, ready, running, blocked }; +// `reaping` = the task has exited and is queued on its core's reap list; its slot must not +// be reused (freeSlot skips it) until the reaper has freed its kernel stack and set it +// `free` (docs/threading-plan.md M8/M9). +const State = enum { free, ready, running, blocked, reaping }; pub const Task = struct { id: u32 = 0, @@ -75,21 +78,25 @@ pub const Task = struct { ipc_wait_endpoint: ?*anyopaque = null, // Physical root of this task's address space, or 0 for a kernel task (which // runs on the shared kernel page tables). A user task carries its own. - aspace: u64 = 0, + address_space: u64 = 0, user_ip: u64 = 0, // user-mode entry point (user task only) user_sp: u64 = 0, // user-mode stack pointer (user task only) - user_arg: u64 = 0, // value delivered in the user's rdi at first entry: 0 for a - // process (its _start ignores it), the closure pointer for a thread (docs/threading.md) + user_arg: u64 = 0, // value delivered in the user's first argument register at first entry + // (rdi on x86_64, via architecture.jumpToUserArg): 0 for a process (its _start ignores + // it), the closure pointer for a thread (docs/threading.md) // The user address this task is blocked on in futex_wait (0 = not futex-waiting). // Cleared to 0 by futexWakeLocked as the "woken, not timed out" signal (docs/threading.md). futex_addr: u64 = 0, - // Next free virtual address in this task's mmap grant arena (0 = uninitialised; - // process.zig lazily seeds it to the arena base on the first mmap). Bumped up - // as the user heap grows; user task only. - heap_next: u64 = 0, - // Next free virtual address in this task's MMIO-grant arena (PML4[226]; 0 = - // uninitialised, process.zig seeds it on the first mmio_map). User task only. - device_map_next: u64 = 0, + // The task id this task is blocked in `thread_join` on (0 = not joining). Woken by + // `wakeJoinersLocked` when that task exits (docs/threading-plan.md M9). + join_target: u32 = 0, + // This task's user-space TLS thread pointer — 0 until set via `set_thread_pointer`. + // Architecture-neutral: the arch layer maps it to the FS base on x86_64, `TPIDR_EL0` on + // aarch64. Restored on every context switch to this task (docs/threading-plan.md M10). + thread_pointer: u64 = 0, + // The mmap / MMIO grant-arena cursors moved from Task to the per-address-space object + // (`AddressSpaceRef`, below) so threads sharing one address space hand out disjoint grants + // — see addressSpaceMmapNextPtr / addressSpaceDeviceMapNextPtr (docs/threading-plan.md M7). // --- synchronous IPC (ipc_sync.zig) --- // Per-process handle table: a small-int handle names a kernel capability object. // Each entry tags its `kind` (an IPC endpoint or a shared-memory object) so the @@ -100,9 +107,9 @@ pub const Task = struct { // receive, cleared when it replies). A client, while blocked in Call, records // its message + reply buffers here and its result lands in `ipc_status`. ipc_client: ?*Task = null, - ipc_send_ptr: u64 = 0, // client: outgoing message (vaddr in this task's AS) + ipc_send_ptr: u64 = 0, // client: outgoing message (virtual_address in this task's address space) ipc_send_len: u64 = 0, - ipc_reply_ptr: u64 = 0, // client: reply buffer (vaddr) + ipc_reply_ptr: u64 = 0, // client: reply buffer (virtual_address) ipc_reply_cap: u64 = 0, ipc_status: i64 = 0, // client: reply length / -errno, written by the replier dma_map_next: u64 = 0, // bump pointer into this task's DMA arena (0 = unseeded) @@ -148,17 +155,57 @@ var tasks = [_]Task{.{}} ** maximum_tasks; /// only when the **last** task on an address space exits. All access is under the big /// kernel lock. There can be no more live address spaces than tasks, so the table is /// sized to the task pool and never overflows in practice. -const AspaceRef = struct { root: u64 = 0, count: u32 = 0 }; -var aspace_refs = [_]AspaceRef{.{}} ** maximum_tasks; -var aspace_destroy_count: u64 = 0; +// The per-address-space kernel object: a reference count plus the grant-arena cursors. +// One live entry per address space; threads sharing an address space share this entry, +// so their mmap/mmio grants bump one cursor and never overlap (docs/threading-plan.md M7). +// `mmap_next`/`device_map_next` are 0 until process.zig seeds them to the arena base. +const AddressSpaceRef = struct { root: u64 = 0, count: u32 = 0, mmap_next: u64 = 0, device_map_next: u64 = 0 }; +var address_space_refs = [_]AddressSpaceRef{.{}} ** maximum_tasks; +var address_space_destroy_count: u64 = 0; + +/// Total bytes of task **kernel** stacks currently allocated from the kernel heap — +/// incremented when a task is created, decremented when the reaper frees a dead task's +/// stack. A test-observable proof that the reaper reclaims every stack (docs/threading- +/// plan.md M8): with no live tasks beyond the baseline, this returns to its baseline. +var live_stack_bytes: usize = 0; + +/// Test-observable: bytes of task kernel stacks currently live (see `live_stack_bytes`). +pub fn liveStackBytes() usize { + return live_stack_bytes; +} + +/// Free a dead task's kernel stack and drop it from `live_stack_bytes`. The task must be +/// off that stack already (killed while not running, or reaped after it switched away). +/// Caller holds the kernel lock. +fn reapStackLocked(t: *Task) void { + if (t.stack.len == 0) return; // boot/idle tasks run on a static stack — nothing to free + live_stack_bytes -= t.stack.len; + heap.allocator().free(t.stack); + t.stack = &.{}; + t.kstack_top = 0; +} + +/// Free every `.reaping` task queued on this core's reap list and mark each `.free` (now +/// its slot may be reused). The tasks are all off their stacks (they switched away), and +/// the caller holds the lock, so freeing is safe (docs/threading-plan.md M8/M9). +fn drainReapListLocked(pc: *PerCpu) void { + var node = pc.reap_list; + pc.reap_list = null; + while (node) |t| { + node = t.next; // save the link before we clear it + t.next = null; + reapStackLocked(t); + t.state = .free; // reusable only now, after the stack is freed + } +} /// Take a reference to address space `root` (0 = a kernel task, which owns none). /// Returns false only if the ref table is full — bounded by `maximum_tasks`, so in /// practice it never is. Caller holds the kernel lock. -fn retainAspace(root: u64) bool { +fn retainAddressSpace(root: u64) bool { if (root == 0) return true; - var free: ?*AspaceRef = null; - for (&aspace_refs) |*entry| { + var free: ?*AddressSpaceRef = null; + for (&address_space_refs) |*entry| { if (entry.count != 0 and entry.root == root) { entry.count += 1; return true; @@ -173,34 +220,54 @@ fn retainAspace(root: u64) bool { /// Drop a reference to `root`; destroy the address space when the **last** one drops. /// A `root` with no entry — never retained, e.g. a hand-built test space — is /// destroyed directly, preserving the pre-refcount behaviour. Caller holds the lock. -fn releaseAspace(root: u64) void { +fn releaseAddressSpace(root: u64) void { if (root == 0) return; - for (&aspace_refs) |*entry| { + for (&address_space_refs) |*entry| { if (entry.count == 0 or entry.root != root) continue; entry.count -= 1; if (entry.count == 0) { entry.root = 0; architecture.destroyAddressSpace(root); - aspace_destroy_count += 1; + address_space_destroy_count += 1; } return; } architecture.destroyAddressSpace(root); - aspace_destroy_count += 1; + address_space_destroy_count += 1; } /// Test-observable: how many address spaces are live (entries with a nonzero count). -pub fn liveAspaceCount() u32 { +pub fn liveAddressSpaceCount() u32 { var live: u32 = 0; - for (&aspace_refs) |*entry| { + for (&address_space_refs) |*entry| { if (entry.count != 0) live += 1; } return live; } /// Test-observable: total address-space destructions since boot. -pub fn aspaceDestroyCount() u64 { - return aspace_destroy_count; +pub fn addressSpaceDestroyCount() u64 { + return address_space_destroy_count; +} + +/// Pointer to the mmap grant-arena cursor for address space `root`, so the mmap syscall +/// can read-and-bump it. Per-address-space (not per-task), so sibling threads get +/// disjoint grants. **Caller holds the kernel lock** (the entry is stable while held). +/// Null only if `root` was never retained — which can't happen for a live user task. +pub fn addressSpaceMmapNextPtr(root: u64) ?*u64 { + for (&address_space_refs) |*entry| { + if (entry.count != 0 and entry.root == root) return &entry.mmap_next; + } + return null; +} + +/// Pointer to the MMIO grant-arena cursor for address space `root` (see +/// `addressSpaceMmapNextPtr`). Caller holds the kernel lock. +pub fn addressSpaceDeviceMapNextPtr(root: u64) ?*u64 { + for (&address_space_refs) |*entry| { + if (entry.count != 0 and entry.root == root) return &entry.device_map_next; + } + return null; } var next_id: u32 = 1; @@ -221,11 +288,18 @@ pub const PerCpu = struct { hw_id: u32 = 0, // the core's hardware id (Local APIC id on x86_64) index: u32 = 0, // dense 0-based core index online: bool = false, // has this core finished bring-up? - loaded_aspace: u64 = 0, // the address-space root currently loaded on this core + loaded_address_space: u64 = 0, // the address-space root currently loaded on this core + loaded_thread_pointer: u64 = 0, // the TLS thread pointer currently loaded on this core (docs/threading-plan.md M10) // Tasks pinned to this core (affinity == index), per priority level + bitmap. pinned_head: [number_priorities]?*Task = .{null} ** number_priorities, pinned_tail: [number_priorities]?*Task = .{null} ** number_priorities, pinned_bitmap: u8 = 0, + // Tasks that ended while running on THIS core: they could not free the kernel stack + // they were standing on, so each pushed itself onto this list (`.reaping` state, linked + // via `Task.next`) and switched away. The next task to run on this core — or the timer + // tick — frees their stacks from its own stack, safely (docs/threading-plan.md M8). A + // *list* (not one slot) so a second death before the first is drained can't lose it. + reap_list: ?*Task = null, }; const maximum_cpus = parameters.maximum_cpus; @@ -257,7 +331,7 @@ var preemption_enabled = true; /// boot, before interrupts are enabled — so no lock is needed here. pub fn init(boot_priority: Priority) void { const pc = &cpus[0]; - pc.* = .{ .index = 0, .online = true, .loaded_aspace = architecture.kernelPageTable() }; + pc.* = .{ .index = 0, .online = true, .loaded_address_space = architecture.kernelPageTable() }; architecture.setCpuLocal(0, @intFromPtr(pc)); tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority }; pc.current = &tasks[0]; @@ -296,7 +370,7 @@ pub fn secondaryMain() callconv(.c) noreturn { pc.current = t; pc.idle = t; pc.online = true; - pc.loaded_aspace = architecture.kernelPageTable(); // the AP adopted the kernel tables at bring-up + pc.loaded_address_space = architecture.kernelPageTable(); // the AP adopted the kernel tables at bring-up sync.leave(flags); architecture.enableInterrupts(); // the timer now preempts this idle context into work @@ -382,7 +456,7 @@ pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool { return ok; } -/// Spawn a **user** task: a task with its own address space (`aspace`) that starts +/// Spawn a **user** task: a task with its own address space (`address_space`) that starts /// in user mode at `entry` on `user_sp`, recorded under `name` (its argv[0]). /// `supervisor` is the id of the spawning process (0 = the kernel) — the kill /// authority — and `exit_endpoint` (an *ipc.Endpoint whose reference the caller @@ -391,23 +465,24 @@ pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool { /// lands in `user_task_trampoline`. /// Returns the new process id, or null (creating nothing) if the table is full or /// out of memory. -/// **Caller must hold the kernel lock** (the loader that builds `aspace` holds it +/// **Caller must hold the kernel lock** (the loader that builds `address_space` holds it /// across the whole spawn, so the address space and the task appear atomically). -pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, user_arg: u64, priority: Priority, task_name: []const u8, supervisor: u32, exit_endpoint: ?*anyopaque) ?u32 { +pub fn spawnUserLocked(address_space: u64, entry: u64, user_sp: u64, user_arg: u64, priority: Priority, task_name: []const u8, supervisor: u32, exit_endpoint: ?*anyopaque) ?u32 { const t = freeSlot() orelse return null; const stack = heap.allocator().alloc(u8, stack_size) catch return null; // Take this task's reference to the address space before we commit the slot, so a - // failure here leaves nothing to unwind (the caller still owns the raw `aspace`). - if (!retainAspace(aspace)) { + // failure here leaves nothing to unwind (the caller still owns the raw `address_space`). + if (!retainAddressSpace(address_space)) { heap.allocator().free(stack); return null; } + live_stack_bytes += stack.len; // the reaper drops this when the task dies (M8) t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack, - .aspace = aspace, + .address_space = address_space, .user_ip = entry, .user_sp = user_sp, .user_arg = user_arg, @@ -445,6 +520,7 @@ fn startUserTask() void { fn create(entry: *const fn () void, priority: Priority, affinity: ?u32) *Task { const t = freeSlot() orelse @panic("sched: task table full"); const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack"); + live_stack_bytes += stack.len; // the reaper drops this when the task dies (M8) t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack, .affinity = affinity }; next_id += 1; const top = @intFromPtr(stack.ptr) + stack.len; @@ -485,7 +561,7 @@ fn schedule() void { /// Make `next` this core's running task: publish its kernel stack (TSS.rsp0, so a /// user-mode interrupt lands on a good stack) and its address space (only when /// it differs from what's loaded — every page-table switch is a full TLB flush), -/// then switch registers/stacks. Kernel tasks (aspace == 0, no kstack_top used +/// then switch registers/stacks. Kernel tasks (address_space == 0, no kstack_top used /// from user mode) resolve to the shared kernel page tables and skip the kernel- /// stack write, so this is a no-op beyond the register switch for a pure-kernel /// workload. The big kernel lock is held and interrupts are off throughout, so no @@ -493,12 +569,25 @@ fn schedule() void { /// `save_sp` receives the outgoing task's stack pointer. fn switchTo(pc: *PerCpu, save_sp: *usize, next: *Task) void { if (next.kstack_top != 0) architecture.setKernelStack(pc.index, next.kstack_top); - const want = if (next.aspace != 0) next.aspace else architecture.kernelPageTable(); - if (want != pc.loaded_aspace) { + const want = if (next.address_space != 0) next.address_space else architecture.kernelPageTable(); + if (want != pc.loaded_address_space) { architecture.loadPageTable(want); - pc.loaded_aspace = want; + pc.loaded_address_space = want; + } + // Restore the next task's user TLS thread pointer — only on change, the same + // conditional-load discipline as CR3 above (docs/threading-plan.md M10). + if (next.thread_pointer != pc.loaded_thread_pointer) { + architecture.setThreadPointer(next.thread_pointer); + pc.loaded_thread_pointer = next.thread_pointer; } architecture.switchContext(save_sp, next.sp); + // Resumed now (switchContext returned into our own switchTo frame). Re-fetch the core + // via thisCpu(): the `pc` parameter is from *our* earlier switchTo call, so it names + // the core we last ran on — stale if we migrated. switchContext only swaps stacks on + // the current core, so thisCpu() is the core the just-dead task died on. If a task + // died switching to us, free its kernel stack: we're on ours so it's safe, and the big + // lock is still held so its slot can't have been reused (docs/threading-plan.md M8). + drainReapListLocked(thisCpu()); } /// Voluntarily give up the CPU to the next ready task. @@ -546,12 +635,12 @@ pub fn futexWaitLocked(addr: u64, timeout_ms: u64) FutexResult { } /// Wake up to `count` tasks blocked in `futex_wait` on `addr` in address space -/// `aspace`. Precondition: the big kernel lock is held. Returns how many woke. -pub fn futexWakeLocked(aspace: u64, addr: u64, count: u32) u32 { +/// `address_space`. Precondition: the big kernel lock is held. Returns how many woke. +pub fn futexWakeLocked(address_space: u64, addr: u64, count: u32) u32 { var woken: u32 = 0; for (&tasks) |*t| { if (woken >= count) break; - if (t.state == .blocked and t.aspace == aspace and t.futex_addr == addr) { + if (t.state == .blocked and t.address_space == address_space and t.futex_addr == addr) { t.futex_addr = 0; // the "woken, not timed out" signal to futexWaitLocked t.wake_at = 0; t.state = .ready; @@ -562,6 +651,55 @@ pub fn futexWakeLocked(aspace: u64, addr: u64, count: u32) u32 { return woken; } +// --- thread join (docs/threading-plan.md M9) -------------------------------- +// +// join needs no per-thread IPC endpoint: `thread_join(tid)` blocks the caller until the +// task with id `tid` has exited, and the exit paths wake any joiner. The caller only ever +// reclaims the joined thread's *user* stack (which the thread vacated the moment it entered +// the kernel to exit), so waking at exit time — not reap time — is safe. + +/// True if a task with id `tid` is still live (has not exited). Caller holds the lock. +fn aliveTid(tid: u32) bool { + for (&tasks) |*t| { + if (t.id == tid and t.state != .free and t.state != .reaping) return true; + } + return false; +} + +/// Block the current task until the task with id `tid` exits (or return at once if it +/// already has / never existed). **Precondition:** the big kernel lock is held; returns +/// with it still held. Woken by `wakeJoinersLocked`. +pub fn joinThreadLocked(tid: u32) void { + while (aliveTid(tid)) { + const t = current(); + t.join_target = tid; + t.state = .blocked; + schedule(); // woken when the joined task exits; lock handed off across the switch + t.join_target = 0; + } +} + +/// Set the calling task's user TLS thread pointer and load it now. Persisted on the Task so +/// context switches restore it (docs/threading-plan.md M10). Caller holds the kernel lock. +pub fn setThreadPointerLocked(addr: u64) void { + const pc = thisCpu(); + pc.current.thread_pointer = addr; + architecture.setThreadPointer(addr); + pc.loaded_thread_pointer = addr; +} + +/// Wake every task blocked in `thread_join` on `tid` — called from the exit paths once the +/// exiting task's state is `.free`. Caller holds the lock. +fn wakeJoinersLocked(tid: u32) void { + for (&tasks) |*t| { + if (t.state == .blocked and t.join_target == tid) { + t.join_target = 0; + t.state = .ready; + enqueue(t); + } + } +} + // --- event-based blocking ------------------------------------------------- // // A WaitQueue is a set of tasks blocked waiting for something (a resource, a @@ -663,7 +801,7 @@ fn removeFrom(head: *[number_priorities]?*Task, tail: *[number_priorities]?*Task /// Precondition: the big kernel lock is held. pub fn taskByIdLocked(id: u32) ?*Task { for (&tasks) |*t| { - if (t.state != .free and t.id == id) return t; + if (t.state != .free and t.state != .reaping and t.id == id) return t; } return null; } @@ -673,7 +811,7 @@ pub fn taskByIdLocked(id: u32) ?*Task { /// Precondition: the big kernel lock is held. pub fn forgetIpcClientLocked(t: *Task) void { for (&tasks) |*other| { - if (other.state != .free and other.ipc_client == t) other.ipc_client = null; + if (other.state != .free and other.state != .reaping and other.ipc_client == t) other.ipc_client = null; } } @@ -765,7 +903,11 @@ pub var reap_task_hook: ?*const fn (*Task) void = null; fn reapKillPendingLocked() void { const pc = thisCpu(); const cur = pc.current; - if (cur.kill_pending and cur.aspace != 0 and !cur.in_system_call) { + // Safety net: if a dying task switched to a *fresh* task (which enters via + // task_trampoline, not switchTo's tail), its stack is still queued here. The dying + // task switched away before this tick, so it is off its stack — drain now (M8). + drainReapListLocked(pc); + if (cur.kill_pending and cur.address_space != 0 and !cur.in_system_call) { if (terminate_current_hook) |hook| hook(); // noreturn } if (reap_task_hook) |hook| { @@ -806,7 +948,11 @@ pub fn setPreemption(enabled: bool) void { pub fn exit() noreturn { _ = sync.enter(); const pc = thisCpu(); - pc.current.state = .free; + // Queue this task for reaping: `.reaping` keeps its slot out of freeSlot until its + // stack is freed; `next` links it on the core's reap list (M8/M9). + pc.current.state = .reaping; + pc.current.next = pc.reap_list; + pc.reap_list = pc.current; const next = dequeueHighest(pc) orelse @panic("sched: no task left to run"); next.state = .running; pc.current = next; @@ -831,17 +977,21 @@ pub fn exitUser() noreturn { pub fn exitUserLocked() noreturn { const pc = thisCpu(); const dying = pc.current; - const as = dying.aspace; + const as = dying.address_space; if (as != 0) { const kroot = architecture.kernelPageTable(); architecture.loadPageTable(kroot); // off the process tables before freeing them - pc.loaded_aspace = kroot; - releaseAspace(as); // destroys only when this was the last task on the space + pc.loaded_address_space = kroot; + releaseAddressSpace(as); // destroys only when this was the last task on the space } - dying.state = .free; - dying.aspace = 0; + dying.state = .reaping; // dead but its slot stays reserved until the stack is freed + wakeJoinersLocked(dying.id); // let any thread_join(dying.id) return (M9) + dying.address_space = 0; dying.kill_pending = false; dying.in_system_call = false; + // Queue for reaping: the task we switch to (or the next tick) frees this stack (M8/M9). + dying.next = pc.reap_list; + pc.reap_list = dying; const next = dequeueHighest(pc) orelse @panic("sched: no task left to run"); next.state = .running; pc.current = next; @@ -857,12 +1007,14 @@ pub fn exitUserLocked() noreturn { /// task isn't running). The kernel stack is leaked, as in `exitUser` (no reaper /// yet). Precondition: the big kernel lock is held. pub fn destroyTaskLocked(t: *Task) void { - if (t.aspace != 0) releaseAspace(t.aspace); // destroys only on the last reference - t.aspace = 0; + if (t.address_space != 0) releaseAddressSpace(t.address_space); // destroys only on the last reference + reapStackLocked(t); // safe to free now: `t` is not running on any core (M8) + t.address_space = 0; t.kill_pending = false; t.in_system_call = false; t.wake_at = 0; t.state = .free; + wakeJoinersLocked(t.id); // a killed thread's joiners must return too (M9) } /// Snapshot the task table into `out` (up to its length), returning the total @@ -876,7 +1028,7 @@ pub fn enumerate(out: []abi.ProcessDescriptor) u64 { defer sync.leave(flags); var total: u64 = 0; for (&tasks) |*t| { - if (t.state == .free) continue; + if (t.state == .free or t.state == .reaping) continue; // reaping = already exited if (total < out.len) { const d = &out[total]; d.* = .{ @@ -886,7 +1038,7 @@ pub fn enumerate(out: []abi.ProcessDescriptor) u64 { .ready => .ready, .running => .running, .blocked => .blocked, - .free => unreachable, + .free, .reaping => unreachable, })), .priority = t.priority, .name_length = t.name_length, @@ -900,7 +1052,7 @@ pub fn enumerate(out: []abi.ProcessDescriptor) u64 { /// Whether the running task is a user process (has its own address space). pub fn currentIsUserProcess() bool { - return current().aspace != 0; + return current().address_space != 0; } pub fn currentId() u32 { diff --git a/system/kernel/tests.zig b/system/kernel/tests.zig index d1c0f61..f6126c5 100644 --- a/system/kernel/tests.zig +++ b/system/kernel/tests.zig @@ -139,8 +139,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void { userPfTest(); } else if (eql(case, "fault-recovery")) { faultRecoveryTest(boot_information); - } else if (eql(case, "aspace-refcount")) { - aspaceRefcountTest(boot_information); + } else if (eql(case, "address-space-refcount")) { + addressSpaceRefcountTest(boot_information); } else if (eql(case, "thread-spawn")) { threadSpawnTest(boot_information); } else if (eql(case, "thread-join")) { @@ -151,6 +151,14 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void { threadMutexTest(boot_information); } else if (eql(case, "thread-id")) { threadIdTest(boot_information); + } else if (eql(case, "thread-alloc")) { + threadAllocTest(boot_information); + } else if (eql(case, "task-reap")) { + taskReapTest(boot_information); + } else if (eql(case, "thread-tls")) { + threadTlsTest(boot_information); + } else if (eql(case, "thread-rwlock")) { + threadRwlockTest(boot_information); } else if (eql(case, "args")) { argsTest(boot_information); } else if (eql(case, "init")) { @@ -779,11 +787,15 @@ fn affinityTest() void { return; } - var spins: u64 = 0; - while (spins < 3_000_000_000) spins +%= 1; // many time slices across the cores + // Let many time slices pass so the scheduler runs the pinned worker across ticks. + // Wait on the wall clock, not a raw iteration count: a fixed-count busy-loop's + // wall-time is a codegen lottery (the optimiser may elide or vectorise it), so an + // unrelated change elsewhere in this file could swing this test from ~4 s to ~50 s. + const run_until = architecture.millis() + 400; + while (architecture.millis() < run_until) {} affinity_running = false; - var settle: u64 = 0; - while (settle < 200_000_000) settle +%= 1; // let the worker see the flag and exit + const settle_until = architecture.millis() + 50; + while (architecture.millis() < settle_until) {} // let the worker see the flag and exit var others: u32 = 0; for (affinity_cores, 0..) |seen, c| { @@ -908,12 +920,12 @@ fn userMemTest() void { log("DANOS-TEST-BEGIN: usermem\n", .{}); const base_free = pmm.stats().free_frames; - const aspace = architecture.createAddressSpace() orelse { + const address_space = architecture.createAddressSpace() orelse { check("created a fresh address space", false); result(); return; }; - check("created a fresh address space", aspace != 0); + check("created a fresh address space", address_space != 0); // Grant three pages into the arena, mapped RW + NX (the mmap contract). const npages = 3; @@ -922,7 +934,7 @@ fn userMemTest() void { var mapped: usize = 0; while (mapped < npages) : (mapped += 1) { frames[mapped] = pmm.alloc() orelse break; - architecture.mapUserPageInto(aspace, arena + mapped * abi.page_size, frames[mapped], true, false); + architecture.mapUserPageInto(address_space, arena + mapped * abi.page_size, frames[mapped], true, false); } check("granted three user pages", mapped == npages); @@ -931,7 +943,7 @@ fn userMemTest() void { var rw_ok = true; for (0..npages) |i| { const va = arena + i * abi.page_size; - const physical = architecture.translate(aspace, va) orelse { + const physical = architecture.translate(address_space, va) orelse { translate_ok = false; continue; }; @@ -946,13 +958,13 @@ fn userMemTest() void { // Release them the way munmap does, then tear down the address space. for (0..npages) |i| { const va = arena + i * abi.page_size; - if (architecture.translate(aspace, va)) |physical| { - architecture.unmapUserPageInto(aspace, va); + if (architecture.translate(address_space, va)) |physical| { + architecture.unmapUserPageInto(address_space, va); pmm.free(physical); } } - check("munmap unmapped every grant", architecture.translate(aspace, arena) == null); - architecture.destroyAddressSpace(aspace); + check("munmap unmapped every grant", architecture.translate(address_space, arena) == null); + architecture.destroyAddressSpace(address_space); check("no frames leaked (free count restored)", pmm.stats().free_frames == base_free); result(); @@ -1120,12 +1132,12 @@ fn dmaTest() void { // Map the run into a fresh address space as coherent DMA and translate each page // back: the same physical run, in order — proving contiguity and the mapping. - const aspace = architecture.createAddressSpace().?; - architecture.mapUserDmaInto(aspace, process.dma_arena_base, phys, frames * abi.page_size); + const address_space = architecture.createAddressSpace().?; + architecture.mapUserDmaInto(address_space, process.dma_arena_base, phys, frames * abi.page_size); var mapped_ok = true; for (0..frames) |i| { const va = process.dma_arena_base + i * abi.page_size; - const got = architecture.translate(aspace, va) orelse { + const got = architecture.translate(address_space, va) orelse { mapped_ok = false; break; }; @@ -1135,7 +1147,7 @@ fn dmaTest() void { // Teardown must reclaim the DMA RAM (the leaves carry no device_grant, so // freeSubtree frees them as ordinary frames) — a driver that just dies leaks none. - architecture.destroyAddressSpace(aspace); + architecture.destroyAddressSpace(address_space); for (0..2) |i| pmm.free(low + i * abi.page_size); check("no frames leaked after DMA teardown", pmm.stats().free_frames == base_free); result(); @@ -1367,9 +1379,9 @@ fn spawnFaultingProcess() ?u32 { const flags = sync.enter(); defer sync.leave(flags); - const aspace = architecture.createAddressSpace() orelse return null; + const address_space = architecture.createAddressSpace() orelse return null; const code_frame = pmm.alloc() orelse { - architecture.destroyAddressSpace(aspace); + architecture.destroyAddressSpace(address_space); return null; }; // Fill through the physmap (the user mapping is read-only); pad with int3 so a @@ -1377,17 +1389,17 @@ fn spawnFaultingProcess() ?u32 { const code: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(code_frame)); @memset(code[0..abi.page_size], 0xCC); @memcpy(code[0..blob.len], blob); - architecture.mapUserPageInto(aspace, process.code_virtual, code_frame, false, true); // RO + X + architecture.mapUserPageInto(address_space, process.code_virtual, code_frame, false, true); // RO + X const stack_frame = pmm.alloc() orelse { - architecture.destroyAddressSpace(aspace); // frees code_frame too — it's mapped + architecture.destroyAddressSpace(address_space); // frees code_frame too — it's mapped return null; }; - architecture.mapUserPageInto(aspace, process.stack_base_virtual, stack_frame, true, false); // RW + NX + architecture.mapUserPageInto(address_space, process.stack_base_virtual, stack_frame, true, false); // RW + NX // Supervised by the calling test task, so exitReasonOf can read the verdict. - const id = scheduler.spawnUserLocked(aspace, process.code_virtual, process.stack_base_virtual + abi.page_size, 0, 4, "fault-probe", scheduler.currentId(), null) orelse { - architecture.destroyAddressSpace(aspace); + const id = scheduler.spawnUserLocked(address_space, process.code_virtual, process.stack_base_virtual + abi.page_size, 0, 4, "fault-probe", scheduler.currentId(), null) orelse { + architecture.destroyAddressSpace(address_space); return null; }; return id; @@ -1448,11 +1460,11 @@ fn faultRecoveryTest(boot_information: *const BootInformation) void { /// address spaces returns to baseline while destructions advance by exactly that many. /// This is the foundation threads (shared address spaces) build on: the refactor must be /// invisible while every space still has exactly one task. -fn aspaceRefcountTest(boot_information: *const BootInformation) void { +fn addressSpaceRefcountTest(boot_information: *const BootInformation) void { _ = boot_information; - log("DANOS-TEST-BEGIN: aspace-refcount\n", .{}); - const base_live = scheduler.liveAspaceCount(); - const base_destroyed = scheduler.aspaceDestroyCount(); + log("DANOS-TEST-BEGIN: address-space-refcount\n", .{}); + const base_live = scheduler.liveAddressSpaceCount(); + const base_destroyed = scheduler.addressSpaceDestroyCount(); const rounds: u32 = 5; var killed: u32 = 0; var round: u32 = 0; @@ -1468,11 +1480,11 @@ fn aspaceRefcountTest(boot_information: *const BootInformation) void { if (process.fault_kill_count >= 1) killed += 1; } check("all probes spawned and were killed", killed == rounds); - check("live address-space count returned to baseline", scheduler.liveAspaceCount() == base_live); - check("each address space destroyed exactly once", scheduler.aspaceDestroyCount() == base_destroyed + rounds); - if (killed == rounds and scheduler.liveAspaceCount() == base_live and - scheduler.aspaceDestroyCount() == base_destroyed + rounds) - log("aspace-refcount: spaces released to baseline ok\n", .{}); + check("live address-space count returned to baseline", scheduler.liveAddressSpaceCount() == base_live); + check("each address space destroyed exactly once", scheduler.addressSpaceDestroyCount() == base_destroyed + rounds); + if (killed == rounds and scheduler.liveAddressSpaceCount() == base_live and + scheduler.addressSpaceDestroyCount() == base_destroyed + rounds) + log("address-space-refcount: spaces released to baseline ok\n", .{}); result(); } @@ -1497,7 +1509,7 @@ fn threadSpawnTest(boot_information: *const BootInformation) void { check("thread-test spawned", spawnNamed(rd, "thread-test")); // Wait for the service's verdict marker (it polls shared memory the worker wrote). - const ok_marker = "thread-test: child ran in shared aspace ok"; + const ok_marker = "thread-test: child ran in shared address space ok"; const fail_marker = "thread-test: FAIL"; scheduler.setPriority(1); const deadline = architecture.millis() + 12000; @@ -1689,6 +1701,172 @@ fn threadIdTest(boot_information: *const BootInformation) void { result(); } +/// Thread-safe allocation (docs/threading-plan.md M7): `thread-test` in alloc mode runs N +/// threads that each do many `alloc`/fill/verify/`free` cycles of varied sizes on the +/// shared runtime heap. If the heap lock or the per-address-space mmap arena were unsafe, +/// two threads' blocks would overlap and a thread would read another's pattern; the +/// verdict marker is emitted only when every thread completes with every block intact. +fn threadAllocTest(boot_information: *const BootInformation) void { + log("DANOS-TEST-BEGIN: thread-alloc\n", .{}); + if (boot_information.initial_ramdisk_len == 0) { + check("bootloader handed over an initial_ramdisk", false); + result(); + return; + } + const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len]; + const rd = initial_ramdisk.Reader.init(image) orelse { + check("initial_ramdisk image is valid", false); + result(); + return; + }; + + var started = false; + var i: u32 = 0; + while (i < rd.count) : (i += 1) { + const item = rd.entry(i) orelse continue; + if (!eql(item.name, "thread-test")) continue; + started = if (process.spawnProcess(item.blob, 4, &.{ "thread-test", "alloc" })) true else |_| false; + break; + } + check("thread-test (alloc mode) spawned", started); + + const ok_marker = "thread-alloc: ok"; + const fail_marker = "thread-alloc: FAIL"; + scheduler.setPriority(1); + const deadline = architecture.millis() + 20000; + while (architecture.millis() < deadline) { + if (bufferHas(ok_marker) or bufferHas(fail_marker)) break; + scheduler.yield(); + } + scheduler.setPriority(4); + + check("concurrent heap allocation stayed corruption-free (shared heap + per-address-space arena)", bufferHas(ok_marker) and !bufferHas(fail_marker)); + result(); +} + +/// Per-thread TLS / FS base (docs/threading-plan.md M10): `thread-test` in tls mode has two +/// threads each set their own FS base and write a unique marker to `%fs:8`, then — after +/// both have written — read it back. If the FS base were not per-thread and restored across +/// context switches, the second write would clobber the first and a thread would read the +/// wrong marker. The verdict marker means both read their own value (no cross-talk). +fn threadTlsTest(boot_information: *const BootInformation) void { + log("DANOS-TEST-BEGIN: thread-tls\n", .{}); + if (boot_information.initial_ramdisk_len == 0) { + check("bootloader handed over an initial_ramdisk", false); + result(); + return; + } + const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len]; + const rd = initial_ramdisk.Reader.init(image) orelse { + check("initial_ramdisk image is valid", false); + result(); + return; + }; + + var started = false; + var i: u32 = 0; + while (i < rd.count) : (i += 1) { + const item = rd.entry(i) orelse continue; + if (!eql(item.name, "thread-test")) continue; + started = if (process.spawnProcess(item.blob, 4, &.{ "thread-test", "tls" })) true else |_| false; + break; + } + check("thread-test (tls mode) spawned", started); + + const ok_marker = "thread-tls: ok"; + const fail_marker = "thread-tls: FAIL"; + scheduler.setPriority(1); + const deadline = architecture.millis() + 12000; + while (architecture.millis() < deadline) { + if (bufferHas(ok_marker) or bufferHas(fail_marker)) break; + scheduler.yield(); + } + scheduler.setPriority(4); + + check("each thread has its own FS-base TLS slot (no cross-talk across switches)", bufferHas(ok_marker) and !bufferHas(fail_marker)); + result(); +} + +/// RwLock (docs/threading-plan.md M11): `thread-test` in rwlock mode runs writers that set +/// two halves of a value under the exclusive lock and readers that check the halves match +/// under the shared lock. If the reader/writer lock were wrong, a reader would observe a +/// half-written value; zero violations across many reads → the lock holds. +fn threadRwlockTest(boot_information: *const BootInformation) void { + log("DANOS-TEST-BEGIN: thread-rwlock\n", .{}); + if (boot_information.initial_ramdisk_len == 0) { + check("bootloader handed over an initial_ramdisk", false); + result(); + return; + } + const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len]; + const rd = initial_ramdisk.Reader.init(image) orelse { + check("initial_ramdisk image is valid", false); + result(); + return; + }; + + var started = false; + var i: u32 = 0; + while (i < rd.count) : (i += 1) { + const item = rd.entry(i) orelse continue; + if (!eql(item.name, "thread-test")) continue; + started = if (process.spawnProcess(item.blob, 4, &.{ "thread-test", "rwlock" })) true else |_| false; + break; + } + check("thread-test (rwlock mode) spawned", started); + + const ok_marker = "thread-rwlock: ok"; + const fail_marker = "thread-rwlock: FAIL"; + scheduler.setPriority(1); + const deadline = architecture.millis() + 20000; + while (architecture.millis() < deadline) { + if (bufferHas(ok_marker) or bufferHas(fail_marker)) break; + scheduler.yield(); + } + scheduler.setPriority(4); + + check("readers/writers over an RwLock never observed a half-written value", bufferHas(ok_marker) and !bufferHas(fail_marker)); + result(); +} + +/// The task reaper (docs/threading-plan.md M8): a dead task's kernel stack used to be +/// leaked ("no reaper yet"). Spawn and kill many ring-3 processes and confirm the total +/// kernel-stack bytes return to baseline — every stack reclaimed, no leak. (Threads exit +/// through the same exitUserLocked path, so this covers them too.) +fn taskReapTest(boot_information: *const BootInformation) void { + _ = boot_information; + log("DANOS-TEST-BEGIN: task-reap\n", .{}); + const base = scheduler.liveStackBytes(); + const rounds: u32 = 12; + var killed: u32 = 0; + var round: u32 = 0; + while (round < rounds) : (round += 1) { + process.fault_kill_count = 0; + const probe = spawnFaultingProcess() orelse break; + _ = probe; + scheduler.setPriority(1); + const deadline = architecture.millis() + 5000; + while (process.fault_kill_count < 1 and architecture.millis() < deadline) scheduler.yield(); + scheduler.setPriority(4); + if (process.fault_kill_count >= 1) killed += 1; + } + // Reaping is asynchronous — a dead task's stack is freed when its core next switches + // or ticks. Poll (bounded) until the live bytes return to baseline: a correct reaper + // gets there in a few ms; a genuine leak never does and this times out. + scheduler.setPriority(1); + const settle_deadline = architecture.millis() + 3000; + while (scheduler.liveStackBytes() != base and architecture.millis() < settle_deadline) scheduler.yield(); + scheduler.setPriority(4); + + const final = scheduler.liveStackBytes(); + log("task-reap: base={d} final={d} killed={d}/{d}\n", .{ base, final, killed, rounds }); + check("all probes spawned and were killed", killed == rounds); + check("kernel stacks reclaimed to baseline (no leak)", final == base); + if (killed == rounds and final == base) + log("task-reap: kernel stacks reclaimed to baseline ok\n", .{}); + result(); +} + /// The full PID-1 path: the bootloader read /system/services/init off the boot volume and /// handed it over; load it as a user ELF and spawn it as a real ring-3 process /// — the same call the normal boot path makes — then confirm it beats. init @@ -3111,20 +3289,20 @@ fn ioPassTest() void { log("DANOS-TEST-BEGIN: iopass\n", .{}); const base_free = pmm.stats().free_frames; - const aspace = architecture.createAddressSpace() orelse { + const address_space = architecture.createAddressSpace() orelse { check("created a fresh address space", false); result(); return; }; const frame = pmm.alloc() orelse { - architecture.destroyAddressSpace(aspace); + architecture.destroyAddressSpace(address_space); check("allocated a frame to grant", false); result(); return; }; // Map it the way mmio_map does (device grant, strong-uncacheable), then tear the space down. - architecture.mapUserDeviceInto(aspace, process.device_arena_base, frame, abi.page_size, false); - architecture.destroyAddressSpace(aspace); + architecture.mapUserDeviceInto(address_space, process.device_arena_base, frame, abi.page_size, false); + architecture.destroyAddressSpace(address_space); // The page tables were reclaimed; the device-granted frame must not have been. check("device-granted frame survived teardown (not reclaimed as RAM)", pmm.stats().free_frames == base_free - 1); @@ -3176,26 +3354,26 @@ fn displayTest(boot_information: *const BootInformation) void { // space, and confirm the leaf's cache type. We never run this space (no CR3 load) — // we only read back the page-table entries — so aliasing the same physical page at // two cache types below is inert. - const aspace = architecture.createAddressSpace() orelse { + const address_space = architecture.createAddressSpace() orelse { check("created a fresh address space", false); result(); return; }; - defer architecture.destroyAddressSpace(aspace); + defer architecture.destroyAddressSpace(address_space); const page_base = fb.base & ~@as(u64, abi.page_size - 1); - architecture.mapUserDeviceInto(aspace, process.device_arena_base, page_base, abi.page_size, true); + architecture.mapUserDeviceInto(address_space, process.device_arena_base, page_base, abi.page_size, true); check( "the framebuffer maps write-combining (PAT entry 4: PAT bit set, PCD/PWT clear)", - architecture.userLeafIsWriteCombining(aspace, process.device_arena_base) == true, + architecture.userLeafIsWriteCombining(address_space, process.device_arena_base) == true, ); // Regression guard: the strong-uncacheable default is still that, so WC is a real // choice the flag makes, not the only behaviour. - architecture.mapUserDeviceInto(aspace, process.device_arena_base + abi.page_size, page_base, abi.page_size, false); + architecture.mapUserDeviceInto(address_space, process.device_arena_base + abi.page_size, page_base, abi.page_size, false); check( "a register window still maps strong-uncacheable", - architecture.userLeafIsWriteCombining(aspace, process.device_arena_base + abi.page_size) == false, + architecture.userLeafIsWriteCombining(address_space, process.device_arena_base + abi.page_size) == false, ); log("display: mapped {d}x{d} pitch {d} (write-combining)\n", .{ fb.width, fb.height, fb.pitch }); diff --git a/system/services/thread-test/thread-test.zig b/system/services/thread-test/thread-test.zig index bd15643..50a72c4 100644 --- a/system/services/thread-test/thread-test.zig +++ b/system/services/thread-test/thread-test.zig @@ -40,7 +40,7 @@ fn runSpawnMode() void { runtime.system.yield(); } if (spawn_done.load(.acquire) == 1 and shared_value == sentinel) { - write("thread-test: child ran in shared aspace ok\n"); + write("thread-test: child ran in shared address space ok\n"); } else { write("thread-test: FAIL worker did not update shared memory\n"); } @@ -74,6 +74,8 @@ fn detachWorker() void { detach_done.store(1, .release); } +fn noopWorker() void {} + fn runJoinMode() void { write("thread-test: join mode starting\n"); @@ -114,7 +116,19 @@ fn runJoinMode() void { return; } - write("thread-test: join ok\n"); // the M3 verdict marker + // Prove join needs no per-thread kernel endpoint (M9): many spawn+join cycles. Under + // the old per-thread-endpoint scheme these leaked handles and would exhaust the + // 16-slot handle table well before 40; here they all succeed. + var cycle: u32 = 0; + while (cycle < 40) : (cycle += 1) { + const th = runtime.Thread.spawn(.{}, noopWorker, .{}) catch { + write("thread-test: FAIL spawn exhausted across join cycles (endpoint leak?)\n"); + return; + }; + th.join(); + } + + write("thread-test: join ok\n"); // the M3/M9 verdict marker } // --- M4: futex mode --------------------------------------------------------- @@ -294,6 +308,167 @@ fn runIdMode() void { write("thread-id: ok\n"); // the M6 verdict marker } +// --- M7: alloc mode (concurrent heap allocation) ---------------------------- + +const alloc_threads: u32 = 4; +const allocs_per_thread: u32 = 500; +var allocs_clean = std.atomic.Value(u32).init(0); + +fn allocWorker(seed: u32) void { + const gpa = runtime.allocator(); + var rng: u32 = seed | 1; + var round: u32 = 0; + while (round < allocs_per_thread) : (round += 1) { + rng = rng *% 1664525 +% 1013904223; // cheap LCG for varied sizes + const size: usize = 16 + (rng % 4080); // 16..4095 bytes + const buf = gpa.alloc(u8, size) catch return; // OOM: don't count this thread clean + const pattern: u8 = @truncate(seed +% round); + @memset(buf, pattern); + // Nothing else should touch our block; if a concurrent allocation overlapped it, + // one of us would read the other's pattern here. + var ok = true; + for (buf) |b| { + if (b != pattern) ok = false; + } + gpa.free(buf); + if (!ok) return; // corruption — leave without counting clean + } + _ = allocs_clean.fetchAdd(1, .monotonic); +} + +fn runAllocMode() void { + write("thread-alloc: starting\n"); + var threads: [alloc_threads]runtime.Thread = undefined; + var n: u32 = 0; + while (n < alloc_threads) : (n += 1) { + threads[n] = runtime.Thread.spawn(.{}, allocWorker, .{n +% 1}) catch { + write("thread-alloc: FAIL spawn\n"); + return; + }; + } + for (threads[0..alloc_threads]) |t| t.join(); + + // Every thread must have completed all rounds with each block intact — proof the + // shared heap and the per-address_space mmap arena are safe under concurrent allocation. + if (allocs_clean.load(.acquire) != alloc_threads) { + write("thread-alloc: FAIL corruption or OOM under concurrent allocation\n"); + return; + } + write("thread-alloc: ok\n"); // the M7 verdict marker +} + +// --- M10: tls mode (per-thread FS base storage) ----------------------------- + +fn writeTlsSlot(value: u64) void { + asm volatile ("movq %[v], %%fs:8" + : + : [v] "r" (value), + : .{ .memory = true }); +} + +fn readTlsSlot() u64 { + return asm volatile ("movq %%fs:8, %[out]" + : [out] "=r" (-> u64), + : + : .{ .memory = true }); +} + +var tls_written = std.atomic.Value(u32).init(0); +var tls_ok = std.atomic.Value(u32).init(0); + +fn tlsWorker(marker: u64) void { + writeTlsSlot(marker); + _ = tls_written.fetchAdd(1, .release); + // Wait until both threads have written their own slot. If the FS base were shared, the + // second write would clobber the first, and the read below would return the wrong + // marker — cross-talk. A per-thread FS base keeps each thread's slot private. + var spins: usize = 0; + while (tls_written.load(.acquire) < 2 and spins < 50_000_000) : (spins += 1) { + runtime.system.yield(); + } + if (readTlsSlot() == marker and runtime.Thread.getCurrentId() != 0) { + _ = tls_ok.fetchAdd(1, .monotonic); + } +} + +fn runTlsMode() void { + write("thread-tls: starting\n"); + const t0 = runtime.Thread.spawn(.{}, tlsWorker, .{@as(u64, 0xAAAA_0000)}) catch { + write("thread-tls: FAIL spawn\n"); + return; + }; + const t1 = runtime.Thread.spawn(.{}, tlsWorker, .{@as(u64, 0xBBBB_0000)}) catch { + write("thread-tls: FAIL spawn\n"); + return; + }; + t0.join(); + t1.join(); + if (tls_ok.load(.acquire) == 2) { + write("thread-tls: ok\n"); // the M10 verdict marker + } else { + write("thread-tls: FAIL cross-talk (FS base not per-thread)\n"); + } +} + +// --- M11: rwlock mode (readers/writers over an RwLock) ---------------------- + +const RwLock = runtime.Thread.RwLock; + +var rwlock = RwLock{}; +var rw_a: u64 = 0; +var rw_b: u64 = 0; // invariant while any lock is held: rw_a == rw_b +var rw_stop = std.atomic.Value(u32).init(0); +var rw_violations = std.atomic.Value(u32).init(0); +var rw_reads = std.atomic.Value(u64).init(0); + +fn rwWriter() void { + var v: u64 = 1; + while (rw_stop.load(.acquire) == 0) : (v +%= 1) { + rwlock.lock(); // exclusive: no reader may observe the gap between the two writes + rw_a = v; + rw_b = v; + rwlock.unlock(); + } +} + +fn rwReader() void { + const reads: u64 = 50_000; + var i: u64 = 0; + while (i < reads) : (i += 1) { + rwlock.lockShared(); + if (rw_a != rw_b) _ = rw_violations.fetchAdd(1, .monotonic); // saw a half-write! + rwlock.unlockShared(); + } + _ = rw_reads.fetchAdd(reads, .monotonic); +} + +fn runRwlockMode() void { + write("thread-rwlock: starting\n"); + var writers: [2]runtime.Thread = undefined; + var readers: [3]runtime.Thread = undefined; + for (&writers) |*w| { + w.* = runtime.Thread.spawn(.{}, rwWriter, .{}) catch { + write("thread-rwlock: FAIL spawn\n"); + return; + }; + } + for (&readers) |*r| { + r.* = runtime.Thread.spawn(.{}, rwReader, .{}) catch { + write("thread-rwlock: FAIL spawn\n"); + return; + }; + } + for (readers) |r| r.join(); + rw_stop.store(1, .release); // readers done → stop the writers + for (writers) |w| w.join(); + + if (rw_violations.load(.acquire) == 0 and rw_reads.load(.acquire) > 0) { + write("thread-rwlock: ok\n"); // the M11 verdict marker + } else { + write("thread-rwlock: FAIL reader observed a half-written value\n"); + } +} + pub fn main(init: runtime.process.Init) void { const mode = init.arguments.get(1) orelse "spawn"; if (std.mem.eql(u8, mode, "join")) { @@ -304,6 +479,12 @@ pub fn main(init: runtime.process.Init) void { runMutexMode(); } else if (std.mem.eql(u8, mode, "id")) { runIdMode(); + } else if (std.mem.eql(u8, mode, "alloc")) { + runAllocMode(); + } else if (std.mem.eql(u8, mode, "tls")) { + runTlsMode(); + } else if (std.mem.eql(u8, mode, "rwlock")) { + runRwlockMode(); } else { runSpawnMode(); } diff --git a/test/qemu_test.py b/test/qemu_test.py index f5753ec..ecca6b2 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -295,8 +295,8 @@ CASES = [ "fail": r"DANOS-TEST-RESULT: FAIL"}, # docs/threading-plan.md M1: address-space refcount — spaces destroyed exactly - # once per process, no leak/double-free (the foundation shared-aspace threads need). - {"name": "aspace-refcount", + # once per process, no leak/double-free (the foundation shared-address-space threads need). + {"name": "address-space-refcount", "timeout": 60, "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, @@ -339,6 +339,38 @@ CASES = [ "timeout": 60, "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, + + # docs/threading-plan.md M7: thread-safe allocation — N threads hammer the shared heap + # (per-aspace mmap arena + locked free list) with no cross-block corruption. + {"name": "thread-alloc", + "smp": 4, + "timeout": 60, + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, + + # docs/threading-plan.md M8: the task reaper — spawn+kill many processes; total kernel + # stack bytes return to baseline (every dead task's stack reclaimed, no leak). + {"name": "task-reap", + "smp": 4, + "timeout": 60, + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, + + # docs/threading-plan.md M10: per-thread fs.base — two threads keep private %fs:8 TLS + # slots across context switches (no cross-talk). + {"name": "thread-tls", + "smp": 4, + "timeout": 60, + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, + + # docs/threading-plan.md M11: RwLock — readers/writers across cores; a reader never + # observes a half-written value (writers hold it exclusively). + {"name": "thread-rwlock", + "smp": 4, + "timeout": 60, + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, # Process arguments: argv arrives on the SysV entry stack (argv[0] = the spawned # name, argv[1..] = the system_spawn argument blob) and echoes back intact. {"name": "args",