Compare commits
20
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
5ab7263c9c | ||
|
|
7f415e724f | ||
|
|
cf140eb772 | ||
|
|
e2dddc941f | ||
|
|
28b3635979 | ||
|
|
6101e429ba | ||
|
|
a4e44e8f31 | ||
|
|
c7e9b5a4f6 | ||
|
|
6bc329456a | ||
|
|
ed3b3f1c45 | ||
|
|
8259678f0a | ||
|
|
f4813c8e99 | ||
|
|
8b7f1d009c | ||
|
|
def34e71fc | ||
|
|
1b33f48acd | ||
|
|
b3a8147bd7 | ||
|
|
0730e77530 | ||
|
|
73df864fd2 | ||
|
|
11e363896f | ||
|
|
6e8b02d771 |
@@ -69,6 +69,36 @@ fn addUserBinary(
|
|||||||
acpi_ids_module: *std.Build.Module,
|
acpi_ids_module: *std.Build.Module,
|
||||||
name: []const u8,
|
name: []const u8,
|
||||||
root: []const u8,
|
root: []const u8,
|
||||||
|
) *std.Build.Step.Compile {
|
||||||
|
return addUserBinaryImpl(b, target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, name, root, false);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// As `addUserBinary`, but built multi-threaded (`single_threaded = false`) so real
|
||||||
|
/// atomics/TLS work — required before a binary may call `runtime.Thread.spawn`
|
||||||
|
/// (docs/threading.md). Threads are a deliberate per-binary opt-in.
|
||||||
|
fn addThreadedUserBinary(
|
||||||
|
b: *std.Build,
|
||||||
|
target: std.Build.ResolvedTarget,
|
||||||
|
runtime_module: *std.Build.Module,
|
||||||
|
mmio_module: *std.Build.Module,
|
||||||
|
xkeyboard_config_module: *std.Build.Module,
|
||||||
|
acpi_ids_module: *std.Build.Module,
|
||||||
|
name: []const u8,
|
||||||
|
root: []const u8,
|
||||||
|
) *std.Build.Step.Compile {
|
||||||
|
return addUserBinaryImpl(b, target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, name, root, true);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn addUserBinaryImpl(
|
||||||
|
b: *std.Build,
|
||||||
|
target: std.Build.ResolvedTarget,
|
||||||
|
runtime_module: *std.Build.Module,
|
||||||
|
mmio_module: *std.Build.Module,
|
||||||
|
xkeyboard_config_module: *std.Build.Module,
|
||||||
|
acpi_ids_module: *std.Build.Module,
|
||||||
|
name: []const u8,
|
||||||
|
root: []const u8,
|
||||||
|
threaded: bool,
|
||||||
) *std.Build.Step.Compile {
|
) *std.Build.Step.Compile {
|
||||||
// Settings (target, optimize, code model, ...) live on the root module only;
|
// Settings (target, optimize, code model, ...) live on the root module only;
|
||||||
// the program and runtime modules leave theirs null and inherit them.
|
// the program and runtime modules leave theirs null and inherit them.
|
||||||
@@ -93,7 +123,7 @@ fn addUserBinary(
|
|||||||
.target = target,
|
.target = target,
|
||||||
.optimize = .ReleaseSmall,
|
.optimize = .ReleaseSmall,
|
||||||
.code_model = .large,
|
.code_model = .large,
|
||||||
.single_threaded = true,
|
.single_threaded = !threaded, // a threaded binary needs real atomics/TLS
|
||||||
.sanitize_c = .off,
|
.sanitize_c = .off,
|
||||||
.stack_check = false,
|
.stack_check = false,
|
||||||
.stack_protector = false,
|
.stack_protector = false,
|
||||||
@@ -505,7 +535,9 @@ pub fn build(b: *std.Build) void {
|
|||||||
// The FAT filesystem server: mounts the block device and serves it into the VFS
|
// The FAT filesystem server: mounts the block device and serves it into the VFS
|
||||||
// at /mnt/usb. Its engine (engine.zig / on-disk.zig) is imported relatively.
|
// at /mnt/usb. Its engine (engine.zig / on-disk.zig) is imported relatively.
|
||||||
const fat_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat", "system/services/fat/fat.zig");
|
const fat_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat", "system/services/fat/fat.zig");
|
||||||
const display_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "display", "system/services/display/display.zig");
|
// Threaded: the display runs a mouse-listener thread alongside its compositor loop
|
||||||
|
// (docs/threading.md, docs/display.md), so it opts into real atomics/TLS.
|
||||||
|
const display_exe = addThreadedUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "display", "system/services/display/display.zig");
|
||||||
const display_demo_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "display-demo", "system/services/display-demo/display-demo.zig");
|
const display_demo_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "display-demo", "system/services/display-demo/display-demo.zig");
|
||||||
const virtio_gpu_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "virtio-gpu", "system/drivers/virtio-gpu/virtio-gpu.zig");
|
const virtio_gpu_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "virtio-gpu", "system/drivers/virtio-gpu/virtio-gpu.zig");
|
||||||
const shm_server_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "shm-server", "system/services/shm-server/shm-server.zig");
|
const shm_server_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "shm-server", "system/services/shm-server/shm-server.zig");
|
||||||
@@ -548,6 +580,9 @@ pub fn build(b: *std.Build) void {
|
|||||||
const args_echo_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "args-echo", "system/services/args-echo/args-echo.zig");
|
const args_echo_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "args-echo", "system/services/args-echo/args-echo.zig");
|
||||||
const process_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "process-test", "system/services/process-test/process-test.zig");
|
const process_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "process-test", "system/services/process-test/process-test.zig");
|
||||||
const log_flush_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "log-flush", "system/services/log-flush/log-flush.zig");
|
const log_flush_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "log-flush", "system/services/log-flush/log-flush.zig");
|
||||||
|
// The first multi-threaded binary: exercises runtime.Thread over the thread ABI
|
||||||
|
// (docs/threading.md). Built threaded so its shared-memory poll is real.
|
||||||
|
const thread_test_exe = addThreadedUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "thread-test", "system/services/thread-test/thread-test.zig");
|
||||||
|
|
||||||
// Pack the user binaries into the initial_ramdisk image with the host-side Python tool
|
// Pack the user binaries into the initial_ramdisk image with the host-side Python tool
|
||||||
// (the container format is trivial, and Python sidesteps std API churn). Args:
|
// (the container format is trivial, and Python sidesteps std API churn). Args:
|
||||||
@@ -591,6 +626,8 @@ pub fn build(b: *std.Build) void {
|
|||||||
mk_run.addFileArg(pci_bus_exe.getEmittedBin());
|
mk_run.addFileArg(pci_bus_exe.getEmittedBin());
|
||||||
mk_run.addArg("crash-test");
|
mk_run.addArg("crash-test");
|
||||||
mk_run.addFileArg(crash_test_exe.getEmittedBin());
|
mk_run.addFileArg(crash_test_exe.getEmittedBin());
|
||||||
|
mk_run.addArg("thread-test");
|
||||||
|
mk_run.addFileArg(thread_test_exe.getEmittedBin());
|
||||||
mk_run.addArg("device-list");
|
mk_run.addArg("device-list");
|
||||||
mk_run.addFileArg(device_list_exe.getEmittedBin());
|
mk_run.addFileArg(device_list_exe.getEmittedBin());
|
||||||
mk_run.addArg("discovery");
|
mk_run.addArg("discovery");
|
||||||
@@ -891,6 +928,22 @@ pub fn build(b: *std.Build) void {
|
|||||||
});
|
});
|
||||||
test_step.dependOn(&b.addRunArtifact(time_tests).step);
|
test_step.dependOn(&b.addRunArtifact(time_tests).step);
|
||||||
|
|
||||||
|
// runtime.Thread's lock/condvar state machines (Mutex/Condition/RwLock/WaitGroup). Its
|
||||||
|
// Futex seam falls back to std.Thread.Futex off the danos target, so the tests exercise
|
||||||
|
// them with real host threads (docs/threading-plan.md M11). Like time.zig it pulls in
|
||||||
|
// system.zig (syscall wrappers), which needs the `abi` module.
|
||||||
|
const thread_tests = b.addTest(.{
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path("library/runtime/thread.zig"),
|
||||||
|
.target = target,
|
||||||
|
.optimize = optimize,
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "abi", .module = abi_module },
|
||||||
|
},
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
test_step.dependOn(&b.addRunArtifact(thread_tests).step);
|
||||||
|
|
||||||
// Convenience: `zig build gen-xkeyboard-config` regenerates the layout tables from the
|
// Convenience: `zig build gen-xkeyboard-config` regenerates the layout tables from the
|
||||||
// vendored data (offline). `fetch` (the network step) stays a manual script run.
|
// vendored data (offline). `fetch` (the network step) stays a manual script run.
|
||||||
const gen_xkb = b.addSystemCommand(&.{ "python3", "tools/make-xkeyboard-config.py", "generate" });
|
const gen_xkb = b.addSystemCommand(&.{ "python3", "tools/make-xkeyboard-config.py", "generate" });
|
||||||
|
|||||||
@@ -104,6 +104,13 @@ Start with the north star:
|
|||||||
port to **one seam** (`std.os.danos`), so we build `runtime.os` (→ that seam) plus a
|
port to **one seam** (`std.os.danos`), so we build `runtime.os` (→ that seam) plus a
|
||||||
thin `runtime.fs`, retire the `posix` shim, and follow a phased path to
|
thin `runtime.fs`, retire the `posix` shim, and follow a phased path to
|
||||||
`zig build-exe hello.zig` running on danos — **not** Linux-ABI emulation.
|
`zig build-exe hello.zig` running on danos — **not** Linux-ABI emulation.
|
||||||
|
- **[threading.md](threading.md) — threads, the std-shaped way.** **Built** (M1–M6):
|
||||||
|
`runtime.Thread` mirrors `std.Thread`'s API (spawn/join/detach, Mutex/Condition/
|
||||||
|
Semaphore) over a **private** thread ABI — several tasks sharing one address space via
|
||||||
|
a `thread_spawn` syscall, futex-backed blocking, address-space refcounting. Why it's the
|
||||||
|
native type and not literal `std.Thread` (the [private ABI](syscall.md)), and why
|
||||||
|
threads stay a narrow opt-in against the [resilience](resilience.md) default. Build
|
||||||
|
plan + gates: [threading-plan.md](threading-plan.md).
|
||||||
- **[vdso.md](vdso.md) — the vDSO, the public system-call boundary.** A design note
|
- **[vdso.md](vdso.md) — the vDSO, the public system-call boundary.** A design note
|
||||||
(not built yet) on keeping `abi.zig` genuinely private: a kernel-supplied, C-ABI
|
(not built yet) on keeping `abi.zig` genuinely private: a kernel-supplied, C-ABI
|
||||||
entry blob mapped into every process as the *only* way into the kernel — so the
|
entry blob mapped into every process as the *only* way into the kernel — so the
|
||||||
|
|||||||
@@ -62,7 +62,7 @@ is the only backend), and `zig build test` stays green.
|
|||||||
- [x] [abi.zig](../system/abi.zig): `shm_create` (34) / `shm_map` (35) syscalls + a
|
- [x] [abi.zig](../system/abi.zig): `shm_create` (34) / `shm_map` (35) syscalls + a
|
||||||
`shm_test` service id. Handlers in process.zig: `shm_create(len)` allocates contiguous,
|
`shm_test` service id. Handlers in process.zig: `shm_create(len)` allocates contiguous,
|
||||||
zeroed, **cacheable** frames, wraps them in a refcounted object, installs a capability
|
zeroed, **cacheable** frames, wraps them in a refcounted object, installs a capability
|
||||||
handle, maps them into the caller's shm arena → returns vaddr + handle; `shm_map(cap)`
|
handle, maps them into the caller's shm arena → returns virtual_address + handle; `shm_map(cap)`
|
||||||
maps the same physical pages into the receiver. Reclaimed on death (see below).
|
maps the same physical pages into the receiver. Reclaimed on death (see below).
|
||||||
- [x] The capability core (ipc-synchronous.zig) is now **kind-tagged**: `scheduler.Task`'s
|
- [x] The capability core (ipc-synchronous.zig) is now **kind-tagged**: `scheduler.Task`'s
|
||||||
handle table holds `HandleObject{kind, ptr}`; `closeHandles` and `shareCapability`
|
handle table holds `HandleObject{kind, ptr}`; `closeHandles` and `shareCapability`
|
||||||
|
|||||||
+2
-2
@@ -77,9 +77,9 @@ deferred (docs/display.md, "What v1 does not do"). v2 builds it: the natural gen
|
|||||||
of M13 capability-passing from *endpoints* to *memory objects* —
|
of M13 capability-passing from *endpoints* to *memory objects* —
|
||||||
|
|
||||||
```
|
```
|
||||||
shm_create(len) -> {handle, vaddr} // a shareable, page-aligned RAM region
|
shm_create(len) -> {handle, virtual_address} // a shareable, page-aligned RAM region
|
||||||
… pass `handle` as the send_cap on an ipc_call …
|
… pass `handle` as the send_cap on an ipc_call …
|
||||||
shm_map(cap) -> vaddr // the receiver maps the same physical pages
|
shm_map(cap) -> virtual_address // the receiver maps the same physical pages
|
||||||
```
|
```
|
||||||
|
|
||||||
The payoff is leverage: the **same** primitive unlocks **both** native GPU drivers *and*
|
The payoff is leverage: the **same** primitive unlocks **both** native GPU drivers *and*
|
||||||
|
|||||||
+48
-7
@@ -211,6 +211,38 @@ with a boot-race retry): `display.info()`, a `Layer` handle with `fill` / `blitT
|
|||||||
`damage`, and `present()`. Application code never issues the raw syscalls — it calls the
|
`damage`, and `present()`. Application code never issues the raw syscalls — it calls the
|
||||||
runtime, as with every other danos service.
|
runtime, as with every other danos service.
|
||||||
|
|
||||||
|
## The cursor: a mouse-listener thread feeding the compositor
|
||||||
|
|
||||||
|
The compositor is the single owner of the framebuffer — only the main `service.run` loop
|
||||||
|
touches the backend and the layer stack. Tracking the mouse without breaking that
|
||||||
|
ownership is the display's first use of [threads](threading.md): the service is built
|
||||||
|
multi-threaded (`addThreadedUserBinary`) and, at startup, spawns a **mouse-listener
|
||||||
|
thread** beside the compositor loop.
|
||||||
|
|
||||||
|
- **Listener thread.** Blocks on the input service's mouse stream
|
||||||
|
(`input.subscribeMouse()`), accumulates the relative `dx`/`dy` motion into an absolute
|
||||||
|
cursor position clamped to the screen, and hands it to the compositor. It never touches
|
||||||
|
the compositor — so no lock guards the framebuffer. A parked `next()` leaves its core
|
||||||
|
free to halt ([halting.md](halting.md)).
|
||||||
|
- **The channel.** A single-slot *latest-value* cell (`CursorChannel`) guarded by a
|
||||||
|
`runtime.Thread.Mutex`: the renderer wants where the cursor *is now*, not a replay of
|
||||||
|
every delta, so a new position overwrites the old. The listener also **pokes** the
|
||||||
|
compositor awake — the main loop is parked in `replyWait`, so the listener posts a
|
||||||
|
zero-payload `ipc.send` to the compositor's endpoint, which arrives as a
|
||||||
|
message-notification ([ipc.md](ipc.md)). The poke is *coalesced*: at most one is queued
|
||||||
|
while the main loop has not drained the last, so a fast mouse cannot flood the endpoint.
|
||||||
|
- **Render.** On the poke, the main loop takes the latest position and moves the cursor —
|
||||||
|
which is just a top-z compositor layer — with the existing `configure` + `present` path
|
||||||
|
(it damages the old and new footprints, so only those two rectangles repaint).
|
||||||
|
|
||||||
|
Two threading facts shape this (both in [threading.md](threading.md)). IPC **handles do
|
||||||
|
not cross threads**, so the listener can't reuse the main loop's endpoint handle — it
|
||||||
|
`ipc.lookup(.display)`s its *own* handle to the same endpoint to poke through. And a
|
||||||
|
multi-threaded service doing concurrent IPC is why the kernel's endpoint-create / register
|
||||||
|
/ lookup syscalls now serialize under the big kernel lock. Shared fate applies: a fault in
|
||||||
|
the listener takes the whole display down, and the supervisor restarts the process
|
||||||
|
([resilience.md](resilience.md)).
|
||||||
|
|
||||||
## What v1 does not do (and why that's fine)
|
## What v1 does not do (and why that's fine)
|
||||||
|
|
||||||
Two capabilities are deliberately out of the first cut. Neither reshapes anything above;
|
Two capabilities are deliberately out of the first cut. Neither reshapes anything above;
|
||||||
@@ -220,8 +252,8 @@ both are clean additions behind the interfaces v1 establishes.
|
|||||||
to render into its *own* buffer and hand the compositor a *reference*, not a stream of
|
to render into its *own* buffer and hand the compositor a *reference*, not a stream of
|
||||||
commands. That needs the missing cross-process shared-memory primitive — best built as
|
commands. That needs the missing cross-process shared-memory primitive — best built as
|
||||||
the natural generalization of the existing M13 [capability passing](driver-model.md)
|
the natural generalization of the existing M13 [capability passing](driver-model.md)
|
||||||
from *endpoints* to *memory objects* (`shm_create(len) → {cap, vaddr}`, pass `cap` on
|
from *endpoints* to *memory objects* (`shm_create(len) → {cap, virtual_address}`, pass `cap` on
|
||||||
an `ipc_call`, receiver `shm_map(cap) → vaddr`). v1 avoids it because server-owned
|
an `ipc_call`, receiver `shm_map(cap) → virtual_address`). v1 avoids it because server-owned
|
||||||
surfaces already prove the whole pipeline.
|
surfaces already prove the whole pipeline.
|
||||||
|
|
||||||
- **Runtime mode-setting (a native backend).** Detecting the EDID mode list and changing
|
- **Runtime mode-setting (a native backend).** Detecting the EDID mode list and changing
|
||||||
@@ -232,7 +264,7 @@ both are clean additions behind the interfaces v1 establishes.
|
|||||||
|
|
||||||
## Verifying it
|
## Verifying it
|
||||||
|
|
||||||
Three QEMU test cases ([tests.zig](../system/kernel/tests.zig), `python3
|
Four QEMU test cases ([tests.zig](../system/kernel/tests.zig), `python3
|
||||||
test/qemu_test.py <case>`), each layering on the last:
|
test/qemu_test.py <case>`), each layering on the last:
|
||||||
|
|
||||||
- **`display`** — the kernel handoff: the seeded `display` device is shaped correctly and
|
- **`display`** — the kernel handoff: the seeded `display` device is shaped correctly and
|
||||||
@@ -245,11 +277,20 @@ test/qemu_test.py <case>`), each layering on the last:
|
|||||||
layer — logging `display: compositor self-check ok`.
|
layer — logging `display: compositor self-check ok`.
|
||||||
- **`display-demo`** — the full pipeline from a separate process: the hardware-free
|
- **`display-demo`** — the full pipeline from a separate process: the hardware-free
|
||||||
[`display-demo`](../system/services/display-demo/) client (the
|
[`display-demo`](../system/services/display-demo/) client (the
|
||||||
[`input-source`](../system/services/input-source/) analog) drives layers — a wallpaper, a
|
[`input-source`](../system/services/input-source/) analog) drives layers — a wallpaper and
|
||||||
sliding rectangle, a cursor — through the layer client API and heartbeats
|
a sliding rectangle — through the layer client API and heartbeats
|
||||||
`display-demo: ok`, proving a frame travelled client → compositor → screen, exactly as
|
`display-demo: ok`, proving a frame travelled client → compositor → screen, exactly as
|
||||||
the [input test](input.md) proves an event travels source → service → subscriber. The
|
the [input test](input.md) proves an event travels source → service → subscriber. It draws
|
||||||
visible motion itself is a screenshot away via `zig build run-x86-64`.
|
no cursor and reads no input — the cursor is the service's own (below), and the demo
|
||||||
|
animates on its own frame timer, independent of the mouse (the test spawns `input`
|
||||||
|
alongside it to keep that independence honest). The visible motion itself is a screenshot
|
||||||
|
away via `zig build run-x86-64`.
|
||||||
|
- **`display-cursor`** — the mouse-listener thread end to end: with the `input` service up,
|
||||||
|
`input-source mouse` publishes pure motion, and the display's listener thread accumulates
|
||||||
|
it into a cursor position handed to the render loop over the `CursorChannel`. Once the
|
||||||
|
cursor has tracked a run of that motion, the service logs
|
||||||
|
`display: cursor tracking mouse ok`. Runs `smp: 4` — the compositor and listener threads
|
||||||
|
execute on different cores, which is what surfaced the IPC-under-lock requirement above.
|
||||||
|
|
||||||
The compositor's pixel math (rectangle clipping, fill, composite, tile blit) and colour
|
The compositor's pixel math (rectangle clipping, fill, composite, tile blit) and colour
|
||||||
packing are additionally covered by pure host unit tests under `zig build test`.
|
packing are additionally covered by pure host unit tests under `zig build test`.
|
||||||
|
|||||||
@@ -240,8 +240,8 @@ once per page, maps writeback-cached, and never reveals a physical address.
|
|||||||
**The fix.**
|
**The fix.**
|
||||||
|
|
||||||
```
|
```
|
||||||
dma_alloc(len, flags) -> vaddr (rax), paddr (rdx)
|
dma_alloc(len, flags) -> virtual_address (rax), physical_address (rdx)
|
||||||
dma_free(vaddr, len) -> 0
|
dma_free(virtual_address, len) -> 0
|
||||||
|
|
||||||
flags: dma_coherent (1) uncacheable; the default and the only one that's portable
|
flags: dma_coherent (1) uncacheable; the default and the only one that's portable
|
||||||
dma_wc (2) write-combining — needs PAT programmed; for framebuffers
|
dma_wc (2) write-combining — needs PAT programmed; for framebuffers
|
||||||
|
|||||||
+1
-1
@@ -74,7 +74,7 @@ The driver syscall numbers (`system/abi.zig`) with the device types they carry
|
|||||||
|---|------|---------|
|
|---|------|---------|
|
||||||
| 11 | `device_enumerate(buf, max) -> total` | Snapshot the device table |
|
| 11 | `device_enumerate(buf, max) -> total` | Snapshot the device table |
|
||||||
| 12 | `device_claim(id) -> ok` | Take **exclusive** ownership |
|
| 12 | `device_claim(id) -> ok` | Take **exclusive** ownership |
|
||||||
| 13 | `mmio_map(id, res_idx) -> vaddr` | Map a claimed device's register window |
|
| 13 | `mmio_map(id, res_idx) -> virtual_address` | Map a claimed device's register window |
|
||||||
| 14 | `irq_bind(id, res_idx, endpoint)` | Deliver that device's IRQ as a notification |
|
| 14 | `irq_bind(id, res_idx, endpoint)` | Deliver that device's IRQ as a notification |
|
||||||
| 15 | `irq_ack(id, res_idx)` | Re-arm the IRQ after servicing the device |
|
| 15 | `irq_ack(id, res_idx)` | Re-arm the IRQ after servicing the device |
|
||||||
| 16 | `device_register(parent_id, desc) -> id` | Publish a child of a device you claimed |
|
| 16 | `device_register(parent_id, desc) -> id` | Publish a child of a device you claimed |
|
||||||
|
|||||||
@@ -0,0 +1,469 @@
|
|||||||
|
# Threading — build plan (`runtime.Thread` over a private thread ABI)
|
||||||
|
|
||||||
|
The ordered, checkpointable build-out for [threading.md](threading.md). Each milestone
|
||||||
|
lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run, like
|
||||||
|
[display-v2-plan.md](display-v2-plan.md). Read threading.md first for the *why*.
|
||||||
|
|
||||||
|
## Locked decisions (do not relitigate)
|
||||||
|
|
||||||
|
- **`runtime.Thread` mirrors `std.Thread`'s API; the implementation is danos-native.**
|
||||||
|
Not literal `std.Thread` — that would break the [private ABI](syscall.md).
|
||||||
|
- **Threads are a narrow, per-binary opt-in.** Default concurrency stays process + IPC
|
||||||
|
([resilience.md](resilience.md)); only a service that asks is built
|
||||||
|
`single_threaded = false`.
|
||||||
|
- **Blocking is futex-backed, never spin-backed** — waiters park in the kernel so an
|
||||||
|
idle core still halts ([halting.md](halting.md)).
|
||||||
|
- **New syscalls are private**: extend [abi.zig](../system/abi.zig) `SystemCall` after
|
||||||
|
`shm_physical = 36` (`thread_spawn = 37`, `thread_exit = 38`, `current_core = 39`,
|
||||||
|
`futex_wait = 40`, `futex_wake = 41`) + a `library/runtime` wrapper; user code never names a number.
|
||||||
|
- **Restart granularity stays the process** — a faulting thread kills its process; the
|
||||||
|
supervisor restarts the process, which respawns its threads.
|
||||||
|
|
||||||
|
## Conventions
|
||||||
|
|
||||||
|
Follow [coding-standards.md](coding-standards.md): spell out non-acronym abbreviations,
|
||||||
|
kebab-case file names, no `Co-Authored-By` trailers. New user binaries go through
|
||||||
|
`addUserBinary` (with the new `threaded` flag where a binary spawns threads) and get
|
||||||
|
packed into the initial-ramdisk; new syscalls extend [abi.zig](../system/abi.zig)
|
||||||
|
`SystemCall` + a `library/runtime` wrapper; test services live beside the code they
|
||||||
|
exercise and register a `ServiceId` if they must be looked up.
|
||||||
|
|
||||||
|
## How to verify along the way
|
||||||
|
|
||||||
|
**Every gate is serial-checkable — no screenshots** (this plan runs unattended). A
|
||||||
|
thread proves it ran by writing to **shared memory** the parent reads back, and proves
|
||||||
|
parallelism by stamping the **core index** it ran on (like the `smp`/`affinity` cases).
|
||||||
|
|
||||||
|
- `zig build test` — host unit tests (closure packing, mutex state machine, futex
|
||||||
|
wrapper encodings).
|
||||||
|
- `python3 test/qemu_test.py <case>` — boots the kernel in QEMU; asserts on serial
|
||||||
|
markers. Thread cases set `smp: true` (real parallelism) and bump `mem` (they boot
|
||||||
|
the process/scheduler stack); each milestone **adds its case to `CASES`** so its gate
|
||||||
|
is runnable.
|
||||||
|
- **Guardrail every milestone:** the concurrency-sensitive existing cases stay green —
|
||||||
|
`smoke`, `sched`, `priority`, `smp`, `affinity`, `process`, `process-kill`,
|
||||||
|
`supervision`, `fault-recovery`, `vfs-client-death`, `ipc`/`ipc-cap`,
|
||||||
|
`display-service`. A threading change that regresses those is rejected.
|
||||||
|
|
||||||
|
## Unattended execution (the loop contract)
|
||||||
|
|
||||||
|
This plan runs to completion **without human input**. Every design choice is already
|
||||||
|
fixed in *Locked decisions*; the checkboxes are the only state. A loop iteration must:
|
||||||
|
|
||||||
|
1. **Resume** at the first milestone that still has an unchecked `- [ ]`. (All earlier
|
||||||
|
milestones are done — do not revisit them.)
|
||||||
|
2. **Work on a branch.** On the first iteration, branch off the current `main` into a new
|
||||||
|
branch (e.g. `threading-phase2` — Phase 1's `threading` is already merged); never
|
||||||
|
commit to `main` directly. Push that **branch** to `origin` after each milestone (step
|
||||||
|
5) so progress is backed up remotely; **do not push `main`** — merging Phase 2 into
|
||||||
|
`main` stays a human step.
|
||||||
|
3. **Implement** every unchecked item in that milestone, including adding its
|
||||||
|
`-Dtest-case` to `CASES` in [test/qemu_test.py](../test/qemu_test.py) (with
|
||||||
|
`smp: true` / a `mem` bump where noted) so the gate is runnable.
|
||||||
|
4. **Run the gate**: `python3 test/qemu_test.py <case>`, then the full **guardrail
|
||||||
|
set**, then `zig build` (clean) and `zig build test` (green).
|
||||||
|
5. **Decide, do not ask:**
|
||||||
|
- **Green** = the milestone's case prints its stated marker(s) and reports `PASS`,
|
||||||
|
the whole guardrail set passes, `zig build` is clean, and host tests are green.
|
||||||
|
→ tick this milestone's boxes **and** its `**Gate:**`-referenced case, `git commit`
|
||||||
|
(`threads(M<n>): <summary>`, no `Co-Authored-By` trailer per
|
||||||
|
[coding-standards.md](coding-standards.md)), then **`git push` the working branch to
|
||||||
|
`origin`** (use `-u` on the first push to set upstream). Continue to the next
|
||||||
|
milestone in the same iteration if budget remains; otherwise let the loop re-fire.
|
||||||
|
- **Red** = anything above fails. Diagnose from the captured serial log
|
||||||
|
(`zig-out/qemu-test/<case>-failed-serial.log`) and fix in place, then re-run — up to
|
||||||
|
**3 fix attempts** for that gate. A concurrency case that fails then passes on a
|
||||||
|
bare re-run is **flaky, not green**: re-run it **twice more** and treat green only
|
||||||
|
if it passes all; otherwise fix the race (a real threading bug), don't paper over
|
||||||
|
it.
|
||||||
|
6. **A genuinely ambiguous fork is not a stop.** Pick the option most consistent with
|
||||||
|
[threading.md](threading.md)'s *Locked decisions*, note the choice in the commit
|
||||||
|
message, and continue. Do not pause for confirmation on in-scope, reversible work —
|
||||||
|
this plan is that authorization.
|
||||||
|
|
||||||
|
**The only stop conditions:**
|
||||||
|
|
||||||
|
- **Done** — every milestone box **in this plan** is checked (M1 through M11), `zig build`
|
||||||
|
clean, the whole `thread-*` suite + guardrail green. Phase 1 (M1–M6) is *already*
|
||||||
|
checked, so do **not** read that as Done: the loop's real work is the first plan section
|
||||||
|
that still has unchecked boxes — Phase 2 (M7–M11). Only stop when M7–M11 are all checked
|
||||||
|
too. Update threading.md's status line, push the final branch state to `origin`, and
|
||||||
|
stop. The branch is on `origin` for review; **merging Phase 2 into `main` is the user's
|
||||||
|
step**, not the loop's.
|
||||||
|
- **Blocked** — a gate is still red after 3 fix attempts, or a step needs something
|
||||||
|
outside the repo (a toolchain change, new hardware, a decision no locked decision
|
||||||
|
covers). Append `> **BLOCKED (M<n>):** <what failed, what was tried, the serial
|
||||||
|
marker missing>` under that milestone, commit **and push** the WIP on the branch, and
|
||||||
|
stop. Do not thrash further and do not silently skip the milestone.
|
||||||
|
|
||||||
|
Nothing else warrants stopping — not "should I proceed?", not "is this right?". The
|
||||||
|
checkboxes + git history are the resumable record; the next iteration picks up from the
|
||||||
|
first unchecked box.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## M1 — Address-space refcount (kernel foundation, no API, no behaviour change) ✅
|
||||||
|
|
||||||
|
The one invariant change threads require, landed and proven **before** anything shares
|
||||||
|
an address space. Today address space is 1:1 with a task and teardown destroys it on any user
|
||||||
|
task's exit; make destruction happen on the **last** exit.
|
||||||
|
|
||||||
|
- [x] A refcount keyed by the address-space root, held in `scheduler.zig`
|
||||||
|
(`address_space_refs`): `retainAddressSpace` takes a reference in `spawnUserLocked` (on the
|
||||||
|
success path, after the slot + stack are secured), all under the big kernel lock.
|
||||||
|
- [x] Both task-teardown paths ([scheduler.zig](../system/kernel/scheduler.zig):
|
||||||
|
`exitUserLocked` and `destroyTaskLocked`) call `releaseAspace`, which decrements
|
||||||
|
and only `destroyAddressSpace`s at **zero**; an unretained space (hand-built test
|
||||||
|
spaces) is destroyed directly, preserving prior behaviour.
|
||||||
|
- [x] `-Dtest-case=address-space-refcount`: spawn and reap several ring-3 processes in sequence
|
||||||
|
and assert (via test-observable `liveAddressSpaceCount`/`addressSpaceDestroyCount`) that the
|
||||||
|
live-space count returns to **baseline** and destructions advance by exactly that
|
||||||
|
many — each space destroyed exactly once, no leak, no double-free. (Refcount
|
||||||
|
observables, not raw frame counts, since kernel stacks are still leaked on exit.)
|
||||||
|
|
||||||
|
**Gate (met):** `python3 test/qemu_test.py address-space-refcount` passes
|
||||||
|
(`address-space-refcount: spaces released to baseline ok` → `DANOS-TEST-RESULT: PASS`), and the
|
||||||
|
full guardrail set passes unchanged — 13/13 (`smoke`, `sched`, `priority`, `smp`,
|
||||||
|
`affinity`, `process`, `process-kill`, `supervision`, `fault-recovery`,
|
||||||
|
`vfs-client-death`, `ipc`, `ipc-cap`, `display-service`); default `zig build` clean,
|
||||||
|
`zig build test` green. The reframing is invisible until an address space is actually shared.
|
||||||
|
|
||||||
|
## M2 — `thread_spawn` + `thread_exit`: a thread runs in the shared address space ✅
|
||||||
|
|
||||||
|
Spawn only — no join yet. Prove a second task executes in the **caller's** address
|
||||||
|
space and exits cleanly.
|
||||||
|
|
||||||
|
- [x] [abi.zig](../system/abi.zig): `thread_spawn = 37`, `thread_exit = 38`. Handlers in
|
||||||
|
process.zig; `thread_spawn` calls `scheduler.spawnThread` (shares the caller's
|
||||||
|
address space, `retainAddressSpace`); `thread_exit` ends the task like a process `exit(0)`
|
||||||
|
(`terminateCurrent` → `releaseAspace`). The closure pointer is delivered in the new
|
||||||
|
thread's **rdi** via a new `jump_to_user_arg` asm path (`t.user_arg`, 0 for a
|
||||||
|
process) — no naked runtime asm.
|
||||||
|
- [x] `library/runtime/thread.zig` (barrel-exported as `runtime.Thread`): `spawn` maps a
|
||||||
|
stack (`mmap`), heap-allocates the `{args}` closure, and calls
|
||||||
|
`thread_spawn(&Closure.entry, stack_top, closure)`; `Closure.entry` (a plain C-ABI
|
||||||
|
Zig fn, closure in rdi) runs the function and calls `thread_exit`. Stack top is
|
||||||
|
16-aligned-minus-8 for the C entry.
|
||||||
|
- [x] A `threaded` flag on the user-binary recipe (`addThreadedUserBinary` →
|
||||||
|
`single_threaded = false`); `thread-test` is the first opt-in binary.
|
||||||
|
- [x] `-Dtest-case=thread-spawn`: `thread-test` spawns a worker that writes a sentinel to
|
||||||
|
a **shared** global and release-stores `done`; the main thread acquire-polls `done`
|
||||||
|
and asserts the shared global holds the sentinel — proof the worker ran in the same
|
||||||
|
address space.
|
||||||
|
|
||||||
|
**Gate (met):** `python3 test/qemu_test.py thread-spawn` passes
|
||||||
|
(`thread-test: child ran in shared address space ok` → `DANOS-TEST-RESULT: PASS`); guardrail set
|
||||||
|
16/16 green (incl. `args`/`init`/`process`, which exercise the new `jump_to_user_arg`
|
||||||
|
process path with arg 0) plus `address-space-refcount`; `zig build` clean, `zig build test`
|
||||||
|
green.
|
||||||
|
|
||||||
|
> **Note (deferred to M3+):** the mmap arena is per-*task* (`heap_next`), so two threads
|
||||||
|
> in one address space that both `mmap` would collide. Fine for M2 (only the parent maps, for the
|
||||||
|
> child's stack); make the arena per-address-space and the runtime heap thread-safe alongside the
|
||||||
|
> `Mutex` work (M5).
|
||||||
|
|
||||||
|
## M3 — `join` + `detach` + real parallelism ✅
|
||||||
|
|
||||||
|
- [x] `join` over the existing exit-notification path
|
||||||
|
([process-lifecycle.md](process-lifecycle.md)): `thread_spawn` gained a 4th arg, an
|
||||||
|
`exit_endpoint` handle (resolved + refcounted like `spawnProcessSupervised`, via
|
||||||
|
`spawnThreadSupervised`); `join` blocks in `ipc_reply_wait` on that endpoint until
|
||||||
|
the child-exit notice for its `tid`, then `munmap`s the stack. `detach` relinquishes
|
||||||
|
the join right (its stack is reclaimed at process exit — kernel-reaper reclaim for
|
||||||
|
detached threads is deferred; see note).
|
||||||
|
- [x] `runtime.Thread.join` / `detach`, plus `Thread.currentCore()` (a new `current_core`
|
||||||
|
= 39 syscall) for the parallelism proof. `getCurrentId` deferred to M6 (TLS), where
|
||||||
|
a lighter self-id fits. The closure now rides the **thread's own stack** (not the
|
||||||
|
heap) — private per thread, so spawn/join touch no shared heap.
|
||||||
|
- [x] `-Dtest-case=thread-join` (`smp: 4`): `thread-test` join mode spawns N=4 workers
|
||||||
|
that each do K=100k `@atomicRmw`-increments on a shared counter and stamp the core
|
||||||
|
they ran on; the main thread joins all N and asserts `counter == N*K` **and**
|
||||||
|
`@popCount(cores_seen) > 1` (genuine cross-core parallelism), then a detached worker
|
||||||
|
proves `detach` runs without a join.
|
||||||
|
|
||||||
|
**Gate (met):** `python3 test/qemu_test.py thread-join` passes (`thread-test: join ok` →
|
||||||
|
`DANOS-TEST-RESULT: PASS`), robust across 4 runs; guardrail 17/17 green (incl. `smp`,
|
||||||
|
`affinity`, `process-kill`, and `args`/`init`/`process` on the exit-endpoint spawn path)
|
||||||
|
plus `address-space-refcount`/`thread-spawn`; `zig build` clean, `zig build test` green.
|
||||||
|
|
||||||
|
> **Note (deferred):** a detached thread's stack is freed only at process exit (not by the
|
||||||
|
> reaper on thread exit) — kernel user-stack tracking + reclaim is a later refinement. And
|
||||||
|
> the runtime heap is still not thread-safe: threads that both allocate concurrently would
|
||||||
|
> race (the thread *machinery* avoids the heap, but worker code sharing an allocator does
|
||||||
|
> not). Both fold into the M5 `Mutex`/allocator work.
|
||||||
|
|
||||||
|
## M4 — Futex: the one blocking primitive ✅
|
||||||
|
|
||||||
|
- [x] [abi.zig](../system/abi.zig): `futex_wait = 40`, `futex_wake = 41`. A waiter is a
|
||||||
|
`.blocked` task tagged with `Task.futex_addr` (no queue linkage);
|
||||||
|
`futex_wait(addr, expected, timeout_ns)` reads the user word under the big lock,
|
||||||
|
parks iff `*addr == expected`, and returns on wake or timeout; `futex_wake(addr,
|
||||||
|
count)` scans the task table and readies up to `count` matching waiters (same
|
||||||
|
address space). No spinning — a parked waiter leaves its core free to `hlt`. A
|
||||||
|
timed wait also sets `wake_at`, so the timer's `wakeExpired` wakes it; `futex_addr`
|
||||||
|
staying non-zero (only `futex_wake` clears it) is how the waiter tells timeout from
|
||||||
|
a real wake.
|
||||||
|
- [x] `runtime.Thread.Futex` (`wait` / `timedWait` / `wake`) over the syscall wrappers.
|
||||||
|
- [x] `-Dtest-case=thread-futex` (`smp: 4`): a waiter thread prints `waiting` and
|
||||||
|
`futex_wait`s on a word; the main thread publishes it, prints `waking`, and
|
||||||
|
`futex_wake`s; the waiter prints `woke`. Then a `timedWait` on an unwoken word
|
||||||
|
reports `error.Timeout`.
|
||||||
|
|
||||||
|
**Gate (met):** `python3 test/qemu_test.py thread-futex` passes, robust across 3 runs —
|
||||||
|
the case's **ordered** regex asserts `waiting → waking → woke → PASS` on the serial
|
||||||
|
stream (the handoff proof), and `thread-futex: timeout ok` confirms the timeout.
|
||||||
|
Guardrail 18/18 green (incl. `sleep`/`event`/`ipc` blocking paths) + `address-space-refcount`,
|
||||||
|
`thread-spawn`, `thread-join`; `zig build` clean, `zig build test` green.
|
||||||
|
|
||||||
|
> **Note:** the kernel test checks only the freshest verdict marker via `bufferHas` (the
|
||||||
|
> in-memory log ring buffer evicts older lines); ordering is asserted against the full
|
||||||
|
> serial stream by the qemu regex instead.
|
||||||
|
|
||||||
|
## M5 — `Mutex` + `Condition` + `Semaphore` ✅
|
||||||
|
|
||||||
|
- [x] `runtime.Thread.Mutex` (three-state futex mutex: CAS fast path, `futex_wait`/`wake`
|
||||||
|
slow path), `Condition` (`wait`/`timedWait`/`signal`/`broadcast`, a futex sequence
|
||||||
|
counter), `Semaphore` (permits over `Mutex`+`Condition`) — the same state machines
|
||||||
|
`std.Thread` uses, ported onto our `Futex`.
|
||||||
|
- [x] `-Dtest-case=thread-mutex` (`smp: 4`): a bounded producer/consumer — 2 producers +
|
||||||
|
2 consumers over one `Mutex` and two `Condition`s move N=2000 unique items through
|
||||||
|
an 8-slot ring; the consumed checksum and tally match exactly (no lost/duplicated
|
||||||
|
item, no overrun) under real cross-core contention. The small ring forces producers
|
||||||
|
to block on full and consumers on empty, exercising `Condition.wait`.
|
||||||
|
|
||||||
|
**Gate (met):** `python3 test/qemu_test.py thread-mutex` passes (`thread-mutex: ok` →
|
||||||
|
`DANOS-TEST-RESULT: PASS`), robust across 3 runs; guardrail 17/17 green (incl.
|
||||||
|
`sleep`/`event`/`ipc`) + all M1–M4 thread cases; `zig build` clean, `zig build test`
|
||||||
|
green.
|
||||||
|
|
||||||
|
> **Deferred (with rationale):**
|
||||||
|
> - **`join` → futex completion word** — the exit-endpoint join (M3) is correct and
|
||||||
|
> tested. A futex-completion join needs the *kernel* to clear+wake a word after the
|
||||||
|
> thread is fully off its stack (a CLONE_CHILD_CLEARTID-style mechanism); doing it in
|
||||||
|
> the thread's own trampoline would let `join` `munmap` the stack while the thread still
|
||||||
|
> runs on it (use-after-free). Left on the exit-endpoint path; the kernel clear-on-exit
|
||||||
|
> is a later, separate refinement.
|
||||||
|
> - **Host unit tests for the state machines** — `Mutex`/`Condition` bottom out in the
|
||||||
|
> `futex_*` syscalls, unavailable on the host without a mockable `Futex` seam. The QEMU
|
||||||
|
> `thread-mutex` gate exercises them under real concurrency instead; a host-side mock is
|
||||||
|
> future work.
|
||||||
|
|
||||||
|
## M6 — `getCurrentId`, docs, and CI wiring ✅
|
||||||
|
|
||||||
|
- [x] `getCurrentId` via a small `thread_self = 42` syscall (`runtime.Thread.getCurrentId`
|
||||||
|
returns the kernel task id). **Per-thread `threadlocal` TLS is deferred** — no
|
||||||
|
consumer needs it, and it would require context-switching the thread pointer per task
|
||||||
|
(real kernel + per-switch cost) for an unused feature; threaded binaries have run fine
|
||||||
|
without it through M2–M5. threading.md's TLS reasoning already scoped it as
|
||||||
|
deferred-unless-needed. When a consumer appears, the shape is: `thread_spawn`
|
||||||
|
allocates a per-thread TLS block, sets the thread pointer, and the context switch saves/
|
||||||
|
restores it.
|
||||||
|
- [x] `RwLock` / `WaitGroup` deferred (no consumer yet); they slot onto the same
|
||||||
|
`Futex`/`Mutex`/`Condition` when wanted.
|
||||||
|
- [x] All `thread-*` cases wired into [test/qemu_test.py](../test/qemu_test.py)
|
||||||
|
(`thread-spawn`/`-join`/`-futex`/`-mutex`/`-id`); threading.md + docs/README.md
|
||||||
|
status updated to **built**; the worked example is threading.md's win-condition.
|
||||||
|
- [x] `-Dtest-case=thread-id` (`smp: 4`): two workers read `getCurrentId`; the main
|
||||||
|
thread confirms all three ids are non-zero and distinct — each thread has its own
|
||||||
|
kernel identity. (Renamed from `thread-tls`, which implied `threadlocal`.)
|
||||||
|
|
||||||
|
**Gate (met):** `python3 test/qemu_test.py thread-id` passes; the whole `thread-*` suite
|
||||||
|
(`thread-spawn`/`-join`/`-futex`/`-mutex`/`-id`) plus the full guardrail set pass; default
|
||||||
|
`zig build` clean, `zig build test` green.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
**Phase 1 (M1–M6): built.** danos has `runtime.Thread` — `spawn`/`join`/`detach`,
|
||||||
|
cross-core parallelism, futex, and `Mutex`/`Condition`/`Semaphore`, all over a private
|
||||||
|
thread ABI behind the runtime.
|
||||||
|
|
||||||
|
**Phase 2 (M7–M11): built.** Thread-safe allocation (M7), a task reaper that reclaims dead
|
||||||
|
tasks' kernel stacks (M8), endpoint-free `thread_join` (M9), the per-thread thread pointer (M10),
|
||||||
|
and `RwLock`/`WaitGroup` + host-testable sync (M11). Two things stay deferred by design
|
||||||
|
(no consumer): the Zig `threadlocal` *compiler* layer (M10) and detached-thread user-stack
|
||||||
|
reclaim (M9) — both noted in place.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Phase 2 — hardening (M7–M11)
|
||||||
|
|
||||||
|
The organising principle, so Phase 2 reinforces danos's goals rather than eroding them:
|
||||||
|
|
||||||
|
- **Everything a thread owns is reclaimed on process death.** Thread stacks, TLS blocks,
|
||||||
|
and futex words live in the process's **address space**, and the kernel's per-process
|
||||||
|
state is keyed by the address-space root — so the M1 refcount + `destroyAddressSpace` already
|
||||||
|
free all of it when the last thread exits. A crashed or killed threaded process leaves
|
||||||
|
**nothing** behind. Phase 2 closes the one thing that is *not* address-space-owned — the
|
||||||
|
per-task **kernel** stack (kernel heap) — with a reaper (M8). This is the
|
||||||
|
[resilience](resilience.md) restart guarantee, extended to threads.
|
||||||
|
- **Kernel owns mechanism; the runtime owns policy.** The kernel maps pages, saves/
|
||||||
|
restores the thread pointer, and reaps dead tasks; the runtime decides allocation, TLS layout,
|
||||||
|
and lock algorithms. Every new kernel entry stays a private syscall behind the runtime
|
||||||
|
([syscall.md](syscall.md)) — the ABI stays renumberable.
|
||||||
|
- **The process is still the isolation and restart boundary.** Threads share fate within
|
||||||
|
one process; Phase 2 never adds a way for one process to reach into another (the
|
||||||
|
cross-process futex stays explicitly out of scope, below).
|
||||||
|
|
||||||
|
### M7 — Thread-safe allocation (the correctness gap) ✅
|
||||||
|
|
||||||
|
Today the mmap arena cursor is per-*task* and the runtime heap is unlocked, so two
|
||||||
|
threads in one process that both allocate corrupt each other. The thread *machinery*
|
||||||
|
avoids this (closure on the stack, stacks mmap'd only by the spawner), but real
|
||||||
|
multi-threaded code would hit it. Closed it:
|
||||||
|
|
||||||
|
- [x] **Kernel — per-address-space mmap arena.** Grew M1's `address_space_refs` entry into the
|
||||||
|
per-address-space object holding the `mmap`/`mmio` arena cursors (moved off `Task`);
|
||||||
|
`scheduler.addressSpaceMmapNextPtr`/`addressSpaceDeviceMapNextPtr` expose them. `systemMmap`
|
||||||
|
reserves a disjoint range under a *brief* lock, then maps **per page** under a
|
||||||
|
short-held lock — not the whole grant — because the big lock is held with interrupts
|
||||||
|
disabled, so pinning it across a multi-MiB memset+map froze other cores (it timed
|
||||||
|
the `affinity` scenario out mid-bring-up). Freed at refcount zero, so the cursors
|
||||||
|
vanish with the process.
|
||||||
|
- [x] **Runtime — thread-safe heap.** The allocator's two free-list mutators
|
||||||
|
(`rawAlloc`/`rawFree`) take a `Thread.Mutex`, gated on
|
||||||
|
`!@import("builtin").single_threaded` so single-threaded binaries compile it out and
|
||||||
|
pay nothing. Uncontended acquisition is a single CAS (no syscall).
|
||||||
|
- [x] `-Dtest-case=thread-alloc` (`smp: 4`): 4 threads each do 500 `alloc`/fill/verify/
|
||||||
|
`free` cycles of varied sizes; each block is filled with a per-thread pattern and
|
||||||
|
verified before free, so any overlap between concurrent allocations is caught.
|
||||||
|
|
||||||
|
**Gate (met):** `thread-alloc` passes (3× non-flaky); full guardrail 23/23 green,
|
||||||
|
`zig build`/`zig build test` clean.
|
||||||
|
|
||||||
|
> **Also fixed here:** the `affinity` guardrail's fixed-count busy-loop (`while (spins <
|
||||||
|
> 3e9)`) had codegen-dependent wall-time — adding a function to `tests.zig` flipped how
|
||||||
|
> the optimiser compiled it, swinging affinity from ~4 s to ~63 s and timing it out.
|
||||||
|
> Reworked it (and the settle loop) to wait on the wall clock instead, so its duration is
|
||||||
|
> independent of unrelated code changes.
|
||||||
|
|
||||||
|
### M8 — The task reaper (cleanup + resilience) ✅
|
||||||
|
|
||||||
|
A dead task's **kernel** stack was leaked ("no reaper yet") — every process *and* thread
|
||||||
|
death lost one, so a crash loop bled kernel memory. The reaper fixes it and serves the
|
||||||
|
[resilience](resilience.md) restart goal directly:
|
||||||
|
|
||||||
|
- [x] A dying task cannot free the kernel stack it runs on, so `exit()`/`exitUserLocked`
|
||||||
|
record it in a **per-core `reap_after_switch` slot** and switch away; the task that
|
||||||
|
resumes on that core frees the stack in `switchTo`'s tail (it's on its own stack, the
|
||||||
|
big lock is still held so the slot can't have been reused). A **tick-time drain**
|
||||||
|
(`reapKillPendingLocked`) is the safety net for the case where the next task is
|
||||||
|
*fresh* (enters via the trampoline, bypassing `switchTo`'s tail). A task killed while
|
||||||
|
*not* running is freed immediately in `destroyTaskLocked`. A `live_stack_bytes`
|
||||||
|
counter is the observable. *(Detached-thread user-stack reclaim moves to M9, which
|
||||||
|
adds the joinable/detached flag.)*
|
||||||
|
- [x] `-Dtest-case=task-reap` (`smp: 4`): spawn and kill 12 processes; poll the
|
||||||
|
test-observable `scheduler.liveStackBytes()` until it returns to **baseline** (a
|
||||||
|
correct reaper gets there in a few ms; a genuine leak times out) — every kernel
|
||||||
|
stack reclaimed, no leak. Threads exit through the same `exitUserLocked`, so covered.
|
||||||
|
|
||||||
|
**Gate (met):** `task-reap` passes (5× isolated + 2× in the full batch); `fault-recovery`,
|
||||||
|
`supervision`, `process-kill`, `address-space-refcount`, `smp`, `affinity` all still green (24/24
|
||||||
|
full guardrail); `zig build`/`zig build test` clean.
|
||||||
|
|
||||||
|
> **Bug found + fixed here (touches every context switch):** the post-`switchContext` reap
|
||||||
|
> first read the `pc` **parameter**, but a task that migrated cores carries a *stale* `pc`
|
||||||
|
> in its saved `switchTo` frame — so it read the wrong core's slot and freed a live stack
|
||||||
|
> (a #GP under SMP). Fixed to re-fetch `thisCpu()` after the switch (the switch only swaps
|
||||||
|
> stacks on the current core).
|
||||||
|
|
||||||
|
### M9 — Futex-completion join (retire the per-thread endpoint)
|
||||||
|
|
||||||
|
With the reaper (M8) able to act *after* a thread is fully off its stack, migrate `join`
|
||||||
|
to the std shape and drop M3's per-thread exit endpoint:
|
||||||
|
|
||||||
|
- [x] A **`thread_join(tid)` syscall** (not a user futex word): it blocks the caller until
|
||||||
|
the task with id `tid` exits, and the exit paths call `wakeJoinersLocked`. `join`
|
||||||
|
only reclaims the joined thread's **user** stack, which the thread vacates the moment
|
||||||
|
it enters the kernel to exit — so waking at *exit* time (not reap time) is safe, and
|
||||||
|
no reaper/address-space juggling or user-memory write is needed. This is equally
|
||||||
|
std-shaped (like `pthread_join`) and much simpler/safer than the planned
|
||||||
|
reaper-written completion word. `thread_spawn` no longer takes an exit endpoint (the
|
||||||
|
runtime passes `no_cap`); the per-thread IPC endpoint is gone.
|
||||||
|
- [x] `thread-join` passes on the new path, and its join mode now runs **40 spawn+join
|
||||||
|
cycles** — under the old per-thread-endpoint scheme those leaked handles would
|
||||||
|
exhaust the 16-slot handle table; here they all succeed, proving join is endpoint-free.
|
||||||
|
|
||||||
|
**Gate (met):** `thread-join` passes (3× isolated) on the `thread_join` path; full
|
||||||
|
guardrail 26/26 (incl. `process-kill`, `supervision`, `fault-recovery`, `task-reap`);
|
||||||
|
`zig build`/`zig build test` clean.
|
||||||
|
|
||||||
|
> **Reaper hardened here (fixes an M8 flake).** M8's single per-core reap slot could be
|
||||||
|
> *overwritten* by a second death on that core before the first drained (a fresh-task/SMP
|
||||||
|
> timing window) — an intermittent one-stack leak (`task-reap` flaked ~20%). Replaced it
|
||||||
|
> with a per-core reap **list** plus a `.reaping` task state so a pending slot can't be
|
||||||
|
> reused before its stack is freed. `task-reap` now 11/11 isolated + 2× in the batch.
|
||||||
|
|
||||||
|
> **Deferred:** detached-thread **user-stack** reclaim (still freed at process exit, as in
|
||||||
|
> M3). Doing it in the reaper needs the saved address space + stack range and a
|
||||||
|
> translate/unmap in a not-currently-loaded address space — real complexity for a bounded leak.
|
||||||
|
> A follow-up when a consumer needs it.
|
||||||
|
|
||||||
|
### M10 — Per-thread TLS: the thread-pointer mechanism ✅
|
||||||
|
|
||||||
|
Give each thread its own thread pointer and private TLS storage — the foundation
|
||||||
|
self-hosting Zig ([zig-self-hosting.md](zig-self-hosting.md)) will build `threadlocal` on.
|
||||||
|
|
||||||
|
- [x] **Kernel** stores `thread_pointer` on `Task` and restores it on every context switch
|
||||||
|
**only when it changes** (the same conditional-load discipline as CR3;
|
||||||
|
`architecture.setThreadPointer` → `wrmsr IA32_FS_BASE` on x86_64). A
|
||||||
|
`set_thread_pointer(addr)` = 44 syscall sets the caller's `thread_pointer` and loads it
|
||||||
|
now. The kernel never touches FS, so there is no swapgs complication.
|
||||||
|
- [x] **Runtime** lays a small per-thread TLS block at the top of each thread's stack
|
||||||
|
(self-pointer at `%fs:0` + scratch slots) and the thread trampoline calls
|
||||||
|
`set_thread_pointer` before any user code — so every spawned thread has a private,
|
||||||
|
switch-stable thread pointer. Reclaimed with the stack.
|
||||||
|
- [x] `-Dtest-case=thread-tls` (`smp: 4`): two threads each write a unique marker to their
|
||||||
|
own `%fs:8` slot and — after both have written — read it back; a shared (non-per-thread)
|
||||||
|
FS base would clobber one and cause cross-talk. Both read their own marker → pass.
|
||||||
|
|
||||||
|
**Gate (met):** `thread-tls` passes (3×); full guardrail 25/25 (the switch-time thread-pointer
|
||||||
|
restore touches every context switch); `zig build`/`zig build test` clean.
|
||||||
|
|
||||||
|
> **Deferred: the Zig `threadlocal` *compiler* layer.** Real `threadlocal` variables need
|
||||||
|
> the ELF **variant-II TLS** surface — `.tdata`/`.tbss` sections + a `PT_TLS` program header
|
||||||
|
> in `user.ld`, a runtime that copies the template with exact negative-offset layout, and
|
||||||
|
> the `.large`-code-model TLS section names — a high-uncertainty lift for a feature with
|
||||||
|
> **no consumer today** (threading.md scopes it "only if a consumer needs it"). What lands
|
||||||
|
> here is the load-bearing piece — the per-thread thread pointer, context-switched — so adding the
|
||||||
|
> compiler layer later is purely runtime+linker work on top, no kernel change. `getCurrentId`
|
||||||
|
> stays the `thread_self` syscall (M6) rather than an fs self-slot (which would need the
|
||||||
|
> main thread's TLS set up in `_start` too).
|
||||||
|
|
||||||
|
**Gate:** `thread-tls` passes; full `thread-*` suite + guardrail green.
|
||||||
|
|
||||||
|
### M11 — `RwLock`, `WaitGroup`, and host-testable sync ✅
|
||||||
|
|
||||||
|
- [x] `runtime.Thread.RwLock` (reader-preferring: `>0` readers / `-1` writer / `0` free,
|
||||||
|
with `lock`/`tryLock`/`unlock` + `lockShared`/`tryLockShared`/`unlockShared`) and
|
||||||
|
`WaitGroup` (`start`/`finish`/`wait`), both on the existing `Mutex`/`Condition`.
|
||||||
|
- [x] A compile-time `Futex` seam gated on `builtin.os.tag == .freestanding`: the futex
|
||||||
|
syscalls on danos, a spin+yield mock off-target (Zig 0.16 has no `std.Thread.Futex`;
|
||||||
|
`wake` is a no-op since the state machines re-check). `thread.zig` is wired into
|
||||||
|
`zig build test`, so `Mutex`/`RwLock`/`WaitGroup` run as **host unit tests** with real
|
||||||
|
`std.Thread` threads (`test` blocks only compile under test).
|
||||||
|
- [x] `-Dtest-case=thread-rwlock` (`smp: 4`): 2 writers set both halves of a value under
|
||||||
|
the exclusive lock while 3 readers check the halves match under the shared lock —
|
||||||
|
zero half-write observations across ~150k reads. Host tests cover the Mutex,
|
||||||
|
RwLock, and WaitGroup state machines.
|
||||||
|
|
||||||
|
**Gate (met):** `zig build test` covers the sync primitives (host threads); `thread-rwlock`
|
||||||
|
passes (3×); full Done gate **26/26** (whole `thread-*` suite + guardrail); `zig build`
|
||||||
|
clean.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Deferred (explicitly not in this plan)
|
||||||
|
|
||||||
|
- **Cross-process shared-memory futex** — the `(address_space, virtual_address)` key can become a
|
||||||
|
physical-address key so two processes share a futex through an [shm](display-v2.md)
|
||||||
|
region. Not needed for intra-process threads.
|
||||||
|
- **Per-thread priorities / affinity distinct from the process** — threads inherit the
|
||||||
|
process priority ([scheduling.md](scheduling.md)); revisit only if it earns its keep.
|
||||||
|
- **Per-thread signal delivery** — signals stay process-scoped
|
||||||
|
([process-lifecycle.md](process-lifecycle.md)).
|
||||||
|
- **A `pthread`/POSIX surface** — the API is `std.Thread`-shaped Zig, nothing more.
|
||||||
|
- **A real `std.Thread` backend** — arrives with self-hosting
|
||||||
|
([zig-self-hosting.md](zig-self-hosting.md)); it sits on these same primitives, so it
|
||||||
|
swaps the impl under `runtime.Thread`, not the call sites.
|
||||||
@@ -0,0 +1,332 @@
|
|||||||
|
# Threading: `runtime.Thread`, a std-shaped API over a private thread ABI
|
||||||
|
|
||||||
|
A note on danos **threads** — several tasks sharing one address space — provided by a
|
||||||
|
`runtime.Thread` type that mirrors the shape of Zig's `std.Thread` while keeping every
|
||||||
|
kernel entry behind the [runtime](../library/runtime). **Built** (M1–M11, see
|
||||||
|
[threading-plan.md](threading-plan.md)): `spawn`/`join`/`detach`, cross-core parallelism,
|
||||||
|
a futex, `Mutex`/`Condition`/`Semaphore`/`RwLock`/`WaitGroup`, `getCurrentId`/`currentCore`,
|
||||||
|
per-thread thread-pointer TLS, thread-safe allocation, and a task reaper that reclaims dead
|
||||||
|
tasks' kernel stacks. Deferred by design (no consumer yet): the Zig `threadlocal`
|
||||||
|
*compiler* layer (the per-thread thread pointer is in place, so it's runtime+linker work on top) and
|
||||||
|
detached-thread user-stack reclaim — see the plan's M9/M10 notes. The analysis is against
|
||||||
|
**Zig 0.16** (the pinned toolchain); `std.Thread`'s internals move between releases, so
|
||||||
|
treat upstream shapes as "0.16.x."
|
||||||
|
|
||||||
|
## The win condition
|
||||||
|
|
||||||
|
A danos service can write
|
||||||
|
|
||||||
|
```zig
|
||||||
|
const t = try runtime.Thread.spawn(.{}, worker, .{ctx});
|
||||||
|
// ... do other work concurrently ...
|
||||||
|
t.join();
|
||||||
|
```
|
||||||
|
|
||||||
|
and get real parallelism across cores — with `runtime.Thread.Mutex`,
|
||||||
|
`runtime.Thread.Condition`, and `runtime.Thread.Semaphore` available for
|
||||||
|
coordination — **without any code path reaching the kernel except through the
|
||||||
|
runtime**. The call sites read exactly like `std.Thread`, so the day danos becomes a
|
||||||
|
real Zig target (see [self-hosting](#the-self-hosting-endgame)) we swap the
|
||||||
|
implementation underneath, not the API above.
|
||||||
|
|
||||||
|
## Locked decisions (do not relitigate)
|
||||||
|
|
||||||
|
- **We build `runtime.Thread`, not literal `std.Thread`.** It mirrors std's *API and
|
||||||
|
features*; the implementation underneath is danos-native. See
|
||||||
|
[Why not literal std.Thread](#why-not-literal-stdthread).
|
||||||
|
- **Threads are a narrow, opt-in capability — not the default concurrency tool.** The
|
||||||
|
default for resilience stays **process + IPC** ([resilience.md](resilience.md),
|
||||||
|
[ipc.md](ipc.md)). See [Where threads fit](#where-threads-fit-the-resilience-tension).
|
||||||
|
- **Blocking synchronization is futex-backed, never spin-backed.** Waiters sleep in
|
||||||
|
the kernel so an idle core still halts ([halting.md](halting.md)).
|
||||||
|
- **Per-binary opt-in to multi-threaded codegen.** Only a service that asks for
|
||||||
|
threads is built `single_threaded = false`; the rest stay lean and single-threaded.
|
||||||
|
- **The thread ABI is private.** New syscalls extend [abi.zig](../system/abi.zig)
|
||||||
|
`SystemCall` and are reached only through `library/runtime` wrappers, exactly like
|
||||||
|
every other danos syscall ([syscall.md](syscall.md)) — numbers stay renumberable.
|
||||||
|
|
||||||
|
## Why not literal `std.Thread`
|
||||||
|
|
||||||
|
danos's ABI invariant is that the **runtime is the sole holder of the syscall ABI**,
|
||||||
|
and that ABI is private and renumberable ([syscall.md](syscall.md) — "unstable
|
||||||
|
private ABI"). That is a security and evolvability asset: no compiled binary can
|
||||||
|
hardcode a syscall number, and the kernel can renumber freely because only the
|
||||||
|
runtime — rebuilt in lockstep — knows the mapping.
|
||||||
|
|
||||||
|
`std.Thread` is incompatible with that invariant on two counts:
|
||||||
|
|
||||||
|
1. **It selects its backend from `builtin.os.tag`, and issues syscalls directly.**
|
||||||
|
danos targets `.os_tag = .freestanding` ([build.zig](../build.zig)), for which
|
||||||
|
`std.Thread` resolves to an unsupported stub that `@compileError`s. Adding a real
|
||||||
|
backend would either bake danos syscall numbers into std (breaking ABI privacy and
|
||||||
|
renumbering) or fork std to route back through the runtime — a permanent rebase
|
||||||
|
cost that buys nothing the native type doesn't.
|
||||||
|
2. **Our user binaries are built `single_threaded = true`** ([build.zig](../build.zig)
|
||||||
|
`addUserBinary`), which compiles threading out entirely and makes atomics and TLS
|
||||||
|
single-threaded. Threads need this flipped per binary regardless.
|
||||||
|
|
||||||
|
So we take the *shape* of `std.Thread`, not the *type*. The cost of replicating the
|
||||||
|
surface (spawn/join/Mutex/Condition) is small; the cost of the std type is the ABI
|
||||||
|
invariant.
|
||||||
|
|
||||||
|
## Where threads fit: the resilience tension
|
||||||
|
|
||||||
|
Threads are in genuine tension with a resilience-first microkernel, and it is worth
|
||||||
|
being explicit so we do not reach for them by reflex.
|
||||||
|
|
||||||
|
The reason danos pays for a microkernel is **fault isolation**
|
||||||
|
([resilience.md](resilience.md)): a component corrupts its own address space, faults,
|
||||||
|
and is **restarted** without touching anyone else — because the boundary *is* the
|
||||||
|
address space. Threads deliberately remove that boundary *within* a process:
|
||||||
|
|
||||||
|
- Threads share one address space, so one thread's stray write corrupts them all —
|
||||||
|
there is no isolation **between** threads.
|
||||||
|
- Threads share fate: a fault in any thread, or a "kill the process" decision, takes
|
||||||
|
down **all** of them. Restartability lives at the process level, not the thread
|
||||||
|
level.
|
||||||
|
- Shared mutable state reintroduces data races — the failure class the
|
||||||
|
isolate-and-message model was chosen to avoid.
|
||||||
|
|
||||||
|
**Therefore:** the default answer to "make X concurrent" stays *another process over
|
||||||
|
IPC* (isolated, independently restartable) or a single event loop with several
|
||||||
|
message sources. Reach for a thread only inside **one** service that needs genuine
|
||||||
|
**shared-memory, low-latency parallelism** and can accept intra-service fate-sharing —
|
||||||
|
e.g. a compositor splitting tile compositing across cores, where per-tile IPC would be
|
||||||
|
too chatty. "Input on one thread, display on another" is *not* that case; it wants two
|
||||||
|
processes. The isolation boundary stays at process granularity.
|
||||||
|
|
||||||
|
## The API surface (mirrors `std.Thread`)
|
||||||
|
|
||||||
|
Lives in `library/runtime/thread.zig`, re-exported as `runtime.Thread`.
|
||||||
|
|
||||||
|
```zig
|
||||||
|
pub const Thread = struct {
|
||||||
|
pub const Id = u32; // the kernel task id
|
||||||
|
pub const SpawnConfig = struct {
|
||||||
|
stack_size: usize = default_stack_size,
|
||||||
|
allocator: ?std.mem.Allocator = null, // for the closure + stack bookkeeping
|
||||||
|
};
|
||||||
|
pub const SpawnError = error{ OutOfMemory, ThreadQuotaExceeded, SystemResources };
|
||||||
|
|
||||||
|
pub fn spawn(config: SpawnConfig, comptime function: anytype, args: anytype) SpawnError!Thread;
|
||||||
|
pub fn join(self: Thread) void; // block until the thread ends, reclaim its stack
|
||||||
|
pub fn detach(self: Thread) void; // give up the right to join; kernel reclaims on exit
|
||||||
|
pub fn getCurrentId() Id;
|
||||||
|
pub fn yield() void; // -> existing `yield` syscall
|
||||||
|
|
||||||
|
pub const Mutex = struct { pub fn lock(*Mutex) void; pub fn tryLock(*Mutex) bool; pub fn unlock(*Mutex) void; };
|
||||||
|
pub const Condition = struct { pub fn wait(*Condition, *Mutex) void; pub fn timedWait(*Condition, *Mutex, u64) error{Timeout}!void; pub fn signal(*Condition) void; pub fn broadcast(*Condition) void; };
|
||||||
|
pub const Semaphore = struct { pub fn wait(*Semaphore) void; pub fn post(*Semaphore) void; };
|
||||||
|
pub const Futex = struct { pub fn wait(*const atomic.Value(u32), u32) void; pub fn timedWait(...) error{Timeout}!void; pub fn wake(*const atomic.Value(u32), u32) void; };
|
||||||
|
// RwLock / ResetEvent / WaitGroup follow the same pattern, added as needed.
|
||||||
|
};
|
||||||
|
```
|
||||||
|
|
||||||
|
Deviations from `std.Thread`, called out honestly:
|
||||||
|
|
||||||
|
- **The thread function's return value is discarded** (as `std.Thread.join` returns
|
||||||
|
`void`). Return data through shared state or a `Semaphore`/`Condition`, not the
|
||||||
|
return.
|
||||||
|
- `getCpuCount()` maps to the existing SMP core count ([smp.md](smp.md)); a service
|
||||||
|
rarely needs it.
|
||||||
|
|
||||||
|
## Kernel primitives (new private syscalls)
|
||||||
|
|
||||||
|
Four new entries extend [abi.zig](../system/abi.zig) `SystemCall` after
|
||||||
|
`shm_physical = 36`, each with a `library/runtime` wrapper:
|
||||||
|
|
||||||
|
| Syscall | Signature | Purpose |
|
||||||
|
|---|---|---|
|
||||||
|
| `thread_spawn` | `(entry, stack_top, arg) -> tid` | create a task sharing the **caller's** address space |
|
||||||
|
| `thread_exit` | `(stack_base, stack_len)` | end the calling thread; hand back its stack range for reclaim |
|
||||||
|
| `futex_wait` | `(addr, expected, timeout_ns) -> status` | block if `*addr == expected`, until woken or timeout |
|
||||||
|
| `futex_wake` | `(addr, count) -> woken` | wake up to `count` waiters on `addr` |
|
||||||
|
|
||||||
|
Plus one invariant change with no new syscall: **address-space reference counting**.
|
||||||
|
|
||||||
|
## Mechanics
|
||||||
|
|
||||||
|
### Address-space reference counting
|
||||||
|
|
||||||
|
Today an address space is 1:1 with a task: `spawnUserLocked` records `address_space` on the
|
||||||
|
Task, and teardown does `destroyAddressSpace(t.address_space)` when **any** user task exits
|
||||||
|
([scheduler.zig](../system/kernel/scheduler.zig)). With threads, several tasks share
|
||||||
|
one `address_space`, so the first to exit would rip the address space out from under its
|
||||||
|
siblings.
|
||||||
|
|
||||||
|
Fix: a small refcount keyed by the address-space root (`createAddressSpace` in
|
||||||
|
[process.zig](../system/kernel/process.zig) sets it to 1). `thread_spawn` increments
|
||||||
|
it; task teardown decrements and only calls `destroyAddressSpace` at **zero**. All of
|
||||||
|
this is already under the big kernel lock, so no new locking. This is the one piece
|
||||||
|
that must land and be proven before anything shares an address space.
|
||||||
|
|
||||||
|
### `thread_spawn` and the trampoline
|
||||||
|
|
||||||
|
The scheduler already accepts an arbitrary `address_space` and does **not** smuggle values
|
||||||
|
through registers — `startUserTask` reads the entry/stack from the Task and
|
||||||
|
`jumpToUser`s ([scheduler.zig](../system/kernel/scheduler.zig)). That makes the thread
|
||||||
|
path clean:
|
||||||
|
|
||||||
|
1. The runtime's `spawn` `mmap`s a stack (syscall `4`), heap-allocates a closure —
|
||||||
|
`{ fn_ptr, args_tuple, completion }`, the std "Instance" pattern — and writes the
|
||||||
|
closure pointer to the **top word of the new stack**.
|
||||||
|
2. It calls `thread_spawn(entry = &threadTrampoline, stack_top, arg = closure_ptr)`.
|
||||||
|
The kernel calls the same `spawnUserLocked` path with the **caller's address space**
|
||||||
|
(refcount++), `entry`, and `user_sp = stack_top`.
|
||||||
|
3. `threadTrampoline` (a small runtime shim) reads the closure off its stack, calls
|
||||||
|
the user function, then calls `thread_exit`. No new register ABI — the closure
|
||||||
|
pointer rides the stack the runtime set up, mirroring how `startUserTask` avoids
|
||||||
|
register smuggling.
|
||||||
|
|
||||||
|
Unlike a process start, there is **no** System V argc/argv/auxv block
|
||||||
|
([sysv.md](sysv.md)) — a thread stack carries only the closure pointer.
|
||||||
|
|
||||||
|
### Lifetime: exit, join, detach, stack reclaim
|
||||||
|
|
||||||
|
- **`thread_exit`** marks the task dead and hands the kernel the thread's user-stack
|
||||||
|
range. The kernel reaps the task on the scheduler (already running on a *kernel*
|
||||||
|
stack, so it can safely unmap the user stack), decrements the address-space refcount, and
|
||||||
|
frees the task slot.
|
||||||
|
- **`join` — Stage 1** reuses the existing exit-notification machinery
|
||||||
|
([process-lifecycle.md](process-lifecycle.md)): `spawn` passes a per-thread
|
||||||
|
`exit_endpoint`, and `join` blocks in `ipc_reply_wait` until the child-exit
|
||||||
|
notification for that `tid` arrives, then `munmap`s the stack. No futex needed to
|
||||||
|
land spawn/join.
|
||||||
|
- **`join` — Stage 2 refinement** migrates to the std shape: a `completion` word in
|
||||||
|
the closure that `thread_exit`'s trampoline `futex_wake`s and `join` `futex_wait`s
|
||||||
|
on — dropping the per-thread endpoint. Kept as a refinement so Stage 1 ships first.
|
||||||
|
- **`detach`** relinquishes the join right; the kernel reclaims the stack and slot on
|
||||||
|
`thread_exit` (a detached thread's stack range is unmapped by the reaper, since no
|
||||||
|
joiner will).
|
||||||
|
|
||||||
|
### Futex, and the sync primitives on top
|
||||||
|
|
||||||
|
`futex_wait`/`futex_wake` are the one blocking primitive; `Mutex`, `Condition`, and
|
||||||
|
`Semaphore` are ordinary user-space state machines over an `atomic.Value(u32)` that
|
||||||
|
call the futex wrappers on the slow path — the same construction `std.Thread` uses,
|
||||||
|
so the algorithms port directly.
|
||||||
|
|
||||||
|
Keying: threads share an address space, so a **virtual address within that address space**
|
||||||
|
identifies a futex uniquely; the kernel keys its wait queue by `(address_space_root, virtual_address)`.
|
||||||
|
Keying by the **physical** address instead (translate `virtual_address -> physical_address` on entry) is a
|
||||||
|
deliberate forward door: it lets two *processes* share a futex through an
|
||||||
|
[shm](display-v2.md) region later, without changing the API. We start with the
|
||||||
|
private-per-address-space key and note the physical-key upgrade.
|
||||||
|
|
||||||
|
No spinning: a contended lock parks the task in the kernel and the core is free to run
|
||||||
|
other work or `hlt` ([halting.md](halting.md)). This is why futex is a locked
|
||||||
|
decision, not a "maybe later."
|
||||||
|
|
||||||
|
### TLS and `getCurrentId`
|
||||||
|
|
||||||
|
danos sets up no thread-pointer TLS today (fine under `single_threaded`). Two scoped needs:
|
||||||
|
|
||||||
|
- **`getCurrentId`** returns the kernel task id — either a trivial syscall or, better,
|
||||||
|
a value the runtime stashes in a per-thread control block.
|
||||||
|
- **`threadlocal` variables** need a real per-thread TLS block and the thread pointer set per
|
||||||
|
thread. `thread_spawn` sets the thread pointer to a runtime-allocated per-thread block; full
|
||||||
|
`threadlocal` support is Stage 3, only if a consumer needs it. Nothing in the core
|
||||||
|
spawn/join/mutex path requires `threadlocal`.
|
||||||
|
|
||||||
|
### Build: multi-threaded codegen, opt-in
|
||||||
|
|
||||||
|
`addUserBinary` gains a `threaded: bool = false` parameter; when set it builds that
|
||||||
|
binary `single_threaded = false` so atomics and (later) TLS are real. Threads and
|
||||||
|
atomics are unsound in a `single_threaded` image, so a binary must opt in **before**
|
||||||
|
it may call `runtime.Thread.spawn`. Everyone else stays single-threaded and lean.
|
||||||
|
|
||||||
|
## Interaction with the rest of the kernel
|
||||||
|
|
||||||
|
- **Scheduler / SMP** ([scheduling.md](scheduling.md), [smp.md](smp.md)): a thread is
|
||||||
|
just another `Task` with an `address_space` shared with its siblings; the existing
|
||||||
|
per-core ready queues, priorities, and affinity apply unchanged. Threads of one
|
||||||
|
process can run on different cores simultaneously — that is the point.
|
||||||
|
- **Halting** ([halting.md](halting.md)): futex-parked waiters keep the "idle core
|
||||||
|
halts" property intact under lock contention — no busy-wait.
|
||||||
|
- **Lifecycle** ([process-lifecycle.md](process-lifecycle.md)): killing a process
|
||||||
|
must kill *all* its threads and only then drop the last address-space ref. The kill path
|
||||||
|
already targets a process; it fans out to every task on that address space.
|
||||||
|
- **Resilience** ([resilience.md](resilience.md)): a faulting thread kills its whole
|
||||||
|
process (shared fate). The supervisor restarts the **process**, which respawns its
|
||||||
|
threads from a known-good state — restart granularity stays the process.
|
||||||
|
- **IPC — two consequences threads forced ([ipc.md](ipc.md)):**
|
||||||
|
- *Handles do not cross threads.* The handle table lives on the `Task`
|
||||||
|
([scheduler.zig](../system/kernel/scheduler.zig)), so a handle number is meaningful
|
||||||
|
only to the thread that created it — thread A's endpoint handle `3` is not thread B's.
|
||||||
|
A thread that needs to reach an endpoint another thread owns looks it up
|
||||||
|
(`ipc.lookup(service)`) to install its **own** handle to the same underlying endpoint.
|
||||||
|
This is how the display's mouse-listener thread reaches the compositor loop's endpoint
|
||||||
|
to poke it awake (docs/display.md).
|
||||||
|
- *IPC syscalls that touch shared kernel state now serialize under the big kernel lock.*
|
||||||
|
`create_ipc_endpoint`/`ipc_register`/`ipc_lookup` allocate from the kernel heap and
|
||||||
|
mutate the global service registry, endpoint refcounts, and handle tables. Those paths
|
||||||
|
were unlocked because a single-threaded process could not race itself; a multi-threaded
|
||||||
|
one can, from two cores at once. They now take `sync.enter()` like `call`/`reply_wait`/
|
||||||
|
`send` already did — the kernel heap has no lock of its own yet (heap.zig: "a lock comes
|
||||||
|
with threads/SMP"), so the big lock is what keeps its callers serialized.
|
||||||
|
|
||||||
|
## Build-out plan (staged, each gate serial-checkable)
|
||||||
|
|
||||||
|
The ordered, `/loop`-runnable milestones live in
|
||||||
|
**[threading-plan.md](threading-plan.md)** (shaped like
|
||||||
|
[display-v2-plan.md](display-v2-plan.md)): every milestone lands on its own and ends in
|
||||||
|
a verifiable gate (`python3 test/qemu_test.py <case>`, asserting serial markers;
|
||||||
|
`zig build test` for host unit tests). The stages below are the shape it expands.
|
||||||
|
|
||||||
|
- **Stage 0 — address-space refcount.** Refcount on the address-space root; teardown destroys
|
||||||
|
at zero. No API yet; nothing shares an address space, so refcount is 1 everywhere.
|
||||||
|
*Gate:* the full QEMU suite stays green (no regression) — proves the reframing is
|
||||||
|
invisible until used.
|
||||||
|
- **Stage 1 — spawn / join / detach.** `thread_spawn` + `thread_exit`, the trampoline,
|
||||||
|
stacks via `mmap`, join over the exit-endpoint, the `threaded` build flag.
|
||||||
|
*Gate:* `-Dtest-case=thread-spawn` — a threaded test service spawns N threads that
|
||||||
|
each `@atomicRmw`-increment a shared counter, the parent joins all N, and asserts
|
||||||
|
the total is exactly N × iterations. Runs `smp` (multi-core) to prove real
|
||||||
|
parallelism.
|
||||||
|
- **Stage 2 — blocking synchronization.** `futex_wait`/`futex_wake` + `Futex`,
|
||||||
|
`Mutex`, `Condition`, `Semaphore`; optionally migrate join to a futex completion
|
||||||
|
word. *Gate:* `-Dtest-case=thread-mutex` — a bounded producer/consumer over a
|
||||||
|
`Mutex` + `Condition` moves K items with no lost wakeups and no busy-wait (assert
|
||||||
|
the consumer blocked, e.g. via a low idle tick count).
|
||||||
|
- **Stage 3 — polish.** Per-thread TLS / thread pointer and `threadlocal` (only if a
|
||||||
|
consumer needs it), `RwLock`/`WaitGroup` as demanded, and this doc's cases wired
|
||||||
|
into [test/qemu_test.py](../test/qemu_test.py).
|
||||||
|
|
||||||
|
## Conventions
|
||||||
|
|
||||||
|
Follow [coding-standards.md](coding-standards.md): spell out non-acronym
|
||||||
|
abbreviations, kebab-case file names, no `Co-Authored-By` trailers. New syscalls
|
||||||
|
extend [abi.zig](../system/abi.zig) `SystemCall` + a `library/runtime` wrapper
|
||||||
|
([syscall.md](syscall.md)). `runtime.Thread` is a first-class runtime module, the same
|
||||||
|
way `runtime.process` ([process-lifecycle.md](process-lifecycle.md)) and `runtime.ipc`
|
||||||
|
are — user code never names a syscall.
|
||||||
|
|
||||||
|
## Non-goals
|
||||||
|
|
||||||
|
- **No preemptive user-space signals delivered to a specific thread.** Signals stay
|
||||||
|
process-scoped ([process-lifecycle.md](process-lifecycle.md)).
|
||||||
|
- **No thread priorities distinct from the process.** Threads inherit the process
|
||||||
|
priority; per-thread priority is a later question if it ever earns its keep.
|
||||||
|
- **No cross-process shared-memory futex yet** — the physical-address key leaves the
|
||||||
|
door open, but the first cut is private-per-address-space.
|
||||||
|
- **No `pthread`/POSIX surface.** The API is `std.Thread`-shaped Zig, nothing more.
|
||||||
|
|
||||||
|
## The self-hosting endgame
|
||||||
|
|
||||||
|
When danos becomes a real Zig target and we (eventually) add a danos backend to std
|
||||||
|
([zig-self-hosting.md](zig-self-hosting.md)), `std.Thread` can sit *on top of* these
|
||||||
|
same kernel primitives — the danos `std.Thread.Impl` would call the very
|
||||||
|
`thread_spawn`/`futex_*` wrappers `runtime.Thread` already uses. Because
|
||||||
|
`runtime.Thread` was built API-compatible from day one, that transition swaps the
|
||||||
|
implementation, not a single call site. Designing to the std shape now is what makes
|
||||||
|
the later self-hosting lift cheap.
|
||||||
|
|
||||||
|
## Further reading
|
||||||
|
|
||||||
|
- [scheduling.md](scheduling.md), [smp.md](smp.md) — the task model these threads join.
|
||||||
|
- [resilience.md](resilience.md), [vision.md](vision.md) — why isolation is the default
|
||||||
|
and threads are the exception.
|
||||||
|
- [syscall.md](syscall.md), [ipc.md](ipc.md) — the private ABI and the messaging model
|
||||||
|
threads sit beside.
|
||||||
|
- [halting.md](halting.md) — the idle/halt property futex-backed blocking preserves.
|
||||||
|
- [zig-self-hosting.md](zig-self-hosting.md) — the target this bends toward.
|
||||||
+1
-1
@@ -119,7 +119,7 @@ One table entry per kernel call, C ABI (System V AMD64), names prefixed
|
|||||||
returns are `u64`, errors return as negative values exactly as today.
|
returns are `u64`, errors return as negative values exactly as today.
|
||||||
|
|
||||||
The calls that return two values in `rax:rdx` today — `dma_alloc`
|
The calls that return two values in `rax:rdx` today — `dma_alloc`
|
||||||
(vaddr + paddr), `msi_bind` (address + data), `shm_create` (vaddr + handle) —
|
(virtual_address + physical_address), `msi_bind` (address + data), `shm_create` (virtual_address + handle) —
|
||||||
become functions returning a two-`u64` struct. The System V ABI returns a
|
become functions returning a two-`u64` struct. The System V ABI returns a
|
||||||
16-byte struct in `rax:rdx`, so the stub is a plain `syscall; ret` — the
|
16-byte struct in `rax:rdx`, so the stub is a plain `syscall; ret` — the
|
||||||
C-ABI spelling of the existing convention, at zero cost.
|
C-ABI spelling of the existing convention, at zero cost.
|
||||||
|
|||||||
@@ -9,15 +9,32 @@
|
|||||||
//! — `grow` asks the kernel for pages via `mmap` instead of mapping frames
|
//! — `grow` asks the kernel for pages via `mmap` instead of mapping frames
|
||||||
//! itself, and the kernel picks the base address.
|
//! itself, and the kernel picks the base address.
|
||||||
//!
|
//!
|
||||||
//! Single-threaded and 16-byte maximum alignment, exactly like the kernel heap; a
|
//! 16-byte maximum alignment, exactly like the kernel heap. The free list is guarded by
|
||||||
//! lock and larger alignments come when user programs gain threads.
|
//! a `Thread.Mutex` **only in multi-threaded binaries** (`addThreadedUserBinary`): the
|
||||||
|
//! guard is gated on `builtin.single_threaded`, so an ordinary single-threaded binary
|
||||||
|
//! compiles it out and pays nothing, while a threaded one can allocate safely from
|
||||||
|
//! several threads at once (docs/threading-plan.md M7). The lock lives at the two
|
||||||
|
//! free-list mutators — `rawAlloc`/`rawFree` — which every entry point funnels through.
|
||||||
|
|
||||||
const std = @import("std");
|
const std = @import("std");
|
||||||
|
const builtin = @import("builtin");
|
||||||
const abi = @import("abi");
|
const abi = @import("abi");
|
||||||
const system_calls = @import("system.zig");
|
const system_calls = @import("system.zig");
|
||||||
|
const Mutex = @import("thread.zig").Thread.Mutex;
|
||||||
|
|
||||||
const page_size = abi.page_size;
|
const page_size = abi.page_size;
|
||||||
|
|
||||||
|
/// Guards `free_list`. A no-op in single-threaded builds (compiled out); a real futex
|
||||||
|
/// mutex in threaded ones. Uncontended acquisition is a single CAS — no syscall.
|
||||||
|
var heap_mutex: Mutex = .{};
|
||||||
|
|
||||||
|
inline fn lockHeap() void {
|
||||||
|
if (comptime !builtin.single_threaded) heap_mutex.lock();
|
||||||
|
}
|
||||||
|
inline fn unlockHeap() void {
|
||||||
|
if (comptime !builtin.single_threaded) heap_mutex.unlock();
|
||||||
|
}
|
||||||
|
|
||||||
/// A block header, at the start of every block; while free it also links the
|
/// A block header, at the start of every block; while free it also links the
|
||||||
/// free list via `next`.
|
/// free list via `next`.
|
||||||
const Block = extern struct {
|
const Block = extern struct {
|
||||||
@@ -84,8 +101,11 @@ fn insertFree(block: *Block) void {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Allocate `len` bytes (16-byte aligned), or null if out of memory.
|
/// Allocate `len` bytes (16-byte aligned), or null if out of memory. Holds the heap lock
|
||||||
|
/// across the free-list search and any `grow` (which also touches the free list).
|
||||||
fn rawAlloc(len: usize) ?[*]u8 {
|
fn rawAlloc(len: usize) ?[*]u8 {
|
||||||
|
lockHeap();
|
||||||
|
defer unlockHeap();
|
||||||
const need = alignUp(header_size + len, 16);
|
const need = alignUp(header_size + len, 16);
|
||||||
|
|
||||||
var attempts: u32 = 0;
|
var attempts: u32 = 0;
|
||||||
@@ -119,6 +139,8 @@ fn rawAlloc(len: usize) ?[*]u8 {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fn rawFree(ptr: [*]u8) void {
|
fn rawFree(ptr: [*]u8) void {
|
||||||
|
lockHeap();
|
||||||
|
defer unlockHeap();
|
||||||
const block: *Block = @ptrFromInt(@intFromPtr(ptr) - header_size);
|
const block: *Block = @ptrFromInt(@intFromPtr(ptr) - header_size);
|
||||||
insertFree(block);
|
insertFree(block);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -70,6 +70,10 @@ pub const panic = start.panic;
|
|||||||
/// Process entry types: the `Init` handed to `main`, and its `Arguments`.
|
/// Process entry types: the `Init` handed to `main`, and its `Arguments`.
|
||||||
pub const process = @import("process.zig");
|
pub const process = @import("process.zig");
|
||||||
|
|
||||||
|
/// Threads: `runtime.Thread`, std.Thread-shaped, over the private thread ABI
|
||||||
|
/// (docs/threading.md). A binary must be built multi-threaded to spawn.
|
||||||
|
pub const Thread = @import("thread.zig").Thread;
|
||||||
|
|
||||||
/// The service harness: one replyWait loop folding requests, signals, and
|
/// The service harness: one replyWait loop folding requests, signals, and
|
||||||
/// notifications into callbacks (docs/process-lifecycle.md).
|
/// notifications into callbacks (docs/process-lifecycle.md).
|
||||||
pub const service = @import("service.zig");
|
pub const service = @import("service.zig");
|
||||||
|
|||||||
@@ -22,7 +22,7 @@ pub const Region = struct {
|
|||||||
};
|
};
|
||||||
|
|
||||||
/// Grant `len` bytes (rounded up to whole pages) of shareable, zeroed, cacheable RAM.
|
/// Grant `len` bytes (rounded up to whole pages) of shareable, zeroed, cacheable RAM.
|
||||||
/// Returns the region or null on failure. Two return values — vaddr in rax, handle in rdx —
|
/// Returns the region or null on failure. Two return values — virtual_address in rax, handle in rdx —
|
||||||
/// so this is a hand-written stub like `dma.alloc`.
|
/// so this is a hand-written stub like `dma.alloc`.
|
||||||
pub fn create(len: usize) ?Region {
|
pub fn create(len: usize) ?Region {
|
||||||
var rax: usize = undefined;
|
var rax: usize = undefined;
|
||||||
|
|||||||
@@ -0,0 +1,484 @@
|
|||||||
|
//! `runtime.Thread` — threads for danos, shaped like Zig's `std.Thread` but built on
|
||||||
|
//! danos's private thread ABI (docs/threading.md). Several tasks share one address
|
||||||
|
//! space; `spawn` starts one, the kernel delivers the closure pointer in the new
|
||||||
|
//! thread's rdi, a plain Zig trampoline runs the user function and calls `thread_exit`,
|
||||||
|
//! and `join` blocks on the thread's exit notification. See docs/threading.md for why
|
||||||
|
//! this mirrors `std.Thread`'s API rather than being the literal type.
|
||||||
|
//!
|
||||||
|
//! The closure (the function's captured args) lives at the **top of the thread's own
|
||||||
|
//! stack**, not the heap — each thread's stack is private, so there is no shared-heap
|
||||||
|
//! concurrency in the spawn/join machinery (the runtime heap is not yet thread-safe).
|
||||||
|
//! A binary must be built multi-threaded (`addThreadedUserBinary`) before it may spawn.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const builtin = @import("builtin");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const sc = @import("system-call.zig");
|
||||||
|
const system = @import("system.zig");
|
||||||
|
|
||||||
|
/// True in a real danos binary; false when this module is compiled for host unit tests.
|
||||||
|
/// The `Futex` seam and the test blocks below branch on it so the lock/condvar state
|
||||||
|
/// machines can be exercised on the host against `std.Thread.Futex` (docs/threading-plan.md
|
||||||
|
/// M11), while the danos build uses the futex syscalls.
|
||||||
|
const on_danos = builtin.os.tag == .freestanding;
|
||||||
|
|
||||||
|
/// A thread stack, if the caller does not override it. 64 KiB of mmap'd, zeroed pages.
|
||||||
|
pub const default_stack_size: usize = 64 * 1024;
|
||||||
|
|
||||||
|
/// Bytes reserved at the top of each thread's stack for its per-thread TLS block (the
|
||||||
|
/// self-pointer plus scratch slots reachable via `%fs`). docs/threading-plan.md M10.
|
||||||
|
const tls_block_size: usize = 64;
|
||||||
|
|
||||||
|
pub const Thread = struct {
|
||||||
|
/// The kernel task id of the spawned thread — what `join` waits on.
|
||||||
|
tid: u32,
|
||||||
|
/// The mmap'd stack, reclaimed by `join` (or at process exit after `detach`).
|
||||||
|
stack_base: usize,
|
||||||
|
stack_size: usize,
|
||||||
|
|
||||||
|
pub const Id = u32;
|
||||||
|
|
||||||
|
pub const SpawnConfig = struct {
|
||||||
|
/// Bytes of stack, rounded up to whole pages by the kernel's mmap.
|
||||||
|
stack_size: usize = default_stack_size,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const SpawnError = error{
|
||||||
|
/// The kernel refused the thread, the stack mmap failed, or no endpoint was free.
|
||||||
|
SystemResources,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Start `function(args...)` on a new thread sharing this address space. Mirrors
|
||||||
|
/// `std.Thread.spawn`. The thread's return value is discarded (as in `std.Thread`);
|
||||||
|
/// return data through shared state.
|
||||||
|
pub fn spawn(config: SpawnConfig, comptime function: anytype, args: anytype) SpawnError!Thread {
|
||||||
|
const Args = @TypeOf(args);
|
||||||
|
const Closure = struct {
|
||||||
|
tls_base: usize,
|
||||||
|
args: Args,
|
||||||
|
/// Entered directly by the kernel with `self` in rdi (C ABI). Establishes this
|
||||||
|
/// thread's TLS pointer, runs the user function, then ends the thread.
|
||||||
|
fn entry(self_addr: usize) callconv(.c) noreturn {
|
||||||
|
const self: *@This() = @ptrFromInt(self_addr);
|
||||||
|
setThreadPointer(self.tls_base); // per-thread thread pointer before any user code
|
||||||
|
@call(.auto, function, self.args);
|
||||||
|
exitThread();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
const base = system.mmap(config.stack_size, system.PROT_READ | system.PROT_WRITE);
|
||||||
|
if (system.mmapFailed(base)) return error.SystemResources;
|
||||||
|
|
||||||
|
// Top of the thread's own stack, downward: the closure, then a small per-thread TLS
|
||||||
|
// block (the thread pointer points here; slot 0 is the variant-II self-pointer, the rest is
|
||||||
|
// scratch for user TLS), then the stack proper (rsp starts below the TLS block, so
|
||||||
|
// the growing stack never overwrites either).
|
||||||
|
var closure_addr = (base + config.stack_size) - @sizeOf(Closure);
|
||||||
|
closure_addr &= ~@as(usize, @alignOf(Closure) - 1); // align the closure down
|
||||||
|
|
||||||
|
const tls_base = (closure_addr - tls_block_size) & ~@as(usize, 15);
|
||||||
|
const tls: [*]usize = @ptrFromInt(tls_base);
|
||||||
|
tls[0] = tls_base; // self-pointer (fs:0), as the x86_64 TLS ABI expects
|
||||||
|
|
||||||
|
const closure: *Closure = @ptrFromInt(closure_addr);
|
||||||
|
closure.* = .{ .tls_base = tls_base, .args = args };
|
||||||
|
|
||||||
|
var stack_top = tls_base & ~@as(usize, 15); // 16-align below the TLS block
|
||||||
|
stack_top -= 8; // ...then rsp % 16 == 8 at the C entry
|
||||||
|
|
||||||
|
const tid = threadSpawn(@intFromPtr(&Closure.entry), stack_top, closure_addr);
|
||||||
|
if (threadSpawnFailed(tid)) {
|
||||||
|
_ = system.munmap(base, config.stack_size);
|
||||||
|
return error.SystemResources;
|
||||||
|
}
|
||||||
|
return .{ .tid = @intCast(tid), .stack_base = base, .stack_size = config.stack_size };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Block until this thread finishes, then reclaim its stack. Mirrors
|
||||||
|
/// `std.Thread.join`. The exit endpoint is private to this thread, so the first
|
||||||
|
/// child-exit notification on it is this thread's.
|
||||||
|
pub fn join(self: Thread) void {
|
||||||
|
_ = sc.systemCall1(.thread_join, self.tid); // block until the thread has exited
|
||||||
|
_ = system.munmap(self.stack_base, self.stack_size); // reclaim its (now-vacated) stack
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Relinquish the right to join: never wait for or reclaim this thread. Its stack is
|
||||||
|
/// reclaimed at process exit (docs/threading-plan.md M3 — kernel-reaper stack reclaim
|
||||||
|
/// for detached threads is a later refinement). Mirrors `std.Thread.detach`.
|
||||||
|
pub fn detach(self: Thread) void {
|
||||||
|
_ = self;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The calling thread's id (its kernel task id). Mirrors `std.Thread.getCurrentId`.
|
||||||
|
pub fn getCurrentId() Id {
|
||||||
|
return @intCast(sc.systemCall0(.thread_self));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The dense 0-based index of the core the calling thread is running on. A danos
|
||||||
|
/// extension beyond `std.Thread`, used to observe genuine cross-core parallelism.
|
||||||
|
pub fn currentCore() Id {
|
||||||
|
return @intCast(sc.systemCall0(.current_core));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `std.Thread.Futex`-shaped block/wake on a `u32` atomic — the primitive the
|
||||||
|
/// blocking `Mutex`/`Condition`/`Semaphore` are built on. Waiters park in the
|
||||||
|
/// kernel (no busy-wait), so an idle core still halts (docs/halting.md).
|
||||||
|
pub const Futex = struct {
|
||||||
|
/// Block while `ptr.* == expect`. Returns when woken by `wake`, or promptly if
|
||||||
|
/// the value already differs (safe against spurious returns, as in std): the
|
||||||
|
/// caller re-checks its condition in a loop.
|
||||||
|
pub fn wait(ptr: *const std.atomic.Value(u32), expect: u32) void {
|
||||||
|
if (comptime on_danos) {
|
||||||
|
_ = futexWait(@intFromPtr(ptr), expect, 0);
|
||||||
|
} else {
|
||||||
|
// Host unit-test mock: spin+yield until the value changes (`wake` is a
|
||||||
|
// no-op — the callers re-check their condition in a loop anyway). Correct,
|
||||||
|
// if busy; fine for the state-machine tests.
|
||||||
|
while (ptr.load(.acquire) == expect) std.Thread.yield() catch {};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// As `wait`, but returns `error.Timeout` if `timeout_ns` elapses first.
|
||||||
|
pub fn timedWait(ptr: *const std.atomic.Value(u32), expect: u32, timeout_ns: u64) error{Timeout}!void {
|
||||||
|
if (comptime on_danos) {
|
||||||
|
if (futexWait(@intFromPtr(ptr), expect, timeout_ns) == abi.futex_timed_out) return error.Timeout;
|
||||||
|
} else {
|
||||||
|
var spins: u64 = 0;
|
||||||
|
const limit = timeout_ns / 1000 + 1;
|
||||||
|
while (ptr.load(.acquire) == expect) : (spins += 1) {
|
||||||
|
if (spins >= limit) return error.Timeout;
|
||||||
|
std.Thread.yield() catch {};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Wake up to `max_waiters` threads blocked on `ptr`.
|
||||||
|
pub fn wake(ptr: *const std.atomic.Value(u32), max_waiters: u32) void {
|
||||||
|
if (comptime on_danos) {
|
||||||
|
_ = futexWake(@intFromPtr(ptr), max_waiters);
|
||||||
|
} else {
|
||||||
|
// host mock: spin-waiters re-check their condition, so no wake is needed.
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A mutual-exclusion lock, `std.Thread.Mutex`-shaped. The classic three-state
|
||||||
|
/// futex mutex (unlocked / locked / contended): the fast path is a single CAS, and
|
||||||
|
/// only a contended lock ever enters the kernel.
|
||||||
|
pub const Mutex = struct {
|
||||||
|
state: std.atomic.Value(u32) = std.atomic.Value(u32).init(unlocked),
|
||||||
|
|
||||||
|
const unlocked: u32 = 0;
|
||||||
|
const locked: u32 = 1;
|
||||||
|
const contended: u32 = 2;
|
||||||
|
|
||||||
|
/// Try to take the lock without blocking; returns whether it was acquired.
|
||||||
|
pub fn tryLock(m: *Mutex) bool {
|
||||||
|
return m.state.cmpxchgStrong(unlocked, locked, .acquire, .monotonic) == null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Acquire the lock, blocking in the kernel while it is contended.
|
||||||
|
pub fn lock(m: *Mutex) void {
|
||||||
|
if (m.state.cmpxchgStrong(unlocked, locked, .acquire, .monotonic) != null) m.lockSlow();
|
||||||
|
}
|
||||||
|
|
||||||
|
fn lockSlow(m: *Mutex) void {
|
||||||
|
@branchHint(.cold);
|
||||||
|
// Mark the lock contended and take it as soon as it falls unlocked; park on
|
||||||
|
// the futex while it stays contended. Marking contended may cause a spurious
|
||||||
|
// wake on unlock (harmless), never a missed one.
|
||||||
|
while (m.state.swap(contended, .acquire) != unlocked) {
|
||||||
|
Futex.wait(&m.state, contended);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Release the lock; wake one waiter if the lock was contended.
|
||||||
|
pub fn unlock(m: *Mutex) void {
|
||||||
|
if (m.state.swap(unlocked, .release) == contended) Futex.wake(&m.state, 1);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A condition variable, `std.Thread.Condition`-shaped. Spurious wakeups are
|
||||||
|
/// allowed — always wait in a predicate loop with the mutex held. Built on a futex
|
||||||
|
/// sequence counter: a waiter samples the seq, drops the mutex, and parks until the
|
||||||
|
/// seq changes (a signal that races the unlock bumps the seq, so it is not missed).
|
||||||
|
pub const Condition = struct {
|
||||||
|
seq: std.atomic.Value(u32) = std.atomic.Value(u32).init(0),
|
||||||
|
|
||||||
|
/// Atomically release `mutex` and block until signalled, then re-acquire it.
|
||||||
|
pub fn wait(c: *Condition, mutex: *Mutex) void {
|
||||||
|
const seq = c.seq.load(.acquire);
|
||||||
|
mutex.unlock();
|
||||||
|
Futex.wait(&c.seq, seq);
|
||||||
|
mutex.lock();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// As `wait`, but returns `error.Timeout` if `timeout_ns` elapses first. The
|
||||||
|
/// mutex is re-acquired either way.
|
||||||
|
pub fn timedWait(c: *Condition, mutex: *Mutex, timeout_ns: u64) error{Timeout}!void {
|
||||||
|
const seq = c.seq.load(.acquire);
|
||||||
|
mutex.unlock();
|
||||||
|
const timed_out = if (Futex.timedWait(&c.seq, seq, timeout_ns)) |_| false else |_| true;
|
||||||
|
mutex.lock();
|
||||||
|
if (timed_out) return error.Timeout;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Wake one waiter.
|
||||||
|
pub fn signal(c: *Condition) void {
|
||||||
|
_ = c.seq.fetchAdd(1, .release);
|
||||||
|
Futex.wake(&c.seq, 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Wake all waiters.
|
||||||
|
pub fn broadcast(c: *Condition) void {
|
||||||
|
_ = c.seq.fetchAdd(1, .release);
|
||||||
|
Futex.wake(&c.seq, std.math.maxInt(u32));
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A counting semaphore, `std.Thread.Semaphore`-shaped: a permit count guarded by a
|
||||||
|
/// `Mutex` + `Condition`.
|
||||||
|
pub const Semaphore = struct {
|
||||||
|
mutex: Mutex = .{},
|
||||||
|
cond: Condition = .{},
|
||||||
|
permits: usize = 0,
|
||||||
|
|
||||||
|
/// Take a permit, blocking until one is available.
|
||||||
|
pub fn wait(s: *Semaphore) void {
|
||||||
|
s.mutex.lock();
|
||||||
|
defer s.mutex.unlock();
|
||||||
|
while (s.permits == 0) s.cond.wait(&s.mutex);
|
||||||
|
s.permits -= 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Return a permit and wake a waiter.
|
||||||
|
pub fn post(s: *Semaphore) void {
|
||||||
|
s.mutex.lock();
|
||||||
|
defer s.mutex.unlock();
|
||||||
|
s.permits += 1;
|
||||||
|
s.cond.signal();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A reader/writer lock, `std.Thread.RwLock`-shaped: many concurrent readers OR one
|
||||||
|
/// exclusive writer. Reader-preferring (a steady stream of readers can delay a writer),
|
||||||
|
/// built on `Mutex` + `Condition` over a signed state: `>0` = that many readers hold
|
||||||
|
/// it, `-1` = a writer holds it, `0` = free.
|
||||||
|
pub const RwLock = struct {
|
||||||
|
mutex: Mutex = .{},
|
||||||
|
cond: Condition = .{},
|
||||||
|
state: i64 = 0,
|
||||||
|
|
||||||
|
/// Acquire shared (read) access, blocking while a writer holds the lock.
|
||||||
|
pub fn lockShared(rw: *RwLock) void {
|
||||||
|
rw.mutex.lock();
|
||||||
|
defer rw.mutex.unlock();
|
||||||
|
while (rw.state < 0) rw.cond.wait(&rw.mutex);
|
||||||
|
rw.state += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Try to acquire shared access without blocking.
|
||||||
|
pub fn tryLockShared(rw: *RwLock) bool {
|
||||||
|
rw.mutex.lock();
|
||||||
|
defer rw.mutex.unlock();
|
||||||
|
if (rw.state < 0) return false;
|
||||||
|
rw.state += 1;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Release shared access; wake a waiting writer once the last reader leaves.
|
||||||
|
pub fn unlockShared(rw: *RwLock) void {
|
||||||
|
rw.mutex.lock();
|
||||||
|
defer rw.mutex.unlock();
|
||||||
|
rw.state -= 1;
|
||||||
|
if (rw.state == 0) rw.cond.broadcast();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Acquire exclusive (write) access, blocking until no readers or writer remain.
|
||||||
|
pub fn lock(rw: *RwLock) void {
|
||||||
|
rw.mutex.lock();
|
||||||
|
defer rw.mutex.unlock();
|
||||||
|
while (rw.state != 0) rw.cond.wait(&rw.mutex);
|
||||||
|
rw.state = -1;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Try to acquire exclusive access without blocking.
|
||||||
|
pub fn tryLock(rw: *RwLock) bool {
|
||||||
|
rw.mutex.lock();
|
||||||
|
defer rw.mutex.unlock();
|
||||||
|
if (rw.state != 0) return false;
|
||||||
|
rw.state = -1;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Release exclusive access; wake all waiters (they re-check their condition).
|
||||||
|
pub fn unlock(rw: *RwLock) void {
|
||||||
|
rw.mutex.lock();
|
||||||
|
defer rw.mutex.unlock();
|
||||||
|
rw.state = 0;
|
||||||
|
rw.cond.broadcast();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A `std.Thread.WaitGroup`-shaped counter: `start` before spawning work, `finish` as
|
||||||
|
/// each unit completes, `wait` blocks until the count returns to zero.
|
||||||
|
pub const WaitGroup = struct {
|
||||||
|
mutex: Mutex = .{},
|
||||||
|
cond: Condition = .{},
|
||||||
|
counter: usize = 0,
|
||||||
|
|
||||||
|
/// Register one pending unit of work.
|
||||||
|
pub fn start(wg: *WaitGroup) void {
|
||||||
|
wg.mutex.lock();
|
||||||
|
defer wg.mutex.unlock();
|
||||||
|
wg.counter += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Mark one unit done; wake waiters if that was the last.
|
||||||
|
pub fn finish(wg: *WaitGroup) void {
|
||||||
|
wg.mutex.lock();
|
||||||
|
defer wg.mutex.unlock();
|
||||||
|
wg.counter -= 1;
|
||||||
|
if (wg.counter == 0) wg.cond.broadcast();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Block until every started unit has finished.
|
||||||
|
pub fn wait(wg: *WaitGroup) void {
|
||||||
|
wg.mutex.lock();
|
||||||
|
defer wg.mutex.unlock();
|
||||||
|
while (wg.counter != 0) wg.cond.wait(&wg.mutex);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
/// thread_spawn(entry, stack_top, arg, exit_endpoint) -> tid, or a wrapped error.
|
||||||
|
fn threadSpawn(entry: usize, stack_top: usize, arg: usize) usize {
|
||||||
|
const exit_endpoint: usize = @intCast(abi.no_cap); // join uses thread_join, not an endpoint
|
||||||
|
return sc.systemCall4(.thread_spawn, entry, stack_top, arg, exit_endpoint);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The kernel returns a real (small) task id on success and a wrapped `-1` on failure;
|
||||||
|
/// no valid task id ever exceeds a u32.
|
||||||
|
inline fn threadSpawnFailed(ret: usize) bool {
|
||||||
|
return ret > std.math.maxInt(u32);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// End the calling thread. Never returns.
|
||||||
|
fn exitThread() noreturn {
|
||||||
|
_ = sc.systemCall0(.thread_exit);
|
||||||
|
unreachable;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Set the calling thread's FS base (its user TLS thread pointer).
|
||||||
|
fn setThreadPointer(addr: usize) void {
|
||||||
|
_ = sc.systemCall1(.set_thread_pointer, addr);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// futex_wait(addr, expect, timeout_ns) -> status (abi.futex_*).
|
||||||
|
fn futexWait(addr: usize, expect: u32, timeout_ns: u64) usize {
|
||||||
|
return sc.systemCall3(.futex_wait, addr, expect, timeout_ns);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// futex_wake(addr, count) -> number woken.
|
||||||
|
fn futexWake(addr: usize, count: u32) usize {
|
||||||
|
return sc.systemCall2(.futex_wake, addr, count);
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- host unit tests (docs/threading-plan.md M11) ---------------------------
|
||||||
|
//
|
||||||
|
// These run under `zig build test` on the host: the `Futex` seam above uses
|
||||||
|
// `std.Thread.Futex` off-danos, so the lock/condvar state machines can be exercised by
|
||||||
|
// real host threads. They are never compiled into a danos binary (test blocks only build
|
||||||
|
// under test), so their `std.Thread` use is fine even though `std.Thread` is unavailable
|
||||||
|
// on the freestanding target.
|
||||||
|
|
||||||
|
test "Mutex serialises concurrent increments across host threads" {
|
||||||
|
var m: Thread.Mutex = .{};
|
||||||
|
var counter: u64 = 0;
|
||||||
|
const workers = 8;
|
||||||
|
const per = 20_000;
|
||||||
|
const Ctx = struct {
|
||||||
|
m: *Thread.Mutex,
|
||||||
|
c: *u64,
|
||||||
|
fn run(ctx: @This()) void {
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < per) : (i += 1) {
|
||||||
|
ctx.m.lock();
|
||||||
|
ctx.c.* += 1;
|
||||||
|
ctx.m.unlock();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
var handles: [workers]std.Thread = undefined;
|
||||||
|
for (&handles) |*h| h.* = try std.Thread.spawn(.{}, Ctx.run, .{Ctx{ .m = &m, .c = &counter }});
|
||||||
|
for (handles) |h| h.join();
|
||||||
|
try std.testing.expectEqual(@as(u64, workers * per), counter);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "RwLock never lets a reader observe a half-written pair" {
|
||||||
|
var rw: Thread.RwLock = .{};
|
||||||
|
var a: u64 = 0;
|
||||||
|
var b: u64 = 0; // invariant while a lock is held: a == b
|
||||||
|
var stop = std.atomic.Value(bool).init(false);
|
||||||
|
var ok = std.atomic.Value(bool).init(true);
|
||||||
|
|
||||||
|
const Writer = struct {
|
||||||
|
rw: *Thread.RwLock,
|
||||||
|
a: *u64,
|
||||||
|
b: *u64,
|
||||||
|
stop: *std.atomic.Value(bool),
|
||||||
|
fn run(w: @This()) void {
|
||||||
|
var v: u64 = 1;
|
||||||
|
while (!w.stop.load(.acquire)) : (v +%= 1) {
|
||||||
|
w.rw.lock();
|
||||||
|
w.a.* = v; // update both halves under the exclusive lock...
|
||||||
|
w.b.* = v;
|
||||||
|
w.rw.unlock();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
const Reader = struct {
|
||||||
|
rw: *Thread.RwLock,
|
||||||
|
a: *u64,
|
||||||
|
b: *u64,
|
||||||
|
ok: *std.atomic.Value(bool),
|
||||||
|
fn run(r: @This()) void {
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < 200_000) : (i += 1) {
|
||||||
|
r.rw.lockShared();
|
||||||
|
if (r.a.* != r.b.*) r.ok.store(false, .release); // ...so a reader must never see them differ
|
||||||
|
r.rw.unlockShared();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
var writers: [2]std.Thread = undefined;
|
||||||
|
for (&writers) |*w| w.* = try std.Thread.spawn(.{}, Writer.run, .{Writer{ .rw = &rw, .a = &a, .b = &b, .stop = &stop }});
|
||||||
|
var readers: [4]std.Thread = undefined;
|
||||||
|
for (&readers) |*rd| rd.* = try std.Thread.spawn(.{}, Reader.run, .{Reader{ .rw = &rw, .a = &a, .b = &b, .ok = &ok }});
|
||||||
|
for (readers) |rd| rd.join();
|
||||||
|
stop.store(true, .release);
|
||||||
|
for (writers) |w| w.join();
|
||||||
|
try std.testing.expect(ok.load(.acquire));
|
||||||
|
}
|
||||||
|
|
||||||
|
test "WaitGroup blocks until every started unit finishes" {
|
||||||
|
var wg: Thread.WaitGroup = .{};
|
||||||
|
var done = std.atomic.Value(u32).init(0);
|
||||||
|
const n = 6;
|
||||||
|
const Ctx = struct {
|
||||||
|
wg: *Thread.WaitGroup,
|
||||||
|
done: *std.atomic.Value(u32),
|
||||||
|
fn run(c: @This()) void {
|
||||||
|
_ = c.done.fetchAdd(1, .monotonic);
|
||||||
|
c.wg.finish();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < n) : (i += 1) wg.start();
|
||||||
|
var handles: [n]std.Thread = undefined;
|
||||||
|
for (&handles) |*h| h.* = try std.Thread.spawn(.{}, Ctx.run, .{Ctx{ .wg = &wg, .done = &done }});
|
||||||
|
wg.wait(); // must not return until all n finished
|
||||||
|
try std.testing.expectEqual(@as(u32, n), done.load(.acquire));
|
||||||
|
for (handles) |h| h.join();
|
||||||
|
}
|
||||||
+19
-6
@@ -39,13 +39,13 @@ pub const SystemCall = enum(u64) {
|
|||||||
ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, receive, cap) -> receive_len (+badge in rdx)
|
ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, receive, cap) -> receive_len (+badge in rdx)
|
||||||
device_enumerate = 11, // device_enumerate(buffer, maximum) -> count: snapshot the device table
|
device_enumerate = 11, // device_enumerate(buffer, maximum) -> count: snapshot the device table
|
||||||
device_claim = 12, // device_claim(id) -> ok: take exclusive ownership of a device
|
device_claim = 12, // device_claim(id) -> ok: take exclusive ownership of a device
|
||||||
mmio_map = 13, // mmio_map(id, resource_index) -> vaddr: map a claimed device's MMIO into this AS
|
mmio_map = 13, // mmio_map(id, resource_index) -> virtual_address: map a claimed device's MMIO into this address space
|
||||||
irq_bind = 14, // irq_bind(id, resource_index, endpoint): deliver a device IRQ as an IPC notification
|
irq_bind = 14, // irq_bind(id, resource_index, endpoint): deliver a device IRQ as an IPC notification
|
||||||
irq_ack = 15, // irq_ack(id, resource_index): re-arm a bound IRQ after servicing it
|
irq_ack = 15, // irq_ack(id, resource_index): re-arm a bound IRQ after servicing it
|
||||||
device_register = 16, // device_register(parent_id, descriptor) -> id: publish a child of a device you claimed
|
device_register = 16, // device_register(parent_id, descriptor) -> id: publish a child of a device you claimed
|
||||||
system_spawn = 17, // system_spawn(name_ptr, name_len, arguments_ptr, arguments_len, exit_endpoint) -> child process id: start a named initial-ramdisk binary as a new ring-3 process
|
system_spawn = 17, // system_spawn(name_ptr, name_len, arguments_ptr, arguments_len, exit_endpoint) -> child process id: start a named initial-ramdisk binary as a new ring-3 process
|
||||||
dma_alloc = 18, // dma_alloc(len, flags) -> vaddr (rax), paddr (rdx): contiguous, pinned, uncacheable DMA memory
|
dma_alloc = 18, // dma_alloc(len, flags) -> virtual_address (rax), physical_address (rdx): contiguous, pinned, uncacheable DMA memory
|
||||||
dma_free = 19, // dma_free(vaddr, len) -> 0: release a prior dma_alloc
|
dma_free = 19, // dma_free(virtual_address, len) -> 0: release a prior dma_alloc
|
||||||
msi_bind = 20, // msi_bind(device_id, endpoint) -> address (rax), data (rdx): a per-device MSI vector for a claimed device
|
msi_bind = 20, // msi_bind(device_id, endpoint) -> address (rax), data (rdx): a per-device MSI vector for a claimed device
|
||||||
io_read = 21, // io_read(device_id, resource_index, offset, width) -> value: read a port in a claimed device's io_port resource
|
io_read = 21, // io_read(device_id, resource_index, offset, width) -> value: read a port in a claimed device's io_port resource
|
||||||
io_write = 22, // io_write(device_id, resource_index, offset, width, value) -> 0: write a port in a claimed device's io_port resource
|
io_write = 22, // io_write(device_id, resource_index, offset, width, value) -> 0: write a port in a claimed device's io_port resource
|
||||||
@@ -60,12 +60,25 @@ pub const SystemCall = enum(u64) {
|
|||||||
timer_bind = 31, // timer_bind(endpoint, ms) -> 0/-errno: one-shot timer — posts a notification when ms elapse
|
timer_bind = 31, // timer_bind(endpoint, ms) -> 0/-errno: one-shot timer — posts a notification when ms elapse
|
||||||
klog_read = 32, // klog_read(offset, ptr, len) -> bytes copied: copy the kernel RAM log buffer out to a user buffer (for persisting the boot log to disk)
|
klog_read = 32, // klog_read(offset, ptr, len) -> bytes copied: copy the kernel RAM log buffer out to a user buffer (for persisting the boot log to disk)
|
||||||
wall_clock = 33, // wall_clock() -> Unix epoch seconds (UTC): the RTC wall-clock time, for filesystem timestamps (mtime). Monotonic time is `clock`.
|
wall_clock = 33, // wall_clock() -> Unix epoch seconds (UTC): the RTC wall-clock time, for filesystem timestamps (mtime). Monotonic time is `clock`.
|
||||||
shm_create = 34, // shm_create(len) -> vaddr (rax), handle (rdx): a shareable, zeroed, cacheable RAM region mapped into this AS; the handle is a capability passed to another process as an ipc_call send_cap (docs/display-v2.md)
|
shm_create = 34, // shm_create(len) -> virtual_address (rax), handle (rdx): a shareable, zeroed, cacheable RAM region mapped into this AS; the handle is a capability passed to another process as an ipc_call send_cap (docs/display-v2.md)
|
||||||
shm_map = 35, // shm_map(cap) -> vaddr: map the shared region named by a received capability into this AS (the same physical pages the creator sees)
|
shm_map = 35, // shm_map(cap) -> virtual_address: map the shared region named by a received capability into this address space (the same physical pages the creator sees)
|
||||||
shm_physical = 36, // shm_physical(cap) -> paddr: the guest-physical base of a shared region held by capability, so a driver can program it into a device (e.g. virtio-gpu attach_backing); the pages are contiguous (docs/display-v2.md)
|
shm_physical = 36, // shm_physical(cap) -> physical_address: the guest-physical base of a shared region held by capability, so a driver can program it into a device (e.g. virtio-gpu attach_backing); the pages are contiguous (docs/display-v2.md)
|
||||||
|
thread_spawn = 37, // thread_spawn(entry, stack_top, arg, exit_endpoint) -> tid: start a task sharing the caller's address space at `entry` on `stack_top`, `arg` in rdi; exit_endpoint (a handle, or no_cap) is notified when it ends — how join waits (docs/threading.md)
|
||||||
|
thread_exit = 38, // thread_exit(): end the calling thread, dropping one reference to its address space (destroyed on the last)
|
||||||
|
current_core = 39, // current_core() -> index: the dense 0-based index of the core the caller is running on (for parallelism/affinity introspection)
|
||||||
|
futex_wait = 40, // futex_wait(addr, expected, timeout_ns) -> status: if *addr == expected, block until woken or the timeout; returns futex_woken/mismatch/timed_out (docs/threading.md)
|
||||||
|
futex_wake = 41, // futex_wake(addr, count) -> woken: wake up to `count` tasks blocked in futex_wait on `addr` in this address space
|
||||||
|
thread_self = 42, // thread_self() -> tid: the calling thread's kernel task id (runtime.Thread.getCurrentId)
|
||||||
|
thread_join = 43, // thread_join(tid) -> 0: block until the thread with id `tid` has exited (runtime.Thread.join; no per-thread IPC endpoint) (docs/threading.md)
|
||||||
|
set_thread_pointer = 44, // set_thread_pointer(addr) -> 0: set the caller's thread pointer (user-space TLS base; x86_64 IA32_FS_BASE, aarch64 TPIDR_EL0); restored per task across context switches (docs/threading-plan.md M10)
|
||||||
_,
|
_,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/// `futex_wait` return codes (in rax).
|
||||||
|
pub const futex_woken: u64 = 0; // woken by a futex_wake
|
||||||
|
pub const futex_mismatch: u64 = 1; // *addr != expected on entry; the caller did not block
|
||||||
|
pub const futex_timed_out: u64 = 2; // the timeout elapsed before a wake
|
||||||
|
|
||||||
/// How a process ended — recorded by the kernel at death, queried by the
|
/// How a process ended — recorded by the kernel at death, queried by the
|
||||||
/// supervisor with `process_exit_reason`, and the input to its restart decision
|
/// supervisor with `process_exit_reason`, and the input to its restart decision
|
||||||
/// (docs/process-lifecycle.md): a clean exit meant to stop, a fault wants a
|
/// (docs/process-lifecycle.md): a clean exit meant to stop, a fault wants a
|
||||||
|
|||||||
@@ -304,6 +304,17 @@ pub fn cpuLocal() usize {
|
|||||||
return pcpu.scheduler();
|
return pcpu.scheduler();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
const ia32_fs_base = 0xC000_0100;
|
||||||
|
|
||||||
|
/// Set the user-space TLS **thread pointer** — the arch-neutral name the generic scheduler
|
||||||
|
/// calls (`architecture.setThreadPointer`). On x86_64 that is the FS-segment base
|
||||||
|
/// (`IA32_FS_BASE`); an aarch64 port implements the same call against `TPIDR_EL0`. The
|
||||||
|
/// kernel never touches FS, so this only affects the user task that runs next, which the
|
||||||
|
/// scheduler restores per task across context switches (docs/threading-plan.md M10).
|
||||||
|
pub fn setThreadPointer(base: u64) void {
|
||||||
|
io.wrmsr(ia32_fs_base, base);
|
||||||
|
}
|
||||||
|
|
||||||
// --- SMP: application-processor bring-up ----------------------------------
|
// --- SMP: application-processor bring-up ----------------------------------
|
||||||
|
|
||||||
/// Record the low (<1 MiB) frame reserved for the AP trampoline. Run once at boot.
|
/// Record the low (<1 MiB) frame reserved for the AP trampoline. Run once at boot.
|
||||||
@@ -687,6 +698,15 @@ pub fn jumpToUser(entry: u64, stack_top: u64) noreturn {
|
|||||||
jump_to_user(entry, stack_top);
|
jump_to_user(entry, stack_top);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// As `jumpToUser`, but delivers `arg0` in the user's `rdi` — how a fresh thread
|
||||||
|
/// receives its closure pointer (docs/threading.md). A normal process is dropped
|
||||||
|
/// with `arg0 = 0`, which its `_start` ignores (it reads argv off the stack).
|
||||||
|
extern fn jump_to_user_arg(rip: u64, rsp: u64, arg0: u64) callconv(.c) noreturn;
|
||||||
|
|
||||||
|
pub fn jumpToUserArg(entry: u64, stack_top: u64, arg0: u64) noreturn {
|
||||||
|
jump_to_user_arg(entry, stack_top, arg0);
|
||||||
|
}
|
||||||
|
|
||||||
/// Route CPU exceptions to `handler`, which receives the trap frame and does not
|
/// Route CPU exceptions to `handler`, which receives the trap frame and does not
|
||||||
/// return. Until set, faults just halt the core.
|
/// return. Until set, faults just halt the core.
|
||||||
pub fn setFaultHandler(handler: *const fn (*const CpuState) noreturn) void {
|
pub fn setFaultHandler(handler: *const fn (*const CpuState) noreturn) void {
|
||||||
|
|||||||
@@ -123,6 +123,22 @@ jump_to_user:
|
|||||||
swapgs # user GS base (isr_common/syscall swap back on entry)
|
swapgs # user GS base (isr_common/syscall swap back on entry)
|
||||||
iretq
|
iretq
|
||||||
|
|
||||||
|
# jump_to_user_arg(rdi = user rip, rsi = user rsp, rdx = user rdi/arg0): as
|
||||||
|
# jump_to_user, but delivers arg0 in the user's rdi — how a fresh **thread**
|
||||||
|
# receives its closure pointer (docs/threading.md). rdi carries the rip only until
|
||||||
|
# it is pushed into the iretq frame, after which we overwrite it with the arg.
|
||||||
|
.global jump_to_user_arg
|
||||||
|
jump_to_user_arg:
|
||||||
|
cli
|
||||||
|
push $0x1B # user SS (0x18 | RPL 3)
|
||||||
|
push %rsi # user RSP
|
||||||
|
push $0x202 # RFLAGS: IF | reserved-1
|
||||||
|
push $0x23 # user CS (0x20 | RPL 3)
|
||||||
|
push %rdi # user RIP (consumes rdi)
|
||||||
|
mov %rdx, %rdi # user rdi = arg0 (the thread's closure pointer)
|
||||||
|
swapgs # user GS base (isr_common/syscall swap back on entry)
|
||||||
|
iretq
|
||||||
|
|
||||||
# --- ring 3 entry/exit ------------------------------------------------------
|
# --- ring 3 entry/exit ------------------------------------------------------
|
||||||
|
|
||||||
# enter_user(rdi = user rip, rsi = user rsp, rdx = &TSS.rsp0)
|
# enter_user(rdi = user rip, rsi = user rsp, rdx = &TSS.rsp0)
|
||||||
|
|||||||
@@ -509,7 +509,7 @@ pub fn unmapInto(pml4: u64, virtual: u64) void {
|
|||||||
/// any address space, not just the live one). Returns null if `virtual` is not
|
/// any address space, not just the live one). Returns null if `virtual` is not
|
||||||
/// mapped at any level. Stops at a 2 MiB huge-page leaf (the physmap uses them),
|
/// mapped at any level. Stops at a 2 MiB huge-page leaf (the physmap uses them),
|
||||||
/// resolving the offset within it. The foundation for cross-address-space copies
|
/// resolving the offset within it. The foundation for cross-address-space copies
|
||||||
/// and for munmap (which needs the frame behind a user vaddr to free it).
|
/// and for munmap (which needs the frame behind a user virtual_address to free it).
|
||||||
pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
||||||
const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
if (pml4e & present == 0) return null;
|
if (pml4e & present == 0) return null;
|
||||||
|
|||||||
@@ -355,7 +355,7 @@ pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_pt
|
|||||||
me.ipc_client = null;
|
me.ipc_client = null;
|
||||||
const n = @min(reply_len, client.ipc_reply_cap);
|
const n = @min(reply_len, client.ipc_reply_cap);
|
||||||
client.ipc_received_cap = abi.no_cap;
|
client.ipc_received_cap = abi.no_cap;
|
||||||
if (!copyAcross(me.aspace, reply_ptr, client.aspace, client.ipc_reply_ptr, n)) {
|
if (!copyAcross(me.address_space, reply_ptr, client.address_space, client.ipc_reply_ptr, n)) {
|
||||||
client.ipc_status = -EFAULT;
|
client.ipc_status = -EFAULT;
|
||||||
} else if (send_cap != abi.no_cap) {
|
} else if (send_cap != abi.no_cap) {
|
||||||
// Transfer the reply's capability into the client. A failure fails the
|
// Transfer the reply's capability into the client. A failure fails the
|
||||||
@@ -383,8 +383,8 @@ pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_pt
|
|||||||
}
|
}
|
||||||
if (popPost(endpoint)) |slot| {
|
if (popPost(endpoint)) |slot| {
|
||||||
const n = @min(@as(usize, slot.length), receive_cap);
|
const n = @min(@as(usize, slot.length), receive_cap);
|
||||||
// Copy from the kernel-resident ring slot (source aspace 0) into the receiver.
|
// Copy from the kernel-resident ring slot (source address_space 0) into the receiver.
|
||||||
if (!copyAcross(0, @intFromPtr(&slot.bytes), me.aspace, receive_ptr, n)) {
|
if (!copyAcross(0, @intFromPtr(&slot.bytes), me.address_space, receive_ptr, n)) {
|
||||||
continue; // bad receive buffer: drop this message, keep serving
|
continue; // bad receive buffer: drop this message, keep serving
|
||||||
}
|
}
|
||||||
out_badge.* = slot.sender_id | notify_badge_bit | notify_message_bit;
|
out_badge.* = slot.sender_id | notify_badge_bit | notify_message_bit;
|
||||||
@@ -392,7 +392,7 @@ pub fn replyWait(endpoint: *Endpoint, reply_ptr: u64, reply_len: u64, receive_pt
|
|||||||
}
|
}
|
||||||
if (dequeueSender(endpoint)) |caller| {
|
if (dequeueSender(endpoint)) |caller| {
|
||||||
const n = @min(caller.ipc_send_len, receive_cap);
|
const n = @min(caller.ipc_send_len, receive_cap);
|
||||||
if (!copyAcross(caller.aspace, caller.ipc_send_ptr, me.aspace, receive_ptr, n)) {
|
if (!copyAcross(caller.address_space, caller.ipc_send_ptr, me.address_space, receive_ptr, n)) {
|
||||||
caller.ipc_status = -EFAULT; // bad sender buffer: fail it, keep serving
|
caller.ipc_status = -EFAULT; // bad sender buffer: fail it, keep serving
|
||||||
scheduler.readyLocked(caller);
|
scheduler.readyLocked(caller);
|
||||||
continue;
|
continue;
|
||||||
|
|||||||
+249
-77
@@ -58,8 +58,9 @@ pub const stack_top_virtual: u64 = stack_base_virtual + parameters.user_stack_pa
|
|||||||
|
|
||||||
/// The mmap grant arena: where `mmap` hands out fresh user pages, above the image
|
/// The mmap grant arena: where `mmap` hands out fresh user pages, above the image
|
||||||
/// and stack but still inside PML4[224] (so no kernel mapping is widened). Each
|
/// and stack but still inside PML4[224] (so no kernel mapping is widened). Each
|
||||||
/// process bump-allocates from `heap_arena_base` upward via `Task.heap_next`; a
|
/// process bump-allocates from `heap_arena_base` upward via a per-address-space cursor
|
||||||
/// 1 GiB window is far more than any user heap needs today.
|
/// (`scheduler.addressSpaceMmapNextPtr`, shared by its threads); a 1 GiB window is far more
|
||||||
|
/// than any user heap needs today.
|
||||||
pub const heap_arena_base: u64 = 0x0000_7000_1000_0000;
|
pub const heap_arena_base: u64 = 0x0000_7000_1000_0000;
|
||||||
pub const heap_arena_end: u64 = heap_arena_base + (1 << 30);
|
pub const heap_arena_end: u64 = heap_arena_base + (1 << 30);
|
||||||
|
|
||||||
@@ -69,8 +70,8 @@ pub const user_half_end: u64 = 0x0000_8000_0000_0000;
|
|||||||
|
|
||||||
/// The MMIO-grant arena: where `mmio_map` places device windows, in PML4[226] —
|
/// The MMIO-grant arena: where `mmio_map` places device windows, in PML4[226] —
|
||||||
/// a user-exclusive region distinct from code/stack/heap (PML4[224]), so mapping
|
/// a user-exclusive region distinct from code/stack/heap (PML4[224]), so mapping
|
||||||
/// device pages user-accessible widens no kernel mapping. Per-process cursor in
|
/// device pages user-accessible widens no kernel mapping. Per-address-space cursor
|
||||||
/// `Task.device_map_next`.
|
/// (`scheduler.addressSpaceDeviceMapNextPtr`).
|
||||||
pub const device_arena_base: u64 = 0x0000_7100_0000_0000;
|
pub const device_arena_base: u64 = 0x0000_7100_0000_0000;
|
||||||
pub const device_arena_end: u64 = device_arena_base + (4 << 30);
|
pub const device_arena_end: u64 = device_arena_base + (4 << 30);
|
||||||
|
|
||||||
@@ -163,7 +164,7 @@ fn fail(state: *architecture.CpuState) void {
|
|||||||
|
|
||||||
fn system_call(state: *architecture.CpuState) void {
|
fn system_call(state: *architecture.CpuState) void {
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
const user = t.aspace != 0;
|
const user = t.address_space != 0;
|
||||||
if (user) {
|
if (user) {
|
||||||
// A condemned process (process_kill caught it running) dies at its next
|
// A condemned process (process_kill caught it running) dies at its next
|
||||||
// kernel entry — before it can spawn, claim, or message anything else.
|
// kernel entry — before it can spawn, claim, or message anything else.
|
||||||
@@ -227,6 +228,22 @@ fn system_call(state: *architecture.CpuState) void {
|
|||||||
.shm_create => systemShmCreate(state),
|
.shm_create => systemShmCreate(state),
|
||||||
.shm_map => systemShmMap(state),
|
.shm_map => systemShmMap(state),
|
||||||
.shm_physical => systemShmPhysical(state),
|
.shm_physical => systemShmPhysical(state),
|
||||||
|
.thread_spawn => systemThreadSpawn(state),
|
||||||
|
.current_core => systemCurrentCore(state),
|
||||||
|
.thread_self => systemThreadSelf(state),
|
||||||
|
.thread_join => systemThreadJoin(state),
|
||||||
|
.set_thread_pointer => systemSetThreadPointer(state),
|
||||||
|
.futex_wait => systemFutexWait(state),
|
||||||
|
.futex_wake => systemFutexWake(state),
|
||||||
|
.thread_exit => {
|
||||||
|
// A thread ends like a process exit(0), but only this task: its
|
||||||
|
// resources are released and its address-space reference dropped (the
|
||||||
|
// space survives while sibling threads hold it). docs/threading.md.
|
||||||
|
if (scheduler.currentIsUserProcess()) {
|
||||||
|
scheduler.current().exit_reason = .exited;
|
||||||
|
terminateCurrent();
|
||||||
|
} else architecture.userExit();
|
||||||
|
},
|
||||||
_ => fail(state),
|
_ => fail(state),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -239,6 +256,13 @@ fn failErr(state: *architecture.CpuState, errno: i64) void {
|
|||||||
/// create_ipc_endpoint() -> handle: allocate an endpoint and install it in the
|
/// create_ipc_endpoint() -> handle: allocate an endpoint and install it in the
|
||||||
/// caller's handle table.
|
/// caller's handle table.
|
||||||
fn systemCreateIpcEndpoint(state: *architecture.CpuState) void {
|
fn systemCreateIpcEndpoint(state: *architecture.CpuState) void {
|
||||||
|
// Under the big kernel lock: this allocates from the kernel heap and mutates the
|
||||||
|
// caller's handle table. A multi-threaded process (e.g. the display's compositor +
|
||||||
|
// mouse-listener threads) can drive this concurrently from two cores, so the endpoint
|
||||||
|
// allocation and every other lock holder must serialize (heap.zig: "a lock comes with
|
||||||
|
// threads/SMP").
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
const endpoint = ipc.createIpcEndpoint() orelse return failErr(state, ipc.ENOMEM);
|
const endpoint = ipc.createIpcEndpoint() orelse return failErr(state, ipc.ENOMEM);
|
||||||
const h = ipc.installHandle(scheduler.current(), endpoint);
|
const h = ipc.installHandle(scheduler.current(), endpoint);
|
||||||
if (h < 0) {
|
if (h < 0) {
|
||||||
@@ -251,6 +275,10 @@ fn systemCreateIpcEndpoint(state: *architecture.CpuState) void {
|
|||||||
/// ipc_register(service_id, handle): publish the caller's endpoint under a
|
/// ipc_register(service_id, handle): publish the caller's endpoint under a
|
||||||
/// well-known id so other processes can find it.
|
/// well-known id so other processes can find it.
|
||||||
fn systemIpcRegister(state: *architecture.CpuState) void {
|
fn systemIpcRegister(state: *architecture.CpuState) void {
|
||||||
|
// Under the big kernel lock: mutates the global service registry and endpoint
|
||||||
|
// refcounts, which threads of the same (or another) process can race.
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
const id: u32 = @truncate(architecture.systemCallArg(state, 0));
|
const id: u32 = @truncate(architecture.systemCallArg(state, 0));
|
||||||
const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 1)) orelse return failErr(state, ipc.EBADF);
|
const endpoint = ipc.resolveHandle(scheduler.current(), architecture.systemCallArg(state, 1)) orelse return failErr(state, ipc.EBADF);
|
||||||
architecture.setSystemCallResult(state, @bitCast(ipc.register(id, endpoint)));
|
architecture.setSystemCallResult(state, @bitCast(ipc.register(id, endpoint)));
|
||||||
@@ -259,6 +287,11 @@ fn systemIpcRegister(state: *architecture.CpuState) void {
|
|||||||
/// ipc_lookup(service_id) -> handle: find a published endpoint and install a
|
/// ipc_lookup(service_id) -> handle: find a published endpoint and install a
|
||||||
/// handle to it in the caller.
|
/// handle to it in the caller.
|
||||||
fn systemIpcLookup(state: *architecture.CpuState) void {
|
fn systemIpcLookup(state: *architecture.CpuState) void {
|
||||||
|
// Under the big kernel lock: reads the global registry, takes an endpoint reference,
|
||||||
|
// and installs a handle — all racy against concurrent threads (this is the path the
|
||||||
|
// display's mouse-listener thread takes to reach the compositor endpoint).
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
const id: u32 = @truncate(architecture.systemCallArg(state, 0));
|
const id: u32 = @truncate(architecture.systemCallArg(state, 0));
|
||||||
const endpoint = ipc.lookup(id) orelse return failErr(state, ipc.ENOENT);
|
const endpoint = ipc.lookup(id) orelse return failErr(state, ipc.ENOENT);
|
||||||
const h = ipc.installHandle(scheduler.current(), endpoint);
|
const h = ipc.installHandle(scheduler.current(), endpoint);
|
||||||
@@ -300,7 +333,7 @@ fn systemIpcReplyWait(state: *architecture.CpuState) void {
|
|||||||
fn systemIpcSend(state: *architecture.CpuState) void {
|
fn systemIpcSend(state: *architecture.CpuState) void {
|
||||||
const me = scheduler.current();
|
const me = scheduler.current();
|
||||||
const endpoint = ipc.resolveHandle(me, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
const endpoint = ipc.resolveHandle(me, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||||
const r = ipc.send(endpoint, me.aspace, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), me.id);
|
const r = ipc.send(endpoint, me.address_space, architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), me.id);
|
||||||
architecture.setSystemCallResult(state, @bitCast(r));
|
architecture.setSystemCallResult(state, @bitCast(r));
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -310,7 +343,7 @@ fn systemDeviceEnumerate(state: *architecture.CpuState) void {
|
|||||||
const buffer_ptr = architecture.systemCallArg(state, 0);
|
const buffer_ptr = architecture.systemCallArg(state, 0);
|
||||||
const maximum = architecture.systemCallArg(state, 1);
|
const maximum = architecture.systemCallArg(state, 1);
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0 or buffer_ptr >= user_half_end) return fail(state);
|
if (t.address_space == 0 or buffer_ptr >= user_half_end) return fail(state);
|
||||||
const sz = @sizeOf(device_abi.DeviceDescriptor);
|
const sz = @sizeOf(device_abi.DeviceDescriptor);
|
||||||
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
|
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
|
||||||
const out: [*]device_abi.DeviceDescriptor = @ptrFromInt(buffer_ptr);
|
const out: [*]device_abi.DeviceDescriptor = @ptrFromInt(buffer_ptr);
|
||||||
@@ -334,14 +367,14 @@ fn systemDeviceClaim(state: *architecture.CpuState) void {
|
|||||||
} else fail(state);
|
} else fail(state);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// mmio_map(device_id, resource_index) -> vaddr: map a claimed device's MMIO window into
|
/// mmio_map(device_id, resource_index) -> virtual_address: map a claimed device's MMIO window into
|
||||||
/// this address space (strong-uncacheable) and return the register base address.
|
/// this address space (strong-uncacheable) and return the register base address.
|
||||||
/// The claim is the capability — a process can only map hardware it owns.
|
/// The claim is the capability — a process can only map hardware it owns.
|
||||||
fn systemMmioMap(state: *architecture.CpuState) void {
|
fn systemMmioMap(state: *architecture.CpuState) void {
|
||||||
const device_id = architecture.systemCallArg(state, 0);
|
const device_id = architecture.systemCallArg(state, 0);
|
||||||
const resource_index = architecture.systemCallArg(state, 1);
|
const resource_index = architecture.systemCallArg(state, 1);
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0) return fail(state);
|
if (t.address_space == 0) return fail(state);
|
||||||
// Read the broker table under the lock: ring-3 device_register (M19) now
|
// Read the broker table under the lock: ring-3 device_register (M19) now
|
||||||
// mutates it concurrently on other cores, so a lock-free read here could
|
// mutates it concurrently on other cores, so a lock-free read here could
|
||||||
// see a torn resource (and a torn length used to panic the arithmetic
|
// see a torn resource (and a torn length used to panic the arithmetic
|
||||||
@@ -359,18 +392,23 @@ fn systemMmioMap(state: *architecture.CpuState) void {
|
|||||||
if (r.len == 0) return fail(state);
|
if (r.len == 0) return fail(state);
|
||||||
if (@addWithOverflow(r.start, r.len)[1] != 0) return fail(state);
|
if (@addWithOverflow(r.start, r.len)[1] != 0) return fail(state);
|
||||||
|
|
||||||
if (t.device_map_next == 0) t.device_map_next = device_arena_base;
|
|
||||||
const first = r.start & ~@as(u64, page_size - 1);
|
const first = r.start & ~@as(u64, page_size - 1);
|
||||||
const last = (r.start + r.len - 1) & ~@as(u64, page_size - 1);
|
const last = (r.start + r.len - 1) & ~@as(u64, page_size - 1);
|
||||||
const pages = (last - first) / page_size + 1;
|
const pages = (last - first) / page_size + 1;
|
||||||
const base_v = t.device_map_next;
|
|
||||||
if (base_v + pages * page_size > device_arena_end) return fail(state);
|
|
||||||
|
|
||||||
// A framebuffer resource asks (via its flag) to be mapped write-combining rather
|
// A framebuffer resource asks (via its flag) to be mapped write-combining rather
|
||||||
// than the strong-uncacheable default that register MMIO needs.
|
// than the strong-uncacheable default that register MMIO needs.
|
||||||
const write_combining = (r.flags & device_abi.resource_flag_write_combining) != 0;
|
const write_combining = (r.flags & device_abi.resource_flag_write_combining) != 0;
|
||||||
architecture.mapUserDeviceInto(t.aspace, base_v, r.start, r.len, write_combining);
|
|
||||||
t.device_map_next = base_v + pages * page_size;
|
// Per-address-space cursor + shared page tables → serialize under the big lock,
|
||||||
|
// same as mmap (docs/threading-plan.md M7).
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
|
const cursor = scheduler.addressSpaceDeviceMapNextPtr(t.address_space) orelse return fail(state);
|
||||||
|
if (cursor.* == 0) cursor.* = device_arena_base; // seed the arena lazily
|
||||||
|
const base_v = cursor.*;
|
||||||
|
if (base_v + pages * page_size > device_arena_end) return fail(state);
|
||||||
|
architecture.mapUserDeviceInto(t.address_space, base_v, r.start, r.len, write_combining);
|
||||||
|
cursor.* = base_v + pages * page_size;
|
||||||
architecture.setSystemCallResult(state, base_v + (r.start & (page_size - 1))); // register base
|
architecture.setSystemCallResult(state, base_v + (r.start & (page_size - 1))); // register base
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -399,7 +437,7 @@ pub fn resolveIoPort(t: *scheduler.Task, device_id: u64, resource_index: u64, of
|
|||||||
/// is fine. See docs/drivers.md.
|
/// is fine. See docs/drivers.md.
|
||||||
fn systemIoRead(state: *architecture.CpuState) void {
|
fn systemIoRead(state: *architecture.CpuState) void {
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0) return fail(state);
|
if (t.address_space == 0) return fail(state);
|
||||||
const width = architecture.systemCallArg(state, 3);
|
const width = architecture.systemCallArg(state, 3);
|
||||||
const port = resolveIoPort(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), width) orelse return fail(state);
|
const port = resolveIoPort(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), width) orelse return fail(state);
|
||||||
architecture.setSystemCallResult(state, architecture.pioRead(@intCast(width), port));
|
architecture.setSystemCallResult(state, architecture.pioRead(@intCast(width), port));
|
||||||
@@ -410,14 +448,14 @@ fn systemIoRead(state: *architecture.CpuState) void {
|
|||||||
/// gate as `io_read`.
|
/// gate as `io_read`.
|
||||||
fn systemIoWrite(state: *architecture.CpuState) void {
|
fn systemIoWrite(state: *architecture.CpuState) void {
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0) return fail(state);
|
if (t.address_space == 0) return fail(state);
|
||||||
const width = architecture.systemCallArg(state, 3);
|
const width = architecture.systemCallArg(state, 3);
|
||||||
const port = resolveIoPort(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), width) orelse return fail(state);
|
const port = resolveIoPort(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1), architecture.systemCallArg(state, 2), width) orelse return fail(state);
|
||||||
architecture.pioWrite(@intCast(width), port, @intCast(architecture.systemCallArg(state, 4)));
|
architecture.pioWrite(@intCast(width), port, @intCast(architecture.systemCallArg(state, 4)));
|
||||||
architecture.setSystemCallResult(state, 0);
|
architecture.setSystemCallResult(state, 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// dma_alloc(len, flags) -> vaddr (rax), paddr (rdx): grant `len` bytes (rounded up to
|
/// dma_alloc(len, flags) -> virtual_address (rax), physical_address (rdx): grant `len` bytes (rounded up to
|
||||||
/// whole pages) of DMA-capable memory — physically contiguous, zeroed, pinned, and
|
/// whole pages) of DMA-capable memory — physically contiguous, zeroed, pinned, and
|
||||||
/// strong-uncacheable (coherent) — mapping it into the caller's DMA arena and handing
|
/// strong-uncacheable (coherent) — mapping it into the caller's DMA arena and handing
|
||||||
/// back both the virtual address to touch and the physical address to program into the
|
/// back both the virtual address to touch and the physical address to program into the
|
||||||
@@ -429,7 +467,7 @@ fn systemDmaAlloc(state: *architecture.CpuState) void {
|
|||||||
const len = architecture.systemCallArg(state, 0);
|
const len = architecture.systemCallArg(state, 0);
|
||||||
const flags = architecture.systemCallArg(state, 1);
|
const flags = architecture.systemCallArg(state, 1);
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0 or len == 0) return fail(state);
|
if (t.address_space == 0 or len == 0) return fail(state);
|
||||||
|
|
||||||
const pages: usize = @intCast((len + page_size - 1) / page_size);
|
const pages: usize = @intCast((len + page_size - 1) / page_size);
|
||||||
const max_phys: u64 = if (flags & abi.dma_below_4g != 0) (@as(u64, 4) << 30) else ~@as(u64, 0);
|
const max_phys: u64 = if (flags & abi.dma_below_4g != 0) (@as(u64, 4) << 30) else ~@as(u64, 0);
|
||||||
@@ -445,14 +483,14 @@ fn systemDmaAlloc(state: *architecture.CpuState) void {
|
|||||||
// Zero through the physmap (the frames aren't mapped in the caller yet), then map.
|
// Zero through the physmap (the frames aren't mapped in the caller yet), then map.
|
||||||
const kernel_view: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(phys));
|
const kernel_view: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(phys));
|
||||||
@memset(kernel_view[0 .. pages * page_size], 0);
|
@memset(kernel_view[0 .. pages * page_size], 0);
|
||||||
architecture.mapUserDmaInto(t.aspace, base_v, phys, pages * page_size);
|
architecture.mapUserDmaInto(t.address_space, base_v, phys, pages * page_size);
|
||||||
|
|
||||||
t.dma_map_next = base_v + pages * page_size;
|
t.dma_map_next = base_v + pages * page_size;
|
||||||
architecture.setSystemCallResult(state, base_v); // virtual address for the CPU
|
architecture.setSystemCallResult(state, base_v); // virtual address for the CPU
|
||||||
architecture.setSystemCallResult2(state, phys); // physical address for the device
|
architecture.setSystemCallResult2(state, phys); // physical address for the device
|
||||||
}
|
}
|
||||||
|
|
||||||
/// dma_free(vaddr, len) -> 0: release a prior `dma_alloc`. Bounded to the DMA arena so
|
/// dma_free(virtual_address, len) -> 0: release a prior `dma_alloc`. Bounded to the DMA arena so
|
||||||
/// it can never unmap-and-free the caller's stack, heap, or an MMIO grant; only pages
|
/// it can never unmap-and-free the caller's stack, heap, or an MMIO grant; only pages
|
||||||
/// actually mapped are freed (an unmapped hole is skipped). Teardown also reclaims any
|
/// actually mapped are freed (an unmapped hole is skipped). Teardown also reclaims any
|
||||||
/// DMA pages left mapped at exit (they carry no `device_grant`, so `freeSubtree` frees
|
/// DMA pages left mapped at exit (they carry no `device_grant`, so `freeSubtree` frees
|
||||||
@@ -461,21 +499,21 @@ fn systemDmaFree(state: *architecture.CpuState) void {
|
|||||||
const base_v = architecture.systemCallArg(state, 0);
|
const base_v = architecture.systemCallArg(state, 0);
|
||||||
const len = architecture.systemCallArg(state, 1);
|
const len = architecture.systemCallArg(state, 1);
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0) return fail(state);
|
if (t.address_space == 0) return fail(state);
|
||||||
const pages: usize = @intCast((len + page_size - 1) / page_size);
|
const pages: usize = @intCast((len + page_size - 1) / page_size);
|
||||||
if (base_v < dma_arena_base or base_v + pages * page_size > dma_arena_end) return fail(state);
|
if (base_v < dma_arena_base or base_v + pages * page_size > dma_arena_end) return fail(state);
|
||||||
|
|
||||||
for (0..pages) |i| {
|
for (0..pages) |i| {
|
||||||
const va = base_v + i * page_size;
|
const va = base_v + i * page_size;
|
||||||
if (architecture.translate(t.aspace, va)) |phys| {
|
if (architecture.translate(t.address_space, va)) |phys| {
|
||||||
architecture.unmapUserPageInto(t.aspace, va);
|
architecture.unmapUserPageInto(t.address_space, va);
|
||||||
pmm.free(phys);
|
pmm.free(phys);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
architecture.setSystemCallResult(state, 0);
|
architecture.setSystemCallResult(state, 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// shm_create(len) -> vaddr (rax), handle (rdx): grant `len` bytes (rounded up to whole
|
/// shm_create(len) -> virtual_address (rax), handle (rdx): grant `len` bytes (rounded up to whole
|
||||||
/// pages) of **shareable, zeroed, cacheable** RAM — contiguous frames mapped into the
|
/// pages) of **shareable, zeroed, cacheable** RAM — contiguous frames mapped into the
|
||||||
/// caller's shm arena — and hand back the virtual address plus a capability handle. Unlike
|
/// caller's shm arena — and hand back the virtual address plus a capability handle. Unlike
|
||||||
/// `dma_alloc` the memory is write-back cacheable (for CPU compositing, not device DMA) and
|
/// `dma_alloc` the memory is write-back cacheable (for CPU compositing, not device DMA) and
|
||||||
@@ -486,7 +524,7 @@ fn systemDmaFree(state: *architecture.CpuState) void {
|
|||||||
fn systemShmCreate(state: *architecture.CpuState) void {
|
fn systemShmCreate(state: *architecture.CpuState) void {
|
||||||
const len = architecture.systemCallArg(state, 0);
|
const len = architecture.systemCallArg(state, 0);
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0 or len == 0) return fail(state);
|
if (t.address_space == 0 or len == 0) return fail(state);
|
||||||
|
|
||||||
const pages: usize = @intCast((len + page_size - 1) / page_size);
|
const pages: usize = @intCast((len + page_size - 1) / page_size);
|
||||||
if (pages == 0 or pages > maximum_shm_pages) return fail(state);
|
if (pages == 0 or pages > maximum_shm_pages) return fail(state);
|
||||||
@@ -511,20 +549,20 @@ fn systemShmCreate(state: *architecture.CpuState) void {
|
|||||||
return fail(state);
|
return fail(state);
|
||||||
}
|
}
|
||||||
|
|
||||||
architecture.mapUserSharedInto(t.aspace, base_v, phys, pages * page_size);
|
architecture.mapUserSharedInto(t.address_space, base_v, phys, pages * page_size);
|
||||||
t.shm_map_next = base_v + pages * page_size;
|
t.shm_map_next = base_v + pages * page_size;
|
||||||
architecture.setSystemCallResult(state, base_v); // vaddr for the CPU
|
architecture.setSystemCallResult(state, base_v); // virtual_address for the CPU
|
||||||
architecture.setSystemCallResult2(state, @intCast(handle)); // capability handle to pass on
|
architecture.setSystemCallResult2(state, @intCast(handle)); // capability handle to pass on
|
||||||
}
|
}
|
||||||
|
|
||||||
/// shm_map(cap) -> vaddr: map the shared region named by a capability handle the caller
|
/// shm_map(cap) -> virtual_address: map the shared region named by a capability handle the caller
|
||||||
/// received (via an `ipc_call` send_cap) into its shm arena — the same physical frames the
|
/// received (via an `ipc_call` send_cap) into its shm arena — the same physical frames the
|
||||||
/// creator sees — returning the virtual address. The handle already holds a reference (taken
|
/// creator sees — returning the virtual address. The handle already holds a reference (taken
|
||||||
/// when the capability was shared), so this only adds a mapping; it never bumps the refcount.
|
/// when the capability was shared), so this only adds a mapping; it never bumps the refcount.
|
||||||
fn systemShmMap(state: *architecture.CpuState) void {
|
fn systemShmMap(state: *architecture.CpuState) void {
|
||||||
const cap = architecture.systemCallArg(state, 0);
|
const cap = architecture.systemCallArg(state, 0);
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0) return fail(state);
|
if (t.address_space == 0) return fail(state);
|
||||||
|
|
||||||
const shm = ipc.resolveShm(t, cap) orelse return fail(state); // not an shm handle we hold
|
const shm = ipc.resolveShm(t, cap) orelse return fail(state); // not an shm handle we hold
|
||||||
if (t.shm_map_next == 0) t.shm_map_next = shm_arena_base;
|
if (t.shm_map_next == 0) t.shm_map_next = shm_arena_base;
|
||||||
@@ -532,12 +570,12 @@ fn systemShmMap(state: *architecture.CpuState) void {
|
|||||||
const size = shm.pages * page_size;
|
const size = shm.pages * page_size;
|
||||||
if (base_v + size > shm_arena_end) return fail(state);
|
if (base_v + size > shm_arena_end) return fail(state);
|
||||||
|
|
||||||
architecture.mapUserSharedInto(t.aspace, base_v, shm.phys, size);
|
architecture.mapUserSharedInto(t.address_space, base_v, shm.phys, size);
|
||||||
t.shm_map_next = base_v + size;
|
t.shm_map_next = base_v + size;
|
||||||
architecture.setSystemCallResult(state, base_v);
|
architecture.setSystemCallResult(state, base_v);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// shm_physical(cap) -> paddr: the guest-physical base of a shared region the caller holds a
|
/// shm_physical(cap) -> physical_address: the guest-physical base of a shared region the caller holds a
|
||||||
/// capability for. The frames are contiguous (allocated by `allocContiguous`), so a single
|
/// capability for. The frames are contiguous (allocated by `allocContiguous`), so a single
|
||||||
/// physical base + length describes the whole region — which is exactly what a driver needs
|
/// physical base + length describes the whole region — which is exactly what a driver needs
|
||||||
/// to hand a shm surface to a device (virtio-gpu `attach_backing`). Only a holder of the
|
/// to hand a shm surface to a device (virtio-gpu `attach_backing`). Only a holder of the
|
||||||
@@ -545,7 +583,7 @@ fn systemShmMap(state: *architecture.CpuState) void {
|
|||||||
fn systemShmPhysical(state: *architecture.CpuState) void {
|
fn systemShmPhysical(state: *architecture.CpuState) void {
|
||||||
const cap = architecture.systemCallArg(state, 0);
|
const cap = architecture.systemCallArg(state, 0);
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0) return fail(state);
|
if (t.address_space == 0) return fail(state);
|
||||||
const shm = ipc.resolveShm(t, cap) orelse return fail(state); // not an shm handle we hold
|
const shm = ipc.resolveShm(t, cap) orelse return fail(state); // not an shm handle we hold
|
||||||
architecture.setSystemCallResult(state, shm.phys);
|
architecture.setSystemCallResult(state, shm.phys);
|
||||||
}
|
}
|
||||||
@@ -566,10 +604,10 @@ fn systemDeviceRegister(state: *architecture.CpuState) void {
|
|||||||
const parent_id = architecture.systemCallArg(state, 0);
|
const parent_id = architecture.systemCallArg(state, 0);
|
||||||
const descriptor_ptr = architecture.systemCallArg(state, 1);
|
const descriptor_ptr = architecture.systemCallArg(state, 1);
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0) return fail(state);
|
if (t.address_space == 0) return fail(state);
|
||||||
|
|
||||||
var descriptor: device_abi.DeviceDescriptor = undefined;
|
var descriptor: device_abi.DeviceDescriptor = undefined;
|
||||||
if (!ipc.copyFromUser(t.aspace, descriptor_ptr, std.mem.asBytes(&descriptor))) return fail(state);
|
if (!ipc.copyFromUser(t.address_space, descriptor_ptr, std.mem.asBytes(&descriptor))) return fail(state);
|
||||||
|
|
||||||
// Under the big kernel lock: the broker's table is also mutated by the
|
// Under the big kernel lock: the broker's table is also mutated by the
|
||||||
// death sweep (releaseAllOwnedBy) and read by enumerate on other cores —
|
// death sweep (releaseAllOwnedBy) and read by enumerate on other cores —
|
||||||
@@ -642,6 +680,125 @@ fn systemSpawn(state: *architecture.CpuState) void {
|
|||||||
fail(state); // no bundled binary by that name
|
fail(state); // no bundled binary by that name
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// thread_spawn(entry, stack_top, arg) -> tid: start a task that shares the **caller's**
|
||||||
|
/// address space (docs/threading.md). The runtime supplies `entry` (its thread
|
||||||
|
/// trampoline), a stack it mmap'd, and the closure pointer, which the kernel delivers in
|
||||||
|
/// the new thread's rdi. The entry and stack must lie in the user half; the new thread is
|
||||||
|
/// supervised by the caller and inherits its priority. Only a user process may spawn.
|
||||||
|
fn systemThreadSpawn(state: *architecture.CpuState) void {
|
||||||
|
const entry = architecture.systemCallArg(state, 0);
|
||||||
|
const stack_top = architecture.systemCallArg(state, 1);
|
||||||
|
const arg = architecture.systemCallArg(state, 2);
|
||||||
|
const exit_handle = architecture.systemCallArg(state, 3);
|
||||||
|
const t = scheduler.current();
|
||||||
|
if (t.address_space == 0) return fail(state); // kernel tasks own no address space to share
|
||||||
|
if (entry == 0 or entry >= user_half_end) return fail(state);
|
||||||
|
if (stack_top == 0 or stack_top > user_half_end) return fail(state);
|
||||||
|
// The endpoint the thread notifies on exit (how join waits), or none.
|
||||||
|
const exit_endpoint: ?*ipc.Endpoint = if (exit_handle == abi.no_cap)
|
||||||
|
null
|
||||||
|
else
|
||||||
|
ipc.resolveHandle(t, exit_handle) orelse return failErr(state, ipc.EBADF);
|
||||||
|
const tid = spawnThreadSupervised(t.address_space, entry, stack_top, arg, t.priority, t.id, exit_endpoint) orelse return fail(state);
|
||||||
|
architecture.setSystemCallResult(state, tid);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Spawn a thread sharing `address_space`, taking the exit-endpoint reference under the **same**
|
||||||
|
/// lock as the spawn (as `spawnProcessSupervised` does), so the thread cannot die before
|
||||||
|
/// its reference exists. Returns the new thread id, or null on resource exhaustion.
|
||||||
|
fn spawnThreadSupervised(address_space: u64, entry: u64, stack_top: u64, arg: u64, priority: scheduler.Priority, supervisor: u32, exit_endpoint: ?*ipc.Endpoint) ?u32 {
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
|
const tid = scheduler.spawnUserLocked(address_space, entry, stack_top, arg, priority, "thread", supervisor, if (exit_endpoint) |e| @ptrCast(e) else null) orelse return null;
|
||||||
|
if (exit_endpoint) |endpoint| endpoint.refcount += 1; // the thread holds it birth-to-death
|
||||||
|
return tid;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// current_core() -> index: the dense 0-based index of the core the caller runs on.
|
||||||
|
fn systemCurrentCore(state: *architecture.CpuState) void {
|
||||||
|
architecture.setSystemCallResult(state, scheduler.currentCpuIndex());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// thread_self() -> tid: the calling thread's kernel task id.
|
||||||
|
fn systemThreadSelf(state: *architecture.CpuState) void {
|
||||||
|
architecture.setSystemCallResult(state, scheduler.currentId());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// set_thread_pointer(addr) -> 0: set the caller's user-space TLS thread pointer. The
|
||||||
|
/// arch layer maps it to IA32_FS_BASE on x86_64, `TPIDR_EL0` on aarch64; the kernel
|
||||||
|
/// never reads it, and the scheduler restores it per task across context switches
|
||||||
|
/// (docs/threading-plan.md M10). `addr` must be a user-half address.
|
||||||
|
fn systemSetThreadPointer(state: *architecture.CpuState) void {
|
||||||
|
const addr = architecture.systemCallArg(state, 0);
|
||||||
|
const t = scheduler.current();
|
||||||
|
if (t.address_space == 0) return fail(state); // kernel tasks have no user TLS
|
||||||
|
if (addr >= user_half_end) return fail(state);
|
||||||
|
const flags = sync.enter();
|
||||||
|
scheduler.setThreadPointerLocked(addr);
|
||||||
|
sync.leave(flags);
|
||||||
|
architecture.setSystemCallResult(state, 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// thread_join(tid) -> 0: block until the thread with id `tid` has exited (docs/threading-
|
||||||
|
/// plan.md M9). Needs no per-thread IPC endpoint. The compare-and-block is one critical
|
||||||
|
/// section, so an exit cannot slip between "is it alive?" and the block.
|
||||||
|
fn systemThreadJoin(state: *architecture.CpuState) void {
|
||||||
|
const tid: u32 = @truncate(architecture.systemCallArg(state, 0));
|
||||||
|
const t = scheduler.current();
|
||||||
|
if (t.address_space == 0) return fail(state); // kernel tasks don't join
|
||||||
|
const flags = sync.enter();
|
||||||
|
scheduler.joinThreadLocked(tid);
|
||||||
|
sync.leave(flags);
|
||||||
|
architecture.setSystemCallResult(state, 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// futex_wait(addr, expected, timeout_ns) -> status (docs/threading.md): if the 4-byte
|
||||||
|
/// user word at `addr` still equals `expected`, block until a futex_wake on `addr` or
|
||||||
|
/// (if timeout_ns > 0) the deadline. The compare and the block are one critical section,
|
||||||
|
/// so a concurrent futex_wake cannot slip between them. Returns futex_woken / mismatch /
|
||||||
|
/// timed_out.
|
||||||
|
fn systemFutexWait(state: *architecture.CpuState) void {
|
||||||
|
const addr = architecture.systemCallArg(state, 0);
|
||||||
|
const expected: u32 = @truncate(architecture.systemCallArg(state, 1));
|
||||||
|
const timeout_ns = architecture.systemCallArg(state, 2);
|
||||||
|
const t = scheduler.current();
|
||||||
|
if (t.address_space == 0) return fail(state);
|
||||||
|
if (addr == 0 or (addr & 3) != 0 or addr + 4 > user_half_end) return fail(state);
|
||||||
|
|
||||||
|
const flags = sync.enter();
|
||||||
|
var word_bytes: [4]u8 = undefined;
|
||||||
|
if (!ipc.copyFromUser(t.address_space, addr, &word_bytes)) {
|
||||||
|
sync.leave(flags);
|
||||||
|
return fail(state);
|
||||||
|
}
|
||||||
|
if (std.mem.readInt(u32, &word_bytes, .little) != expected) {
|
||||||
|
sync.leave(flags);
|
||||||
|
architecture.setSystemCallResult(state, abi.futex_mismatch);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const timeout_ms = if (timeout_ns == 0) 0 else (timeout_ns + 999_999) / 1_000_000;
|
||||||
|
const result = scheduler.futexWaitLocked(addr, timeout_ms);
|
||||||
|
sync.leave(flags);
|
||||||
|
architecture.setSystemCallResult(state, switch (result) {
|
||||||
|
.woken => abi.futex_woken,
|
||||||
|
.timed_out => abi.futex_timed_out,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// futex_wake(addr, count) -> woken: wake up to `count` tasks blocked in futex_wait on
|
||||||
|
/// `addr` in the caller's address space.
|
||||||
|
fn systemFutexWake(state: *architecture.CpuState) void {
|
||||||
|
const addr = architecture.systemCallArg(state, 0);
|
||||||
|
const count: u32 = @truncate(architecture.systemCallArg(state, 1));
|
||||||
|
const t = scheduler.current();
|
||||||
|
if (t.address_space == 0) return fail(state);
|
||||||
|
if (addr == 0 or (addr & 3) != 0 or addr + 4 > user_half_end) return fail(state);
|
||||||
|
const flags = sync.enter();
|
||||||
|
const woken = scheduler.futexWakeLocked(t.address_space, addr, count);
|
||||||
|
sync.leave(flags);
|
||||||
|
architecture.setSystemCallResult(state, woken);
|
||||||
|
}
|
||||||
|
|
||||||
/// process_enumerate(buffer, maximum) -> total: snapshot the task table into the
|
/// process_enumerate(buffer, maximum) -> total: snapshot the task table into the
|
||||||
/// caller's buffer (up to `maximum` `abi.ProcessDescriptor` entries), returning
|
/// caller's buffer (up to `maximum` `abi.ProcessDescriptor` entries), returning
|
||||||
/// the total live-task count — the exact shape of `device_enumerate`, so a `ps`
|
/// the total live-task count — the exact shape of `device_enumerate`, so a `ps`
|
||||||
@@ -652,7 +809,7 @@ fn systemProcessEnumerate(state: *architecture.CpuState) void {
|
|||||||
const buffer_ptr = architecture.systemCallArg(state, 0);
|
const buffer_ptr = architecture.systemCallArg(state, 0);
|
||||||
const maximum = architecture.systemCallArg(state, 1);
|
const maximum = architecture.systemCallArg(state, 1);
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0 or buffer_ptr >= user_half_end) return fail(state);
|
if (t.address_space == 0 or buffer_ptr >= user_half_end) return fail(state);
|
||||||
const sz = @sizeOf(abi.ProcessDescriptor);
|
const sz = @sizeOf(abi.ProcessDescriptor);
|
||||||
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
|
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
|
||||||
const out: [*]abi.ProcessDescriptor = @ptrFromInt(buffer_ptr);
|
const out: [*]abi.ProcessDescriptor = @ptrFromInt(buffer_ptr);
|
||||||
@@ -665,7 +822,7 @@ fn systemProcessEnumerate(state: *architecture.CpuState) void {
|
|||||||
/// cannot be a weapon (ids are never reused, so a stale one just misses).
|
/// cannot be a weapon (ids are never reused, so a stale one just misses).
|
||||||
fn systemProcessKill(state: *architecture.CpuState) void {
|
fn systemProcessKill(state: *architecture.CpuState) void {
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0) return fail(state);
|
if (t.address_space == 0) return fail(state);
|
||||||
const id = architecture.systemCallArg(state, 0);
|
const id = architecture.systemCallArg(state, 0);
|
||||||
if (id > std.math.maxInt(u32)) return failErr(state, ipc.ESRCH);
|
if (id > std.math.maxInt(u32)) return failErr(state, ipc.ESRCH);
|
||||||
const r = killProcess(t.id, @intCast(id));
|
const r = killProcess(t.id, @intCast(id));
|
||||||
@@ -798,7 +955,7 @@ pub fn killProcess(caller_id: u32, target_id: u32) i64 {
|
|||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
const target = scheduler.taskByIdLocked(target_id) orelse return -ipc.ESRCH;
|
const target = scheduler.taskByIdLocked(target_id) orelse return -ipc.ESRCH;
|
||||||
if (target.aspace == 0) return -ipc.ESRCH; // kernel tasks are not processes
|
if (target.address_space == 0) return -ipc.ESRCH; // kernel tasks are not processes
|
||||||
if (target.supervisor != caller_id) return -ipc.EPERM;
|
if (target.supervisor != caller_id) return -ipc.EPERM;
|
||||||
target.exit_reason = .killed;
|
target.exit_reason = .killed;
|
||||||
if (target.state == .running) {
|
if (target.state == .running) {
|
||||||
@@ -868,7 +1025,7 @@ var exit_subscribers: [exit_subscriber_capacity]?ExitSubscriber = .{null} ** exi
|
|||||||
/// secret between cooperating processes. -ENOSPC when the table is full.
|
/// secret between cooperating processes. -ENOSPC when the table is full.
|
||||||
fn systemProcessSubscribe(state: *architecture.CpuState) void {
|
fn systemProcessSubscribe(state: *architecture.CpuState) void {
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0) return fail(state);
|
if (t.address_space == 0) return fail(state);
|
||||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
@@ -888,7 +1045,7 @@ fn systemProcessSubscribe(state: *architecture.CpuState) void {
|
|||||||
/// delivered immediately on bind, coalesced into one notification.
|
/// delivered immediately on bind, coalesced into one notification.
|
||||||
fn systemSignalBind(state: *architecture.CpuState) void {
|
fn systemSignalBind(state: *architecture.CpuState) void {
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0) return fail(state);
|
if (t.address_space == 0) return fail(state);
|
||||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
@@ -908,7 +1065,7 @@ fn systemSignalBind(state: *architecture.CpuState) void {
|
|||||||
/// targets accumulate the signal in their pending mask.
|
/// targets accumulate the signal in their pending mask.
|
||||||
fn systemProcessSignal(state: *architecture.CpuState) void {
|
fn systemProcessSignal(state: *architecture.CpuState) void {
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0) return fail(state);
|
if (t.address_space == 0) return fail(state);
|
||||||
const id = architecture.systemCallArg(state, 0);
|
const id = architecture.systemCallArg(state, 0);
|
||||||
const signal = architecture.systemCallArg(state, 1);
|
const signal = architecture.systemCallArg(state, 1);
|
||||||
if (id > std.math.maxInt(u32)) return failErr(state, ipc.ESRCH);
|
if (id > std.math.maxInt(u32)) return failErr(state, ipc.ESRCH);
|
||||||
@@ -916,7 +1073,7 @@ fn systemProcessSignal(state: *architecture.CpuState) void {
|
|||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
const target = scheduler.taskByIdLocked(@intCast(id)) orelse return failErr(state, ipc.ESRCH);
|
const target = scheduler.taskByIdLocked(@intCast(id)) orelse return failErr(state, ipc.ESRCH);
|
||||||
if (target.aspace == 0) return failErr(state, ipc.ESRCH);
|
if (target.address_space == 0) return failErr(state, ipc.ESRCH);
|
||||||
if (target.supervisor != t.id and target.id != t.id) return failErr(state, ipc.EPERM);
|
if (target.supervisor != t.id and target.id != t.id) return failErr(state, ipc.EPERM);
|
||||||
target.pending_signals |= @as(u32, 1) << @intCast(signal);
|
target.pending_signals |= @as(u32, 1) << @intCast(signal);
|
||||||
if (target.signal_endpoint) |raw| {
|
if (target.signal_endpoint) |raw| {
|
||||||
@@ -953,7 +1110,7 @@ fn timerSweepLocked() void {
|
|||||||
/// timer_bind(endpoint, ms): arm a one-shot timer. -ENOSPC when the table is full.
|
/// timer_bind(endpoint, ms): arm a one-shot timer. -ENOSPC when the table is full.
|
||||||
fn systemTimerBind(state: *architecture.CpuState) void {
|
fn systemTimerBind(state: *architecture.CpuState) void {
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0) return fail(state);
|
if (t.address_space == 0) return fail(state);
|
||||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 0)) orelse return failErr(state, ipc.EBADF);
|
||||||
const ms = architecture.systemCallArg(state, 1);
|
const ms = architecture.systemCallArg(state, 1);
|
||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
@@ -970,7 +1127,7 @@ fn systemTimerBind(state: *architecture.CpuState) void {
|
|||||||
|
|
||||||
fn systemProcessExitReason(state: *architecture.CpuState) void {
|
fn systemProcessExitReason(state: *architecture.CpuState) void {
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0) return fail(state);
|
if (t.address_space == 0) return fail(state);
|
||||||
const id = architecture.systemCallArg(state, 0);
|
const id = architecture.systemCallArg(state, 0);
|
||||||
if (id > std.math.maxInt(u32)) return failErr(state, ipc.ESRCH);
|
if (id > std.math.maxInt(u32)) return failErr(state, ipc.ESRCH);
|
||||||
const r = exitReasonOf(t.id, @intCast(id));
|
const r = exitReasonOf(t.id, @intCast(id));
|
||||||
@@ -996,7 +1153,7 @@ fn ownedGsi(t: *scheduler.Task, device_id: u64, resource_index: u64) ?u32 {
|
|||||||
/// IPC_ReplyWait and is woken by the ISR; see system/kernel/irq.zig for the cycle.
|
/// IPC_ReplyWait and is woken by the ISR; see system/kernel/irq.zig for the cycle.
|
||||||
fn systemIrqBind(state: *architecture.CpuState) void {
|
fn systemIrqBind(state: *architecture.CpuState) void {
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0) return fail(state);
|
if (t.address_space == 0) return fail(state);
|
||||||
const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse
|
const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse
|
||||||
return fail(state);
|
return fail(state);
|
||||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 2)) orelse return fail(state);
|
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 2)) orelse return fail(state);
|
||||||
@@ -1016,7 +1173,7 @@ fn systemIrqBind(state: *architecture.CpuState) void {
|
|||||||
fn systemMsiBind(state: *architecture.CpuState) void {
|
fn systemMsiBind(state: *architecture.CpuState) void {
|
||||||
const device_id = architecture.systemCallArg(state, 0);
|
const device_id = architecture.systemCallArg(state, 0);
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0) return fail(state);
|
if (t.address_space == 0) return fail(state);
|
||||||
const owner = devices_broker.ownerOf(device_id) orelse return fail(state);
|
const owner = devices_broker.ownerOf(device_id) orelse return fail(state);
|
||||||
if (owner != t.id) return fail(state); // not claimed by this process
|
if (owner != t.id) return fail(state); // not claimed by this process
|
||||||
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 1)) orelse return failErr(state, ipc.EBADF);
|
const endpoint = ipc.resolveHandle(t, architecture.systemCallArg(state, 1)) orelse return failErr(state, ipc.EBADF);
|
||||||
@@ -1035,7 +1192,7 @@ fn systemMsiBind(state: *architecture.CpuState) void {
|
|||||||
/// more arrives until the driver says it has serviced the hardware.
|
/// more arrives until the driver says it has serviced the hardware.
|
||||||
fn systemIrqAck(state: *architecture.CpuState) void {
|
fn systemIrqAck(state: *architecture.CpuState) void {
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0) return fail(state);
|
if (t.address_space == 0) return fail(state);
|
||||||
const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse
|
const gsi = ownedGsi(t, architecture.systemCallArg(state, 0), architecture.systemCallArg(state, 1)) orelse
|
||||||
return fail(state);
|
return fail(state);
|
||||||
|
|
||||||
@@ -1123,37 +1280,52 @@ fn systemKlogRead(state: *architecture.CpuState) void {
|
|||||||
fn systemMmap(state: *architecture.CpuState) void {
|
fn systemMmap(state: *architecture.CpuState) void {
|
||||||
const len = architecture.systemCallArg(state, 0);
|
const len = architecture.systemCallArg(state, 0);
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0) return fail(state); // not a user process — nothing to map into
|
if (t.address_space == 0) return fail(state); // not a user process — nothing to map into
|
||||||
const pages = (len + page_size - 1) / page_size;
|
const pages = (len + page_size - 1) / page_size;
|
||||||
if (pages == 0 or pages > maximum_mmap_pages) return fail(state);
|
if (pages == 0 or pages > maximum_mmap_pages) return fail(state);
|
||||||
|
|
||||||
if (t.heap_next == 0) t.heap_next = heap_arena_base; // seed the arena lazily
|
// Reserve a disjoint range under a *brief* lock (the cursor is shared by every thread
|
||||||
const base = t.heap_next;
|
// in this address space). The mapping below then takes the lock **per page**, not for
|
||||||
if (base + pages * page_size > heap_arena_end) return fail(state); // arena exhausted
|
// the whole grant: the big lock is held with interrupts disabled, so pinning it across
|
||||||
|
// a multi-MiB memset+map would freeze every other core on its next tick — which timed
|
||||||
|
// the `affinity` scenario out (docs/threading-plan.md M7).
|
||||||
|
const base = reserve: {
|
||||||
|
const flags = sync.enter();
|
||||||
|
defer sync.leave(flags);
|
||||||
|
const cursor = scheduler.addressSpaceMmapNextPtr(t.address_space) orelse return fail(state);
|
||||||
|
if (cursor.* == 0) cursor.* = heap_arena_base; // seed the arena lazily
|
||||||
|
const b = cursor.*;
|
||||||
|
if (b + pages * page_size > heap_arena_end) return fail(state); // arena exhausted
|
||||||
|
cursor.* = b + pages * page_size; // reserve now, so concurrent grants can't overlap
|
||||||
|
break :reserve b;
|
||||||
|
};
|
||||||
|
|
||||||
// Map page by page. On mid-way frame exhaustion, roll back the pages already mapped
|
// Map the reserved range page by page, each page under a short-held lock (the range is
|
||||||
// (unmap + free) so no partial grant leaks into the address space — the same
|
// already reserved, so pages can't overlap another thread's; the lock only serializes
|
||||||
// all-or-nothing guarantee as before, but without a fixed scratch array, so the
|
// the shared page-table walk). On mid-way frame exhaustion, roll back the mapped pages
|
||||||
// per-call size can be a multi-MiB framebuffer.
|
// so no partial grant leaks — the reserved-but-unmapped tail of the arena is left
|
||||||
|
// fallow (a rare, bounded address-space leak, not a memory leak).
|
||||||
var mapped: usize = 0;
|
var mapped: usize = 0;
|
||||||
while (mapped < pages) : (mapped += 1) {
|
while (mapped < pages) : (mapped += 1) {
|
||||||
|
const flags = sync.enter();
|
||||||
const frame = pmm.alloc() orelse {
|
const frame = pmm.alloc() orelse {
|
||||||
var i: usize = 0;
|
var i: usize = 0;
|
||||||
while (i < mapped) : (i += 1) {
|
while (i < mapped) : (i += 1) {
|
||||||
const va = base + i * page_size;
|
const va = base + i * page_size;
|
||||||
if (architecture.translate(t.aspace, va)) |physical| {
|
if (architecture.translate(t.address_space, va)) |physical| {
|
||||||
architecture.unmapUserPageInto(t.aspace, va);
|
architecture.unmapUserPageInto(t.address_space, va);
|
||||||
pmm.free(physical);
|
pmm.free(physical);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
sync.leave(flags);
|
||||||
return fail(state);
|
return fail(state);
|
||||||
};
|
};
|
||||||
const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(frame));
|
const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(frame));
|
||||||
@memset(destination[0..page_size], 0); // hand out zeroed memory
|
@memset(destination[0..page_size], 0); // hand out zeroed memory
|
||||||
architecture.mapUserPageInto(t.aspace, base + mapped * page_size, frame, true, false); // RW + NX
|
architecture.mapUserPageInto(t.address_space, base + mapped * page_size, frame, true, false); // RW + NX
|
||||||
|
sync.leave(flags);
|
||||||
}
|
}
|
||||||
t.heap_next = base + pages * page_size;
|
architecture.setSystemCallResult(state, base); // the cursor was already advanced at reserve
|
||||||
architecture.setSystemCallResult(state, base);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// munmap(base, len): release a range previously handed out by `mmap`. Unmaps
|
/// munmap(base, len): release a range previously handed out by `mmap`. Unmaps
|
||||||
@@ -1165,14 +1337,14 @@ fn systemMunmap(state: *architecture.CpuState) void {
|
|||||||
const base = architecture.systemCallArg(state, 0);
|
const base = architecture.systemCallArg(state, 0);
|
||||||
const len = architecture.systemCallArg(state, 1);
|
const len = architecture.systemCallArg(state, 1);
|
||||||
const t = scheduler.current();
|
const t = scheduler.current();
|
||||||
if (t.aspace == 0 or base % page_size != 0) return fail(state);
|
if (t.address_space == 0 or base % page_size != 0) return fail(state);
|
||||||
const pages = (len + page_size - 1) / page_size;
|
const pages = (len + page_size - 1) / page_size;
|
||||||
if (base < heap_arena_base or base + pages * page_size > heap_arena_end) return fail(state);
|
if (base < heap_arena_base or base + pages * page_size > heap_arena_end) return fail(state);
|
||||||
|
|
||||||
for (0..pages) |i| {
|
for (0..pages) |i| {
|
||||||
const va = base + i * page_size;
|
const va = base + i * page_size;
|
||||||
if (architecture.translate(t.aspace, va)) |physical| {
|
if (architecture.translate(t.address_space, va)) |physical| {
|
||||||
architecture.unmapUserPageInto(t.aspace, va);
|
architecture.unmapUserPageInto(t.address_space, va);
|
||||||
pmm.free(physical);
|
pmm.free(physical);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1236,7 +1408,7 @@ const maximum_segments = 16;
|
|||||||
const maximum_pages = 256; // 1 MiB loader budget; the user region caps at 2 MiB anyway
|
const maximum_pages = 256; // 1 MiB loader budget; the user region caps at 2 MiB anyway
|
||||||
|
|
||||||
const Segment = struct {
|
const Segment = struct {
|
||||||
vaddr: u64,
|
virtual_address: u64,
|
||||||
memsz: u64,
|
memsz: u64,
|
||||||
filesz: u64,
|
filesz: u64,
|
||||||
off: u64,
|
off: u64,
|
||||||
@@ -1284,7 +1456,7 @@ fn parseSegments(image: []const u8, segs: *[maximum_segments]Segment) InitError!
|
|||||||
if (w and x) return error.BadSegment; // W^X, even for init
|
if (w and x) return error.BadSegment; // W^X, even for init
|
||||||
|
|
||||||
const seg = Segment{
|
const seg = Segment{
|
||||||
.vaddr = phdr.p_vaddr,
|
.virtual_address = phdr.p_vaddr,
|
||||||
.memsz = phdr.p_memsz,
|
.memsz = phdr.p_memsz,
|
||||||
.filesz = phdr.p_filesz,
|
.filesz = phdr.p_filesz,
|
||||||
.off = phdr.p_offset,
|
.off = phdr.p_offset,
|
||||||
@@ -1293,9 +1465,9 @@ fn parseSegments(image: []const u8, segs: *[maximum_segments]Segment) InitError!
|
|||||||
};
|
};
|
||||||
// No overlap with any earlier segment (page-granular, since mapping is).
|
// No overlap with any earlier segment (page-granular, since mapping is).
|
||||||
for (segs[0..count]) |other| {
|
for (segs[0..count]) |other| {
|
||||||
const a_end = seg.vaddr + seg.pages() * page_size;
|
const a_end = seg.virtual_address + seg.pages() * page_size;
|
||||||
const b_end = other.vaddr + other.pages() * page_size;
|
const b_end = other.virtual_address + other.pages() * page_size;
|
||||||
if (seg.vaddr < b_end and other.vaddr < a_end) return error.BadSegment;
|
if (seg.virtual_address < b_end and other.virtual_address < a_end) return error.BadSegment;
|
||||||
}
|
}
|
||||||
total_pages += seg.pages();
|
total_pages += seg.pages();
|
||||||
if (total_pages > maximum_pages) return error.ProgramTooBig;
|
if (total_pages > maximum_pages) return error.ProgramTooBig;
|
||||||
@@ -1306,17 +1478,17 @@ fn parseSegments(image: []const u8, segs: *[maximum_segments]Segment) InitError!
|
|||||||
|
|
||||||
// The entry point must land inside an executable segment.
|
// The entry point must land inside an executable segment.
|
||||||
for (segs[0..count]) |seg| {
|
for (segs[0..count]) |seg| {
|
||||||
if (seg.executable and ehdr.e_entry >= seg.vaddr and ehdr.e_entry < seg.vaddr + seg.memsz)
|
if (seg.executable and ehdr.e_entry >= seg.virtual_address and ehdr.e_entry < seg.virtual_address + seg.memsz)
|
||||||
return .{ .count = count, .entry = ehdr.e_entry };
|
return .{ .count = count, .entry = ehdr.e_entry };
|
||||||
}
|
}
|
||||||
return error.BadEntry;
|
return error.BadEntry;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Load one page of a segment into address space `aspace`: a fresh frame, zeroed
|
/// Load one page of a segment into address space `address_space`: a fresh frame, zeroed
|
||||||
/// and filled through the physmap, mapped user-accessible with the segment's W^X.
|
/// and filled through the physmap, mapped user-accessible with the segment's W^X.
|
||||||
/// On a later failure the whole address space is torn down, which frees every
|
/// On a later failure the whole address space is torn down, which frees every
|
||||||
/// frame mapped into it — so no per-page rollback list is needed here.
|
/// frame mapped into it — so no per-page rollback list is needed here.
|
||||||
fn loadPageInto(aspace: u64, image: []const u8, seg: Segment, page_index: u64) InitError!void {
|
fn loadPageInto(address_space: u64, image: []const u8, seg: Segment, page_index: u64) InitError!void {
|
||||||
const frame = pmm.alloc() orelse return error.OutOfMemory;
|
const frame = pmm.alloc() orelse return error.OutOfMemory;
|
||||||
const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(frame));
|
const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(frame));
|
||||||
@memset(destination[0..page_size], 0);
|
@memset(destination[0..page_size], 0);
|
||||||
@@ -1325,7 +1497,7 @@ fn loadPageInto(aspace: u64, image: []const u8, seg: Segment, page_index: u64) I
|
|||||||
const n = @min(page_size, seg.filesz - page_off);
|
const n = @min(page_size, seg.filesz - page_off);
|
||||||
@memcpy(destination[0..n], image[seg.off + page_off ..][0..n]);
|
@memcpy(destination[0..n], image[seg.off + page_off ..][0..n]);
|
||||||
}
|
}
|
||||||
architecture.mapUserPageInto(aspace, seg.vaddr + page_off, frame, seg.writable, seg.executable);
|
architecture.mapUserPageInto(address_space, seg.virtual_address + page_off, frame, seg.writable, seg.executable);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Build the System V AMD64 process-entry block at the top of a process's stack
|
/// Build the System V AMD64 process-entry block at the top of a process's stack
|
||||||
@@ -1412,11 +1584,11 @@ pub fn spawnProcessSupervised(image: []const u8, priority: u3, argv: []const []c
|
|||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
|
|
||||||
const aspace = architecture.createAddressSpace() orelse return error.OutOfMemory;
|
const address_space = architecture.createAddressSpace() orelse return error.OutOfMemory;
|
||||||
errdefer architecture.destroyAddressSpace(aspace);
|
errdefer architecture.destroyAddressSpace(address_space);
|
||||||
|
|
||||||
for (segs[0..parsed.count]) |seg| {
|
for (segs[0..parsed.count]) |seg| {
|
||||||
for (0..seg.pages()) |i| try loadPageInto(aspace, image, seg, i);
|
for (0..seg.pages()) |i| try loadPageInto(address_space, image, seg, i);
|
||||||
}
|
}
|
||||||
|
|
||||||
// The stack: `user_stack_pages` zeroed pages below stack_top_virtual, RW + NX.
|
// The stack: `user_stack_pages` zeroed pages below stack_top_virtual, RW + NX.
|
||||||
@@ -1430,10 +1602,10 @@ pub fn spawnProcessSupervised(image: []const u8, priority: u3, argv: []const []c
|
|||||||
const page_virtual = stack_base_virtual + i * page_size;
|
const page_virtual = stack_base_virtual + i * page_size;
|
||||||
if (i == parameters.user_stack_pages - 1)
|
if (i == parameters.user_stack_pages - 1)
|
||||||
user_sp = buildEntryStack(stack_page, page_virtual, argv);
|
user_sp = buildEntryStack(stack_page, page_virtual, argv);
|
||||||
architecture.mapUserPageInto(aspace, page_virtual, stack_frame, true, false); // RW + NX
|
architecture.mapUserPageInto(address_space, page_virtual, stack_frame, true, false); // RW + NX
|
||||||
}
|
}
|
||||||
|
|
||||||
const child = scheduler.spawnUserLocked(aspace, parsed.entry, user_sp, priority, argv[0], supervisor, if (exit_endpoint) |endpoint| @ptrCast(endpoint) else null) orelse
|
const child = scheduler.spawnUserLocked(address_space, parsed.entry, user_sp, 0, priority, argv[0], supervisor, if (exit_endpoint) |endpoint| @ptrCast(endpoint) else null) orelse
|
||||||
return error.OutOfMemory;
|
return error.OutOfMemory;
|
||||||
// The child holds a reference to its exit endpoint from birth to death. Taken
|
// The child holds a reference to its exit endpoint from birth to death. Taken
|
||||||
// only now, after nothing can fail; the lock is still held, so the child
|
// only now, after nothing can fail; the lock is still held, so the child
|
||||||
|
|||||||
+305
-37
@@ -31,7 +31,10 @@ const number_priorities = 8;
|
|||||||
const stack_size = parameters.kernel_stack_size; // each task's kernel stack
|
const stack_size = parameters.kernel_stack_size; // each task's kernel stack
|
||||||
const maximum_tasks = parameters.maximum_tasks; // maximum tasks alive at once (static pool)
|
const maximum_tasks = parameters.maximum_tasks; // maximum tasks alive at once (static pool)
|
||||||
|
|
||||||
const State = enum { free, ready, running, blocked };
|
// `reaping` = the task has exited and is queued on its core's reap list; its slot must not
|
||||||
|
// be reused (freeSlot skips it) until the reaper has freed its kernel stack and set it
|
||||||
|
// `free` (docs/threading-plan.md M8/M9).
|
||||||
|
const State = enum { free, ready, running, blocked, reaping };
|
||||||
|
|
||||||
pub const Task = struct {
|
pub const Task = struct {
|
||||||
id: u32 = 0,
|
id: u32 = 0,
|
||||||
@@ -75,16 +78,25 @@ pub const Task = struct {
|
|||||||
ipc_wait_endpoint: ?*anyopaque = null,
|
ipc_wait_endpoint: ?*anyopaque = null,
|
||||||
// Physical root of this task's address space, or 0 for a kernel task (which
|
// Physical root of this task's address space, or 0 for a kernel task (which
|
||||||
// runs on the shared kernel page tables). A user task carries its own.
|
// runs on the shared kernel page tables). A user task carries its own.
|
||||||
aspace: u64 = 0,
|
address_space: u64 = 0,
|
||||||
user_ip: u64 = 0, // user-mode entry point (user task only)
|
user_ip: u64 = 0, // user-mode entry point (user task only)
|
||||||
user_sp: u64 = 0, // user-mode stack pointer (user task only)
|
user_sp: u64 = 0, // user-mode stack pointer (user task only)
|
||||||
// Next free virtual address in this task's mmap grant arena (0 = uninitialised;
|
user_arg: u64 = 0, // value delivered in the user's first argument register at first entry
|
||||||
// process.zig lazily seeds it to the arena base on the first mmap). Bumped up
|
// (rdi on x86_64, via architecture.jumpToUserArg): 0 for a process (its _start ignores
|
||||||
// as the user heap grows; user task only.
|
// it), the closure pointer for a thread (docs/threading.md)
|
||||||
heap_next: u64 = 0,
|
// The user address this task is blocked on in futex_wait (0 = not futex-waiting).
|
||||||
// Next free virtual address in this task's MMIO-grant arena (PML4[226]; 0 =
|
// Cleared to 0 by futexWakeLocked as the "woken, not timed out" signal (docs/threading.md).
|
||||||
// uninitialised, process.zig seeds it on the first mmio_map). User task only.
|
futex_addr: u64 = 0,
|
||||||
device_map_next: u64 = 0,
|
// The task id this task is blocked in `thread_join` on (0 = not joining). Woken by
|
||||||
|
// `wakeJoinersLocked` when that task exits (docs/threading-plan.md M9).
|
||||||
|
join_target: u32 = 0,
|
||||||
|
// This task's user-space TLS thread pointer — 0 until set via `set_thread_pointer`.
|
||||||
|
// Architecture-neutral: the arch layer maps it to the FS base on x86_64, `TPIDR_EL0` on
|
||||||
|
// aarch64. Restored on every context switch to this task (docs/threading-plan.md M10).
|
||||||
|
thread_pointer: u64 = 0,
|
||||||
|
// The mmap / MMIO grant-arena cursors moved from Task to the per-address-space object
|
||||||
|
// (`AddressSpaceRef`, below) so threads sharing one address space hand out disjoint grants
|
||||||
|
// — see addressSpaceMmapNextPtr / addressSpaceDeviceMapNextPtr (docs/threading-plan.md M7).
|
||||||
// --- synchronous IPC (ipc_sync.zig) ---
|
// --- synchronous IPC (ipc_sync.zig) ---
|
||||||
// Per-process handle table: a small-int handle names a kernel capability object.
|
// Per-process handle table: a small-int handle names a kernel capability object.
|
||||||
// Each entry tags its `kind` (an IPC endpoint or a shared-memory object) so the
|
// Each entry tags its `kind` (an IPC endpoint or a shared-memory object) so the
|
||||||
@@ -95,9 +107,9 @@ pub const Task = struct {
|
|||||||
// receive, cleared when it replies). A client, while blocked in Call, records
|
// receive, cleared when it replies). A client, while blocked in Call, records
|
||||||
// its message + reply buffers here and its result lands in `ipc_status`.
|
// its message + reply buffers here and its result lands in `ipc_status`.
|
||||||
ipc_client: ?*Task = null,
|
ipc_client: ?*Task = null,
|
||||||
ipc_send_ptr: u64 = 0, // client: outgoing message (vaddr in this task's AS)
|
ipc_send_ptr: u64 = 0, // client: outgoing message (virtual_address in this task's address space)
|
||||||
ipc_send_len: u64 = 0,
|
ipc_send_len: u64 = 0,
|
||||||
ipc_reply_ptr: u64 = 0, // client: reply buffer (vaddr)
|
ipc_reply_ptr: u64 = 0, // client: reply buffer (virtual_address)
|
||||||
ipc_reply_cap: u64 = 0,
|
ipc_reply_cap: u64 = 0,
|
||||||
ipc_status: i64 = 0, // client: reply length / -errno, written by the replier
|
ipc_status: i64 = 0, // client: reply length / -errno, written by the replier
|
||||||
dma_map_next: u64 = 0, // bump pointer into this task's DMA arena (0 = unseeded)
|
dma_map_next: u64 = 0, // bump pointer into this task's DMA arena (0 = unseeded)
|
||||||
@@ -136,6 +148,127 @@ pub const ipc_maximum_handles = 16;
|
|||||||
pub const HandleObject = struct { kind: u8, ptr: *anyopaque };
|
pub const HandleObject = struct { kind: u8, ptr: *anyopaque };
|
||||||
|
|
||||||
var tasks = [_]Task{.{}} ** maximum_tasks;
|
var tasks = [_]Task{.{}} ** maximum_tasks;
|
||||||
|
|
||||||
|
/// Address-space reference counts: one live entry per address space, counting the
|
||||||
|
/// tasks that share it. An address space is 1:1 with a process today; threads
|
||||||
|
/// (docs/threading.md) will push a count above 1, and `destroyAddressSpace` must run
|
||||||
|
/// only when the **last** task on an address space exits. All access is under the big
|
||||||
|
/// kernel lock. There can be no more live address spaces than tasks, so the table is
|
||||||
|
/// sized to the task pool and never overflows in practice.
|
||||||
|
// The per-address-space kernel object: a reference count plus the grant-arena cursors.
|
||||||
|
// One live entry per address space; threads sharing an address space share this entry,
|
||||||
|
// so their mmap/mmio grants bump one cursor and never overlap (docs/threading-plan.md M7).
|
||||||
|
// `mmap_next`/`device_map_next` are 0 until process.zig seeds them to the arena base.
|
||||||
|
const AddressSpaceRef = struct { root: u64 = 0, count: u32 = 0, mmap_next: u64 = 0, device_map_next: u64 = 0 };
|
||||||
|
var address_space_refs = [_]AddressSpaceRef{.{}} ** maximum_tasks;
|
||||||
|
var address_space_destroy_count: u64 = 0;
|
||||||
|
|
||||||
|
/// Total bytes of task **kernel** stacks currently allocated from the kernel heap —
|
||||||
|
/// incremented when a task is created, decremented when the reaper frees a dead task's
|
||||||
|
/// stack. A test-observable proof that the reaper reclaims every stack (docs/threading-
|
||||||
|
/// plan.md M8): with no live tasks beyond the baseline, this returns to its baseline.
|
||||||
|
var live_stack_bytes: usize = 0;
|
||||||
|
|
||||||
|
/// Test-observable: bytes of task kernel stacks currently live (see `live_stack_bytes`).
|
||||||
|
pub fn liveStackBytes() usize {
|
||||||
|
return live_stack_bytes;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Free a dead task's kernel stack and drop it from `live_stack_bytes`. The task must be
|
||||||
|
/// off that stack already (killed while not running, or reaped after it switched away).
|
||||||
|
/// Caller holds the kernel lock.
|
||||||
|
fn reapStackLocked(t: *Task) void {
|
||||||
|
if (t.stack.len == 0) return; // boot/idle tasks run on a static stack — nothing to free
|
||||||
|
live_stack_bytes -= t.stack.len;
|
||||||
|
heap.allocator().free(t.stack);
|
||||||
|
t.stack = &.{};
|
||||||
|
t.kstack_top = 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Free every `.reaping` task queued on this core's reap list and mark each `.free` (now
|
||||||
|
/// its slot may be reused). The tasks are all off their stacks (they switched away), and
|
||||||
|
/// the caller holds the lock, so freeing is safe (docs/threading-plan.md M8/M9).
|
||||||
|
fn drainReapListLocked(pc: *PerCpu) void {
|
||||||
|
var node = pc.reap_list;
|
||||||
|
pc.reap_list = null;
|
||||||
|
while (node) |t| {
|
||||||
|
node = t.next; // save the link before we clear it
|
||||||
|
t.next = null;
|
||||||
|
reapStackLocked(t);
|
||||||
|
t.state = .free; // reusable only now, after the stack is freed
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Take a reference to address space `root` (0 = a kernel task, which owns none).
|
||||||
|
/// Returns false only if the ref table is full — bounded by `maximum_tasks`, so in
|
||||||
|
/// practice it never is. Caller holds the kernel lock.
|
||||||
|
fn retainAddressSpace(root: u64) bool {
|
||||||
|
if (root == 0) return true;
|
||||||
|
var free: ?*AddressSpaceRef = null;
|
||||||
|
for (&address_space_refs) |*entry| {
|
||||||
|
if (entry.count != 0 and entry.root == root) {
|
||||||
|
entry.count += 1;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
if (entry.count == 0 and free == null) free = entry;
|
||||||
|
}
|
||||||
|
const slot = free orelse return false;
|
||||||
|
slot.* = .{ .root = root, .count = 1 };
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Drop a reference to `root`; destroy the address space when the **last** one drops.
|
||||||
|
/// A `root` with no entry — never retained, e.g. a hand-built test space — is
|
||||||
|
/// destroyed directly, preserving the pre-refcount behaviour. Caller holds the lock.
|
||||||
|
fn releaseAddressSpace(root: u64) void {
|
||||||
|
if (root == 0) return;
|
||||||
|
for (&address_space_refs) |*entry| {
|
||||||
|
if (entry.count == 0 or entry.root != root) continue;
|
||||||
|
entry.count -= 1;
|
||||||
|
if (entry.count == 0) {
|
||||||
|
entry.root = 0;
|
||||||
|
architecture.destroyAddressSpace(root);
|
||||||
|
address_space_destroy_count += 1;
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
architecture.destroyAddressSpace(root);
|
||||||
|
address_space_destroy_count += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Test-observable: how many address spaces are live (entries with a nonzero count).
|
||||||
|
pub fn liveAddressSpaceCount() u32 {
|
||||||
|
var live: u32 = 0;
|
||||||
|
for (&address_space_refs) |*entry| {
|
||||||
|
if (entry.count != 0) live += 1;
|
||||||
|
}
|
||||||
|
return live;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Test-observable: total address-space destructions since boot.
|
||||||
|
pub fn addressSpaceDestroyCount() u64 {
|
||||||
|
return address_space_destroy_count;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Pointer to the mmap grant-arena cursor for address space `root`, so the mmap syscall
|
||||||
|
/// can read-and-bump it. Per-address-space (not per-task), so sibling threads get
|
||||||
|
/// disjoint grants. **Caller holds the kernel lock** (the entry is stable while held).
|
||||||
|
/// Null only if `root` was never retained — which can't happen for a live user task.
|
||||||
|
pub fn addressSpaceMmapNextPtr(root: u64) ?*u64 {
|
||||||
|
for (&address_space_refs) |*entry| {
|
||||||
|
if (entry.count != 0 and entry.root == root) return &entry.mmap_next;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Pointer to the MMIO grant-arena cursor for address space `root` (see
|
||||||
|
/// `addressSpaceMmapNextPtr`). Caller holds the kernel lock.
|
||||||
|
pub fn addressSpaceDeviceMapNextPtr(root: u64) ?*u64 {
|
||||||
|
for (&address_space_refs) |*entry| {
|
||||||
|
if (entry.count != 0 and entry.root == root) return &entry.device_map_next;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
var next_id: u32 = 1;
|
var next_id: u32 = 1;
|
||||||
|
|
||||||
/// Per-CPU scheduler state: the task each core is running, its own idle task, and a
|
/// Per-CPU scheduler state: the task each core is running, its own idle task, and a
|
||||||
@@ -155,11 +288,18 @@ pub const PerCpu = struct {
|
|||||||
hw_id: u32 = 0, // the core's hardware id (Local APIC id on x86_64)
|
hw_id: u32 = 0, // the core's hardware id (Local APIC id on x86_64)
|
||||||
index: u32 = 0, // dense 0-based core index
|
index: u32 = 0, // dense 0-based core index
|
||||||
online: bool = false, // has this core finished bring-up?
|
online: bool = false, // has this core finished bring-up?
|
||||||
loaded_aspace: u64 = 0, // the address-space root currently loaded on this core
|
loaded_address_space: u64 = 0, // the address-space root currently loaded on this core
|
||||||
|
loaded_thread_pointer: u64 = 0, // the TLS thread pointer currently loaded on this core (docs/threading-plan.md M10)
|
||||||
// Tasks pinned to this core (affinity == index), per priority level + bitmap.
|
// Tasks pinned to this core (affinity == index), per priority level + bitmap.
|
||||||
pinned_head: [number_priorities]?*Task = .{null} ** number_priorities,
|
pinned_head: [number_priorities]?*Task = .{null} ** number_priorities,
|
||||||
pinned_tail: [number_priorities]?*Task = .{null} ** number_priorities,
|
pinned_tail: [number_priorities]?*Task = .{null} ** number_priorities,
|
||||||
pinned_bitmap: u8 = 0,
|
pinned_bitmap: u8 = 0,
|
||||||
|
// Tasks that ended while running on THIS core: they could not free the kernel stack
|
||||||
|
// they were standing on, so each pushed itself onto this list (`.reaping` state, linked
|
||||||
|
// via `Task.next`) and switched away. The next task to run on this core — or the timer
|
||||||
|
// tick — frees their stacks from its own stack, safely (docs/threading-plan.md M8). A
|
||||||
|
// *list* (not one slot) so a second death before the first is drained can't lose it.
|
||||||
|
reap_list: ?*Task = null,
|
||||||
};
|
};
|
||||||
|
|
||||||
const maximum_cpus = parameters.maximum_cpus;
|
const maximum_cpus = parameters.maximum_cpus;
|
||||||
@@ -191,7 +331,7 @@ var preemption_enabled = true;
|
|||||||
/// boot, before interrupts are enabled — so no lock is needed here.
|
/// boot, before interrupts are enabled — so no lock is needed here.
|
||||||
pub fn init(boot_priority: Priority) void {
|
pub fn init(boot_priority: Priority) void {
|
||||||
const pc = &cpus[0];
|
const pc = &cpus[0];
|
||||||
pc.* = .{ .index = 0, .online = true, .loaded_aspace = architecture.kernelPageTable() };
|
pc.* = .{ .index = 0, .online = true, .loaded_address_space = architecture.kernelPageTable() };
|
||||||
architecture.setCpuLocal(0, @intFromPtr(pc));
|
architecture.setCpuLocal(0, @intFromPtr(pc));
|
||||||
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
|
tasks[0] = .{ .id = 0, .state = .running, .priority = boot_priority };
|
||||||
pc.current = &tasks[0];
|
pc.current = &tasks[0];
|
||||||
@@ -230,7 +370,7 @@ pub fn secondaryMain() callconv(.c) noreturn {
|
|||||||
pc.current = t;
|
pc.current = t;
|
||||||
pc.idle = t;
|
pc.idle = t;
|
||||||
pc.online = true;
|
pc.online = true;
|
||||||
pc.loaded_aspace = architecture.kernelPageTable(); // the AP adopted the kernel tables at bring-up
|
pc.loaded_address_space = architecture.kernelPageTable(); // the AP adopted the kernel tables at bring-up
|
||||||
sync.leave(flags);
|
sync.leave(flags);
|
||||||
|
|
||||||
architecture.enableInterrupts(); // the timer now preempts this idle context into work
|
architecture.enableInterrupts(); // the timer now preempts this idle context into work
|
||||||
@@ -316,7 +456,7 @@ pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
|
|||||||
return ok;
|
return ok;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Spawn a **user** task: a task with its own address space (`aspace`) that starts
|
/// Spawn a **user** task: a task with its own address space (`address_space`) that starts
|
||||||
/// in user mode at `entry` on `user_sp`, recorded under `name` (its argv[0]).
|
/// in user mode at `entry` on `user_sp`, recorded under `name` (its argv[0]).
|
||||||
/// `supervisor` is the id of the spawning process (0 = the kernel) — the kill
|
/// `supervisor` is the id of the spawning process (0 = the kernel) — the kill
|
||||||
/// authority — and `exit_endpoint` (an *ipc.Endpoint whose reference the caller
|
/// authority — and `exit_endpoint` (an *ipc.Endpoint whose reference the caller
|
||||||
@@ -325,19 +465,27 @@ pub fn spawnOn(entry: *const fn () void, priority: Priority, cpu: u32) bool {
|
|||||||
/// lands in `user_task_trampoline`.
|
/// lands in `user_task_trampoline`.
|
||||||
/// Returns the new process id, or null (creating nothing) if the table is full or
|
/// Returns the new process id, or null (creating nothing) if the table is full or
|
||||||
/// out of memory.
|
/// out of memory.
|
||||||
/// **Caller must hold the kernel lock** (the loader that builds `aspace` holds it
|
/// **Caller must hold the kernel lock** (the loader that builds `address_space` holds it
|
||||||
/// across the whole spawn, so the address space and the task appear atomically).
|
/// across the whole spawn, so the address space and the task appear atomically).
|
||||||
pub fn spawnUserLocked(aspace: u64, entry: u64, user_sp: u64, priority: Priority, task_name: []const u8, supervisor: u32, exit_endpoint: ?*anyopaque) ?u32 {
|
pub fn spawnUserLocked(address_space: u64, entry: u64, user_sp: u64, user_arg: u64, priority: Priority, task_name: []const u8, supervisor: u32, exit_endpoint: ?*anyopaque) ?u32 {
|
||||||
const t = freeSlot() orelse return null;
|
const t = freeSlot() orelse return null;
|
||||||
const stack = heap.allocator().alloc(u8, stack_size) catch return null;
|
const stack = heap.allocator().alloc(u8, stack_size) catch return null;
|
||||||
|
// Take this task's reference to the address space before we commit the slot, so a
|
||||||
|
// failure here leaves nothing to unwind (the caller still owns the raw `address_space`).
|
||||||
|
if (!retainAddressSpace(address_space)) {
|
||||||
|
heap.allocator().free(stack);
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
live_stack_bytes += stack.len; // the reaper drops this when the task dies (M8)
|
||||||
t.* = .{
|
t.* = .{
|
||||||
.id = next_id,
|
.id = next_id,
|
||||||
.state = .ready,
|
.state = .ready,
|
||||||
.priority = priority,
|
.priority = priority,
|
||||||
.stack = stack,
|
.stack = stack,
|
||||||
.aspace = aspace,
|
.address_space = address_space,
|
||||||
.user_ip = entry,
|
.user_ip = entry,
|
||||||
.user_sp = user_sp,
|
.user_sp = user_sp,
|
||||||
|
.user_arg = user_arg,
|
||||||
.supervisor = supervisor,
|
.supervisor = supervisor,
|
||||||
.exit_endpoint = exit_endpoint,
|
.exit_endpoint = exit_endpoint,
|
||||||
};
|
};
|
||||||
@@ -363,7 +511,7 @@ fn startUserTask() void {
|
|||||||
// No serial chatter here: this runs on every spawn, unserialized against
|
// No serial chatter here: this runs on every spawn, unserialized against
|
||||||
// user-space writes, and its output used to shear concurrent log lines in
|
// user-space writes, and its output used to shear concurrent log lines in
|
||||||
// half — the largest source of corrupted markers in the QEMU scenarios.
|
// half — the largest source of corrupted markers in the QEMU scenarios.
|
||||||
architecture.jumpToUser(t.user_ip, t.user_sp); // noreturn
|
architecture.jumpToUserArg(t.user_ip, t.user_sp, t.user_arg); // noreturn (arg0 = 0 for a process)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the
|
/// The unlocked task-creation primitive. Caller must hold the kernel lock (or be the
|
||||||
@@ -372,6 +520,7 @@ fn startUserTask() void {
|
|||||||
fn create(entry: *const fn () void, priority: Priority, affinity: ?u32) *Task {
|
fn create(entry: *const fn () void, priority: Priority, affinity: ?u32) *Task {
|
||||||
const t = freeSlot() orelse @panic("sched: task table full");
|
const t = freeSlot() orelse @panic("sched: task table full");
|
||||||
const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack");
|
const stack = heap.allocator().alloc(u8, stack_size) catch @panic("sched: no memory for task stack");
|
||||||
|
live_stack_bytes += stack.len; // the reaper drops this when the task dies (M8)
|
||||||
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack, .affinity = affinity };
|
t.* = .{ .id = next_id, .state = .ready, .priority = priority, .stack = stack, .affinity = affinity };
|
||||||
next_id += 1;
|
next_id += 1;
|
||||||
const top = @intFromPtr(stack.ptr) + stack.len;
|
const top = @intFromPtr(stack.ptr) + stack.len;
|
||||||
@@ -412,7 +561,7 @@ fn schedule() void {
|
|||||||
/// Make `next` this core's running task: publish its kernel stack (TSS.rsp0, so a
|
/// Make `next` this core's running task: publish its kernel stack (TSS.rsp0, so a
|
||||||
/// user-mode interrupt lands on a good stack) and its address space (only when
|
/// user-mode interrupt lands on a good stack) and its address space (only when
|
||||||
/// it differs from what's loaded — every page-table switch is a full TLB flush),
|
/// it differs from what's loaded — every page-table switch is a full TLB flush),
|
||||||
/// then switch registers/stacks. Kernel tasks (aspace == 0, no kstack_top used
|
/// then switch registers/stacks. Kernel tasks (address_space == 0, no kstack_top used
|
||||||
/// from user mode) resolve to the shared kernel page tables and skip the kernel-
|
/// from user mode) resolve to the shared kernel page tables and skip the kernel-
|
||||||
/// stack write, so this is a no-op beyond the register switch for a pure-kernel
|
/// stack write, so this is a no-op beyond the register switch for a pure-kernel
|
||||||
/// workload. The big kernel lock is held and interrupts are off throughout, so no
|
/// workload. The big kernel lock is held and interrupts are off throughout, so no
|
||||||
@@ -420,12 +569,25 @@ fn schedule() void {
|
|||||||
/// `save_sp` receives the outgoing task's stack pointer.
|
/// `save_sp` receives the outgoing task's stack pointer.
|
||||||
fn switchTo(pc: *PerCpu, save_sp: *usize, next: *Task) void {
|
fn switchTo(pc: *PerCpu, save_sp: *usize, next: *Task) void {
|
||||||
if (next.kstack_top != 0) architecture.setKernelStack(pc.index, next.kstack_top);
|
if (next.kstack_top != 0) architecture.setKernelStack(pc.index, next.kstack_top);
|
||||||
const want = if (next.aspace != 0) next.aspace else architecture.kernelPageTable();
|
const want = if (next.address_space != 0) next.address_space else architecture.kernelPageTable();
|
||||||
if (want != pc.loaded_aspace) {
|
if (want != pc.loaded_address_space) {
|
||||||
architecture.loadPageTable(want);
|
architecture.loadPageTable(want);
|
||||||
pc.loaded_aspace = want;
|
pc.loaded_address_space = want;
|
||||||
|
}
|
||||||
|
// Restore the next task's user TLS thread pointer — only on change, the same
|
||||||
|
// conditional-load discipline as CR3 above (docs/threading-plan.md M10).
|
||||||
|
if (next.thread_pointer != pc.loaded_thread_pointer) {
|
||||||
|
architecture.setThreadPointer(next.thread_pointer);
|
||||||
|
pc.loaded_thread_pointer = next.thread_pointer;
|
||||||
}
|
}
|
||||||
architecture.switchContext(save_sp, next.sp);
|
architecture.switchContext(save_sp, next.sp);
|
||||||
|
// Resumed now (switchContext returned into our own switchTo frame). Re-fetch the core
|
||||||
|
// via thisCpu(): the `pc` parameter is from *our* earlier switchTo call, so it names
|
||||||
|
// the core we last ran on — stale if we migrated. switchContext only swaps stacks on
|
||||||
|
// the current core, so thisCpu() is the core the just-dead task died on. If a task
|
||||||
|
// died switching to us, free its kernel stack: we're on ours so it's safe, and the big
|
||||||
|
// lock is still held so its slot can't have been reused (docs/threading-plan.md M8).
|
||||||
|
drainReapListLocked(thisCpu());
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Voluntarily give up the CPU to the next ready task.
|
/// Voluntarily give up the CPU to the next ready task.
|
||||||
@@ -446,6 +608,98 @@ pub fn sleep(ms: u64) void {
|
|||||||
sync.leave(flags);
|
sync.leave(flags);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- futex: block/wake on a user address (docs/threading.md) ----------------
|
||||||
|
//
|
||||||
|
// A futex waiter is not linked into any queue — it is simply a `.blocked` task
|
||||||
|
// tagged with the address it waits on (`futex_addr`). Waking scans the task table
|
||||||
|
// (bounded) for matching waiters. A timed wait also sets `wake_at`, so the timer's
|
||||||
|
// `wakeExpired` can wake it; `futex_addr` stays non-zero in that case, which is how
|
||||||
|
// the waiter tells a timeout from a real wake.
|
||||||
|
|
||||||
|
pub const FutexResult = enum { woken, timed_out };
|
||||||
|
|
||||||
|
/// Block the current task on futex `addr` until woken, or (if `timeout_ms > 0`) the
|
||||||
|
/// deadline. **Precondition:** the big kernel lock is held and the caller has already
|
||||||
|
/// checked, under this same lock, that the futex word equals the expected value — so
|
||||||
|
/// no wake can be missed. Returns with the lock still held.
|
||||||
|
pub fn futexWaitLocked(addr: u64, timeout_ms: u64) FutexResult {
|
||||||
|
const t = current();
|
||||||
|
t.futex_addr = addr;
|
||||||
|
t.wake_at = if (timeout_ms > 0) architecture.millis() + timeout_ms else 0;
|
||||||
|
t.state = .blocked;
|
||||||
|
schedule(); // woken by futexWakeLocked (clears futex_addr) or wakeExpired (timeout)
|
||||||
|
const woken = t.futex_addr == 0;
|
||||||
|
t.futex_addr = 0;
|
||||||
|
t.wake_at = 0;
|
||||||
|
return if (woken) .woken else .timed_out;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Wake up to `count` tasks blocked in `futex_wait` on `addr` in address space
|
||||||
|
/// `address_space`. Precondition: the big kernel lock is held. Returns how many woke.
|
||||||
|
pub fn futexWakeLocked(address_space: u64, addr: u64, count: u32) u32 {
|
||||||
|
var woken: u32 = 0;
|
||||||
|
for (&tasks) |*t| {
|
||||||
|
if (woken >= count) break;
|
||||||
|
if (t.state == .blocked and t.address_space == address_space and t.futex_addr == addr) {
|
||||||
|
t.futex_addr = 0; // the "woken, not timed out" signal to futexWaitLocked
|
||||||
|
t.wake_at = 0;
|
||||||
|
t.state = .ready;
|
||||||
|
enqueue(t);
|
||||||
|
woken += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return woken;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- thread join (docs/threading-plan.md M9) --------------------------------
|
||||||
|
//
|
||||||
|
// join needs no per-thread IPC endpoint: `thread_join(tid)` blocks the caller until the
|
||||||
|
// task with id `tid` has exited, and the exit paths wake any joiner. The caller only ever
|
||||||
|
// reclaims the joined thread's *user* stack (which the thread vacated the moment it entered
|
||||||
|
// the kernel to exit), so waking at exit time — not reap time — is safe.
|
||||||
|
|
||||||
|
/// True if a task with id `tid` is still live (has not exited). Caller holds the lock.
|
||||||
|
fn aliveTid(tid: u32) bool {
|
||||||
|
for (&tasks) |*t| {
|
||||||
|
if (t.id == tid and t.state != .free and t.state != .reaping) return true;
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Block the current task until the task with id `tid` exits (or return at once if it
|
||||||
|
/// already has / never existed). **Precondition:** the big kernel lock is held; returns
|
||||||
|
/// with it still held. Woken by `wakeJoinersLocked`.
|
||||||
|
pub fn joinThreadLocked(tid: u32) void {
|
||||||
|
while (aliveTid(tid)) {
|
||||||
|
const t = current();
|
||||||
|
t.join_target = tid;
|
||||||
|
t.state = .blocked;
|
||||||
|
schedule(); // woken when the joined task exits; lock handed off across the switch
|
||||||
|
t.join_target = 0;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Set the calling task's user TLS thread pointer and load it now. Persisted on the Task so
|
||||||
|
/// context switches restore it (docs/threading-plan.md M10). Caller holds the kernel lock.
|
||||||
|
pub fn setThreadPointerLocked(addr: u64) void {
|
||||||
|
const pc = thisCpu();
|
||||||
|
pc.current.thread_pointer = addr;
|
||||||
|
architecture.setThreadPointer(addr);
|
||||||
|
pc.loaded_thread_pointer = addr;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Wake every task blocked in `thread_join` on `tid` — called from the exit paths once the
|
||||||
|
/// exiting task's state is `.free`. Caller holds the lock.
|
||||||
|
fn wakeJoinersLocked(tid: u32) void {
|
||||||
|
for (&tasks) |*t| {
|
||||||
|
if (t.state == .blocked and t.join_target == tid) {
|
||||||
|
t.join_target = 0;
|
||||||
|
t.state = .ready;
|
||||||
|
enqueue(t);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// --- event-based blocking -------------------------------------------------
|
// --- event-based blocking -------------------------------------------------
|
||||||
//
|
//
|
||||||
// A WaitQueue is a set of tasks blocked waiting for something (a resource, a
|
// A WaitQueue is a set of tasks blocked waiting for something (a resource, a
|
||||||
@@ -547,7 +801,7 @@ fn removeFrom(head: *[number_priorities]?*Task, tail: *[number_priorities]?*Task
|
|||||||
/// Precondition: the big kernel lock is held.
|
/// Precondition: the big kernel lock is held.
|
||||||
pub fn taskByIdLocked(id: u32) ?*Task {
|
pub fn taskByIdLocked(id: u32) ?*Task {
|
||||||
for (&tasks) |*t| {
|
for (&tasks) |*t| {
|
||||||
if (t.state != .free and t.id == id) return t;
|
if (t.state != .free and t.state != .reaping and t.id == id) return t;
|
||||||
}
|
}
|
||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
@@ -557,7 +811,7 @@ pub fn taskByIdLocked(id: u32) ?*Task {
|
|||||||
/// Precondition: the big kernel lock is held.
|
/// Precondition: the big kernel lock is held.
|
||||||
pub fn forgetIpcClientLocked(t: *Task) void {
|
pub fn forgetIpcClientLocked(t: *Task) void {
|
||||||
for (&tasks) |*other| {
|
for (&tasks) |*other| {
|
||||||
if (other.state != .free and other.ipc_client == t) other.ipc_client = null;
|
if (other.state != .free and other.state != .reaping and other.ipc_client == t) other.ipc_client = null;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -649,7 +903,11 @@ pub var reap_task_hook: ?*const fn (*Task) void = null;
|
|||||||
fn reapKillPendingLocked() void {
|
fn reapKillPendingLocked() void {
|
||||||
const pc = thisCpu();
|
const pc = thisCpu();
|
||||||
const cur = pc.current;
|
const cur = pc.current;
|
||||||
if (cur.kill_pending and cur.aspace != 0 and !cur.in_system_call) {
|
// Safety net: if a dying task switched to a *fresh* task (which enters via
|
||||||
|
// task_trampoline, not switchTo's tail), its stack is still queued here. The dying
|
||||||
|
// task switched away before this tick, so it is off its stack — drain now (M8).
|
||||||
|
drainReapListLocked(pc);
|
||||||
|
if (cur.kill_pending and cur.address_space != 0 and !cur.in_system_call) {
|
||||||
if (terminate_current_hook) |hook| hook(); // noreturn
|
if (terminate_current_hook) |hook| hook(); // noreturn
|
||||||
}
|
}
|
||||||
if (reap_task_hook) |hook| {
|
if (reap_task_hook) |hook| {
|
||||||
@@ -690,7 +948,11 @@ pub fn setPreemption(enabled: bool) void {
|
|||||||
pub fn exit() noreturn {
|
pub fn exit() noreturn {
|
||||||
_ = sync.enter();
|
_ = sync.enter();
|
||||||
const pc = thisCpu();
|
const pc = thisCpu();
|
||||||
pc.current.state = .free;
|
// Queue this task for reaping: `.reaping` keeps its slot out of freeSlot until its
|
||||||
|
// stack is freed; `next` links it on the core's reap list (M8/M9).
|
||||||
|
pc.current.state = .reaping;
|
||||||
|
pc.current.next = pc.reap_list;
|
||||||
|
pc.reap_list = pc.current;
|
||||||
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
|
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
|
||||||
next.state = .running;
|
next.state = .running;
|
||||||
pc.current = next;
|
pc.current = next;
|
||||||
@@ -715,17 +977,21 @@ pub fn exitUser() noreturn {
|
|||||||
pub fn exitUserLocked() noreturn {
|
pub fn exitUserLocked() noreturn {
|
||||||
const pc = thisCpu();
|
const pc = thisCpu();
|
||||||
const dying = pc.current;
|
const dying = pc.current;
|
||||||
const as = dying.aspace;
|
const as = dying.address_space;
|
||||||
if (as != 0) {
|
if (as != 0) {
|
||||||
const kroot = architecture.kernelPageTable();
|
const kroot = architecture.kernelPageTable();
|
||||||
architecture.loadPageTable(kroot); // off the process tables before freeing them
|
architecture.loadPageTable(kroot); // off the process tables before freeing them
|
||||||
pc.loaded_aspace = kroot;
|
pc.loaded_address_space = kroot;
|
||||||
architecture.destroyAddressSpace(as);
|
releaseAddressSpace(as); // destroys only when this was the last task on the space
|
||||||
}
|
}
|
||||||
dying.state = .free;
|
dying.state = .reaping; // dead but its slot stays reserved until the stack is freed
|
||||||
dying.aspace = 0;
|
wakeJoinersLocked(dying.id); // let any thread_join(dying.id) return (M9)
|
||||||
|
dying.address_space = 0;
|
||||||
dying.kill_pending = false;
|
dying.kill_pending = false;
|
||||||
dying.in_system_call = false;
|
dying.in_system_call = false;
|
||||||
|
// Queue for reaping: the task we switch to (or the next tick) frees this stack (M8/M9).
|
||||||
|
dying.next = pc.reap_list;
|
||||||
|
pc.reap_list = dying;
|
||||||
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
|
const next = dequeueHighest(pc) orelse @panic("sched: no task left to run");
|
||||||
next.state = .running;
|
next.state = .running;
|
||||||
pc.current = next;
|
pc.current = next;
|
||||||
@@ -741,12 +1007,14 @@ pub fn exitUserLocked() noreturn {
|
|||||||
/// task isn't running). The kernel stack is leaked, as in `exitUser` (no reaper
|
/// task isn't running). The kernel stack is leaked, as in `exitUser` (no reaper
|
||||||
/// yet). Precondition: the big kernel lock is held.
|
/// yet). Precondition: the big kernel lock is held.
|
||||||
pub fn destroyTaskLocked(t: *Task) void {
|
pub fn destroyTaskLocked(t: *Task) void {
|
||||||
if (t.aspace != 0) architecture.destroyAddressSpace(t.aspace);
|
if (t.address_space != 0) releaseAddressSpace(t.address_space); // destroys only on the last reference
|
||||||
t.aspace = 0;
|
reapStackLocked(t); // safe to free now: `t` is not running on any core (M8)
|
||||||
|
t.address_space = 0;
|
||||||
t.kill_pending = false;
|
t.kill_pending = false;
|
||||||
t.in_system_call = false;
|
t.in_system_call = false;
|
||||||
t.wake_at = 0;
|
t.wake_at = 0;
|
||||||
t.state = .free;
|
t.state = .free;
|
||||||
|
wakeJoinersLocked(t.id); // a killed thread's joiners must return too (M9)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Snapshot the task table into `out` (up to its length), returning the total
|
/// Snapshot the task table into `out` (up to its length), returning the total
|
||||||
@@ -760,7 +1028,7 @@ pub fn enumerate(out: []abi.ProcessDescriptor) u64 {
|
|||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
var total: u64 = 0;
|
var total: u64 = 0;
|
||||||
for (&tasks) |*t| {
|
for (&tasks) |*t| {
|
||||||
if (t.state == .free) continue;
|
if (t.state == .free or t.state == .reaping) continue; // reaping = already exited
|
||||||
if (total < out.len) {
|
if (total < out.len) {
|
||||||
const d = &out[total];
|
const d = &out[total];
|
||||||
d.* = .{
|
d.* = .{
|
||||||
@@ -770,7 +1038,7 @@ pub fn enumerate(out: []abi.ProcessDescriptor) u64 {
|
|||||||
.ready => .ready,
|
.ready => .ready,
|
||||||
.running => .running,
|
.running => .running,
|
||||||
.blocked => .blocked,
|
.blocked => .blocked,
|
||||||
.free => unreachable,
|
.free, .reaping => unreachable,
|
||||||
})),
|
})),
|
||||||
.priority = t.priority,
|
.priority = t.priority,
|
||||||
.name_length = t.name_length,
|
.name_length = t.name_length,
|
||||||
@@ -784,7 +1052,7 @@ pub fn enumerate(out: []abi.ProcessDescriptor) u64 {
|
|||||||
|
|
||||||
/// Whether the running task is a user process (has its own address space).
|
/// Whether the running task is a user process (has its own address space).
|
||||||
pub fn currentIsUserProcess() bool {
|
pub fn currentIsUserProcess() bool {
|
||||||
return current().aspace != 0;
|
return current().address_space != 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn currentId() u32 {
|
pub fn currentId() u32 {
|
||||||
|
|||||||
+531
-33
@@ -101,6 +101,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
|||||||
displayServiceTest(boot_information);
|
displayServiceTest(boot_information);
|
||||||
} else if (eql(case, "display-demo")) {
|
} else if (eql(case, "display-demo")) {
|
||||||
displayDemoTest(boot_information);
|
displayDemoTest(boot_information);
|
||||||
|
} else if (eql(case, "display-cursor")) {
|
||||||
|
displayCursorTest(boot_information);
|
||||||
} else if (eql(case, "shm")) {
|
} else if (eql(case, "shm")) {
|
||||||
shmTest(boot_information);
|
shmTest(boot_information);
|
||||||
} else if (eql(case, "virtio-gpu")) {
|
} else if (eql(case, "virtio-gpu")) {
|
||||||
@@ -139,6 +141,26 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
|||||||
userPfTest();
|
userPfTest();
|
||||||
} else if (eql(case, "fault-recovery")) {
|
} else if (eql(case, "fault-recovery")) {
|
||||||
faultRecoveryTest(boot_information);
|
faultRecoveryTest(boot_information);
|
||||||
|
} else if (eql(case, "address-space-refcount")) {
|
||||||
|
addressSpaceRefcountTest(boot_information);
|
||||||
|
} else if (eql(case, "thread-spawn")) {
|
||||||
|
threadSpawnTest(boot_information);
|
||||||
|
} else if (eql(case, "thread-join")) {
|
||||||
|
threadJoinTest(boot_information);
|
||||||
|
} else if (eql(case, "thread-futex")) {
|
||||||
|
threadFutexTest(boot_information);
|
||||||
|
} else if (eql(case, "thread-mutex")) {
|
||||||
|
threadMutexTest(boot_information);
|
||||||
|
} else if (eql(case, "thread-id")) {
|
||||||
|
threadIdTest(boot_information);
|
||||||
|
} else if (eql(case, "thread-alloc")) {
|
||||||
|
threadAllocTest(boot_information);
|
||||||
|
} else if (eql(case, "task-reap")) {
|
||||||
|
taskReapTest(boot_information);
|
||||||
|
} else if (eql(case, "thread-tls")) {
|
||||||
|
threadTlsTest(boot_information);
|
||||||
|
} else if (eql(case, "thread-rwlock")) {
|
||||||
|
threadRwlockTest(boot_information);
|
||||||
} else if (eql(case, "args")) {
|
} else if (eql(case, "args")) {
|
||||||
argsTest(boot_information);
|
argsTest(boot_information);
|
||||||
} else if (eql(case, "init")) {
|
} else if (eql(case, "init")) {
|
||||||
@@ -767,11 +789,15 @@ fn affinityTest() void {
|
|||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
var spins: u64 = 0;
|
// Let many time slices pass so the scheduler runs the pinned worker across ticks.
|
||||||
while (spins < 3_000_000_000) spins +%= 1; // many time slices across the cores
|
// Wait on the wall clock, not a raw iteration count: a fixed-count busy-loop's
|
||||||
|
// wall-time is a codegen lottery (the optimiser may elide or vectorise it), so an
|
||||||
|
// unrelated change elsewhere in this file could swing this test from ~4 s to ~50 s.
|
||||||
|
const run_until = architecture.millis() + 400;
|
||||||
|
while (architecture.millis() < run_until) {}
|
||||||
affinity_running = false;
|
affinity_running = false;
|
||||||
var settle: u64 = 0;
|
const settle_until = architecture.millis() + 50;
|
||||||
while (settle < 200_000_000) settle +%= 1; // let the worker see the flag and exit
|
while (architecture.millis() < settle_until) {} // let the worker see the flag and exit
|
||||||
|
|
||||||
var others: u32 = 0;
|
var others: u32 = 0;
|
||||||
for (affinity_cores, 0..) |seen, c| {
|
for (affinity_cores, 0..) |seen, c| {
|
||||||
@@ -896,12 +922,12 @@ fn userMemTest() void {
|
|||||||
log("DANOS-TEST-BEGIN: usermem\n", .{});
|
log("DANOS-TEST-BEGIN: usermem\n", .{});
|
||||||
const base_free = pmm.stats().free_frames;
|
const base_free = pmm.stats().free_frames;
|
||||||
|
|
||||||
const aspace = architecture.createAddressSpace() orelse {
|
const address_space = architecture.createAddressSpace() orelse {
|
||||||
check("created a fresh address space", false);
|
check("created a fresh address space", false);
|
||||||
result();
|
result();
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
check("created a fresh address space", aspace != 0);
|
check("created a fresh address space", address_space != 0);
|
||||||
|
|
||||||
// Grant three pages into the arena, mapped RW + NX (the mmap contract).
|
// Grant three pages into the arena, mapped RW + NX (the mmap contract).
|
||||||
const npages = 3;
|
const npages = 3;
|
||||||
@@ -910,7 +936,7 @@ fn userMemTest() void {
|
|||||||
var mapped: usize = 0;
|
var mapped: usize = 0;
|
||||||
while (mapped < npages) : (mapped += 1) {
|
while (mapped < npages) : (mapped += 1) {
|
||||||
frames[mapped] = pmm.alloc() orelse break;
|
frames[mapped] = pmm.alloc() orelse break;
|
||||||
architecture.mapUserPageInto(aspace, arena + mapped * abi.page_size, frames[mapped], true, false);
|
architecture.mapUserPageInto(address_space, arena + mapped * abi.page_size, frames[mapped], true, false);
|
||||||
}
|
}
|
||||||
check("granted three user pages", mapped == npages);
|
check("granted three user pages", mapped == npages);
|
||||||
|
|
||||||
@@ -919,7 +945,7 @@ fn userMemTest() void {
|
|||||||
var rw_ok = true;
|
var rw_ok = true;
|
||||||
for (0..npages) |i| {
|
for (0..npages) |i| {
|
||||||
const va = arena + i * abi.page_size;
|
const va = arena + i * abi.page_size;
|
||||||
const physical = architecture.translate(aspace, va) orelse {
|
const physical = architecture.translate(address_space, va) orelse {
|
||||||
translate_ok = false;
|
translate_ok = false;
|
||||||
continue;
|
continue;
|
||||||
};
|
};
|
||||||
@@ -934,13 +960,13 @@ fn userMemTest() void {
|
|||||||
// Release them the way munmap does, then tear down the address space.
|
// Release them the way munmap does, then tear down the address space.
|
||||||
for (0..npages) |i| {
|
for (0..npages) |i| {
|
||||||
const va = arena + i * abi.page_size;
|
const va = arena + i * abi.page_size;
|
||||||
if (architecture.translate(aspace, va)) |physical| {
|
if (architecture.translate(address_space, va)) |physical| {
|
||||||
architecture.unmapUserPageInto(aspace, va);
|
architecture.unmapUserPageInto(address_space, va);
|
||||||
pmm.free(physical);
|
pmm.free(physical);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
check("munmap unmapped every grant", architecture.translate(aspace, arena) == null);
|
check("munmap unmapped every grant", architecture.translate(address_space, arena) == null);
|
||||||
architecture.destroyAddressSpace(aspace);
|
architecture.destroyAddressSpace(address_space);
|
||||||
|
|
||||||
check("no frames leaked (free count restored)", pmm.stats().free_frames == base_free);
|
check("no frames leaked (free count restored)", pmm.stats().free_frames == base_free);
|
||||||
result();
|
result();
|
||||||
@@ -1108,12 +1134,12 @@ fn dmaTest() void {
|
|||||||
|
|
||||||
// Map the run into a fresh address space as coherent DMA and translate each page
|
// Map the run into a fresh address space as coherent DMA and translate each page
|
||||||
// back: the same physical run, in order — proving contiguity and the mapping.
|
// back: the same physical run, in order — proving contiguity and the mapping.
|
||||||
const aspace = architecture.createAddressSpace().?;
|
const address_space = architecture.createAddressSpace().?;
|
||||||
architecture.mapUserDmaInto(aspace, process.dma_arena_base, phys, frames * abi.page_size);
|
architecture.mapUserDmaInto(address_space, process.dma_arena_base, phys, frames * abi.page_size);
|
||||||
var mapped_ok = true;
|
var mapped_ok = true;
|
||||||
for (0..frames) |i| {
|
for (0..frames) |i| {
|
||||||
const va = process.dma_arena_base + i * abi.page_size;
|
const va = process.dma_arena_base + i * abi.page_size;
|
||||||
const got = architecture.translate(aspace, va) orelse {
|
const got = architecture.translate(address_space, va) orelse {
|
||||||
mapped_ok = false;
|
mapped_ok = false;
|
||||||
break;
|
break;
|
||||||
};
|
};
|
||||||
@@ -1123,7 +1149,7 @@ fn dmaTest() void {
|
|||||||
|
|
||||||
// Teardown must reclaim the DMA RAM (the leaves carry no device_grant, so
|
// Teardown must reclaim the DMA RAM (the leaves carry no device_grant, so
|
||||||
// freeSubtree frees them as ordinary frames) — a driver that just dies leaks none.
|
// freeSubtree frees them as ordinary frames) — a driver that just dies leaks none.
|
||||||
architecture.destroyAddressSpace(aspace);
|
architecture.destroyAddressSpace(address_space);
|
||||||
for (0..2) |i| pmm.free(low + i * abi.page_size);
|
for (0..2) |i| pmm.free(low + i * abi.page_size);
|
||||||
check("no frames leaked after DMA teardown", pmm.stats().free_frames == base_free);
|
check("no frames leaked after DMA teardown", pmm.stats().free_frames == base_free);
|
||||||
result();
|
result();
|
||||||
@@ -1355,9 +1381,9 @@ fn spawnFaultingProcess() ?u32 {
|
|||||||
const flags = sync.enter();
|
const flags = sync.enter();
|
||||||
defer sync.leave(flags);
|
defer sync.leave(flags);
|
||||||
|
|
||||||
const aspace = architecture.createAddressSpace() orelse return null;
|
const address_space = architecture.createAddressSpace() orelse return null;
|
||||||
const code_frame = pmm.alloc() orelse {
|
const code_frame = pmm.alloc() orelse {
|
||||||
architecture.destroyAddressSpace(aspace);
|
architecture.destroyAddressSpace(address_space);
|
||||||
return null;
|
return null;
|
||||||
};
|
};
|
||||||
// Fill through the physmap (the user mapping is read-only); pad with int3 so a
|
// Fill through the physmap (the user mapping is read-only); pad with int3 so a
|
||||||
@@ -1365,17 +1391,17 @@ fn spawnFaultingProcess() ?u32 {
|
|||||||
const code: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(code_frame));
|
const code: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(code_frame));
|
||||||
@memset(code[0..abi.page_size], 0xCC);
|
@memset(code[0..abi.page_size], 0xCC);
|
||||||
@memcpy(code[0..blob.len], blob);
|
@memcpy(code[0..blob.len], blob);
|
||||||
architecture.mapUserPageInto(aspace, process.code_virtual, code_frame, false, true); // RO + X
|
architecture.mapUserPageInto(address_space, process.code_virtual, code_frame, false, true); // RO + X
|
||||||
|
|
||||||
const stack_frame = pmm.alloc() orelse {
|
const stack_frame = pmm.alloc() orelse {
|
||||||
architecture.destroyAddressSpace(aspace); // frees code_frame too — it's mapped
|
architecture.destroyAddressSpace(address_space); // frees code_frame too — it's mapped
|
||||||
return null;
|
return null;
|
||||||
};
|
};
|
||||||
architecture.mapUserPageInto(aspace, process.stack_base_virtual, stack_frame, true, false); // RW + NX
|
architecture.mapUserPageInto(address_space, process.stack_base_virtual, stack_frame, true, false); // RW + NX
|
||||||
|
|
||||||
// Supervised by the calling test task, so exitReasonOf can read the verdict.
|
// Supervised by the calling test task, so exitReasonOf can read the verdict.
|
||||||
const id = scheduler.spawnUserLocked(aspace, process.code_virtual, process.stack_base_virtual + abi.page_size, 4, "fault-probe", scheduler.currentId(), null) orelse {
|
const id = scheduler.spawnUserLocked(address_space, process.code_virtual, process.stack_base_virtual + abi.page_size, 0, 4, "fault-probe", scheduler.currentId(), null) orelse {
|
||||||
architecture.destroyAddressSpace(aspace);
|
architecture.destroyAddressSpace(address_space);
|
||||||
return null;
|
return null;
|
||||||
};
|
};
|
||||||
return id;
|
return id;
|
||||||
@@ -1429,6 +1455,420 @@ fn faultRecoveryTest(boot_information: *const BootInformation) void {
|
|||||||
result();
|
result();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Address-space refcount (docs/threading-plan.md M1): every process holds exactly one
|
||||||
|
/// reference to its address space, released when it dies, so `destroyAddressSpace` runs
|
||||||
|
/// exactly once per space — no leak, no double-free. Spawn and kill several ring-3
|
||||||
|
/// processes (the faulting probe, reaped by the kernel) and confirm the count of live
|
||||||
|
/// address spaces returns to baseline while destructions advance by exactly that many.
|
||||||
|
/// This is the foundation threads (shared address spaces) build on: the refactor must be
|
||||||
|
/// invisible while every space still has exactly one task.
|
||||||
|
fn addressSpaceRefcountTest(boot_information: *const BootInformation) void {
|
||||||
|
_ = boot_information;
|
||||||
|
log("DANOS-TEST-BEGIN: address-space-refcount\n", .{});
|
||||||
|
const base_live = scheduler.liveAddressSpaceCount();
|
||||||
|
const base_destroyed = scheduler.addressSpaceDestroyCount();
|
||||||
|
const rounds: u32 = 5;
|
||||||
|
var killed: u32 = 0;
|
||||||
|
var round: u32 = 0;
|
||||||
|
while (round < rounds) : (round += 1) {
|
||||||
|
process.fault_kill_count = 0;
|
||||||
|
const probe = spawnFaultingProcess() orelse break;
|
||||||
|
_ = probe;
|
||||||
|
// Let the probe fault on its first instruction and be reaped.
|
||||||
|
scheduler.setPriority(1);
|
||||||
|
const deadline = architecture.millis() + 5000;
|
||||||
|
while (process.fault_kill_count < 1 and architecture.millis() < deadline) scheduler.yield();
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
if (process.fault_kill_count >= 1) killed += 1;
|
||||||
|
}
|
||||||
|
check("all probes spawned and were killed", killed == rounds);
|
||||||
|
check("live address-space count returned to baseline", scheduler.liveAddressSpaceCount() == base_live);
|
||||||
|
check("each address space destroyed exactly once", scheduler.addressSpaceDestroyCount() == base_destroyed + rounds);
|
||||||
|
if (killed == rounds and scheduler.liveAddressSpaceCount() == base_live and
|
||||||
|
scheduler.addressSpaceDestroyCount() == base_destroyed + rounds)
|
||||||
|
log("address-space-refcount: spaces released to baseline ok\n", .{});
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Thread spawn (docs/threading-plan.md M2): the `thread-test` service spawns a worker
|
||||||
|
/// thread that writes a shared global; the main thread, polling that memory, observes the
|
||||||
|
/// write — proving `runtime.Thread.spawn` started a task in the **same** address space
|
||||||
|
/// (a separate process could not touch it). The service's own marker is the verdict.
|
||||||
|
fn threadSpawnTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: thread-spawn\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
check("thread-test spawned", spawnNamed(rd, "thread-test"));
|
||||||
|
|
||||||
|
// Wait for the service's verdict marker (it polls shared memory the worker wrote).
|
||||||
|
const ok_marker = "thread-test: child ran in shared address space ok";
|
||||||
|
const fail_marker = "thread-test: FAIL";
|
||||||
|
scheduler.setPriority(1);
|
||||||
|
const deadline = architecture.millis() + 12000;
|
||||||
|
while (architecture.millis() < deadline) {
|
||||||
|
if (bufferHas(ok_marker) or bufferHas(fail_marker)) break;
|
||||||
|
scheduler.yield();
|
||||||
|
}
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
|
||||||
|
check("a worker thread ran in the shared address space (shared write observed)", bufferHas(ok_marker));
|
||||||
|
check("the thread path reported no failure", !bufferHas(fail_marker));
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Thread join + parallelism (docs/threading-plan.md M3): `thread-test` in join mode
|
||||||
|
/// spawns N workers that each do K atomic increments on a shared counter and stamp the
|
||||||
|
/// core they ran on; it `join`s all N and asserts the total is exactly N*K (every worker
|
||||||
|
/// ran, join waited for each) and that >1 core was used (genuine parallelism), then a
|
||||||
|
/// detached worker proves `detach`. Its single verdict marker is the case result.
|
||||||
|
fn threadJoinTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: thread-join\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
// Spawn thread-test in join mode (argv selects the mode).
|
||||||
|
var started = false;
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < rd.count) : (i += 1) {
|
||||||
|
const item = rd.entry(i) orelse continue;
|
||||||
|
if (!eql(item.name, "thread-test")) continue;
|
||||||
|
started = if (process.spawnProcess(item.blob, 4, &.{ "thread-test", "join" })) true else |_| false;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
check("thread-test (join mode) spawned", started);
|
||||||
|
|
||||||
|
const ok_marker = "thread-test: join ok";
|
||||||
|
const fail_marker = "thread-test: FAIL";
|
||||||
|
scheduler.setPriority(1);
|
||||||
|
const deadline = architecture.millis() + 15000;
|
||||||
|
while (architecture.millis() < deadline) {
|
||||||
|
if (bufferHas(ok_marker) or bufferHas(fail_marker)) break;
|
||||||
|
scheduler.yield();
|
||||||
|
}
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
|
||||||
|
check("N worker threads joined; counter exact (N*K) and >1 core used", bufferHas(ok_marker));
|
||||||
|
check("no thread failure reported", !bufferHas(fail_marker));
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Futex (docs/threading-plan.md M4): `thread-test` in futex mode has a waiter thread
|
||||||
|
/// block in `futex_wait` on a word; the main thread publishes the word and `futex_wake`s
|
||||||
|
/// it. The serial order `waiting → waking → woke` shows the kernel handoff (the waiter
|
||||||
|
/// parked and was woken, not spun), and a `timedWait` on an unwoken word reports a
|
||||||
|
/// timeout. The verdict marker is emitted only after both hold.
|
||||||
|
fn threadFutexTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: thread-futex\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
var started = false;
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < rd.count) : (i += 1) {
|
||||||
|
const item = rd.entry(i) orelse continue;
|
||||||
|
if (!eql(item.name, "thread-test")) continue;
|
||||||
|
started = if (process.spawnProcess(item.blob, 4, &.{ "thread-test", "futex" })) true else |_| false;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
check("thread-test (futex mode) spawned", started);
|
||||||
|
|
||||||
|
const ok_marker = "thread-futex: ok";
|
||||||
|
const fail_marker = "thread-futex: FAIL";
|
||||||
|
scheduler.setPriority(1);
|
||||||
|
const deadline = architecture.millis() + 15000;
|
||||||
|
while (architecture.millis() < deadline) {
|
||||||
|
if (bufferHas(ok_marker) or bufferHas(fail_marker)) break;
|
||||||
|
scheduler.yield();
|
||||||
|
}
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
|
||||||
|
// Only the freshest marker is checked here — the kernel's log ring buffer may have
|
||||||
|
// evicted the earlier ones by now. The waiting/waking/woke ordering (the handoff
|
||||||
|
// proof) is asserted against the full serial stream by the qemu case's regex; the
|
||||||
|
// verdict marker is emitted by thread-test only after the wake AND the timeout hold.
|
||||||
|
check("futex handoff + timeout completed (verdict reached, no failure)", bufferHas(ok_marker) and !bufferHas(fail_marker));
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Mutex + Condition (docs/threading-plan.md M5): `thread-test` in mutex mode runs a
|
||||||
|
/// bounded producer/consumer — P producers and C consumers over one `Mutex` and two
|
||||||
|
/// `Condition`s move N unique items through a small ring. Every item is produced once;
|
||||||
|
/// if the lock and condition variables are correct under real cross-core contention,
|
||||||
|
/// the consumed checksum and tally match exactly (no lost or duplicated item, no
|
||||||
|
/// overrun). The verdict marker is emitted only when both match.
|
||||||
|
fn threadMutexTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: thread-mutex\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
var started = false;
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < rd.count) : (i += 1) {
|
||||||
|
const item = rd.entry(i) orelse continue;
|
||||||
|
if (!eql(item.name, "thread-test")) continue;
|
||||||
|
started = if (process.spawnProcess(item.blob, 4, &.{ "thread-test", "mutex" })) true else |_| false;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
check("thread-test (mutex mode) spawned", started);
|
||||||
|
|
||||||
|
const ok_marker = "thread-mutex: ok";
|
||||||
|
const fail_marker = "thread-mutex: FAIL";
|
||||||
|
scheduler.setPriority(1);
|
||||||
|
const deadline = architecture.millis() + 20000;
|
||||||
|
while (architecture.millis() < deadline) {
|
||||||
|
if (bufferHas(ok_marker) or bufferHas(fail_marker)) break;
|
||||||
|
scheduler.yield();
|
||||||
|
}
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
|
||||||
|
check("producer/consumer over Mutex+Condition moved every item exactly once", bufferHas(ok_marker) and !bufferHas(fail_marker));
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Thread identity (docs/threading-plan.md M6): `thread-test` in id mode spawns two
|
||||||
|
/// workers that each read `runtime.Thread.getCurrentId`; the main thread confirms all
|
||||||
|
/// three ids are non-zero and distinct — proof each thread has its own kernel identity.
|
||||||
|
fn threadIdTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: thread-id\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
var started = false;
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < rd.count) : (i += 1) {
|
||||||
|
const item = rd.entry(i) orelse continue;
|
||||||
|
if (!eql(item.name, "thread-test")) continue;
|
||||||
|
started = if (process.spawnProcess(item.blob, 4, &.{ "thread-test", "id" })) true else |_| false;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
check("thread-test (id mode) spawned", started);
|
||||||
|
|
||||||
|
const ok_marker = "thread-id: ok";
|
||||||
|
const fail_marker = "thread-id: FAIL";
|
||||||
|
scheduler.setPriority(1);
|
||||||
|
const deadline = architecture.millis() + 12000;
|
||||||
|
while (architecture.millis() < deadline) {
|
||||||
|
if (bufferHas(ok_marker) or bufferHas(fail_marker)) break;
|
||||||
|
scheduler.yield();
|
||||||
|
}
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
|
||||||
|
check("each thread has a distinct, non-zero getCurrentId", bufferHas(ok_marker) and !bufferHas(fail_marker));
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Thread-safe allocation (docs/threading-plan.md M7): `thread-test` in alloc mode runs N
|
||||||
|
/// threads that each do many `alloc`/fill/verify/`free` cycles of varied sizes on the
|
||||||
|
/// shared runtime heap. If the heap lock or the per-address-space mmap arena were unsafe,
|
||||||
|
/// two threads' blocks would overlap and a thread would read another's pattern; the
|
||||||
|
/// verdict marker is emitted only when every thread completes with every block intact.
|
||||||
|
fn threadAllocTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: thread-alloc\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
var started = false;
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < rd.count) : (i += 1) {
|
||||||
|
const item = rd.entry(i) orelse continue;
|
||||||
|
if (!eql(item.name, "thread-test")) continue;
|
||||||
|
started = if (process.spawnProcess(item.blob, 4, &.{ "thread-test", "alloc" })) true else |_| false;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
check("thread-test (alloc mode) spawned", started);
|
||||||
|
|
||||||
|
const ok_marker = "thread-alloc: ok";
|
||||||
|
const fail_marker = "thread-alloc: FAIL";
|
||||||
|
scheduler.setPriority(1);
|
||||||
|
const deadline = architecture.millis() + 20000;
|
||||||
|
while (architecture.millis() < deadline) {
|
||||||
|
if (bufferHas(ok_marker) or bufferHas(fail_marker)) break;
|
||||||
|
scheduler.yield();
|
||||||
|
}
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
|
||||||
|
check("concurrent heap allocation stayed corruption-free (shared heap + per-address-space arena)", bufferHas(ok_marker) and !bufferHas(fail_marker));
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Per-thread TLS / FS base (docs/threading-plan.md M10): `thread-test` in tls mode has two
|
||||||
|
/// threads each set their own FS base and write a unique marker to `%fs:8`, then — after
|
||||||
|
/// both have written — read it back. If the FS base were not per-thread and restored across
|
||||||
|
/// context switches, the second write would clobber the first and a thread would read the
|
||||||
|
/// wrong marker. The verdict marker means both read their own value (no cross-talk).
|
||||||
|
fn threadTlsTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: thread-tls\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
var started = false;
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < rd.count) : (i += 1) {
|
||||||
|
const item = rd.entry(i) orelse continue;
|
||||||
|
if (!eql(item.name, "thread-test")) continue;
|
||||||
|
started = if (process.spawnProcess(item.blob, 4, &.{ "thread-test", "tls" })) true else |_| false;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
check("thread-test (tls mode) spawned", started);
|
||||||
|
|
||||||
|
const ok_marker = "thread-tls: ok";
|
||||||
|
const fail_marker = "thread-tls: FAIL";
|
||||||
|
scheduler.setPriority(1);
|
||||||
|
const deadline = architecture.millis() + 12000;
|
||||||
|
while (architecture.millis() < deadline) {
|
||||||
|
if (bufferHas(ok_marker) or bufferHas(fail_marker)) break;
|
||||||
|
scheduler.yield();
|
||||||
|
}
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
|
||||||
|
check("each thread has its own FS-base TLS slot (no cross-talk across switches)", bufferHas(ok_marker) and !bufferHas(fail_marker));
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// RwLock (docs/threading-plan.md M11): `thread-test` in rwlock mode runs writers that set
|
||||||
|
/// two halves of a value under the exclusive lock and readers that check the halves match
|
||||||
|
/// under the shared lock. If the reader/writer lock were wrong, a reader would observe a
|
||||||
|
/// half-written value; zero violations across many reads → the lock holds.
|
||||||
|
fn threadRwlockTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: thread-rwlock\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
var started = false;
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < rd.count) : (i += 1) {
|
||||||
|
const item = rd.entry(i) orelse continue;
|
||||||
|
if (!eql(item.name, "thread-test")) continue;
|
||||||
|
started = if (process.spawnProcess(item.blob, 4, &.{ "thread-test", "rwlock" })) true else |_| false;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
check("thread-test (rwlock mode) spawned", started);
|
||||||
|
|
||||||
|
const ok_marker = "thread-rwlock: ok";
|
||||||
|
const fail_marker = "thread-rwlock: FAIL";
|
||||||
|
scheduler.setPriority(1);
|
||||||
|
const deadline = architecture.millis() + 20000;
|
||||||
|
while (architecture.millis() < deadline) {
|
||||||
|
if (bufferHas(ok_marker) or bufferHas(fail_marker)) break;
|
||||||
|
scheduler.yield();
|
||||||
|
}
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
|
||||||
|
check("readers/writers over an RwLock never observed a half-written value", bufferHas(ok_marker) and !bufferHas(fail_marker));
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The task reaper (docs/threading-plan.md M8): a dead task's kernel stack used to be
|
||||||
|
/// leaked ("no reaper yet"). Spawn and kill many ring-3 processes and confirm the total
|
||||||
|
/// kernel-stack bytes return to baseline — every stack reclaimed, no leak. (Threads exit
|
||||||
|
/// through the same exitUserLocked path, so this covers them too.)
|
||||||
|
fn taskReapTest(boot_information: *const BootInformation) void {
|
||||||
|
_ = boot_information;
|
||||||
|
log("DANOS-TEST-BEGIN: task-reap\n", .{});
|
||||||
|
const base = scheduler.liveStackBytes();
|
||||||
|
const rounds: u32 = 12;
|
||||||
|
var killed: u32 = 0;
|
||||||
|
var round: u32 = 0;
|
||||||
|
while (round < rounds) : (round += 1) {
|
||||||
|
process.fault_kill_count = 0;
|
||||||
|
const probe = spawnFaultingProcess() orelse break;
|
||||||
|
_ = probe;
|
||||||
|
scheduler.setPriority(1);
|
||||||
|
const deadline = architecture.millis() + 5000;
|
||||||
|
while (process.fault_kill_count < 1 and architecture.millis() < deadline) scheduler.yield();
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
if (process.fault_kill_count >= 1) killed += 1;
|
||||||
|
}
|
||||||
|
// Reaping is asynchronous — a dead task's stack is freed when its core next switches
|
||||||
|
// or ticks. Poll (bounded) until the live bytes return to baseline: a correct reaper
|
||||||
|
// gets there in a few ms; a genuine leak never does and this times out.
|
||||||
|
scheduler.setPriority(1);
|
||||||
|
const settle_deadline = architecture.millis() + 3000;
|
||||||
|
while (scheduler.liveStackBytes() != base and architecture.millis() < settle_deadline) scheduler.yield();
|
||||||
|
scheduler.setPriority(4);
|
||||||
|
|
||||||
|
const final = scheduler.liveStackBytes();
|
||||||
|
log("task-reap: base={d} final={d} killed={d}/{d}\n", .{ base, final, killed, rounds });
|
||||||
|
check("all probes spawned and were killed", killed == rounds);
|
||||||
|
check("kernel stacks reclaimed to baseline (no leak)", final == base);
|
||||||
|
if (killed == rounds and final == base)
|
||||||
|
log("task-reap: kernel stacks reclaimed to baseline ok\n", .{});
|
||||||
|
result();
|
||||||
|
}
|
||||||
|
|
||||||
/// The full PID-1 path: the bootloader read /system/services/init off the boot volume and
|
/// The full PID-1 path: the bootloader read /system/services/init off the boot volume and
|
||||||
/// handed it over; load it as a user ELF and spawn it as a real ring-3 process
|
/// handed it over; load it as a user ELF and spawn it as a real ring-3 process
|
||||||
/// — the same call the normal boot path makes — then confirm it beats. init
|
/// — the same call the normal boot path makes — then confirm it beats. init
|
||||||
@@ -2344,6 +2784,46 @@ fn displayServiceTest(boot_information: *const BootInformation) void {
|
|||||||
while (true) scheduler.yield();
|
while (true) scheduler.yield();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The threaded compositor tracks a mouse (docs/threading.md, docs/display.md). Spawn the
|
||||||
|
/// `input` fan-out service, the display (which runs a mouse-listener thread alongside its
|
||||||
|
/// compositor loop and draws a top-z cursor), and `input-source` in `mouse` mode — a
|
||||||
|
/// synthetic source publishing pure motion. The display's own marker,
|
||||||
|
/// `display: cursor tracking mouse ok`, is printed once the cursor has tracked a run of
|
||||||
|
/// motion end to end (source -> input service -> listener thread -> channel -> render), so
|
||||||
|
/// like the other display cases we match on serial rather than poll in-kernel.
|
||||||
|
fn displayCursorTest(boot_information: *const BootInformation) void {
|
||||||
|
log("DANOS-TEST-BEGIN: display-cursor\n", .{});
|
||||||
|
if (boot_information.initial_ramdisk_len == 0) {
|
||||||
|
check("bootloader handed over an initial_ramdisk", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||||
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||||
|
check("initial_ramdisk image is valid", false);
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
if (!spawnNamed(rd, "input")) {
|
||||||
|
log("display-cursor: could not spawn the input service\n", .{});
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (!spawnNamed(rd, "display")) {
|
||||||
|
log("display-cursor: could not spawn the display service\n", .{});
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (!spawnNamedWithArg(rd, "input-source", "mouse")) {
|
||||||
|
log("display-cursor: could not spawn the mouse source\n", .{});
|
||||||
|
result();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
scheduler.setPriority(1); // below the services, so they run
|
||||||
|
while (true) scheduler.yield();
|
||||||
|
}
|
||||||
|
|
||||||
/// D4 — a separate process drives the compositor. Spawn the display service and the
|
/// D4 — a separate process drives the compositor. Spawn the display service and the
|
||||||
/// hardware-free `display-demo` client, which creates a wallpaper, a moving rectangle,
|
/// hardware-free `display-demo` client, which creates a wallpaper, a moving rectangle,
|
||||||
/// and a cursor and presents a run of frames. Its `display-demo: ok` heartbeat — printed
|
/// and a cursor and presents a run of frames. Its `display-demo: ok` heartbeat — printed
|
||||||
@@ -2370,6 +2850,11 @@ fn displayDemoTest(boot_information: *const BootInformation) void {
|
|||||||
result();
|
result();
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
// Spawn the input service too — real boot has it, and it guards the demo's
|
||||||
|
// independence from input: the demo must animate to `display-demo: ok` on its own
|
||||||
|
// frame timer even with the input service available (a client that blocks its
|
||||||
|
// animation loop on a mouse read would stall here, never reaching the marker).
|
||||||
|
_ = spawnNamed(rd, "input");
|
||||||
_ = spawnNamed(rd, "display-demo");
|
_ = spawnNamed(rd, "display-demo");
|
||||||
scheduler.setPriority(1); // below the service + demo, so they run
|
scheduler.setPriority(1); // below the service + demo, so they run
|
||||||
while (true) scheduler.yield();
|
while (true) scheduler.yield();
|
||||||
@@ -2591,6 +3076,19 @@ fn spawnNamed(rd: initial_ramdisk.Reader, name: []const u8) bool {
|
|||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// As `spawnNamed`, but passes one extra argv entry (argv[1]) — e.g. a mode selector like
|
||||||
|
/// `input-source mouse`.
|
||||||
|
fn spawnNamedWithArg(rd: initial_ramdisk.Reader, name: []const u8, arg: []const u8) bool {
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < rd.count) : (i += 1) {
|
||||||
|
const item = rd.entry(i) orelse continue;
|
||||||
|
if (eql(item.name, name)) {
|
||||||
|
return if (process.spawnProcess(item.blob, 4, &.{ item.name, arg })) true else |_| false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
/// The GSI discovery recorded for the HPET, from the same device table drivers see.
|
/// The GSI discovery recorded for the HPET, from the same device table drivers see.
|
||||||
fn hpetGsi() ?u32 {
|
fn hpetGsi() ?u32 {
|
||||||
var buffer: [16]device_abi.DeviceDescriptor = undefined;
|
var buffer: [16]device_abi.DeviceDescriptor = undefined;
|
||||||
@@ -2851,20 +3349,20 @@ fn ioPassTest() void {
|
|||||||
log("DANOS-TEST-BEGIN: iopass\n", .{});
|
log("DANOS-TEST-BEGIN: iopass\n", .{});
|
||||||
const base_free = pmm.stats().free_frames;
|
const base_free = pmm.stats().free_frames;
|
||||||
|
|
||||||
const aspace = architecture.createAddressSpace() orelse {
|
const address_space = architecture.createAddressSpace() orelse {
|
||||||
check("created a fresh address space", false);
|
check("created a fresh address space", false);
|
||||||
result();
|
result();
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
const frame = pmm.alloc() orelse {
|
const frame = pmm.alloc() orelse {
|
||||||
architecture.destroyAddressSpace(aspace);
|
architecture.destroyAddressSpace(address_space);
|
||||||
check("allocated a frame to grant", false);
|
check("allocated a frame to grant", false);
|
||||||
result();
|
result();
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
// Map it the way mmio_map does (device grant, strong-uncacheable), then tear the space down.
|
// Map it the way mmio_map does (device grant, strong-uncacheable), then tear the space down.
|
||||||
architecture.mapUserDeviceInto(aspace, process.device_arena_base, frame, abi.page_size, false);
|
architecture.mapUserDeviceInto(address_space, process.device_arena_base, frame, abi.page_size, false);
|
||||||
architecture.destroyAddressSpace(aspace);
|
architecture.destroyAddressSpace(address_space);
|
||||||
|
|
||||||
// The page tables were reclaimed; the device-granted frame must not have been.
|
// The page tables were reclaimed; the device-granted frame must not have been.
|
||||||
check("device-granted frame survived teardown (not reclaimed as RAM)", pmm.stats().free_frames == base_free - 1);
|
check("device-granted frame survived teardown (not reclaimed as RAM)", pmm.stats().free_frames == base_free - 1);
|
||||||
@@ -2916,26 +3414,26 @@ fn displayTest(boot_information: *const BootInformation) void {
|
|||||||
// space, and confirm the leaf's cache type. We never run this space (no CR3 load) —
|
// space, and confirm the leaf's cache type. We never run this space (no CR3 load) —
|
||||||
// we only read back the page-table entries — so aliasing the same physical page at
|
// we only read back the page-table entries — so aliasing the same physical page at
|
||||||
// two cache types below is inert.
|
// two cache types below is inert.
|
||||||
const aspace = architecture.createAddressSpace() orelse {
|
const address_space = architecture.createAddressSpace() orelse {
|
||||||
check("created a fresh address space", false);
|
check("created a fresh address space", false);
|
||||||
result();
|
result();
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
defer architecture.destroyAddressSpace(aspace);
|
defer architecture.destroyAddressSpace(address_space);
|
||||||
|
|
||||||
const page_base = fb.base & ~@as(u64, abi.page_size - 1);
|
const page_base = fb.base & ~@as(u64, abi.page_size - 1);
|
||||||
architecture.mapUserDeviceInto(aspace, process.device_arena_base, page_base, abi.page_size, true);
|
architecture.mapUserDeviceInto(address_space, process.device_arena_base, page_base, abi.page_size, true);
|
||||||
check(
|
check(
|
||||||
"the framebuffer maps write-combining (PAT entry 4: PAT bit set, PCD/PWT clear)",
|
"the framebuffer maps write-combining (PAT entry 4: PAT bit set, PCD/PWT clear)",
|
||||||
architecture.userLeafIsWriteCombining(aspace, process.device_arena_base) == true,
|
architecture.userLeafIsWriteCombining(address_space, process.device_arena_base) == true,
|
||||||
);
|
);
|
||||||
|
|
||||||
// Regression guard: the strong-uncacheable default is still that, so WC is a real
|
// Regression guard: the strong-uncacheable default is still that, so WC is a real
|
||||||
// choice the flag makes, not the only behaviour.
|
// choice the flag makes, not the only behaviour.
|
||||||
architecture.mapUserDeviceInto(aspace, process.device_arena_base + abi.page_size, page_base, abi.page_size, false);
|
architecture.mapUserDeviceInto(address_space, process.device_arena_base + abi.page_size, page_base, abi.page_size, false);
|
||||||
check(
|
check(
|
||||||
"a register window still maps strong-uncacheable",
|
"a register window still maps strong-uncacheable",
|
||||||
architecture.userLeafIsWriteCombining(aspace, process.device_arena_base + abi.page_size) == false,
|
architecture.userLeafIsWriteCombining(address_space, process.device_arena_base + abi.page_size) == false,
|
||||||
);
|
);
|
||||||
|
|
||||||
log("display: mapped {d}x{d} pitch {d} (write-combining)\n", .{ fb.width, fb.height, fb.pitch });
|
log("display: mapped {d}x{d} pitch {d} (write-combining)\n", .{ fb.width, fb.height, fb.pitch });
|
||||||
|
|||||||
@@ -1,15 +1,19 @@
|
|||||||
//! system/services/display-demo — a hardware-free client of the display service, the
|
//! system/services/display-demo — a hardware-free client of the display service, the
|
||||||
//! `input-source` analog for the compositor. It creates a wallpaper, a rectangle it moves
|
//! `input-source` analog for the compositor. It creates a wallpaper and a rectangle it
|
||||||
//! each frame, and a small cursor, then drives the compositor in a present loop — proof
|
//! slides each frame, then drives the compositor in a present loop — proof that a
|
||||||
//! that a *separate process* can compose a moving scene through the display service over
|
//! *separate process* can compose a moving scene through the display service over IPC,
|
||||||
//! IPC, exercising the layer client API and damage-driven present end to end
|
//! exercising the layer client API and damage-driven present end to end
|
||||||
//! (docs/display.md). It logs `display-demo: ok` once it has driven a run of frames.
|
//! (docs/display.md). It logs `display-demo: ok` once it has driven a run of frames.
|
||||||
|
//!
|
||||||
|
//! It draws no cursor and reads no input: the on-screen cursor is the display service's
|
||||||
|
//! own, tracked by the service's mouse-listener thread (docs/display.md). The demo's job
|
||||||
|
//! is only to prove client-driven animation, so its loop runs on its own frame timer and
|
||||||
|
//! is deliberately independent of the mouse.
|
||||||
|
|
||||||
const runtime = @import("runtime");
|
const runtime = @import("runtime");
|
||||||
const display = runtime.display;
|
const display = runtime.display;
|
||||||
const system = runtime.system;
|
const system = runtime.system;
|
||||||
const time = runtime.time;
|
const time = runtime.time;
|
||||||
const input = runtime.input;
|
|
||||||
|
|
||||||
pub fn main() void {
|
pub fn main() void {
|
||||||
const mode = display.info() orelse {
|
const mode = display.info() orelse {
|
||||||
@@ -28,15 +32,6 @@ pub fn main() void {
|
|||||||
const box = display.createLayer(0, box_y, box_w, box_h, 1) orelse return createFailed();
|
const box = display.createLayer(0, box_y, box_w, box_h, 1) orelse return createFailed();
|
||||||
_ = box.fill(0, 0, box_w, box_h, display.color(0xE0, 0x60, 0x40));
|
_ = box.fill(0, 0, box_w, box_h, display.color(0xE0, 0x60, 0x40));
|
||||||
|
|
||||||
// A little cursor on top. Its position is signed (the layer API is i32) and clamped to
|
|
||||||
// the screen; mouse motion arrives as relative deltas we accumulate below.
|
|
||||||
var cursor_x: i32 = @intCast(mode.width / 2);
|
|
||||||
var cursor_y: i32 = @intCast(mode.height / 2);
|
|
||||||
const cursor_max_x: i32 = @as(i32, @intCast(mode.width)) - 12;
|
|
||||||
const cursor_max_y: i32 = @as(i32, @intCast(mode.height)) - 12;
|
|
||||||
const cursor = display.createLayer(cursor_x, cursor_y, 12, 12, 2) orelse return createFailed();
|
|
||||||
_ = cursor.fill(0, 0, 12, 12, display.color(0xF0, 0xF0, 0xF0));
|
|
||||||
|
|
||||||
_ = display.present();
|
_ = display.present();
|
||||||
_ = system.write("display-demo: scene up; animating\n");
|
_ = system.write("display-demo: scene up; animating\n");
|
||||||
|
|
||||||
@@ -45,18 +40,7 @@ pub fn main() void {
|
|||||||
var dx: i32 = 8;
|
var dx: i32 = 8;
|
||||||
var frame: u32 = 0;
|
var frame: u32 = 0;
|
||||||
|
|
||||||
var mouse = input.subscribeMouse(); // type: ?input.MouseSubscriber
|
|
||||||
if (mouse == null) _ = system.write("display-demo: no mouse; animating without it\n");
|
|
||||||
|
|
||||||
while (true) : (frame += 1) {
|
while (true) : (frame += 1) {
|
||||||
if (mouse) |*ms| {
|
|
||||||
if (ms.next()) |event| {
|
|
||||||
cursor_x = clamp(cursor_x + event.dx, 0, cursor_max_x);
|
|
||||||
cursor_y = clamp(cursor_y + event.dy, 0, cursor_max_y);
|
|
||||||
_ = cursor.configure(cursor_x, cursor_y, 2, true);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
x += dx;
|
x += dx;
|
||||||
if (x <= 0) {
|
if (x <= 0) {
|
||||||
x = 0;
|
x = 0;
|
||||||
@@ -74,13 +58,6 @@ pub fn main() void {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Clamp `v` to the inclusive range [lo, hi].
|
|
||||||
fn clamp(v: i32, lo: i32, hi: i32) i32 {
|
|
||||||
if (v < lo) return lo;
|
|
||||||
if (v > hi) return hi;
|
|
||||||
return v;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn createFailed() void {
|
fn createFailed() void {
|
||||||
_ = system.write("display-demo: create failed\n");
|
_ = system.write("display-demo: create failed\n");
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -21,6 +21,8 @@ const backend_mod = @import("backend.zig");
|
|||||||
const protocol = runtime.display_protocol;
|
const protocol = runtime.display_protocol;
|
||||||
const ipc = runtime.ipc;
|
const ipc = runtime.ipc;
|
||||||
const system = runtime.system;
|
const system = runtime.system;
|
||||||
|
const input = runtime.input;
|
||||||
|
const Thread = runtime.Thread;
|
||||||
const Rect = compositor.Rect;
|
const Rect = compositor.Rect;
|
||||||
const Surface = compositor.Surface;
|
const Surface = compositor.Surface;
|
||||||
|
|
||||||
@@ -328,6 +330,157 @@ fn fail_check(_: []const u8) void {
|
|||||||
_ = system.write("display: compositor self-check FAILED (setup)\n");
|
_ = system.write("display: compositor self-check FAILED (setup)\n");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- cursor + mouse-input thread --------------------------------------------
|
||||||
|
//
|
||||||
|
// The compositor is the single owner of the framebuffer: only the main service
|
||||||
|
// loop touches `backend` and the layer stack. A dedicated listener thread (spawned
|
||||||
|
// in `initialise`) blocks on the input service's mouse stream, accumulates relative
|
||||||
|
// motion into an absolute cursor position, and hands that position to the main loop
|
||||||
|
// through `cursor_channel` — a single-slot latest-value cell (the renderer wants
|
||||||
|
// where the cursor *is*, not a replay of every delta). The listener never touches
|
||||||
|
// the compositor; it only writes the channel and pokes the main loop awake with a
|
||||||
|
// self-directed `ipc.send`, which arrives as a message-notification in the service
|
||||||
|
// loop (docs/threading.md, docs/display.md). Shared fate: a fault in the listener
|
||||||
|
// takes the whole display down and the supervisor restarts it (docs/resilience.md).
|
||||||
|
|
||||||
|
const cursor_size = 10; // a small square sprite — enough to prove tracking
|
||||||
|
const cursor_z = 0xFFFF_FFFF; // always above client layers
|
||||||
|
const cursor_report_threshold = 5; // px of travel before the tracking marker latches
|
||||||
|
|
||||||
|
var cursor_layer: ?u32 = null;
|
||||||
|
var cursor_origin_x: i32 = 0;
|
||||||
|
var cursor_origin_y: i32 = 0;
|
||||||
|
/// Latched once the cursor has demonstrably tracked a run of motion end to end
|
||||||
|
/// (source -> input service -> listener -> channel -> render): the `display-cursor`
|
||||||
|
/// test's success marker.
|
||||||
|
var cursor_tracking_reported: bool = false;
|
||||||
|
|
||||||
|
const poke_byte = [_]u8{0}; // the poke carries no payload; the value lives in the channel
|
||||||
|
|
||||||
|
/// Shared between the listener thread (producer) and the main loop (consumer).
|
||||||
|
/// Latest-value semantics with a coalesced wake: at most one poke is queued while
|
||||||
|
/// the main loop has not drained the last one, so a fast mouse cannot flood the
|
||||||
|
/// service endpoint.
|
||||||
|
const CursorChannel = struct {
|
||||||
|
lock: Thread.Mutex = .{},
|
||||||
|
poke_endpoint: ipc.Handle = 0,
|
||||||
|
x: i32 = 0,
|
||||||
|
y: i32 = 0,
|
||||||
|
buttons: u32 = 0,
|
||||||
|
dirty: bool = false,
|
||||||
|
poke_pending: bool = false,
|
||||||
|
|
||||||
|
const Snapshot = struct { x: i32, y: i32, buttons: u32 };
|
||||||
|
|
||||||
|
/// Producer (listener thread): record the newest position and, unless a wake is
|
||||||
|
/// already queued, poke the main loop awake.
|
||||||
|
fn publish(self: *CursorChannel, x: i32, y: i32, buttons: u32) void {
|
||||||
|
self.lock.lock();
|
||||||
|
self.x = x;
|
||||||
|
self.y = y;
|
||||||
|
self.buttons = buttons;
|
||||||
|
self.dirty = true;
|
||||||
|
const need_poke = !self.poke_pending;
|
||||||
|
if (need_poke) self.poke_pending = true;
|
||||||
|
self.lock.unlock();
|
||||||
|
if (need_poke) _ = ipc.send(self.poke_endpoint, &poke_byte);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Consumer (main loop): take the latest position, or null if nothing changed
|
||||||
|
/// since the last take. Clears the wake latch so the next publish pokes again.
|
||||||
|
fn take(self: *CursorChannel) ?Snapshot {
|
||||||
|
self.lock.lock();
|
||||||
|
defer self.lock.unlock();
|
||||||
|
self.poke_pending = false;
|
||||||
|
if (!self.dirty) return null;
|
||||||
|
self.dirty = false;
|
||||||
|
return .{ .x = self.x, .y = self.y, .buttons = self.buttons };
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
var cursor_channel: CursorChannel = .{};
|
||||||
|
|
||||||
|
fn clampAxis(value: i32, max: i32) i32 {
|
||||||
|
if (value < 0) return 0;
|
||||||
|
if (value > max) return max;
|
||||||
|
return value;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The mouse-listener thread. Blocks on the input service's mouse stream, accumulates
|
||||||
|
/// relative motion into an absolute position clamped to the screen, and publishes each
|
||||||
|
/// update. Runs for the life of the process; a parked `next()` leaves the core free to
|
||||||
|
/// halt (docs/halting.md). It reads only its own state and the channel — never the
|
||||||
|
/// compositor — so no lock guards the framebuffer.
|
||||||
|
fn mouseListener(width: u32, height: u32) void {
|
||||||
|
var mouse = input.subscribeMouse() orelse {
|
||||||
|
_ = system.write("display: mouse subscribe failed\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
// Our own handle to the compositor's endpoint. IPC handles are per-thread, so we
|
||||||
|
// cannot reuse the main thread's service handle — we look the service up to install a
|
||||||
|
// handle in this thread's table. A poke posted here wakes the compositor loop parked
|
||||||
|
// in replyWait (docs/threading.md: handles do not cross threads).
|
||||||
|
cursor_channel.poke_endpoint = ipc.lookup(.display) orelse {
|
||||||
|
_ = system.write("display: mouse listener could not reach the compositor endpoint\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
const max_x: i32 = @as(i32, @intCast(width)) - 1;
|
||||||
|
const max_y: i32 = @as(i32, @intCast(height)) - 1;
|
||||||
|
var x: i32 = @divTrunc(max_x, 2);
|
||||||
|
var y: i32 = @divTrunc(max_y, 2);
|
||||||
|
var buttons: u32 = 0;
|
||||||
|
while (true) {
|
||||||
|
const event = mouse.next() orelse continue;
|
||||||
|
// Switch on the raw kind (not @enumFromInt, which would panic on a scroll or
|
||||||
|
// future kind): motion moves the cursor, anything else just updates buttons.
|
||||||
|
if (event.kind == @intFromEnum(input.MouseEventKind.motion)) {
|
||||||
|
x = clampAxis(x + event.dx, max_x);
|
||||||
|
y = clampAxis(y + event.dy, max_y);
|
||||||
|
} else {
|
||||||
|
buttons = event.buttons;
|
||||||
|
}
|
||||||
|
cursor_channel.publish(x, y, buttons);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Consume the latest cursor position from the channel and repaint the cursor layer at
|
||||||
|
/// it. Runs on the main loop (the compositor owner) in response to a listener poke.
|
||||||
|
/// `configureLayer` damages both the old and new footprints, so a plain `present`
|
||||||
|
/// repaints exactly the two rectangles that changed.
|
||||||
|
fn renderCursor() void {
|
||||||
|
const snapshot = cursor_channel.take() orelse return;
|
||||||
|
const id = cursor_layer orelse return;
|
||||||
|
_ = configureLayer(id, snapshot.x, snapshot.y, cursor_z, true);
|
||||||
|
present();
|
||||||
|
if (!cursor_tracking_reported and
|
||||||
|
@abs(snapshot.x - cursor_origin_x) >= cursor_report_threshold and
|
||||||
|
@abs(snapshot.y - cursor_origin_y) >= cursor_report_threshold)
|
||||||
|
{
|
||||||
|
cursor_tracking_reported = true;
|
||||||
|
_ = system.write("display: cursor tracking mouse ok\n");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Create the cursor sprite (a top-z square) at screen centre and spawn the listener
|
||||||
|
/// thread. Called from `initialise` once the backend is up. If either step fails the
|
||||||
|
/// display still serves drawing clients — it just has no cursor.
|
||||||
|
fn startCursorTracking() void {
|
||||||
|
const mode = backend.info();
|
||||||
|
cursor_origin_x = @divTrunc(@as(i32, @intCast(mode.width)), 2);
|
||||||
|
cursor_origin_y = @divTrunc(@as(i32, @intCast(mode.height)), 2);
|
||||||
|
const id = createLayer(cursor_origin_x, cursor_origin_y, cursor_size, cursor_size, cursor_z, true) orelse {
|
||||||
|
_ = system.write("display: could not create cursor layer\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
cursor_layer = id;
|
||||||
|
_ = fillLayer(id, Rect.init(0, 0, cursor_size, cursor_size), protocol.pack(mode.format, 0xF0, 0xF0, 0xF0));
|
||||||
|
present(); // show the cursor at its start position
|
||||||
|
|
||||||
|
_ = Thread.spawn(.{}, mouseListener, .{ mode.width, mode.height }) catch {
|
||||||
|
_ = system.write("display: could not spawn mouse listener\n");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
// --- service ----------------------------------------------------------------
|
// --- service ----------------------------------------------------------------
|
||||||
|
|
||||||
fn initialise(endpoint: ipc.Handle) bool {
|
fn initialise(endpoint: ipc.Handle) bool {
|
||||||
@@ -350,6 +503,9 @@ fn initialise(endpoint: ipc.Handle) bool {
|
|||||||
_ = system.write("display: presented frame 0\n");
|
_ = system.write("display: presented frame 0\n");
|
||||||
|
|
||||||
selfCheck();
|
selfCheck();
|
||||||
|
|
||||||
|
// Bring up the cursor and the mouse-listener thread now that the backend is live.
|
||||||
|
startCursorTracking();
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -435,11 +591,16 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Han
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The only notification the compositor arms is the post-attach present timer: repaint the
|
/// Two notification sources reach the compositor. A **message-notification** is a poke
|
||||||
/// screen into the freshly attached native surface, verify the frame landed, then run the
|
/// from the mouse-listener thread (a buffered self-`ipc.send`, `notify_message_bit`):
|
||||||
/// one-shot mode-set self-check (V5).
|
/// repaint the cursor at its latest channel position. Anything else is the post-attach
|
||||||
|
/// present **timer**: repaint into the freshly attached native surface, verify the frame
|
||||||
|
/// landed, then run the one-shot mode-set self-check (V5).
|
||||||
fn onNotification(badge: u64) void {
|
fn onNotification(badge: u64) void {
|
||||||
_ = badge;
|
if (badge & ipc.notify_message_bit != 0) {
|
||||||
|
renderCursor();
|
||||||
|
return;
|
||||||
|
}
|
||||||
present(); // native present + verify (first timer fire after the upgrade)
|
present(); // native present + verify (first timer fire after the upgrade)
|
||||||
if (pending_modeset_check) {
|
if (pending_modeset_check) {
|
||||||
pending_modeset_check = false;
|
pending_modeset_check = false;
|
||||||
|
|||||||
@@ -10,17 +10,38 @@
|
|||||||
//! keyboard and mouse drivers publish their own synthetic streams today; swapping in
|
//! keyboard and mouse drivers publish their own synthetic streams today; swapping in
|
||||||
//! decoded hardware is a follow-up (see docs/input.md).
|
//! decoded hardware is a follow-up (see docs/input.md).
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
const runtime = @import("runtime");
|
const runtime = @import("runtime");
|
||||||
const input = runtime.input;
|
const input = runtime.input;
|
||||||
const system = runtime.system;
|
const system = runtime.system;
|
||||||
|
|
||||||
pub fn main() void {
|
pub fn main(init: runtime.process.Init) void {
|
||||||
var source = input.connectSource() orelse {
|
var source = input.connectSource() orelse {
|
||||||
_ = system.write("input-source: input service unavailable\n");
|
_ = system.write("input-source: input service unavailable\n");
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
_ = system.write("input-source: publishing synthetic input events\n");
|
|
||||||
|
|
||||||
|
// "mouse" mode publishes a steady stream of pure motion (dx=dy=+1), for driving a
|
||||||
|
// cursor (the `display-cursor` test). The default "rotate" mode cycles all device
|
||||||
|
// classes to exercise the service's per-device routing (the `input` test).
|
||||||
|
const mode = init.arguments.get(1) orelse "rotate";
|
||||||
|
if (std.mem.eql(u8, mode, "mouse")) {
|
||||||
|
_ = system.write("input-source: publishing synthetic mouse motion\n");
|
||||||
|
while (true) {
|
||||||
|
_ = source.publishMouseEvent(.{
|
||||||
|
.kind = @intFromEnum(input.MouseEventKind.motion),
|
||||||
|
.button = 0,
|
||||||
|
.dx = 1,
|
||||||
|
.dy = 1,
|
||||||
|
.scroll_x = 0,
|
||||||
|
.scroll_y = 0,
|
||||||
|
.buttons = 0,
|
||||||
|
});
|
||||||
|
system.sleep(20); // ~50 events/sec: moves the cursor briskly
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
_ = system.write("input-source: publishing synthetic input events\n");
|
||||||
var step: usize = 0;
|
var step: usize = 0;
|
||||||
while (true) : (step +%= 1) {
|
while (true) : (step +%= 1) {
|
||||||
// Rotate across the device classes so every publish path (and the service's
|
// Rotate across the device classes so every publish path (and the service's
|
||||||
|
|||||||
@@ -0,0 +1,491 @@
|
|||||||
|
//! thread-test — danos's multi-threaded exerciser (docs/threading-plan.md M2, M3).
|
||||||
|
//!
|
||||||
|
//! Two modes, chosen by argv[1] (default "spawn"):
|
||||||
|
//! spawn — M2: one worker writes a shared global; the main thread observes it, proving
|
||||||
|
//! `runtime.Thread.spawn` started a task in the **same** address space.
|
||||||
|
//! join — M3: N workers each do K atomic increments on a shared counter and stamp the
|
||||||
|
//! core they ran on; the main thread `join`s all N and checks the total is
|
||||||
|
//! exactly N*K (every worker ran, join waited) and that >1 core was used
|
||||||
|
//! (genuine parallelism). Then a detached worker proves `detach` runs and
|
||||||
|
//! needs no join.
|
||||||
|
//!
|
||||||
|
//! Built multi-threaded (`addThreadedUserBinary`) so atomics/shared reads are real.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const runtime = @import("runtime");
|
||||||
|
|
||||||
|
fn write(comptime s: []const u8) void {
|
||||||
|
_ = runtime.system.write(s);
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- M2: spawn mode ---------------------------------------------------------
|
||||||
|
|
||||||
|
var shared_value: u32 = 0;
|
||||||
|
var spawn_done = std.atomic.Value(u32).init(0);
|
||||||
|
const sentinel: u32 = 0xA5A5;
|
||||||
|
|
||||||
|
fn spawnWorker() void {
|
||||||
|
shared_value = sentinel;
|
||||||
|
spawn_done.store(1, .release);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn runSpawnMode() void {
|
||||||
|
write("thread-test: starting\n");
|
||||||
|
_ = runtime.Thread.spawn(.{}, spawnWorker, .{}) catch {
|
||||||
|
write("thread-test: FAIL spawn refused\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
var spins: usize = 0;
|
||||||
|
while (spawn_done.load(.acquire) == 0 and spins < 50_000_000) : (spins += 1) {
|
||||||
|
runtime.system.yield();
|
||||||
|
}
|
||||||
|
if (spawn_done.load(.acquire) == 1 and shared_value == sentinel) {
|
||||||
|
write("thread-test: child ran in shared address space ok\n");
|
||||||
|
} else {
|
||||||
|
write("thread-test: FAIL worker did not update shared memory\n");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- M3: join mode ----------------------------------------------------------
|
||||||
|
|
||||||
|
const worker_count: u32 = 4;
|
||||||
|
const iterations: u64 = 100_000;
|
||||||
|
|
||||||
|
var counter = std.atomic.Value(u64).init(0);
|
||||||
|
var cores_seen = std.atomic.Value(u32).init(0);
|
||||||
|
|
||||||
|
fn joinWorker() void {
|
||||||
|
var i: u64 = 0;
|
||||||
|
while (i < iterations) : (i += 1) {
|
||||||
|
_ = counter.fetchAdd(1, .monotonic);
|
||||||
|
if (i % 1000 == 0) stampCore(); // periodic: catches cross-core migration too
|
||||||
|
}
|
||||||
|
stampCore();
|
||||||
|
}
|
||||||
|
|
||||||
|
fn stampCore() void {
|
||||||
|
const core = runtime.Thread.currentCore();
|
||||||
|
if (core < 32) _ = cores_seen.fetchOr(@as(u32, 1) << @intCast(core), .monotonic);
|
||||||
|
}
|
||||||
|
|
||||||
|
var detach_done = std.atomic.Value(u32).init(0);
|
||||||
|
|
||||||
|
fn detachWorker() void {
|
||||||
|
detach_done.store(1, .release);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn noopWorker() void {}
|
||||||
|
|
||||||
|
fn runJoinMode() void {
|
||||||
|
write("thread-test: join mode starting\n");
|
||||||
|
|
||||||
|
var threads: [worker_count]runtime.Thread = undefined;
|
||||||
|
var spawned: u32 = 0;
|
||||||
|
while (spawned < worker_count) : (spawned += 1) {
|
||||||
|
threads[spawned] = runtime.Thread.spawn(.{}, joinWorker, .{}) catch break;
|
||||||
|
}
|
||||||
|
if (spawned != worker_count) {
|
||||||
|
write("thread-test: FAIL could not spawn all workers\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
for (threads[0..spawned]) |t| t.join();
|
||||||
|
|
||||||
|
const total = counter.load(.acquire);
|
||||||
|
const cores = @popCount(cores_seen.load(.acquire));
|
||||||
|
if (total != worker_count * iterations) {
|
||||||
|
write("thread-test: FAIL counter mismatch (a worker was lost or join did not wait)\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (cores <= 1) {
|
||||||
|
write("thread-test: FAIL workers never ran on more than one core\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
// detach: the worker runs and we never join it.
|
||||||
|
const dt = runtime.Thread.spawn(.{}, detachWorker, .{}) catch {
|
||||||
|
write("thread-test: FAIL detach spawn refused\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
dt.detach();
|
||||||
|
var spins: usize = 0;
|
||||||
|
while (detach_done.load(.acquire) == 0 and spins < 50_000_000) : (spins += 1) {
|
||||||
|
runtime.system.yield();
|
||||||
|
}
|
||||||
|
if (detach_done.load(.acquire) != 1) {
|
||||||
|
write("thread-test: FAIL detached worker did not run\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Prove join needs no per-thread kernel endpoint (M9): many spawn+join cycles. Under
|
||||||
|
// the old per-thread-endpoint scheme these leaked handles and would exhaust the
|
||||||
|
// 16-slot handle table well before 40; here they all succeed.
|
||||||
|
var cycle: u32 = 0;
|
||||||
|
while (cycle < 40) : (cycle += 1) {
|
||||||
|
const th = runtime.Thread.spawn(.{}, noopWorker, .{}) catch {
|
||||||
|
write("thread-test: FAIL spawn exhausted across join cycles (endpoint leak?)\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
th.join();
|
||||||
|
}
|
||||||
|
|
||||||
|
write("thread-test: join ok\n"); // the M3/M9 verdict marker
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- M4: futex mode ---------------------------------------------------------
|
||||||
|
|
||||||
|
const Futex = runtime.Thread.Futex;
|
||||||
|
|
||||||
|
var futex_word = std.atomic.Value(u32).init(0);
|
||||||
|
var waiter_parked = std.atomic.Value(u32).init(0);
|
||||||
|
|
||||||
|
fn futexWaiter() void {
|
||||||
|
write("thread-futex: waiting\n");
|
||||||
|
waiter_parked.store(1, .release);
|
||||||
|
// Block while the word is still 0; the waker sets it to 1 and wakes us.
|
||||||
|
while (futex_word.load(.acquire) == 0) {
|
||||||
|
Futex.wait(&futex_word, 0);
|
||||||
|
}
|
||||||
|
write("thread-futex: woke\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
fn runFutexMode() void {
|
||||||
|
write("thread-futex: starting\n");
|
||||||
|
|
||||||
|
const waiter = runtime.Thread.spawn(.{}, futexWaiter, .{}) catch {
|
||||||
|
write("thread-futex: FAIL spawn refused\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
// Let the waiter reach its wait, then give it a beat to actually park in-kernel.
|
||||||
|
var spins: usize = 0;
|
||||||
|
while (waiter_parked.load(.acquire) == 0 and spins < 50_000_000) : (spins += 1) {
|
||||||
|
runtime.system.yield();
|
||||||
|
}
|
||||||
|
runtime.system.sleep(50);
|
||||||
|
|
||||||
|
// The handshake: publish the value, then wake the parked waiter.
|
||||||
|
futex_word.store(1, .release);
|
||||||
|
write("thread-futex: waking\n");
|
||||||
|
Futex.wake(&futex_word, 1);
|
||||||
|
|
||||||
|
waiter.join(); // returns once the waiter woke and printed "woke"
|
||||||
|
|
||||||
|
// Timeout: nobody ever wakes this word, so timedWait must report a timeout.
|
||||||
|
var lonely = std.atomic.Value(u32).init(0);
|
||||||
|
if (Futex.timedWait(&lonely, 0, 100_000_000)) |_| {
|
||||||
|
write("thread-futex: FAIL timedWait did not time out\n");
|
||||||
|
return;
|
||||||
|
} else |_| {}
|
||||||
|
write("thread-futex: timeout ok\n");
|
||||||
|
|
||||||
|
write("thread-futex: ok\n"); // the M4 verdict marker
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- M5: mutex mode (bounded producer/consumer over Mutex + Condition) ------
|
||||||
|
|
||||||
|
const Mutex = runtime.Thread.Mutex;
|
||||||
|
const Condition = runtime.Thread.Condition;
|
||||||
|
|
||||||
|
const producers: u32 = 2;
|
||||||
|
const consumers: u32 = 2;
|
||||||
|
const per_producer: u32 = 1000;
|
||||||
|
const per_consumer: u32 = 1000; // producers*per_producer == consumers*per_consumer (balanced)
|
||||||
|
const total_items: u32 = producers * per_producer;
|
||||||
|
const ring_cap: usize = 8; // small, so producers block on full and consumers on empty
|
||||||
|
|
||||||
|
var ring: [ring_cap]u32 = undefined;
|
||||||
|
var ring_count: usize = 0;
|
||||||
|
var ring_head: usize = 0;
|
||||||
|
var ring_tail: usize = 0;
|
||||||
|
|
||||||
|
var pc_mutex = Mutex{};
|
||||||
|
var not_full = Condition{};
|
||||||
|
var not_empty = Condition{};
|
||||||
|
|
||||||
|
// Verified outside the lock: the checksum and tally of everything consumed.
|
||||||
|
var consumed_sum = std.atomic.Value(u64).init(0);
|
||||||
|
var consumed_count = std.atomic.Value(u32).init(0);
|
||||||
|
|
||||||
|
fn producer(base: u32) void {
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < per_producer) : (i += 1) {
|
||||||
|
const item = base + i;
|
||||||
|
pc_mutex.lock();
|
||||||
|
while (ring_count == ring_cap) not_full.wait(&pc_mutex);
|
||||||
|
ring[ring_tail] = item;
|
||||||
|
ring_tail = (ring_tail + 1) % ring_cap;
|
||||||
|
ring_count += 1;
|
||||||
|
pc_mutex.unlock();
|
||||||
|
not_empty.signal();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn consumer() void {
|
||||||
|
var i: u32 = 0;
|
||||||
|
while (i < per_consumer) : (i += 1) {
|
||||||
|
pc_mutex.lock();
|
||||||
|
while (ring_count == 0) not_empty.wait(&pc_mutex);
|
||||||
|
const item = ring[ring_head];
|
||||||
|
ring_head = (ring_head + 1) % ring_cap;
|
||||||
|
ring_count -= 1;
|
||||||
|
pc_mutex.unlock();
|
||||||
|
not_full.signal();
|
||||||
|
_ = consumed_sum.fetchAdd(item, .monotonic);
|
||||||
|
_ = consumed_count.fetchAdd(1, .monotonic);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn runMutexMode() void {
|
||||||
|
write("thread-mutex: starting\n");
|
||||||
|
|
||||||
|
var threads: [producers + consumers]runtime.Thread = undefined;
|
||||||
|
var n: usize = 0;
|
||||||
|
var p: u32 = 0;
|
||||||
|
while (p < producers) : (p += 1) {
|
||||||
|
threads[n] = runtime.Thread.spawn(.{}, producer, .{p * per_producer}) catch {
|
||||||
|
write("thread-mutex: FAIL producer spawn\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
n += 1;
|
||||||
|
}
|
||||||
|
var c: u32 = 0;
|
||||||
|
while (c < consumers) : (c += 1) {
|
||||||
|
threads[n] = runtime.Thread.spawn(.{}, consumer, .{}) catch {
|
||||||
|
write("thread-mutex: FAIL consumer spawn\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
n += 1;
|
||||||
|
}
|
||||||
|
for (threads[0..n]) |t| t.join();
|
||||||
|
|
||||||
|
// Every item 0..total_items-1 was produced exactly once; if the mutex/condition are
|
||||||
|
// correct, each was consumed exactly once, so the checksum matches.
|
||||||
|
const expected_sum: u64 = @as(u64, total_items) * (total_items - 1) / 2;
|
||||||
|
if (consumed_count.load(.acquire) != total_items) {
|
||||||
|
write("thread-mutex: FAIL wrong number of items consumed\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (consumed_sum.load(.acquire) != expected_sum) {
|
||||||
|
write("thread-mutex: FAIL checksum mismatch (item lost or duplicated)\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
write("thread-mutex: ok\n"); // the M5 verdict marker
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- M6: id mode (getCurrentId identity) ------------------------------------
|
||||||
|
|
||||||
|
var worker_ids: [2]std.atomic.Value(u32) = .{ std.atomic.Value(u32).init(0), std.atomic.Value(u32).init(0) };
|
||||||
|
|
||||||
|
fn idWorker(slot: usize) void {
|
||||||
|
worker_ids[slot].store(runtime.Thread.getCurrentId(), .release);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn runIdMode() void {
|
||||||
|
write("thread-id: starting\n");
|
||||||
|
const main_id = runtime.Thread.getCurrentId();
|
||||||
|
|
||||||
|
const t0 = runtime.Thread.spawn(.{}, idWorker, .{@as(usize, 0)}) catch {
|
||||||
|
write("thread-id: FAIL spawn\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
const t1 = runtime.Thread.spawn(.{}, idWorker, .{@as(usize, 1)}) catch {
|
||||||
|
write("thread-id: FAIL spawn\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
t0.join();
|
||||||
|
t1.join();
|
||||||
|
|
||||||
|
const id0 = worker_ids[0].load(.acquire);
|
||||||
|
const id1 = worker_ids[1].load(.acquire);
|
||||||
|
// Each thread has its own kernel task id: all three distinct and non-zero.
|
||||||
|
if (main_id == 0 or id0 == 0 or id1 == 0) {
|
||||||
|
write("thread-id: FAIL a thread reported id 0\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (id0 == id1 or id0 == main_id or id1 == main_id) {
|
||||||
|
write("thread-id: FAIL thread ids collided\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
write("thread-id: ok\n"); // the M6 verdict marker
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- M7: alloc mode (concurrent heap allocation) ----------------------------
|
||||||
|
|
||||||
|
const alloc_threads: u32 = 4;
|
||||||
|
const allocs_per_thread: u32 = 500;
|
||||||
|
var allocs_clean = std.atomic.Value(u32).init(0);
|
||||||
|
|
||||||
|
fn allocWorker(seed: u32) void {
|
||||||
|
const gpa = runtime.allocator();
|
||||||
|
var rng: u32 = seed | 1;
|
||||||
|
var round: u32 = 0;
|
||||||
|
while (round < allocs_per_thread) : (round += 1) {
|
||||||
|
rng = rng *% 1664525 +% 1013904223; // cheap LCG for varied sizes
|
||||||
|
const size: usize = 16 + (rng % 4080); // 16..4095 bytes
|
||||||
|
const buf = gpa.alloc(u8, size) catch return; // OOM: don't count this thread clean
|
||||||
|
const pattern: u8 = @truncate(seed +% round);
|
||||||
|
@memset(buf, pattern);
|
||||||
|
// Nothing else should touch our block; if a concurrent allocation overlapped it,
|
||||||
|
// one of us would read the other's pattern here.
|
||||||
|
var ok = true;
|
||||||
|
for (buf) |b| {
|
||||||
|
if (b != pattern) ok = false;
|
||||||
|
}
|
||||||
|
gpa.free(buf);
|
||||||
|
if (!ok) return; // corruption — leave without counting clean
|
||||||
|
}
|
||||||
|
_ = allocs_clean.fetchAdd(1, .monotonic);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn runAllocMode() void {
|
||||||
|
write("thread-alloc: starting\n");
|
||||||
|
var threads: [alloc_threads]runtime.Thread = undefined;
|
||||||
|
var n: u32 = 0;
|
||||||
|
while (n < alloc_threads) : (n += 1) {
|
||||||
|
threads[n] = runtime.Thread.spawn(.{}, allocWorker, .{n +% 1}) catch {
|
||||||
|
write("thread-alloc: FAIL spawn\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
}
|
||||||
|
for (threads[0..alloc_threads]) |t| t.join();
|
||||||
|
|
||||||
|
// Every thread must have completed all rounds with each block intact — proof the
|
||||||
|
// shared heap and the per-address_space mmap arena are safe under concurrent allocation.
|
||||||
|
if (allocs_clean.load(.acquire) != alloc_threads) {
|
||||||
|
write("thread-alloc: FAIL corruption or OOM under concurrent allocation\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
write("thread-alloc: ok\n"); // the M7 verdict marker
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- M10: tls mode (per-thread FS base storage) -----------------------------
|
||||||
|
|
||||||
|
fn writeTlsSlot(value: u64) void {
|
||||||
|
asm volatile ("movq %[v], %%fs:8"
|
||||||
|
:
|
||||||
|
: [v] "r" (value),
|
||||||
|
: .{ .memory = true });
|
||||||
|
}
|
||||||
|
|
||||||
|
fn readTlsSlot() u64 {
|
||||||
|
return asm volatile ("movq %%fs:8, %[out]"
|
||||||
|
: [out] "=r" (-> u64),
|
||||||
|
:
|
||||||
|
: .{ .memory = true });
|
||||||
|
}
|
||||||
|
|
||||||
|
var tls_written = std.atomic.Value(u32).init(0);
|
||||||
|
var tls_ok = std.atomic.Value(u32).init(0);
|
||||||
|
|
||||||
|
fn tlsWorker(marker: u64) void {
|
||||||
|
writeTlsSlot(marker);
|
||||||
|
_ = tls_written.fetchAdd(1, .release);
|
||||||
|
// Wait until both threads have written their own slot. If the FS base were shared, the
|
||||||
|
// second write would clobber the first, and the read below would return the wrong
|
||||||
|
// marker — cross-talk. A per-thread FS base keeps each thread's slot private.
|
||||||
|
var spins: usize = 0;
|
||||||
|
while (tls_written.load(.acquire) < 2 and spins < 50_000_000) : (spins += 1) {
|
||||||
|
runtime.system.yield();
|
||||||
|
}
|
||||||
|
if (readTlsSlot() == marker and runtime.Thread.getCurrentId() != 0) {
|
||||||
|
_ = tls_ok.fetchAdd(1, .monotonic);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn runTlsMode() void {
|
||||||
|
write("thread-tls: starting\n");
|
||||||
|
const t0 = runtime.Thread.spawn(.{}, tlsWorker, .{@as(u64, 0xAAAA_0000)}) catch {
|
||||||
|
write("thread-tls: FAIL spawn\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
const t1 = runtime.Thread.spawn(.{}, tlsWorker, .{@as(u64, 0xBBBB_0000)}) catch {
|
||||||
|
write("thread-tls: FAIL spawn\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
t0.join();
|
||||||
|
t1.join();
|
||||||
|
if (tls_ok.load(.acquire) == 2) {
|
||||||
|
write("thread-tls: ok\n"); // the M10 verdict marker
|
||||||
|
} else {
|
||||||
|
write("thread-tls: FAIL cross-talk (FS base not per-thread)\n");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- M11: rwlock mode (readers/writers over an RwLock) ----------------------
|
||||||
|
|
||||||
|
const RwLock = runtime.Thread.RwLock;
|
||||||
|
|
||||||
|
var rwlock = RwLock{};
|
||||||
|
var rw_a: u64 = 0;
|
||||||
|
var rw_b: u64 = 0; // invariant while any lock is held: rw_a == rw_b
|
||||||
|
var rw_stop = std.atomic.Value(u32).init(0);
|
||||||
|
var rw_violations = std.atomic.Value(u32).init(0);
|
||||||
|
var rw_reads = std.atomic.Value(u64).init(0);
|
||||||
|
|
||||||
|
fn rwWriter() void {
|
||||||
|
var v: u64 = 1;
|
||||||
|
while (rw_stop.load(.acquire) == 0) : (v +%= 1) {
|
||||||
|
rwlock.lock(); // exclusive: no reader may observe the gap between the two writes
|
||||||
|
rw_a = v;
|
||||||
|
rw_b = v;
|
||||||
|
rwlock.unlock();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn rwReader() void {
|
||||||
|
const reads: u64 = 50_000;
|
||||||
|
var i: u64 = 0;
|
||||||
|
while (i < reads) : (i += 1) {
|
||||||
|
rwlock.lockShared();
|
||||||
|
if (rw_a != rw_b) _ = rw_violations.fetchAdd(1, .monotonic); // saw a half-write!
|
||||||
|
rwlock.unlockShared();
|
||||||
|
}
|
||||||
|
_ = rw_reads.fetchAdd(reads, .monotonic);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn runRwlockMode() void {
|
||||||
|
write("thread-rwlock: starting\n");
|
||||||
|
var writers: [2]runtime.Thread = undefined;
|
||||||
|
var readers: [3]runtime.Thread = undefined;
|
||||||
|
for (&writers) |*w| {
|
||||||
|
w.* = runtime.Thread.spawn(.{}, rwWriter, .{}) catch {
|
||||||
|
write("thread-rwlock: FAIL spawn\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
}
|
||||||
|
for (&readers) |*r| {
|
||||||
|
r.* = runtime.Thread.spawn(.{}, rwReader, .{}) catch {
|
||||||
|
write("thread-rwlock: FAIL spawn\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
}
|
||||||
|
for (readers) |r| r.join();
|
||||||
|
rw_stop.store(1, .release); // readers done → stop the writers
|
||||||
|
for (writers) |w| w.join();
|
||||||
|
|
||||||
|
if (rw_violations.load(.acquire) == 0 and rw_reads.load(.acquire) > 0) {
|
||||||
|
write("thread-rwlock: ok\n"); // the M11 verdict marker
|
||||||
|
} else {
|
||||||
|
write("thread-rwlock: FAIL reader observed a half-written value\n");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main(init: runtime.process.Init) void {
|
||||||
|
const mode = init.arguments.get(1) orelse "spawn";
|
||||||
|
if (std.mem.eql(u8, mode, "join")) {
|
||||||
|
runJoinMode();
|
||||||
|
} else if (std.mem.eql(u8, mode, "futex")) {
|
||||||
|
runFutexMode();
|
||||||
|
} else if (std.mem.eql(u8, mode, "mutex")) {
|
||||||
|
runMutexMode();
|
||||||
|
} else if (std.mem.eql(u8, mode, "id")) {
|
||||||
|
runIdMode();
|
||||||
|
} else if (std.mem.eql(u8, mode, "alloc")) {
|
||||||
|
runAllocMode();
|
||||||
|
} else if (std.mem.eql(u8, mode, "tls")) {
|
||||||
|
runTlsMode();
|
||||||
|
} else if (std.mem.eql(u8, mode, "rwlock")) {
|
||||||
|
runRwlockMode();
|
||||||
|
} else {
|
||||||
|
runSpawnMode();
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -185,6 +185,16 @@ CASES = [
|
|||||||
{"name": "display-demo",
|
{"name": "display-demo",
|
||||||
"expect": r"display-demo: scene up[\s\S]*display-demo: ok",
|
"expect": r"display-demo: scene up[\s\S]*display-demo: ok",
|
||||||
"fail": r"display-demo: (no display|create failed)|display: could not|CPU EXCEPTION|KERNEL PANIC"},
|
"fail": r"display-demo: (no display|create failed)|display: could not|CPU EXCEPTION|KERNEL PANIC"},
|
||||||
|
# Threaded compositor tracks a mouse (docs/threading.md, docs/display.md): the display
|
||||||
|
# runs a mouse-listener thread alongside its compositor loop. `input-source mouse`
|
||||||
|
# publishes pure motion -> the input service fans it to the display's listener -> the
|
||||||
|
# listener accumulates it into a cursor position handed to the render loop over a
|
||||||
|
# single-slot channel. `display: cursor tracking mouse ok` latches once the cursor has
|
||||||
|
# tracked a run of that motion end to end.
|
||||||
|
{"name": "display-cursor",
|
||||||
|
"smp": 4,
|
||||||
|
"expect": r"display: online \d+x\d+[\s\S]*display: cursor tracking mouse ok",
|
||||||
|
"fail": r"display: (could not|mouse subscribe failed)|CPU EXCEPTION|KERNEL PANIC"},
|
||||||
# Shared memory (v2 V2): shm-client creates a region, writes a pattern, and passes its
|
# Shared memory (v2 V2): shm-client creates a region, writes a pattern, and passes its
|
||||||
# capability to shm-server, which maps it and confirms the same bytes — proving
|
# capability to shm-server, which maps it and confirms the same bytes — proving
|
||||||
# cross-process shared pages over the extended capability passing.
|
# cross-process shared pages over the extended capability passing.
|
||||||
@@ -293,6 +303,84 @@ CASES = [
|
|||||||
"timeout": 60,
|
"timeout": 60,
|
||||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
|
||||||
|
# docs/threading-plan.md M1: address-space refcount — spaces destroyed exactly
|
||||||
|
# once per process, no leak/double-free (the foundation shared-address-space threads need).
|
||||||
|
{"name": "address-space-refcount",
|
||||||
|
"timeout": 60,
|
||||||
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
|
||||||
|
# docs/threading-plan.md M2: runtime.Thread.spawn — a worker thread runs in the
|
||||||
|
# caller's address space (a shared-memory write, observed by the main thread).
|
||||||
|
{"name": "thread-spawn",
|
||||||
|
"timeout": 60,
|
||||||
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
|
||||||
|
# docs/threading-plan.md M3: join + parallelism — N workers each do K atomic
|
||||||
|
# increments (total exactly N*K after join) and run on >1 core; plus detach.
|
||||||
|
{"name": "thread-join",
|
||||||
|
"smp": 4,
|
||||||
|
"timeout": 60,
|
||||||
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
|
||||||
|
# docs/threading-plan.md M4: futex — a thread parks in futex_wait and is woken by
|
||||||
|
# futex_wake (serial order waiting/waking/woke), and timedWait reports a timeout.
|
||||||
|
{"name": "thread-futex",
|
||||||
|
"smp": 4,
|
||||||
|
"timeout": 60,
|
||||||
|
"expect": r"thread-futex: waiting[\s\S]*thread-futex: waking[\s\S]*thread-futex: woke[\s\S]*DANOS-TEST-RESULT: PASS",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
|
||||||
|
# docs/threading-plan.md M5: Mutex + Condition — a bounded producer/consumer moves
|
||||||
|
# N unique items across cores; the consumed checksum matches exactly (no loss).
|
||||||
|
{"name": "thread-mutex",
|
||||||
|
"smp": 4,
|
||||||
|
"timeout": 60,
|
||||||
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
|
||||||
|
# docs/threading-plan.md M6: thread identity — getCurrentId is distinct and non-zero
|
||||||
|
# for the main thread and two workers.
|
||||||
|
{"name": "thread-id",
|
||||||
|
"smp": 4,
|
||||||
|
"timeout": 60,
|
||||||
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
|
||||||
|
# docs/threading-plan.md M7: thread-safe allocation — N threads hammer the shared heap
|
||||||
|
# (per-aspace mmap arena + locked free list) with no cross-block corruption.
|
||||||
|
{"name": "thread-alloc",
|
||||||
|
"smp": 4,
|
||||||
|
"timeout": 60,
|
||||||
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
|
||||||
|
# docs/threading-plan.md M8: the task reaper — spawn+kill many processes; total kernel
|
||||||
|
# stack bytes return to baseline (every dead task's stack reclaimed, no leak).
|
||||||
|
{"name": "task-reap",
|
||||||
|
"smp": 4,
|
||||||
|
"timeout": 60,
|
||||||
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
|
||||||
|
# docs/threading-plan.md M10: per-thread fs.base — two threads keep private %fs:8 TLS
|
||||||
|
# slots across context switches (no cross-talk).
|
||||||
|
{"name": "thread-tls",
|
||||||
|
"smp": 4,
|
||||||
|
"timeout": 60,
|
||||||
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
|
|
||||||
|
# docs/threading-plan.md M11: RwLock — readers/writers across cores; a reader never
|
||||||
|
# observes a half-written value (writers hold it exclusively).
|
||||||
|
{"name": "thread-rwlock",
|
||||||
|
"smp": 4,
|
||||||
|
"timeout": 60,
|
||||||
|
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||||
|
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||||
# Process arguments: argv arrives on the SysV entry stack (argv[0] = the spawned
|
# Process arguments: argv arrives on the SysV entry stack (argv[0] = the spawned
|
||||||
# name, argv[1..] = the system_spawn argument blob) and echoes back intact.
|
# name, argv[1..] = the system_spawn argument blob) and echoes back intact.
|
||||||
{"name": "args",
|
{"name": "args",
|
||||||
|
|||||||
Reference in New Issue
Block a user