threads(M3): join, detach, and cross-core parallelism
thread_spawn takes a 4th arg, an exit-endpoint handle: spawnThreadSupervised resolves and refcounts it under the spawn lock (like spawnProcessSupervised), so a thread's death posts a child-exit notification carrying its tid. runtime Thread.join blocks in replyWait on that (private) endpoint for its tid, then munmaps the stack; detach relinquishes the join (stack reclaimed at process exit, for now). New current_core=39 syscall + Thread.currentCore() lets a worker observe which core it ran on. The closure now lives at the top of the thread's own (private) stack instead of the heap, so spawn/join never touch the not-yet-thread-safe runtime heap. thread-test gains a join mode: 4 workers x 100k atomic increments, joined, with counter == N*K and >1 core stamped (real parallelism), plus a detached worker. Gate thread-join PASS (4x, non-flaky); 17 guardrail/M1/M2 cases green; build + host tests clean.
This commit is contained in:
@@ -1,47 +1,123 @@
|
||||
//! thread-test — the first multi-threaded danos binary (docs/threading-plan.md M2).
|
||||
//! thread-test — danos's multi-threaded exerciser (docs/threading-plan.md M2, M3).
|
||||
//!
|
||||
//! Proves `runtime.Thread.spawn` starts a task in the **same address space**: the main
|
||||
//! thread spawns a worker, the worker writes a shared global and signals `done`, and the
|
||||
//! main thread — polling that shared memory — observes the write. Seeing the write proves
|
||||
//! the two tasks share one address space (a separate process could not touch this memory).
|
||||
//! The `thread-test: child ran in shared aspace ok` line is the case's marker.
|
||||
//! Two modes, chosen by argv[1] (default "spawn"):
|
||||
//! spawn — M2: one worker writes a shared global; the main thread observes it, proving
|
||||
//! `runtime.Thread.spawn` started a task in the **same** address space.
|
||||
//! join — M3: N workers each do K atomic increments on a shared counter and stamp the
|
||||
//! core they ran on; the main thread `join`s all N and checks the total is
|
||||
//! exactly N*K (every worker ran, join waited) and that >1 core was used
|
||||
//! (genuine parallelism). Then a detached worker proves `detach` runs and
|
||||
//! needs no join.
|
||||
//!
|
||||
//! Built multi-threaded (`addThreadedUserBinary`), so the poll below is a real atomic
|
||||
//! load the compiler must re-read — under a single-threaded build it could be hoisted.
|
||||
//! Built multi-threaded (`addThreadedUserBinary`) so atomics/shared reads are real.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
|
||||
/// Written by the worker thread, read by main — the shared-address-space evidence.
|
||||
var shared_value: u32 = 0;
|
||||
/// Release/acquire handshake: publishes the `shared_value` write to the reader.
|
||||
var done = std.atomic.Value(u32).init(0);
|
||||
fn write(comptime s: []const u8) void {
|
||||
_ = runtime.system.write(s);
|
||||
}
|
||||
|
||||
// --- M2: spawn mode ---------------------------------------------------------
|
||||
|
||||
var shared_value: u32 = 0;
|
||||
var spawn_done = std.atomic.Value(u32).init(0);
|
||||
const sentinel: u32 = 0xA5A5;
|
||||
|
||||
fn worker() void {
|
||||
shared_value = sentinel; // a plain write to a global we share with main
|
||||
done.store(1, .release); // ...published by this release store
|
||||
fn spawnWorker() void {
|
||||
shared_value = sentinel;
|
||||
spawn_done.store(1, .release);
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
_ = runtime.system.write("thread-test: starting\n");
|
||||
|
||||
_ = runtime.Thread.spawn(.{}, worker, .{}) catch {
|
||||
_ = runtime.system.write("thread-test: FAIL spawn refused\n");
|
||||
fn runSpawnMode() void {
|
||||
write("thread-test: starting\n");
|
||||
_ = runtime.Thread.spawn(.{}, spawnWorker, .{}) catch {
|
||||
write("thread-test: FAIL spawn refused\n");
|
||||
return;
|
||||
};
|
||||
|
||||
// Bounded wait for the worker to run and publish. yield() keeps the core useful;
|
||||
// the acquire load pairs with the worker's release store.
|
||||
var spins: usize = 0;
|
||||
while (done.load(.acquire) == 0 and spins < 50_000_000) : (spins += 1) {
|
||||
while (spawn_done.load(.acquire) == 0 and spins < 50_000_000) : (spins += 1) {
|
||||
runtime.system.yield();
|
||||
}
|
||||
|
||||
if (done.load(.acquire) == 1 and shared_value == sentinel) {
|
||||
_ = runtime.system.write("thread-test: child ran in shared aspace ok\n");
|
||||
if (spawn_done.load(.acquire) == 1 and shared_value == sentinel) {
|
||||
write("thread-test: child ran in shared aspace ok\n");
|
||||
} else {
|
||||
_ = runtime.system.write("thread-test: FAIL worker did not update shared memory\n");
|
||||
write("thread-test: FAIL worker did not update shared memory\n");
|
||||
}
|
||||
}
|
||||
|
||||
// --- M3: join mode ----------------------------------------------------------
|
||||
|
||||
const worker_count: u32 = 4;
|
||||
const iterations: u64 = 100_000;
|
||||
|
||||
var counter = std.atomic.Value(u64).init(0);
|
||||
var cores_seen = std.atomic.Value(u32).init(0);
|
||||
|
||||
fn joinWorker() void {
|
||||
var i: u64 = 0;
|
||||
while (i < iterations) : (i += 1) {
|
||||
_ = counter.fetchAdd(1, .monotonic);
|
||||
if (i % 1000 == 0) stampCore(); // periodic: catches cross-core migration too
|
||||
}
|
||||
stampCore();
|
||||
}
|
||||
|
||||
fn stampCore() void {
|
||||
const core = runtime.Thread.currentCore();
|
||||
if (core < 32) _ = cores_seen.fetchOr(@as(u32, 1) << @intCast(core), .monotonic);
|
||||
}
|
||||
|
||||
var detach_done = std.atomic.Value(u32).init(0);
|
||||
|
||||
fn detachWorker() void {
|
||||
detach_done.store(1, .release);
|
||||
}
|
||||
|
||||
fn runJoinMode() void {
|
||||
write("thread-test: join mode starting\n");
|
||||
|
||||
var threads: [worker_count]runtime.Thread = undefined;
|
||||
var spawned: u32 = 0;
|
||||
while (spawned < worker_count) : (spawned += 1) {
|
||||
threads[spawned] = runtime.Thread.spawn(.{}, joinWorker, .{}) catch break;
|
||||
}
|
||||
if (spawned != worker_count) {
|
||||
write("thread-test: FAIL could not spawn all workers\n");
|
||||
return;
|
||||
}
|
||||
for (threads[0..spawned]) |t| t.join();
|
||||
|
||||
const total = counter.load(.acquire);
|
||||
const cores = @popCount(cores_seen.load(.acquire));
|
||||
if (total != worker_count * iterations) {
|
||||
write("thread-test: FAIL counter mismatch (a worker was lost or join did not wait)\n");
|
||||
return;
|
||||
}
|
||||
if (cores <= 1) {
|
||||
write("thread-test: FAIL workers never ran on more than one core\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// detach: the worker runs and we never join it.
|
||||
const dt = runtime.Thread.spawn(.{}, detachWorker, .{}) catch {
|
||||
write("thread-test: FAIL detach spawn refused\n");
|
||||
return;
|
||||
};
|
||||
dt.detach();
|
||||
var spins: usize = 0;
|
||||
while (detach_done.load(.acquire) == 0 and spins < 50_000_000) : (spins += 1) {
|
||||
runtime.system.yield();
|
||||
}
|
||||
if (detach_done.load(.acquire) != 1) {
|
||||
write("thread-test: FAIL detached worker did not run\n");
|
||||
return;
|
||||
}
|
||||
|
||||
write("thread-test: join ok\n"); // the M3 verdict marker
|
||||
}
|
||||
|
||||
pub fn main(init: runtime.process.Init) void {
|
||||
const mode = init.arguments.get(1) orelse "spawn";
|
||||
if (std.mem.eql(u8, mode, "join")) runJoinMode() else runSpawnMode();
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user