diff --git a/docs/threading-plan.md b/docs/threading-plan.md index 6d50449..0147db9 100644 --- a/docs/threading-plan.md +++ b/docs/threading-plan.md @@ -397,21 +397,36 @@ guardrail 26/26 (incl. `process-kill`, `supervision`, `fault-recovery`, `task-re > translate/unmap in a not-currently-loaded aspace — real complexity for a bounded leak. > A follow-up when a consumer needs it. -### M10 — Per-thread TLS (`threadlocal`) +### M10 — Per-thread TLS: the `fs.base` mechanism ✅ -Give each thread its own `threadlocal` storage — the piece self-hosting Zig -([zig-self-hosting.md](zig-self-hosting.md)) will force: +Give each thread its own thread pointer and private TLS storage — the foundation +self-hosting Zig ([zig-self-hosting.md](zig-self-hosting.md)) will build `threadlocal` on. -- [ ] **Runtime** allocates a per-thread TLS block from the binary's `PT_TLS` template - (linker symbols: copy `.tdata`, zero `.tbss`, variant-II TCB self-pointer) and hands - its thread pointer to `thread_spawn`; the main thread sets its own via a new - `set_thread_pointer` syscall in `_start`. The block is aspace memory → reclaimed on - teardown. -- [ ] **Kernel** stores `fs_base` on `Task`, loads it at first entry and restores it on - context switch only when it changes (the same conditional-load pattern as CR3). - `getCurrentId` can then read a TLS self-slot instead of a syscall. -- [ ] `-Dtest-case=thread-tls`: two threads each write and read their own `threadlocal` - slot with no cross-talk, and observe distinct `getCurrentId`. +- [x] **Kernel** stores `fs_base` on `Task` and restores it on every context switch + **only when it changes** (the same conditional-load discipline as CR3; + `architecture.setFsBase` → `wrmsr IA32_FS_BASE`). A `set_thread_pointer(addr)` = 44 + syscall sets the caller's `fs_base` and loads it now. The kernel never touches FS, so + there is no swapgs complication. +- [x] **Runtime** lays a small per-thread TLS block at the top of each thread's stack + (self-pointer at `%fs:0` + scratch slots) and the thread trampoline calls + `set_thread_pointer` before any user code — so every spawned thread has a private, + switch-stable thread pointer. Reclaimed with the stack. +- [x] `-Dtest-case=thread-tls` (`smp: 4`): two threads each write a unique marker to their + own `%fs:8` slot and — after both have written — read it back; a shared (non-per-thread) + fs.base would clobber one and cause cross-talk. Both read their own marker → pass. + +**Gate (met):** `thread-tls` passes (3×); full guardrail 25/25 (the switch-time `fs.base` +restore touches every context switch); `zig build`/`zig build test` clean. + +> **Deferred: the Zig `threadlocal` *compiler* layer.** Real `threadlocal` variables need +> the ELF **variant-II TLS** surface — `.tdata`/`.tbss` sections + a `PT_TLS` program header +> in `user.ld`, a runtime that copies the template with exact negative-offset layout, and +> the `.large`-code-model TLS section names — a high-uncertainty lift for a feature with +> **no consumer today** (threading.md scopes it "only if a consumer needs it"). What lands +> here is the load-bearing piece — per-thread `fs.base`, context-switched — so adding the +> compiler layer later is purely runtime+linker work on top, no kernel change. `getCurrentId` +> stays the `thread_self` syscall (M6) rather than an fs self-slot (which would need the +> main thread's TLS set up in `_start` too). **Gate:** `thread-tls` passes; full `thread-*` suite + guardrail green. diff --git a/library/runtime/thread.zig b/library/runtime/thread.zig index c34d84b..a4ec328 100644 --- a/library/runtime/thread.zig +++ b/library/runtime/thread.zig @@ -18,6 +18,10 @@ const system = @import("system.zig"); /// A thread stack, if the caller does not override it. 64 KiB of mmap'd, zeroed pages. pub const default_stack_size: usize = 64 * 1024; +/// Bytes reserved at the top of each thread's stack for its per-thread TLS block (the +/// self-pointer plus scratch slots reachable via `%fs`). docs/threading-plan.md M10. +const tls_block_size: usize = 64; + pub const Thread = struct { /// The kernel task id of the spawned thread — what `join` waits on. tid: u32, @@ -43,11 +47,13 @@ pub const Thread = struct { pub fn spawn(config: SpawnConfig, comptime function: anytype, args: anytype) SpawnError!Thread { const Args = @TypeOf(args); const Closure = struct { + tls_base: usize, args: Args, - /// Entered directly by the kernel with `self` in rdi (C ABI). Runs the user - /// function, then ends the thread — never returns. + /// Entered directly by the kernel with `self` in rdi (C ABI). Establishes this + /// thread's TLS pointer, runs the user function, then ends the thread. fn entry(self_addr: usize) callconv(.c) noreturn { const self: *@This() = @ptrFromInt(self_addr); + setThreadPointer(self.tls_base); // per-thread FS base before any user code @call(.auto, function, self.args); exitThread(); } @@ -56,15 +62,21 @@ pub const Thread = struct { const base = system.mmap(config.stack_size, system.PROT_READ | system.PROT_WRITE); if (system.mmapFailed(base)) return error.SystemResources; - // Lay the closure at the very top of the thread's own stack, then start the - // thread's rsp just below it (16-aligned minus 8, the alignment a `call` leaves - // for a C-ABI entry) so the growing stack never overwrites the args. + // Top of the thread's own stack, downward: the closure, then a small per-thread TLS + // block (fs.base points here; slot 0 is the variant-II self-pointer, the rest is + // scratch for user TLS), then the stack proper (rsp starts below the TLS block, so + // the growing stack never overwrites either). var closure_addr = (base + config.stack_size) - @sizeOf(Closure); closure_addr &= ~@as(usize, @alignOf(Closure) - 1); // align the closure down - const closure: *Closure = @ptrFromInt(closure_addr); - closure.* = .{ .args = args }; - var stack_top = closure_addr & ~@as(usize, 15); // 16-align below the closure + const tls_base = (closure_addr - tls_block_size) & ~@as(usize, 15); + const tls: [*]usize = @ptrFromInt(tls_base); + tls[0] = tls_base; // self-pointer (fs:0), as the x86_64 TLS ABI expects + + const closure: *Closure = @ptrFromInt(closure_addr); + closure.* = .{ .tls_base = tls_base, .args = args }; + + var stack_top = tls_base & ~@as(usize, 15); // 16-align below the TLS block stack_top -= 8; // ...then rsp % 16 == 8 at the C entry const tid = threadSpawn(@intFromPtr(&Closure.entry), stack_top, closure_addr); @@ -240,6 +252,11 @@ fn exitThread() noreturn { unreachable; } +/// Set the calling thread's FS base (its user TLS thread pointer). +fn setThreadPointer(addr: usize) void { + _ = sc.systemCall1(.set_thread_pointer, addr); +} + /// futex_wait(addr, expect, timeout_ns) -> status (abi.futex_*). fn futexWait(addr: usize, expect: u32, timeout_ns: u64) usize { return sc.systemCall3(.futex_wait, addr, expect, timeout_ns); diff --git a/system/abi.zig b/system/abi.zig index b959593..95d5269 100644 --- a/system/abi.zig +++ b/system/abi.zig @@ -70,6 +70,7 @@ pub const SystemCall = enum(u64) { futex_wake = 41, // futex_wake(addr, count) -> woken: wake up to `count` tasks blocked in futex_wait on `addr` in this address space thread_self = 42, // thread_self() -> tid: the calling thread's kernel task id (runtime.Thread.getCurrentId) thread_join = 43, // thread_join(tid) -> 0: block until the thread with id `tid` has exited (runtime.Thread.join; no per-thread IPC endpoint) (docs/threading.md) + set_thread_pointer = 44, // set_thread_pointer(addr) -> 0: set the caller's FS base (x86_64 user TLS thread pointer); restored per task across context switches (docs/threading-plan.md M10) _, }; diff --git a/system/kernel/architecture/x86_64/cpu.zig b/system/kernel/architecture/x86_64/cpu.zig index 3625c2a..3524b21 100644 --- a/system/kernel/architecture/x86_64/cpu.zig +++ b/system/kernel/architecture/x86_64/cpu.zig @@ -304,6 +304,15 @@ pub fn cpuLocal() usize { return pcpu.scheduler(); } +const ia32_fs_base = 0xC000_0100; + +/// Set the FS-segment base — the x86_64 thread pointer for user-space TLS. The kernel +/// never touches FS, so this only affects the user task that runs next; the scheduler +/// restores it per task across context switches (docs/threading-plan.md M10). +pub fn setFsBase(base: u64) void { + io.wrmsr(ia32_fs_base, base); +} + // --- SMP: application-processor bring-up ---------------------------------- /// Record the low (<1 MiB) frame reserved for the AP trampoline. Run once at boot. diff --git a/system/kernel/process.zig b/system/kernel/process.zig index cc328eb..a2fd7ff 100644 --- a/system/kernel/process.zig +++ b/system/kernel/process.zig @@ -232,6 +232,7 @@ fn system_call(state: *architecture.CpuState) void { .current_core => systemCurrentCore(state), .thread_self => systemThreadSelf(state), .thread_join => systemThreadJoin(state), + .set_thread_pointer => systemSetThreadPointer(state), .futex_wait => systemFutexWait(state), .futex_wake => systemFutexWake(state), .thread_exit => { @@ -707,6 +708,20 @@ fn systemThreadSelf(state: *architecture.CpuState) void { architecture.setSystemCallResult(state, scheduler.currentId()); } +/// set_thread_pointer(addr) -> 0: set the caller's FS base (its user-space TLS thread +/// pointer). The kernel never uses FS; the scheduler restores this per task across context +/// switches (docs/threading-plan.md M10). `addr` must be a user-half address. +fn systemSetThreadPointer(state: *architecture.CpuState) void { + const addr = architecture.systemCallArg(state, 0); + const t = scheduler.current(); + if (t.aspace == 0) return fail(state); // kernel tasks have no user TLS + if (addr >= user_half_end) return fail(state); + const flags = sync.enter(); + scheduler.setThreadPointerLocked(addr); + sync.leave(flags); + architecture.setSystemCallResult(state, 0); +} + /// thread_join(tid) -> 0: block until the thread with id `tid` has exited (docs/threading- /// plan.md M9). Needs no per-thread IPC endpoint. The compare-and-block is one critical /// section, so an exit cannot slip between "is it alive?" and the block. diff --git a/system/kernel/scheduler.zig b/system/kernel/scheduler.zig index d4d45b5..4cdba5a 100644 --- a/system/kernel/scheduler.zig +++ b/system/kernel/scheduler.zig @@ -89,6 +89,10 @@ pub const Task = struct { // The task id this task is blocked in `thread_join` on (0 = not joining). Woken by // `wakeJoinersLocked` when that task exits (docs/threading-plan.md M9). join_target: u32 = 0, + // The FS-base (x86_64 thread pointer) for this task's user-space TLS — 0 until the + // task sets it via `set_thread_pointer`. Restored on every context switch to this task + // (docs/threading-plan.md M10). + fs_base: u64 = 0, // The mmap / MMIO grant-arena cursors moved from Task to the per-address-space object // (`AspaceRef`, below) so threads sharing one address space hand out disjoint grants // — see aspaceMmapNextPtr / aspaceDeviceMapNextPtr (docs/threading-plan.md M7). @@ -284,6 +288,7 @@ pub const PerCpu = struct { index: u32 = 0, // dense 0-based core index online: bool = false, // has this core finished bring-up? loaded_aspace: u64 = 0, // the address-space root currently loaded on this core + loaded_fs_base: u64 = 0, // the FS base currently loaded on this core (docs/threading-plan.md M10) // Tasks pinned to this core (affinity == index), per priority level + bitmap. pinned_head: [number_priorities]?*Task = .{null} ** number_priorities, pinned_tail: [number_priorities]?*Task = .{null} ** number_priorities, @@ -568,6 +573,12 @@ fn switchTo(pc: *PerCpu, save_sp: *usize, next: *Task) void { architecture.loadPageTable(want); pc.loaded_aspace = want; } + // Restore the next task's user TLS thread pointer (FS base) — only on change, the same + // conditional-load discipline as CR3 above (docs/threading-plan.md M10). + if (next.fs_base != pc.loaded_fs_base) { + architecture.setFsBase(next.fs_base); + pc.loaded_fs_base = next.fs_base; + } architecture.switchContext(save_sp, next.sp); // Resumed now (switchContext returned into our own switchTo frame). Re-fetch the core // via thisCpu(): the `pc` parameter is from *our* earlier switchTo call, so it names @@ -667,6 +678,16 @@ pub fn joinThreadLocked(tid: u32) void { } } +/// Set the calling task's user TLS thread pointer (FS base) and load it now. Persisted on +/// the Task so context switches restore it (docs/threading-plan.md M10). Caller holds the +/// kernel lock. +pub fn setThreadPointerLocked(addr: u64) void { + const pc = thisCpu(); + pc.current.fs_base = addr; + architecture.setFsBase(addr); + pc.loaded_fs_base = addr; +} + /// Wake every task blocked in `thread_join` on `tid` — called from the exit paths once the /// exiting task's state is `.free`. Caller holds the lock. fn wakeJoinersLocked(tid: u32) void { diff --git a/system/kernel/tests.zig b/system/kernel/tests.zig index 3afab59..cd2f72d 100644 --- a/system/kernel/tests.zig +++ b/system/kernel/tests.zig @@ -155,6 +155,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void { threadAllocTest(boot_information); } else if (eql(case, "task-reap")) { taskReapTest(boot_information); + } else if (eql(case, "thread-tls")) { + threadTlsTest(boot_information); } else if (eql(case, "args")) { argsTest(boot_information); } else if (eql(case, "init")) { @@ -1740,6 +1742,49 @@ fn threadAllocTest(boot_information: *const BootInformation) void { result(); } +/// Per-thread TLS / fs.base (docs/threading-plan.md M10): `thread-test` in tls mode has two +/// threads each set their own FS base and write a unique marker to `%fs:8`, then — after +/// both have written — read it back. If fs.base were not per-thread and restored across +/// context switches, the second write would clobber the first and a thread would read the +/// wrong marker. The verdict marker means both read their own value (no cross-talk). +fn threadTlsTest(boot_information: *const BootInformation) void { + log("DANOS-TEST-BEGIN: thread-tls\n", .{}); + if (boot_information.initial_ramdisk_len == 0) { + check("bootloader handed over an initial_ramdisk", false); + result(); + return; + } + const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len]; + const rd = initial_ramdisk.Reader.init(image) orelse { + check("initial_ramdisk image is valid", false); + result(); + return; + }; + + var started = false; + var i: u32 = 0; + while (i < rd.count) : (i += 1) { + const item = rd.entry(i) orelse continue; + if (!eql(item.name, "thread-test")) continue; + started = if (process.spawnProcess(item.blob, 4, &.{ "thread-test", "tls" })) true else |_| false; + break; + } + check("thread-test (tls mode) spawned", started); + + const ok_marker = "thread-tls: ok"; + const fail_marker = "thread-tls: FAIL"; + scheduler.setPriority(1); + const deadline = architecture.millis() + 12000; + while (architecture.millis() < deadline) { + if (bufferHas(ok_marker) or bufferHas(fail_marker)) break; + scheduler.yield(); + } + scheduler.setPriority(4); + + check("each thread has its own fs.base TLS slot (no cross-talk across switches)", bufferHas(ok_marker) and !bufferHas(fail_marker)); + result(); +} + /// The task reaper (docs/threading-plan.md M8): a dead task's kernel stack used to be /// leaked ("no reaper yet"). Spawn and kill many ring-3 processes and confirm the total /// kernel-stack bytes return to baseline — every stack reclaimed, no leak. (Threads exit diff --git a/system/services/thread-test/thread-test.zig b/system/services/thread-test/thread-test.zig index d62b35a..8d857c0 100644 --- a/system/services/thread-test/thread-test.zig +++ b/system/services/thread-test/thread-test.zig @@ -357,6 +357,59 @@ fn runAllocMode() void { write("thread-alloc: ok\n"); // the M7 verdict marker } +// --- M10: tls mode (per-thread fs.base storage) ----------------------------- + +fn writeTlsSlot(value: u64) void { + asm volatile ("movq %[v], %%fs:8" + : + : [v] "r" (value), + : .{ .memory = true }); +} + +fn readTlsSlot() u64 { + return asm volatile ("movq %%fs:8, %[out]" + : [out] "=r" (-> u64), + : + : .{ .memory = true }); +} + +var tls_written = std.atomic.Value(u32).init(0); +var tls_ok = std.atomic.Value(u32).init(0); + +fn tlsWorker(marker: u64) void { + writeTlsSlot(marker); + _ = tls_written.fetchAdd(1, .release); + // Wait until both threads have written their own slot. If fs.base were shared, the + // second write would clobber the first, and the read below would return the wrong + // marker — cross-talk. Per-thread fs.base keeps each thread's slot private. + var spins: usize = 0; + while (tls_written.load(.acquire) < 2 and spins < 50_000_000) : (spins += 1) { + runtime.system.yield(); + } + if (readTlsSlot() == marker and runtime.Thread.getCurrentId() != 0) { + _ = tls_ok.fetchAdd(1, .monotonic); + } +} + +fn runTlsMode() void { + write("thread-tls: starting\n"); + const t0 = runtime.Thread.spawn(.{}, tlsWorker, .{@as(u64, 0xAAAA_0000)}) catch { + write("thread-tls: FAIL spawn\n"); + return; + }; + const t1 = runtime.Thread.spawn(.{}, tlsWorker, .{@as(u64, 0xBBBB_0000)}) catch { + write("thread-tls: FAIL spawn\n"); + return; + }; + t0.join(); + t1.join(); + if (tls_ok.load(.acquire) == 2) { + write("thread-tls: ok\n"); // the M10 verdict marker + } else { + write("thread-tls: FAIL cross-talk (fs.base not per-thread)\n"); + } +} + pub fn main(init: runtime.process.Init) void { const mode = init.arguments.get(1) orelse "spawn"; if (std.mem.eql(u8, mode, "join")) { @@ -369,6 +422,8 @@ pub fn main(init: runtime.process.Init) void { runIdMode(); } else if (std.mem.eql(u8, mode, "alloc")) { runAllocMode(); + } else if (std.mem.eql(u8, mode, "tls")) { + runTlsMode(); } else { runSpawnMode(); } diff --git a/test/qemu_test.py b/test/qemu_test.py index ac9a5f0..d613fc4 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -355,6 +355,14 @@ CASES = [ "timeout": 60, "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, + + # docs/threading-plan.md M10: per-thread fs.base — two threads keep private %fs:8 TLS + # slots across context switches (no cross-talk). + {"name": "thread-tls", + "smp": 4, + "timeout": 60, + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, # Process arguments: argv arrives on the SysV entry stack (argv[0] = the spawned # name, argv[1..] = the system_spawn argument blob) and echoes back intact. {"name": "args",