threads(M10): per-thread fs.base — the TLS thread-pointer mechanism

Each thread gets its own x86_64 thread pointer (FS base) for user-space TLS.
Task.fs_base is restored on every context switch only when it changes (same
conditional-load discipline as CR3; architecture.setFsBase -> wrmsr
IA32_FS_BASE). New set_thread_pointer=44 syscall sets the caller's fs_base and
loads it now. The kernel never touches FS, so no swapgs complication.

The runtime lays a small per-thread TLS block at the top of each thread's stack
(self-pointer at %fs:0 + scratch) and the thread trampoline calls
set_thread_pointer before any user code — so every spawned thread has a private,
switch-stable thread pointer, reclaimed with the stack.

thread-test tls mode: two threads write unique markers to their own %fs:8 and,
after both wrote, read back — a shared fs.base would clobber one (cross-talk).

Deferred: the Zig threadlocal *compiler* layer (ELF variant-II PT_TLS + linker
sections + template copy) — high-uncertainty, no consumer today; this lands the
load-bearing per-thread fs.base it builds on. See docs/threading-plan.md M10.

Gate thread-tls PASS (3x); full guardrail 25/25; build + host tests clean.
This commit is contained in:
2026-07-20 23:55:51 +01:00
parent 6bc329456a
commit c7e9b5a4f6
9 changed files with 207 additions and 21 deletions
@@ -304,6 +304,15 @@ pub fn cpuLocal() usize {
return pcpu.scheduler();
}
const ia32_fs_base = 0xC000_0100;
/// Set the FS-segment base — the x86_64 thread pointer for user-space TLS. The kernel
/// never touches FS, so this only affects the user task that runs next; the scheduler
/// restores it per task across context switches (docs/threading-plan.md M10).
pub fn setFsBase(base: u64) void {
io.wrmsr(ia32_fs_base, base);
}
// --- SMP: application-processor bring-up ----------------------------------
/// Record the low (<1 MiB) frame reserved for the AP trampoline. Run once at boot.
+15
View File
@@ -232,6 +232,7 @@ fn system_call(state: *architecture.CpuState) void {
.current_core => systemCurrentCore(state),
.thread_self => systemThreadSelf(state),
.thread_join => systemThreadJoin(state),
.set_thread_pointer => systemSetThreadPointer(state),
.futex_wait => systemFutexWait(state),
.futex_wake => systemFutexWake(state),
.thread_exit => {
@@ -707,6 +708,20 @@ fn systemThreadSelf(state: *architecture.CpuState) void {
architecture.setSystemCallResult(state, scheduler.currentId());
}
/// set_thread_pointer(addr) -> 0: set the caller's FS base (its user-space TLS thread
/// pointer). The kernel never uses FS; the scheduler restores this per task across context
/// switches (docs/threading-plan.md M10). `addr` must be a user-half address.
fn systemSetThreadPointer(state: *architecture.CpuState) void {
const addr = architecture.systemCallArg(state, 0);
const t = scheduler.current();
if (t.aspace == 0) return fail(state); // kernel tasks have no user TLS
if (addr >= user_half_end) return fail(state);
const flags = sync.enter();
scheduler.setThreadPointerLocked(addr);
sync.leave(flags);
architecture.setSystemCallResult(state, 0);
}
/// thread_join(tid) -> 0: block until the thread with id `tid` has exited (docs/threading-
/// plan.md M9). Needs no per-thread IPC endpoint. The compare-and-block is one critical
/// section, so an exit cannot slip between "is it alive?" and the block.
+21
View File
@@ -89,6 +89,10 @@ pub const Task = struct {
// The task id this task is blocked in `thread_join` on (0 = not joining). Woken by
// `wakeJoinersLocked` when that task exits (docs/threading-plan.md M9).
join_target: u32 = 0,
// The FS-base (x86_64 thread pointer) for this task's user-space TLS — 0 until the
// task sets it via `set_thread_pointer`. Restored on every context switch to this task
// (docs/threading-plan.md M10).
fs_base: u64 = 0,
// The mmap / MMIO grant-arena cursors moved from Task to the per-address-space object
// (`AspaceRef`, below) so threads sharing one address space hand out disjoint grants
// — see aspaceMmapNextPtr / aspaceDeviceMapNextPtr (docs/threading-plan.md M7).
@@ -284,6 +288,7 @@ pub const PerCpu = struct {
index: u32 = 0, // dense 0-based core index
online: bool = false, // has this core finished bring-up?
loaded_aspace: u64 = 0, // the address-space root currently loaded on this core
loaded_fs_base: u64 = 0, // the FS base currently loaded on this core (docs/threading-plan.md M10)
// Tasks pinned to this core (affinity == index), per priority level + bitmap.
pinned_head: [number_priorities]?*Task = .{null} ** number_priorities,
pinned_tail: [number_priorities]?*Task = .{null} ** number_priorities,
@@ -568,6 +573,12 @@ fn switchTo(pc: *PerCpu, save_sp: *usize, next: *Task) void {
architecture.loadPageTable(want);
pc.loaded_aspace = want;
}
// Restore the next task's user TLS thread pointer (FS base) — only on change, the same
// conditional-load discipline as CR3 above (docs/threading-plan.md M10).
if (next.fs_base != pc.loaded_fs_base) {
architecture.setFsBase(next.fs_base);
pc.loaded_fs_base = next.fs_base;
}
architecture.switchContext(save_sp, next.sp);
// Resumed now (switchContext returned into our own switchTo frame). Re-fetch the core
// via thisCpu(): the `pc` parameter is from *our* earlier switchTo call, so it names
@@ -667,6 +678,16 @@ pub fn joinThreadLocked(tid: u32) void {
}
}
/// Set the calling task's user TLS thread pointer (FS base) and load it now. Persisted on
/// the Task so context switches restore it (docs/threading-plan.md M10). Caller holds the
/// kernel lock.
pub fn setThreadPointerLocked(addr: u64) void {
const pc = thisCpu();
pc.current.fs_base = addr;
architecture.setFsBase(addr);
pc.loaded_fs_base = addr;
}
/// Wake every task blocked in `thread_join` on `tid` — called from the exit paths once the
/// exiting task's state is `.free`. Caller holds the lock.
fn wakeJoinersLocked(tid: u32) void {
+45
View File
@@ -155,6 +155,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
threadAllocTest(boot_information);
} else if (eql(case, "task-reap")) {
taskReapTest(boot_information);
} else if (eql(case, "thread-tls")) {
threadTlsTest(boot_information);
} else if (eql(case, "args")) {
argsTest(boot_information);
} else if (eql(case, "init")) {
@@ -1740,6 +1742,49 @@ fn threadAllocTest(boot_information: *const BootInformation) void {
result();
}
/// Per-thread TLS / fs.base (docs/threading-plan.md M10): `thread-test` in tls mode has two
/// threads each set their own FS base and write a unique marker to `%fs:8`, then — after
/// both have written — read it back. If fs.base were not per-thread and restored across
/// context switches, the second write would clobber the first and a thread would read the
/// wrong marker. The verdict marker means both read their own value (no cross-talk).
fn threadTlsTest(boot_information: *const BootInformation) void {
log("DANOS-TEST-BEGIN: thread-tls\n", .{});
if (boot_information.initial_ramdisk_len == 0) {
check("bootloader handed over an initial_ramdisk", false);
result();
return;
}
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
const rd = initial_ramdisk.Reader.init(image) orelse {
check("initial_ramdisk image is valid", false);
result();
return;
};
var started = false;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(item.name, "thread-test")) continue;
started = if (process.spawnProcess(item.blob, 4, &.{ "thread-test", "tls" })) true else |_| false;
break;
}
check("thread-test (tls mode) spawned", started);
const ok_marker = "thread-tls: ok";
const fail_marker = "thread-tls: FAIL";
scheduler.setPriority(1);
const deadline = architecture.millis() + 12000;
while (architecture.millis() < deadline) {
if (bufferHas(ok_marker) or bufferHas(fail_marker)) break;
scheduler.yield();
}
scheduler.setPriority(4);
check("each thread has its own fs.base TLS slot (no cross-talk across switches)", bufferHas(ok_marker) and !bufferHas(fail_marker));
result();
}
/// The task reaper (docs/threading-plan.md M8): a dead task's kernel stack used to be
/// leaked ("no reaper yet"). Spawn and kill many ring-3 processes and confirm the total
/// kernel-stack bytes return to baseline — every stack reclaimed, no leak. (Threads exit