threads(M7): thread-safe allocation (per-aspace mmap arena + locked heap)

Move the mmap/mmio grant-arena cursors off Task into the per-address-space object
(scheduler aspace_refs, exposed via aspaceMmapNextPtr/aspaceDeviceMapNextPtr), so
sibling threads in one address space hand out disjoint grants. systemMmap reserves
a range under a brief lock then maps per page under a short-held lock (not the
whole grant): the big lock runs with interrupts disabled, so pinning it across a
multi-MiB memset+map froze other cores. Guard the runtime heap's rawAlloc/rawFree
with a Thread.Mutex, gated on !single_threaded so ordinary binaries compile it out.

thread-test gains an alloc mode: 4 threads x 500 alloc/fill/verify/free cycles;
any overlap between concurrent allocations is caught by the pattern check.

Also fix the affinity guardrail: its 3-billion-iteration busy-loop had
codegen-dependent wall-time (adding a function to tests.zig swung it ~4s -> ~63s
and timed it out). Reworked to wait on the wall clock instead.

Gate thread-alloc PASS (3x); full guardrail 23/23 green; build + host tests clean.
This commit is contained in:
2026-07-20 22:57:11 +01:00
parent f4813c8e99
commit 8259678f0a
7 changed files with 228 additions and 47 deletions
+39 -18
View File
@@ -58,8 +58,9 @@ pub const stack_top_virtual: u64 = stack_base_virtual + parameters.user_stack_pa
/// The mmap grant arena: where `mmap` hands out fresh user pages, above the image
/// and stack but still inside PML4[224] (so no kernel mapping is widened). Each
/// process bump-allocates from `heap_arena_base` upward via `Task.heap_next`; a
/// 1 GiB window is far more than any user heap needs today.
/// process bump-allocates from `heap_arena_base` upward via a per-address-space cursor
/// (`scheduler.aspaceMmapNextPtr`, shared by its threads); a 1 GiB window is far more
/// than any user heap needs today.
pub const heap_arena_base: u64 = 0x0000_7000_1000_0000;
pub const heap_arena_end: u64 = heap_arena_base + (1 << 30);
@@ -69,8 +70,8 @@ pub const user_half_end: u64 = 0x0000_8000_0000_0000;
/// The MMIO-grant arena: where `mmio_map` places device windows, in PML4[226] —
/// a user-exclusive region distinct from code/stack/heap (PML4[224]), so mapping
/// device pages user-accessible widens no kernel mapping. Per-process cursor in
/// `Task.device_map_next`.
/// device pages user-accessible widens no kernel mapping. Per-address-space cursor
/// (`scheduler.aspaceDeviceMapNextPtr`).
pub const device_arena_base: u64 = 0x0000_7100_0000_0000;
pub const device_arena_end: u64 = device_arena_base + (4 << 30);
@@ -373,18 +374,23 @@ fn systemMmioMap(state: *architecture.CpuState) void {
if (r.len == 0) return fail(state);
if (@addWithOverflow(r.start, r.len)[1] != 0) return fail(state);
if (t.device_map_next == 0) t.device_map_next = device_arena_base;
const first = r.start & ~@as(u64, page_size - 1);
const last = (r.start + r.len - 1) & ~@as(u64, page_size - 1);
const pages = (last - first) / page_size + 1;
const base_v = t.device_map_next;
if (base_v + pages * page_size > device_arena_end) return fail(state);
// A framebuffer resource asks (via its flag) to be mapped write-combining rather
// than the strong-uncacheable default that register MMIO needs.
const write_combining = (r.flags & device_abi.resource_flag_write_combining) != 0;
// Per-address-space cursor + shared page tables → serialize under the big lock,
// same as mmap (docs/threading-plan.md M7).
const flags = sync.enter();
defer sync.leave(flags);
const cursor = scheduler.aspaceDeviceMapNextPtr(t.aspace) orelse return fail(state);
if (cursor.* == 0) cursor.* = device_arena_base; // seed the arena lazily
const base_v = cursor.*;
if (base_v + pages * page_size > device_arena_end) return fail(state);
architecture.mapUserDeviceInto(t.aspace, base_v, r.start, r.len, write_combining);
t.device_map_next = base_v + pages * page_size;
cursor.* = base_v + pages * page_size;
architecture.setSystemCallResult(state, base_v + (r.start & (page_size - 1))); // register base
}
@@ -1232,16 +1238,30 @@ fn systemMmap(state: *architecture.CpuState) void {
const pages = (len + page_size - 1) / page_size;
if (pages == 0 or pages > maximum_mmap_pages) return fail(state);
if (t.heap_next == 0) t.heap_next = heap_arena_base; // seed the arena lazily
const base = t.heap_next;
if (base + pages * page_size > heap_arena_end) return fail(state); // arena exhausted
// Reserve a disjoint range under a *brief* lock (the cursor is shared by every thread
// in this address space). The mapping below then takes the lock **per page**, not for
// the whole grant: the big lock is held with interrupts disabled, so pinning it across
// a multi-MiB memset+map would freeze every other core on its next tick — which timed
// the `affinity` scenario out (docs/threading-plan.md M7).
const base = reserve: {
const flags = sync.enter();
defer sync.leave(flags);
const cursor = scheduler.aspaceMmapNextPtr(t.aspace) orelse return fail(state);
if (cursor.* == 0) cursor.* = heap_arena_base; // seed the arena lazily
const b = cursor.*;
if (b + pages * page_size > heap_arena_end) return fail(state); // arena exhausted
cursor.* = b + pages * page_size; // reserve now, so concurrent grants can't overlap
break :reserve b;
};
// Map page by page. On mid-way frame exhaustion, roll back the pages already mapped
// (unmap + free) so no partial grant leaks into the address space — the same
// all-or-nothing guarantee as before, but without a fixed scratch array, so the
// per-call size can be a multi-MiB framebuffer.
// Map the reserved range page by page, each page under a short-held lock (the range is
// already reserved, so pages can't overlap another thread's; the lock only serializes
// the shared page-table walk). On mid-way frame exhaustion, roll back the mapped pages
// so no partial grant leaks — the reserved-but-unmapped tail of the arena is left
// fallow (a rare, bounded address-space leak, not a memory leak).
var mapped: usize = 0;
while (mapped < pages) : (mapped += 1) {
const flags = sync.enter();
const frame = pmm.alloc() orelse {
var i: usize = 0;
while (i < mapped) : (i += 1) {
@@ -1251,14 +1271,15 @@ fn systemMmap(state: *architecture.CpuState) void {
pmm.free(physical);
}
}
sync.leave(flags);
return fail(state);
};
const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(frame));
@memset(destination[0..page_size], 0); // hand out zeroed memory
architecture.mapUserPageInto(t.aspace, base + mapped * page_size, frame, true, false); // RW + NX
sync.leave(flags);
}
t.heap_next = base + pages * page_size;
architecture.setSystemCallResult(state, base);
architecture.setSystemCallResult(state, base); // the cursor was already advanced at reserve
}
/// munmap(base, len): release a range previously handed out by `mmap`. Unmaps
+28 -8
View File
@@ -83,13 +83,9 @@ pub const Task = struct {
// The user address this task is blocked on in futex_wait (0 = not futex-waiting).
// Cleared to 0 by futexWakeLocked as the "woken, not timed out" signal (docs/threading.md).
futex_addr: u64 = 0,
// Next free virtual address in this task's mmap grant arena (0 = uninitialised;
// process.zig lazily seeds it to the arena base on the first mmap). Bumped up
// as the user heap grows; user task only.
heap_next: u64 = 0,
// Next free virtual address in this task's MMIO-grant arena (PML4[226]; 0 =
// uninitialised, process.zig seeds it on the first mmio_map). User task only.
device_map_next: u64 = 0,
// The mmap / MMIO grant-arena cursors moved from Task to the per-address-space object
// (`AspaceRef`, below) so threads sharing one address space hand out disjoint grants
// — see aspaceMmapNextPtr / aspaceDeviceMapNextPtr (docs/threading-plan.md M7).
// --- synchronous IPC (ipc_sync.zig) ---
// Per-process handle table: a small-int handle names a kernel capability object.
// Each entry tags its `kind` (an IPC endpoint or a shared-memory object) so the
@@ -148,7 +144,11 @@ var tasks = [_]Task{.{}} ** maximum_tasks;
/// only when the **last** task on an address space exits. All access is under the big
/// kernel lock. There can be no more live address spaces than tasks, so the table is
/// sized to the task pool and never overflows in practice.
const AspaceRef = struct { root: u64 = 0, count: u32 = 0 };
// The per-address-space kernel object: a reference count plus the grant-arena cursors.
// One live entry per address space; threads sharing an address space share this entry,
// so their mmap/mmio grants bump one cursor and never overlap (docs/threading-plan.md M7).
// `mmap_next`/`device_map_next` are 0 until process.zig seeds them to the arena base.
const AspaceRef = struct { root: u64 = 0, count: u32 = 0, mmap_next: u64 = 0, device_map_next: u64 = 0 };
var aspace_refs = [_]AspaceRef{.{}} ** maximum_tasks;
var aspace_destroy_count: u64 = 0;
@@ -202,6 +202,26 @@ pub fn liveAspaceCount() u32 {
pub fn aspaceDestroyCount() u64 {
return aspace_destroy_count;
}
/// Pointer to the mmap grant-arena cursor for address space `root`, so the mmap syscall
/// can read-and-bump it. Per-address-space (not per-task), so sibling threads get
/// disjoint grants. **Caller holds the kernel lock** (the entry is stable while held).
/// Null only if `root` was never retained — which can't happen for a live user task.
pub fn aspaceMmapNextPtr(root: u64) ?*u64 {
for (&aspace_refs) |*entry| {
if (entry.count != 0 and entry.root == root) return &entry.mmap_next;
}
return null;
}
/// Pointer to the MMIO grant-arena cursor for address space `root` (see
/// `aspaceMmapNextPtr`). Caller holds the kernel lock.
pub fn aspaceDeviceMapNextPtr(root: u64) ?*u64 {
for (&aspace_refs) |*entry| {
if (entry.count != 0 and entry.root == root) return &entry.device_map_next;
}
return null;
}
var next_id: u32 = 1;
/// Per-CPU scheduler state: the task each core is running, its own idle task, and a
+53 -4
View File
@@ -151,6 +151,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
threadMutexTest(boot_information);
} else if (eql(case, "thread-id")) {
threadIdTest(boot_information);
} else if (eql(case, "thread-alloc")) {
threadAllocTest(boot_information);
} else if (eql(case, "args")) {
argsTest(boot_information);
} else if (eql(case, "init")) {
@@ -779,11 +781,15 @@ fn affinityTest() void {
return;
}
var spins: u64 = 0;
while (spins < 3_000_000_000) spins +%= 1; // many time slices across the cores
// Let many time slices pass so the scheduler runs the pinned worker across ticks.
// Wait on the wall clock, not a raw iteration count: a fixed-count busy-loop's
// wall-time is a codegen lottery (the optimiser may elide or vectorise it), so an
// unrelated change elsewhere in this file could swing this test from ~4 s to ~50 s.
const run_until = architecture.millis() + 400;
while (architecture.millis() < run_until) {}
affinity_running = false;
var settle: u64 = 0;
while (settle < 200_000_000) settle +%= 1; // let the worker see the flag and exit
const settle_until = architecture.millis() + 50;
while (architecture.millis() < settle_until) {} // let the worker see the flag and exit
var others: u32 = 0;
for (affinity_cores, 0..) |seen, c| {
@@ -1689,6 +1695,49 @@ fn threadIdTest(boot_information: *const BootInformation) void {
result();
}
/// Thread-safe allocation (docs/threading-plan.md M7): `thread-test` in alloc mode runs N
/// threads that each do many `alloc`/fill/verify/`free` cycles of varied sizes on the
/// shared runtime heap. If the heap lock or the per-address-space mmap arena were unsafe,
/// two threads' blocks would overlap and a thread would read another's pattern; the
/// verdict marker is emitted only when every thread completes with every block intact.
fn threadAllocTest(boot_information: *const BootInformation) void {
log("DANOS-TEST-BEGIN: thread-alloc\n", .{});
if (boot_information.initial_ramdisk_len == 0) {
check("bootloader handed over an initial_ramdisk", false);
result();
return;
}
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
const rd = initial_ramdisk.Reader.init(image) orelse {
check("initial_ramdisk image is valid", false);
result();
return;
};
var started = false;
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!eql(item.name, "thread-test")) continue;
started = if (process.spawnProcess(item.blob, 4, &.{ "thread-test", "alloc" })) true else |_| false;
break;
}
check("thread-test (alloc mode) spawned", started);
const ok_marker = "thread-alloc: ok";
const fail_marker = "thread-alloc: FAIL";
scheduler.setPriority(1);
const deadline = architecture.millis() + 20000;
while (architecture.millis() < deadline) {
if (bufferHas(ok_marker) or bufferHas(fail_marker)) break;
scheduler.yield();
}
scheduler.setPriority(4);
check("concurrent heap allocation stayed corruption-free (shared heap + per-aspace arena)", bufferHas(ok_marker) and !bufferHas(fail_marker));
result();
}
/// The full PID-1 path: the bootloader read /system/services/init off the boot volume and
/// handed it over; load it as a user ELF and spawn it as a real ring-3 process
/// — the same call the normal boot path makes — then confirm it beats. init