threads(M7): thread-safe allocation (per-aspace mmap arena + locked heap)
Move the mmap/mmio grant-arena cursors off Task into the per-address-space object (scheduler aspace_refs, exposed via aspaceMmapNextPtr/aspaceDeviceMapNextPtr), so sibling threads in one address space hand out disjoint grants. systemMmap reserves a range under a brief lock then maps per page under a short-held lock (not the whole grant): the big lock runs with interrupts disabled, so pinning it across a multi-MiB memset+map froze other cores. Guard the runtime heap's rawAlloc/rawFree with a Thread.Mutex, gated on !single_threaded so ordinary binaries compile it out. thread-test gains an alloc mode: 4 threads x 500 alloc/fill/verify/free cycles; any overlap between concurrent allocations is caught by the pattern check. Also fix the affinity guardrail: its 3-billion-iteration busy-loop had codegen-dependent wall-time (adding a function to tests.zig swung it ~4s -> ~63s and timed it out). Reworked to wait on the wall clock instead. Gate thread-alloc PASS (3x); full guardrail 23/23 green; build + host tests clean.
This commit is contained in:
+39
-18
@@ -58,8 +58,9 @@ pub const stack_top_virtual: u64 = stack_base_virtual + parameters.user_stack_pa
|
||||
|
||||
/// The mmap grant arena: where `mmap` hands out fresh user pages, above the image
|
||||
/// and stack but still inside PML4[224] (so no kernel mapping is widened). Each
|
||||
/// process bump-allocates from `heap_arena_base` upward via `Task.heap_next`; a
|
||||
/// 1 GiB window is far more than any user heap needs today.
|
||||
/// process bump-allocates from `heap_arena_base` upward via a per-address-space cursor
|
||||
/// (`scheduler.aspaceMmapNextPtr`, shared by its threads); a 1 GiB window is far more
|
||||
/// than any user heap needs today.
|
||||
pub const heap_arena_base: u64 = 0x0000_7000_1000_0000;
|
||||
pub const heap_arena_end: u64 = heap_arena_base + (1 << 30);
|
||||
|
||||
@@ -69,8 +70,8 @@ pub const user_half_end: u64 = 0x0000_8000_0000_0000;
|
||||
|
||||
/// The MMIO-grant arena: where `mmio_map` places device windows, in PML4[226] —
|
||||
/// a user-exclusive region distinct from code/stack/heap (PML4[224]), so mapping
|
||||
/// device pages user-accessible widens no kernel mapping. Per-process cursor in
|
||||
/// `Task.device_map_next`.
|
||||
/// device pages user-accessible widens no kernel mapping. Per-address-space cursor
|
||||
/// (`scheduler.aspaceDeviceMapNextPtr`).
|
||||
pub const device_arena_base: u64 = 0x0000_7100_0000_0000;
|
||||
pub const device_arena_end: u64 = device_arena_base + (4 << 30);
|
||||
|
||||
@@ -373,18 +374,23 @@ fn systemMmioMap(state: *architecture.CpuState) void {
|
||||
if (r.len == 0) return fail(state);
|
||||
if (@addWithOverflow(r.start, r.len)[1] != 0) return fail(state);
|
||||
|
||||
if (t.device_map_next == 0) t.device_map_next = device_arena_base;
|
||||
const first = r.start & ~@as(u64, page_size - 1);
|
||||
const last = (r.start + r.len - 1) & ~@as(u64, page_size - 1);
|
||||
const pages = (last - first) / page_size + 1;
|
||||
const base_v = t.device_map_next;
|
||||
if (base_v + pages * page_size > device_arena_end) return fail(state);
|
||||
|
||||
// A framebuffer resource asks (via its flag) to be mapped write-combining rather
|
||||
// than the strong-uncacheable default that register MMIO needs.
|
||||
const write_combining = (r.flags & device_abi.resource_flag_write_combining) != 0;
|
||||
|
||||
// Per-address-space cursor + shared page tables → serialize under the big lock,
|
||||
// same as mmap (docs/threading-plan.md M7).
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const cursor = scheduler.aspaceDeviceMapNextPtr(t.aspace) orelse return fail(state);
|
||||
if (cursor.* == 0) cursor.* = device_arena_base; // seed the arena lazily
|
||||
const base_v = cursor.*;
|
||||
if (base_v + pages * page_size > device_arena_end) return fail(state);
|
||||
architecture.mapUserDeviceInto(t.aspace, base_v, r.start, r.len, write_combining);
|
||||
t.device_map_next = base_v + pages * page_size;
|
||||
cursor.* = base_v + pages * page_size;
|
||||
architecture.setSystemCallResult(state, base_v + (r.start & (page_size - 1))); // register base
|
||||
}
|
||||
|
||||
@@ -1232,16 +1238,30 @@ fn systemMmap(state: *architecture.CpuState) void {
|
||||
const pages = (len + page_size - 1) / page_size;
|
||||
if (pages == 0 or pages > maximum_mmap_pages) return fail(state);
|
||||
|
||||
if (t.heap_next == 0) t.heap_next = heap_arena_base; // seed the arena lazily
|
||||
const base = t.heap_next;
|
||||
if (base + pages * page_size > heap_arena_end) return fail(state); // arena exhausted
|
||||
// Reserve a disjoint range under a *brief* lock (the cursor is shared by every thread
|
||||
// in this address space). The mapping below then takes the lock **per page**, not for
|
||||
// the whole grant: the big lock is held with interrupts disabled, so pinning it across
|
||||
// a multi-MiB memset+map would freeze every other core on its next tick — which timed
|
||||
// the `affinity` scenario out (docs/threading-plan.md M7).
|
||||
const base = reserve: {
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
const cursor = scheduler.aspaceMmapNextPtr(t.aspace) orelse return fail(state);
|
||||
if (cursor.* == 0) cursor.* = heap_arena_base; // seed the arena lazily
|
||||
const b = cursor.*;
|
||||
if (b + pages * page_size > heap_arena_end) return fail(state); // arena exhausted
|
||||
cursor.* = b + pages * page_size; // reserve now, so concurrent grants can't overlap
|
||||
break :reserve b;
|
||||
};
|
||||
|
||||
// Map page by page. On mid-way frame exhaustion, roll back the pages already mapped
|
||||
// (unmap + free) so no partial grant leaks into the address space — the same
|
||||
// all-or-nothing guarantee as before, but without a fixed scratch array, so the
|
||||
// per-call size can be a multi-MiB framebuffer.
|
||||
// Map the reserved range page by page, each page under a short-held lock (the range is
|
||||
// already reserved, so pages can't overlap another thread's; the lock only serializes
|
||||
// the shared page-table walk). On mid-way frame exhaustion, roll back the mapped pages
|
||||
// so no partial grant leaks — the reserved-but-unmapped tail of the arena is left
|
||||
// fallow (a rare, bounded address-space leak, not a memory leak).
|
||||
var mapped: usize = 0;
|
||||
while (mapped < pages) : (mapped += 1) {
|
||||
const flags = sync.enter();
|
||||
const frame = pmm.alloc() orelse {
|
||||
var i: usize = 0;
|
||||
while (i < mapped) : (i += 1) {
|
||||
@@ -1251,14 +1271,15 @@ fn systemMmap(state: *architecture.CpuState) void {
|
||||
pmm.free(physical);
|
||||
}
|
||||
}
|
||||
sync.leave(flags);
|
||||
return fail(state);
|
||||
};
|
||||
const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(frame));
|
||||
@memset(destination[0..page_size], 0); // hand out zeroed memory
|
||||
architecture.mapUserPageInto(t.aspace, base + mapped * page_size, frame, true, false); // RW + NX
|
||||
sync.leave(flags);
|
||||
}
|
||||
t.heap_next = base + pages * page_size;
|
||||
architecture.setSystemCallResult(state, base);
|
||||
architecture.setSystemCallResult(state, base); // the cursor was already advanced at reserve
|
||||
}
|
||||
|
||||
/// munmap(base, len): release a range previously handed out by `mmap`. Unmaps
|
||||
|
||||
Reference in New Issue
Block a user