kernel: M3 shared-fate — shared-memory frames live while any mapping does

Each address space that maps a shared-memory region now holds its own
reference, recorded on the AddressSpaceRef and dropped when the space is
destroyed — so 'last reference' means no handles AND no mappings, and a
region's frames can no longer be freed out from under a sibling thread (or
any other live mapper) when the handle-holding task dies. The group-death
notification still posts after every mapping release. (docs/shared-fate-plan.md M3)
This commit is contained in:
Daniel Samson
2026-07-22 10:34:03 +01:00
parent b09a62bc36
commit 08e139ebba
3 changed files with 94 additions and 5 deletions
+12 -3
View File
@@ -173,9 +173,18 @@ pub fn createSharedMemory(phys: u64, pages: usize) ?*SharedMemoryObject {
return shared_memory; return shared_memory;
} }
/// Drop a shared-memory reference; when the last one goes, return its frames to the /// Take a shared-memory reference — a MAPPING's reference (docs/shared-fate-plan.md
/// allocator and free the object. (The mappings themselves are torn down with each /// M3): each address space that maps the region holds one, recorded on the space
/// sharer's address space; `device_grant` keeps that from freeing the frames early.) /// and dropped at its destruction. Caller holds the big kernel lock.
pub fn retainSharedMemory(shared_memory: *SharedMemoryObject) void {
shared_memory.refcount += 1;
}
/// Drop a shared-memory reference; when the last one goes — no handles AND no
/// mappings left — return its frames to the allocator and free the object.
/// (The mappings themselves are torn down with each sharer's address space;
/// `device_grant` keeps that sweep from freeing the frames, and the space's
/// recorded mapping references keep this drop from freeing them early.)
pub fn dropSharedMemoryReference(shared_memory: *SharedMemoryObject) void { pub fn dropSharedMemoryReference(shared_memory: *SharedMemoryObject) void {
if (shared_memory.refcount > 1) { if (shared_memory.refcount > 1) {
shared_memory.refcount -= 1; shared_memory.refcount -= 1;
+38 -1
View File
@@ -168,6 +168,7 @@ pub fn init() void {
scheduler.reap_task_hook = reapTaskLocked; scheduler.reap_task_hook = reapTaskLocked;
scheduler.timer_tick_hook = timerSweepLocked; scheduler.timer_tick_hook = timerSweepLocked;
scheduler.group_exit_hook = groupExitLocked; scheduler.group_exit_hook = groupExitLocked;
scheduler.space_mapping_release_hook = dropSpaceMappingHook;
} }
/// Return -1 (as an unsigned bit pattern) in the system_call result register. /// Return -1 (as an unsigned bit pattern) in the system_call result register.
@@ -570,9 +571,26 @@ fn systemSharedMemoryCreate(state: *architecture.CpuState) void {
for (0..pages) |i| pmm.free(phys + i * page_size); for (0..pages) |i| pmm.free(phys + i * page_size);
return fail(state); return fail(state);
}; };
// The creator's mapping holds its own reference, recorded on the space
// (docs/shared-fate-plan.md M3): frames must outlive every MAPPING, not just
// every handle — a sibling thread keeps using the region after the
// handle-holding thread dies.
{
const flags = sync.enter();
if (!scheduler.recordSpaceMappingLocked(t.address_space, @ptrCast(shared_memory))) {
sync.leave(flags);
ipc.dropSharedMemoryReference(shared_memory); // last ref: frees the object AND its frames
return fail(state);
}
ipc.retainSharedMemory(shared_memory);
sync.leave(flags);
}
const handle = ipc.installSharedMemoryHandle(t, shared_memory); const handle = ipc.installSharedMemoryHandle(t, shared_memory);
if (handle < 0) { if (handle < 0) {
ipc.dropSharedMemoryReference(shared_memory); // last ref: frees the object and its frames // The mapping record keeps one reference; drop only the creator's. The
// object (and frames) now live until this space is destroyed — the
// region was never named, so nothing else can reach it.
ipc.dropSharedMemoryReference(shared_memory);
return fail(state); return fail(state);
} }
@@ -597,6 +615,18 @@ fn systemSharedMemoryMap(state: *architecture.CpuState) void {
const size = shared_memory.pages * page_size; const size = shared_memory.pages * page_size;
if (base_v + size > shared_memory_arena_end) return fail(state); if (base_v + size > shared_memory_arena_end) return fail(state);
// This mapping holds its own reference, recorded on the space and dropped at
// its destruction (docs/shared-fate-plan.md M3) — the handle's reference is
// separate and may be closed while the mapping lives on.
{
const flags = sync.enter();
if (!scheduler.recordSpaceMappingLocked(t.address_space, @ptrCast(shared_memory))) {
sync.leave(flags);
return fail(state); // mapping table full: refuse rather than map unrecorded
}
ipc.retainSharedMemory(shared_memory);
sync.leave(flags);
}
architecture.mapUserSharedInto(t.address_space, base_v, shared_memory.phys, size); architecture.mapUserSharedInto(t.address_space, base_v, shared_memory.phys, size);
t.shared_memory_map_next = base_v + size; t.shared_memory_map_next = base_v + size;
architecture.setSystemCallResult(state, base_v); architecture.setSystemCallResult(state, base_v);
@@ -1126,6 +1156,13 @@ fn groupExitLocked(leader: u32, supervisor: u32, reason: abi.ExitReason, exit_en
} }
} }
/// scheduler.space_mapping_release_hook: drop one shared-memory MAPPING
/// reference when the space that held it is destroyed (docs/shared-fate-plan.md
/// M3). Lock held by the release path.
fn dropSpaceMappingHook(object: *anyopaque) void {
ipc.dropSharedMemoryReference(@ptrCast(@alignCast(object)));
}
/// Overwrite dead task `id`'s exit record with the group reason, or re-append it /// Overwrite dead task `id`'s exit record with the group reason, or re-append it
/// if the group's death burst already evicted it — a supervisor must always be /// if the group's death burst already evicted it — a supervisor must always be
/// able to read the reason for a notification it just received. Lock held. /// able to read the reason for a notification it just received. Lock held.
+44 -1
View File
@@ -185,7 +185,18 @@ const AddressSpaceRef = struct {
group_supervisor: u32 = 0, group_supervisor: u32 = 0,
group_reason: abi.ExitReason = .exited, group_reason: abi.ExitReason = .exited,
group_exit_endpoint: ?*anyopaque = null, group_exit_endpoint: ?*anyopaque = null,
// Shared-memory objects mapped into this space (docs/shared-fate-plan.md M3).
// Each mapping holds one reference to its object, dropped through
// `space_mapping_release_hook` when the space is destroyed — so "last
// reference" means no handles AND no mappings, and frames can never be freed
// while a live space still maps them. Opaque: the object type is the IPC
// layer's.
mappings: [maximum_space_mappings]?*anyopaque = .{null} ** maximum_space_mappings,
}; };
/// Shared-memory mappings one address space can hold — matches the per-task
/// handle table's order of magnitude; `shared_memory_map` fails when full.
const maximum_space_mappings = 16;
var address_space_refs = [_]AddressSpaceRef{.{}} ** maximum_tasks; var address_space_refs = [_]AddressSpaceRef{.{}} ** maximum_tasks;
var address_space_destroy_count: u64 = 0; var address_space_destroy_count: u64 = 0;
@@ -265,11 +276,19 @@ fn releaseAddressSpace(root: u64) void {
const supervisor = entry.group_supervisor; const supervisor = entry.group_supervisor;
const reason = entry.group_reason; const reason = entry.group_reason;
const endpoint = entry.group_exit_endpoint; const endpoint = entry.group_exit_endpoint;
const mappings = entry.mappings;
entry.* = .{}; entry.* = .{};
architecture.destroyAddressSpace(root); architecture.destroyAddressSpace(root);
address_space_destroy_count += 1; address_space_destroy_count += 1;
// Release the mapping references now that no mapping exists —
// `device_grant`-tagged leaves kept destroyAddressSpace's sweep off
// the frames, so this drop is what may actually free them (M3).
if (space_mapping_release_hook) |release| {
for (mappings) |slot| if (slot) |object| release(object);
}
// The group-death moment: the space is gone, every member is dead. // The group-death moment: the space is gone, every member is dead.
// process.zig posts the leader's deferred exit publication here. // process.zig posts the leader's deferred exit publication here —
// last, so the supervisor's notification postdates every release.
if (was_dying) if (group_exit_hook) |hook| hook(leader, supervisor, reason, endpoint); if (was_dying) if (group_exit_hook) |hook| hook(leader, supervisor, reason, endpoint);
} }
return; return;
@@ -304,6 +323,30 @@ pub fn markGroupDyingLocked(root: u64, leader: u32, supervisor: u32, reason: abi
return false; return false;
} }
/// Called (lock held) once per recorded shared-memory mapping when an address
/// space is destroyed — drops the mapping's object reference. Registered by
/// process.zig (the object type lives in the IPC layer).
pub var space_mapping_release_hook: ?*const fn (*anyopaque) void = null;
/// Record a shared-memory mapping on `root`'s space; its reference is dropped
/// via `space_mapping_release_hook` at space destruction. Returns false —
/// recording nothing — if the space has no live entry or its mapping table is
/// full. On true, the caller has transferred one object reference to the space.
/// Caller holds the lock.
pub fn recordSpaceMappingLocked(root: u64, object: *anyopaque) bool {
for (&address_space_refs) |*entry| {
if (entry.count == 0 or entry.root != root) continue;
for (&entry.mappings) |*slot| {
if (slot.* == null) {
slot.* = object;
return true;
}
}
return false;
}
return false;
}
/// Whether `root`'s group is already dying. Caller holds the lock. /// Whether `root`'s group is already dying. Caller holds the lock.
pub fn groupDyingLocked(root: u64) bool { pub fn groupDyingLocked(root: u64) bool {
for (&address_space_refs) |*entry| { for (&address_space_refs) |*entry| {