kernel: shared-fate review fixes — lock the shm/dma walks, contract edges
The adversarial review of the branch confirmed the big one: the shm/DMA page-table walks and their pmm/heap calls ran outside the big kernel lock — pre-existing, but fatal once the per-space cursors invited sibling threads to race them (two concurrent creates could orphan a page table: one thread's region silently unmapped, the frame leaked — a plausible root for the long-standing intermittent AP ring-3 fault at the shm base). All three paths now follow the mmap discipline: allocation, object build, record, and handle under one lock hold with full rollback; the map itself per-page under brief holds; dma_free's translate/unmap/free per-page likewise. Contract edges from the same review: thread_spawn into a dying group returns -ESRCH (was generic -1); process_kill during the condemned window answers from the latch's stashed supervisor (0 or -EPERM, was -ESRCH once the leader slot was reaped); exit derives the group reason from its own argument rather than the racy exit_code global; a worker's thread_exit no longer overwrites a concurrent group-kill stamp; checkGroupDead now asserts exactly-one notification via the drained ring. Full suite: 100/100.
This commit is contained in:
+11
-8
@@ -3233,11 +3233,14 @@ fn awaitExitBadge(endpoint: *ipcsync.Endpoint) u64 {
|
||||
|
||||
/// A group death is one notification, badged with the LEADER, arriving only
|
||||
/// after every member (and the address space) is gone — asserted by every
|
||||
/// shared-fate case below.
|
||||
fn checkGroupDead(me: u32, leader: u32, badge: u64, reason: abi.ExitReason) void {
|
||||
/// shared-fate case below. "Exactly one": after a settling sleep, a second
|
||||
/// (buggy, double-posted) badge would still sit in the endpoint's notify ring.
|
||||
fn checkGroupDead(me: u32, leader: u32, badge: u64, reason: abi.ExitReason, endpoint: *ipcsync.Endpoint) void {
|
||||
check("one exit notification, badged with the leader", badge == abi.notify_badge_bit | abi.notify_exit_bit | leader);
|
||||
check("the leader's recorded reason is the group reason", process.exitReasonOf(me, leader) == @intFromEnum(reason));
|
||||
check("no group member is listed after the death", !groupListed(leader));
|
||||
scheduler.sleep(200);
|
||||
check("exactly one exit notification (ring drained)", endpoint.notify_head == endpoint.notify_tail);
|
||||
}
|
||||
|
||||
fn threadFaultGroupTest(boot_information: *const BootInformation) void {
|
||||
@@ -3258,7 +3261,7 @@ fn threadFaultGroupTest(boot_information: *const BootInformation) void {
|
||||
return;
|
||||
}
|
||||
const badge = awaitExitBadge(endpoint);
|
||||
checkGroupDead(me, child, badge, .segmentation_fault);
|
||||
checkGroupDead(me, child, badge, .segmentation_fault, endpoint);
|
||||
check("one fault kill for the whole group", process.fault_kill_count == 1);
|
||||
const deadline = architecture.millis() + 5000;
|
||||
while (scheduler.liveStackBytes() > stacks_base and architecture.millis() < deadline) scheduler.yield();
|
||||
@@ -3293,7 +3296,7 @@ fn killThreadedGroupTest(boot_information: *const BootInformation) void {
|
||||
scheduler.sleep(100); // let the worker really be running on another core
|
||||
check("the supervisor's kill is accepted", process.killProcess(me, child) == 0);
|
||||
const badge = awaitExitBadge(endpoint);
|
||||
checkGroupDead(me, child, badge, .killed);
|
||||
checkGroupDead(me, child, badge, .killed, endpoint);
|
||||
check("the leader's claim was released before the notification", devices_broker.ownerOf(0) == null);
|
||||
check("the worker's claim was released before the notification", devices_broker.ownerOf(1) == null);
|
||||
check("a dead group stays dead (-ESRCH)", process.killProcess(me, child) == -ipcsync.ESRCH);
|
||||
@@ -3319,7 +3322,7 @@ fn killViaWorkerTidTest(boot_information: *const BootInformation) void {
|
||||
check("a non-supervisor aiming at the worker is refused (-EPERM)", process.killProcess(me + 12345, worker) == -ipcsync.EPERM);
|
||||
check("the supervisor's kill aimed at the WORKER id is accepted", process.killProcess(me, worker) == 0);
|
||||
const badge = awaitExitBadge(endpoint);
|
||||
checkGroupDead(me, child, badge, .killed);
|
||||
checkGroupDead(me, child, badge, .killed, endpoint);
|
||||
result();
|
||||
}
|
||||
|
||||
@@ -3339,7 +3342,7 @@ fn racingTriggersTest(boot_information: *const BootInformation) void {
|
||||
return;
|
||||
}
|
||||
const badge = awaitExitBadge(endpoint);
|
||||
checkGroupDead(me, child, badge, .segmentation_fault);
|
||||
checkGroupDead(me, child, badge, .segmentation_fault, endpoint);
|
||||
check("two racing faults counted as ONE group kill", process.fault_kill_count == 1);
|
||||
result();
|
||||
}
|
||||
@@ -3359,7 +3362,7 @@ fn exitGroupTest(boot_information: *const BootInformation) void {
|
||||
return;
|
||||
}
|
||||
const badge = awaitExitBadge(endpoint);
|
||||
checkGroupDead(me, child, badge, .aborted);
|
||||
checkGroupDead(me, child, badge, .aborted, endpoint);
|
||||
result();
|
||||
}
|
||||
|
||||
@@ -3381,7 +3384,7 @@ fn threadTestMarkerCase(boot_information: *const BootInformation, case_name: []c
|
||||
return;
|
||||
}
|
||||
const badge = awaitExitBadge(endpoint);
|
||||
checkGroupDead(me, child, badge, .exited);
|
||||
checkGroupDead(me, child, badge, .exited, endpoint);
|
||||
result();
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user