kernel+tests: M4 shared-fate — nine group-death test cases, per-space arena cursors
Eight new QEMU cases (thread-fault-group, kill-threaded-group, kill-via-worker-tid, racing-triggers, exit-group, leader-thread-exit, thread-exit-solo, shm-mapping-ref) driving seven new thread-test modes; a shared checkGroupDead asserts the contract everywhere: one notification, badged with the leader, reason on the leader's record, no member listed, claims released first. The shm-mapping-ref case flushed out the per-task DMA/shared-memory arena cursor bug directly (a sibling's regions mapped over the worker's), so both cursors moved to the AddressSpaceRef like the mmap/MMIO cursors before them (threading M7 pattern). Runtime gains Thread.tryExitCurrent for the leader -EPERM refusal path. Docs updated: threading.md's shared-fate gap is closed, process-management.md and process-lifecycle.md describe the leader re-key, plan status = implemented. Full suite: 100/100.
This commit is contained in:
@@ -146,6 +146,22 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
faultRecoveryTest(boot_information);
|
||||
} else if (eql(case, "address-space-refcount")) {
|
||||
addressSpaceRefcountTest(boot_information);
|
||||
} else if (eql(case, "thread-fault-group")) {
|
||||
threadFaultGroupTest(boot_information);
|
||||
} else if (eql(case, "kill-threaded-group")) {
|
||||
killThreadedGroupTest(boot_information);
|
||||
} else if (eql(case, "kill-via-worker-tid")) {
|
||||
killViaWorkerTidTest(boot_information);
|
||||
} else if (eql(case, "racing-triggers")) {
|
||||
racingTriggersTest(boot_information);
|
||||
} else if (eql(case, "exit-group")) {
|
||||
exitGroupTest(boot_information);
|
||||
} else if (eql(case, "leader-thread-exit")) {
|
||||
threadTestMarkerCase(boot_information, "leader-thread-exit", "leader-exit");
|
||||
} else if (eql(case, "thread-exit-solo")) {
|
||||
threadTestMarkerCase(boot_information, "thread-exit-solo", "solo");
|
||||
} else if (eql(case, "shm-mapping-ref")) {
|
||||
threadTestMarkerCase(boot_information, "shm-mapping-ref", "shm-worker");
|
||||
} else if (eql(case, "thread-spawn")) {
|
||||
threadSpawnTest(boot_information);
|
||||
} else if (eql(case, "thread-join")) {
|
||||
@@ -3167,6 +3183,208 @@ fn kernelVfsTest(boot_information: *const BootInformation) void {
|
||||
result();
|
||||
}
|
||||
|
||||
// --- shared-fate helpers (docs/shared-fate-plan.md M4) -----------------------
|
||||
|
||||
/// Spawn the ramdisk's thread-test with `mode` as argv[1], supervised by the
|
||||
/// calling test task on `endpoint`. Returns the child (leader) id, or 0.
|
||||
fn spawnThreadTestSupervised(boot_information: *const BootInformation, mode: []const u8, endpoint: *ipcsync.Endpoint) u32 {
|
||||
const ramdisk = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(ramdisk) orelse return 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(initial_ramdisk.basename(item.name), "thread-test")) continue;
|
||||
return process.spawnProcessSupervised(item.blob, 4, &.{ "thread-test", mode }, scheduler.currentId(), endpoint) catch 0;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// Whether any live task still belongs to `leader`'s group.
|
||||
fn groupListed(leader: u32) bool {
|
||||
var table: [32]abi.ProcessDescriptor = undefined;
|
||||
const total = scheduler.enumerate(&table);
|
||||
for (table[0..@min(total, table.len)]) |descriptor| {
|
||||
if (descriptor.id == leader or descriptor.leader == leader) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Poll enumerate for a WORKER of `leader` (same leader, different id) until
|
||||
/// `deadline` (millis); returns its id, or 0.
|
||||
fn findWorkerOf(leader: u32, deadline: u64) u32 {
|
||||
while (architecture.millis() < deadline) {
|
||||
var table: [32]abi.ProcessDescriptor = undefined;
|
||||
const total = scheduler.enumerate(&table);
|
||||
for (table[0..@min(total, table.len)]) |descriptor| {
|
||||
if (descriptor.leader == leader and descriptor.id != leader) return descriptor.id;
|
||||
}
|
||||
scheduler.yield();
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// Block on `endpoint` for the next notification and return its badge.
|
||||
fn awaitExitBadge(endpoint: *ipcsync.Endpoint) u64 {
|
||||
var badge: u64 = 0;
|
||||
var received_cap: u64 = 0;
|
||||
_ = ipcsync.replyWait(endpoint, 0, 0, 0, 0, abi.no_cap, &badge, &received_cap);
|
||||
return badge;
|
||||
}
|
||||
|
||||
/// A group death is one notification, badged with the LEADER, arriving only
|
||||
/// after every member (and the address space) is gone — asserted by every
|
||||
/// shared-fate case below.
|
||||
fn checkGroupDead(me: u32, leader: u32, badge: u64, reason: abi.ExitReason) void {
|
||||
check("one exit notification, badged with the leader", badge == abi.notify_badge_bit | abi.notify_exit_bit | leader);
|
||||
check("the leader's recorded reason is the group reason", process.exitReasonOf(me, leader) == @intFromEnum(reason));
|
||||
check("no group member is listed after the death", !groupListed(leader));
|
||||
}
|
||||
|
||||
fn threadFaultGroupTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: thread-fault-group\n", .{});
|
||||
const me = scheduler.currentId();
|
||||
const endpoint = ipcsync.createIpcEndpoint() orelse {
|
||||
check("exit endpoint allocated", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
const stacks_base = scheduler.liveStackBytes();
|
||||
const spaces_base = scheduler.liveAddressSpaceCount();
|
||||
process.fault_kill_count = 0;
|
||||
const child = spawnThreadTestSupervised(boot_information, "fault-worker", endpoint);
|
||||
check("thread-test spawned (fault-worker)", child != 0);
|
||||
if (child == 0) {
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const badge = awaitExitBadge(endpoint);
|
||||
checkGroupDead(me, child, badge, .segmentation_fault);
|
||||
check("one fault kill for the whole group", process.fault_kill_count == 1);
|
||||
const deadline = architecture.millis() + 5000;
|
||||
while (scheduler.liveStackBytes() > stacks_base and architecture.millis() < deadline) scheduler.yield();
|
||||
check("address spaces returned to base", scheduler.liveAddressSpaceCount() == spaces_base);
|
||||
check("kernel stacks returned to base", scheduler.liveStackBytes() == stacks_base);
|
||||
result();
|
||||
}
|
||||
|
||||
fn killThreadedGroupTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: kill-threaded-group\n", .{});
|
||||
const me = scheduler.currentId();
|
||||
var buffer: [2]device_abi.DeviceDescriptor = undefined;
|
||||
check("the device tree is seeded (>= 2 devices)", devices_broker.enumerate(&buffer) >= 2);
|
||||
const endpoint = ipcsync.createIpcEndpoint() orelse {
|
||||
check("exit endpoint allocated", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
const child = spawnThreadTestSupervised(boot_information, "spin-forever", endpoint);
|
||||
check("thread-test spawned (spin-forever)", child != 0);
|
||||
if (child == 0) {
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const worker = findWorkerOf(child, architecture.millis() + 8000);
|
||||
check("the spinning worker is enumerable with leader = the child", worker != 0);
|
||||
// Claims for BOTH members: group death must release every member's claims
|
||||
// before the supervisor hears anything — the worker's by the deferred
|
||||
// (condemned) path.
|
||||
check("device 0 claimed for the leader", devices_broker.claim(0, child));
|
||||
check("device 1 claimed for the worker", devices_broker.claim(1, worker));
|
||||
scheduler.sleep(100); // let the worker really be running on another core
|
||||
check("the supervisor's kill is accepted", process.killProcess(me, child) == 0);
|
||||
const badge = awaitExitBadge(endpoint);
|
||||
checkGroupDead(me, child, badge, .killed);
|
||||
check("the leader's claim was released before the notification", devices_broker.ownerOf(0) == null);
|
||||
check("the worker's claim was released before the notification", devices_broker.ownerOf(1) == null);
|
||||
check("a dead group stays dead (-ESRCH)", process.killProcess(me, child) == -ipcsync.ESRCH);
|
||||
result();
|
||||
}
|
||||
|
||||
fn killViaWorkerTidTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: kill-via-worker-tid\n", .{});
|
||||
const me = scheduler.currentId();
|
||||
const endpoint = ipcsync.createIpcEndpoint() orelse {
|
||||
check("exit endpoint allocated", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
const child = spawnThreadTestSupervised(boot_information, "spin-forever", endpoint);
|
||||
check("thread-test spawned (spin-forever)", child != 0);
|
||||
if (child == 0) {
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const worker = findWorkerOf(child, architecture.millis() + 8000);
|
||||
check("the spinning worker is enumerable", worker != 0);
|
||||
check("a non-supervisor aiming at the worker is refused (-EPERM)", process.killProcess(me + 12345, worker) == -ipcsync.EPERM);
|
||||
check("the supervisor's kill aimed at the WORKER id is accepted", process.killProcess(me, worker) == 0);
|
||||
const badge = awaitExitBadge(endpoint);
|
||||
checkGroupDead(me, child, badge, .killed);
|
||||
result();
|
||||
}
|
||||
|
||||
fn racingTriggersTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: racing-triggers\n", .{});
|
||||
const me = scheduler.currentId();
|
||||
const endpoint = ipcsync.createIpcEndpoint() orelse {
|
||||
check("exit endpoint allocated", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
process.fault_kill_count = 0;
|
||||
const child = spawnThreadTestSupervised(boot_information, "race", endpoint);
|
||||
check("thread-test spawned (race)", child != 0);
|
||||
if (child == 0) {
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const badge = awaitExitBadge(endpoint);
|
||||
checkGroupDead(me, child, badge, .segmentation_fault);
|
||||
check("two racing faults counted as ONE group kill", process.fault_kill_count == 1);
|
||||
result();
|
||||
}
|
||||
|
||||
fn exitGroupTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: exit-group\n", .{});
|
||||
const me = scheduler.currentId();
|
||||
const endpoint = ipcsync.createIpcEndpoint() orelse {
|
||||
check("exit endpoint allocated", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
const child = spawnThreadTestSupervised(boot_information, "exit-worker", endpoint);
|
||||
check("thread-test spawned (exit-worker)", child != 0);
|
||||
if (child == 0) {
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const badge = awaitExitBadge(endpoint);
|
||||
checkGroupDead(me, child, badge, .aborted);
|
||||
result();
|
||||
}
|
||||
|
||||
/// leader-thread-exit, thread-exit-solo, and shm-mapping-ref share one shape:
|
||||
/// the child asserts its own property, prints a marker the harness matches, and
|
||||
/// exits clean — the kernel side asserts the clean group death.
|
||||
fn threadTestMarkerCase(boot_information: *const BootInformation, case_name: []const u8, mode: []const u8) void {
|
||||
log("DANOS-TEST-BEGIN: {s}\n", .{case_name});
|
||||
const me = scheduler.currentId();
|
||||
const endpoint = ipcsync.createIpcEndpoint() orelse {
|
||||
check("exit endpoint allocated", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
const child = spawnThreadTestSupervised(boot_information, mode, endpoint);
|
||||
check("thread-test spawned", child != 0);
|
||||
if (child == 0) {
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const badge = awaitExitBadge(endpoint);
|
||||
checkGroupDead(me, child, badge, .exited);
|
||||
result();
|
||||
}
|
||||
|
||||
fn spawnNamed(rd: initial_ramdisk.Reader, name: []const u8) bool {
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
|
||||
Reference in New Issue
Block a user