test fault-on-AP and affinity; report the faulting core

fault-ap-df pins a #DF to an AP (its own IST must catch it); affinity checks a pinned task never migrates. onException now names the core, so an AP fault is attributed and shown contained. Both teeth-checked.
This commit is contained in:
Daniel Samson
2026-07-08 14:45:28 +01:00
parent 37f72a4df8
commit 26ac97df31
4 changed files with 127 additions and 8 deletions
+91
View File
@@ -74,6 +74,8 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void {
ipcTest();
} else if (eql(case, "smp")) {
smpTest();
} else if (eql(case, "affinity")) {
affinityTest();
} else if (eql(case, "smp-stress")) {
stressTest();
} else if (eql(case, "smp-retry")) {
@@ -84,6 +86,8 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void {
faultPageFault();
} else if (eql(case, "fault-df")) {
faultDoubleFault();
} else if (eql(case, "fault-ap-df")) {
faultApTest();
} else if (eql(case, "fault-nx")) {
faultNoExecute();
} else if (eql(case, "fault-null")) {
@@ -552,6 +556,52 @@ fn smpTest() void {
result();
}
// --- affinity: a pinned task never migrates -------------------------------
var affinity_cores = [_]bool{false} ** 8;
var affinity_running: bool = true;
fn affinityWorker() void {
const p: *volatile bool = &affinity_running;
while (p.*) {
const c = sched.currentCpuIndex();
if (c < affinity_cores.len) affinity_cores[c] = true;
}
sched.exit();
}
/// A task pinned to a core must run **only** on that core. Pin a busy worker to
/// core 1 and let it run through many preemptions; it must have stamped core 1 and no
/// other. An *unpinned* task scatters across cores (that's what the smp test shows),
/// so a broken pin fails this deterministically — over this many time slices a
/// free-floating task will land on some other core.
fn affinityTest() void {
log("DANOS-TEST-BEGIN: affinity\n", .{});
affinity_cores = .{false} ** 8;
affinity_running = true;
if (!sched.spawnOn(affinityWorker, 4, 1)) {
check("worker pinned to core 1 (run with -smp)", false);
result();
return;
}
var spins: u64 = 0;
while (spins < 3_000_000_000) spins +%= 1; // many time slices across the cores
affinity_running = false;
var settle: u64 = 0;
while (settle < 200_000_000) settle +%= 1; // let the worker see the flag and exit
var others: u32 = 0;
for (affinity_cores, 0..) |seen, c| {
if (seen and c != 1) others += 1;
}
log("DANOS-AFFINITY: pinned worker touched core 1={}, other cores={d}\n", .{ affinity_cores[1], others });
check("pinned task ran on its core (1)", affinity_cores[1]);
check("pinned task never migrated to another core", others == 0);
result();
}
// --- SMP stress: hammer the big kernel lock across cores ------------------
const stress_pairs = 4; // producer/consumer pairs (8 tasks; fits the 16-task pool)
@@ -704,3 +754,44 @@ fn faultDoubleFault() void {
);
bad_sp += 0;
}
var ap_reached_fault: bool = false;
/// A task that faults with a #DF *on whatever core it's pinned to*. Announces the
/// core, then triggers the same double fault as `faultDoubleFault` — which is only
/// survivable on IST1, so it exercises that core's own TSS.
fn apDoubleFaultTask() void {
log("DANOS-AP: task running on core {d}, triggering #DF\n", .{sched.currentCpuIndex()});
@atomicStore(bool, &ap_reached_fault, true, .release);
arch.disableInterrupts();
var bad_sp: u64 = 0x5000000000;
asm volatile (
\\mov %[sp], %%rsp
\\ud2
:
: [sp] "r" (bad_sp),
: .{ .memory = true }
);
bad_sp += 0;
}
/// Fault on an application processor. Pins a double-faulting task to core 1, so the
/// fault is taken and handled by *that core's own* IDT and TSS/IST — not the BSP's.
/// The harness matches "core N: double fault (vector 8)" with N ≥ 1, which can only
/// appear if the AP caught the #DF on its IST1 (a broken per-core TSS would
/// triple-fault and reset instead). We then show the BSP still runs afterwards, so
/// the fault was *contained* to the AP, not fatal to the system.
fn faultApTest() void {
log("DANOS-TEST-BEGIN: fault-ap-df\n", .{});
if (!sched.spawnOn(apDoubleFaultTask, 6, 1)) {
log("DANOS-AP: could not pin to core 1 (run with -smp) - FAIL\n", .{});
arch.halt();
}
// Wait until the AP is about to fault, then keep running to prove containment.
var spins: u64 = 0;
while (!@atomicLoad(bool, &ap_reached_fault, .acquire) and spins < 5_000_000_000) spins +%= 1;
var settle: u64 = 0;
while (settle < 500_000_000) settle +%= 1; // let the AP take + report the fault
log("DANOS-BSP: core {d} still running after the AP fault (contained)\n", .{sched.currentCpuIndex()});
arch.halt();
}