diff --git a/docs/smp.md b/docs/smp.md index 36bba61..e11760b 100644 --- a/docs/smp.md +++ b/docs/smp.md @@ -194,7 +194,15 @@ next lands. core online, and enters the run loop. With interrupts on, each core's own timer tick preempts its idle context into whatever the global ready queue offers — so all cores pull real work in parallel. The `smp` test spawns CPU-bound workers and confirms they - execute on all four cores at once. + execute on all four cores at once, and `fault-ap-df` pins a #DF to an AP and checks + that core catches it on **its own** IST (a broken per-core TSS would triple-fault) — + reported as "core N: …", so a fault is always attributed to the core it happened on, + and is contained to that core (the rest of the system keeps running). +- **Thread affinity** — `spawnOn(entry, priority, cpu)` pins a task to a core (its own + per-core pinned queue, merged with the global queue at selection; see + [scheduling.md](scheduling.md#affinity-pinning-a-task-to-a-core)). The `affinity` + test confirms a pinned task never migrates. This is the mechanism the fault-on-AP + test rides on, and the *explicit-affinity* real-time-predictable model. - **`single_threaded` off** — the kernel was built `single_threaded = true`, which compiles `std.atomic` down to plain non-atomic ops. Harmless on one core, but it quietly breaks the big kernel lock across cores; it's now `false`. @@ -214,8 +222,13 @@ next lands. - **IPIs** — cross-core wake/preempt. Not needed for correctness: an idle core wakes on its own timer tick and pulls ready work then; IPIs only cut that latency from ≤1 ms to near-instant. -- **Per-core run queues + thread affinity** — the Fiasco.OC direction, if the single - global queue's lock contention ever bites (and the more real-time-predictable model). +- **Per-core run queues** — the Fiasco.OC direction, if the single global queue's lock + contention ever bites. (Thread *affinity* already exists — see above; this is the + further step of giving each core its own primary run queue for load distribution.) +- **Fault recovery** — today a fault halts (only) the faulting core. Turning that into + "kill the task, keep the core running" is the [resilience](resilience.md) track (it + needs the task's lock/resource state handled), and for taking a core fully offline, + its tasks migrated first. ## Further reading diff --git a/src/kernel/main.zig b/src/kernel/main.zig index bdf89fe..f62e063 100644 --- a/src/kernel/main.zig +++ b/src/kernel/main.zig @@ -320,21 +320,25 @@ fn kib(frames: u64) u64 { return frames * danos.page_size / (1024); } -/// Report a CPU exception and halt. There's no fault recovery yet, so any -/// exception is terminal — but it reports what and where (to every output sink, -/// plus a POST code and a persistent breadcrumb) instead of silently resetting. +/// Report a CPU exception and halt **this core**. There's no fault recovery yet, so +/// the faulting core is terminal — but the fault is *contained* to it: on an +/// application processor only that core stops, and the rest of the system keeps +/// running (full recovery — kill the task, keep the core — is the resilience track, +/// see docs/resilience.md). The report names the core so an AP fault is attributed, +/// and goes to every output sink plus a POST code and a persistent breadcrumb. fn onException(state: *const arch.CpuState) noreturn { log.checkpoint(cp_exception); + const core = scheduler.currentCpuIndex(); // A fault is user-facing enough to paint on screen too (via statusPrint), on // top of the diagnostic log. - statusPrint("\nCPU EXCEPTION: {s} (vector {d})\n", .{ arch.vectorName(state.vector), state.vector }); + statusPrint("\nCPU EXCEPTION on core {d}: {s} (vector {d})\n", .{ core, arch.vectorName(state.vector), state.vector }); statusPrint(" error code : 0x{x}\n", .{state.error_code}); statusPrint(" RIP : 0x{x:0>16}\n", .{state.rip}); statusPrint(" RSP : 0x{x:0>16}\n", .{state.rsp}); if (state.vector == 14) statusPrint(" CR2 (addr) : 0x{x:0>16}\n", .{arch.readCr2()}); var buf: [128]u8 = undefined; - log.recordPanic(std.fmt.bufPrint(&buf, "CPU exception {s} (vector {d}) at RIP 0x{x}", .{ arch.vectorName(state.vector), state.vector, state.rip }) catch "cpu exception"); + log.recordPanic(std.fmt.bufPrint(&buf, "CPU exception {s} (vector {d}) on core {d} at RIP 0x{x}", .{ arch.vectorName(state.vector), state.vector, core, state.rip }) catch "cpu exception"); arch.halt(); } diff --git a/src/kernel/tests.zig b/src/kernel/tests.zig index 08fcf9e..ba225e4 100644 --- a/src/kernel/tests.zig +++ b/src/kernel/tests.zig @@ -74,6 +74,8 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void { ipcTest(); } else if (eql(case, "smp")) { smpTest(); + } else if (eql(case, "affinity")) { + affinityTest(); } else if (eql(case, "smp-stress")) { stressTest(); } else if (eql(case, "smp-retry")) { @@ -84,6 +86,8 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void { faultPageFault(); } else if (eql(case, "fault-df")) { faultDoubleFault(); + } else if (eql(case, "fault-ap-df")) { + faultApTest(); } else if (eql(case, "fault-nx")) { faultNoExecute(); } else if (eql(case, "fault-null")) { @@ -552,6 +556,52 @@ fn smpTest() void { result(); } +// --- affinity: a pinned task never migrates ------------------------------- + +var affinity_cores = [_]bool{false} ** 8; +var affinity_running: bool = true; + +fn affinityWorker() void { + const p: *volatile bool = &affinity_running; + while (p.*) { + const c = sched.currentCpuIndex(); + if (c < affinity_cores.len) affinity_cores[c] = true; + } + sched.exit(); +} + +/// A task pinned to a core must run **only** on that core. Pin a busy worker to +/// core 1 and let it run through many preemptions; it must have stamped core 1 and no +/// other. An *unpinned* task scatters across cores (that's what the smp test shows), +/// so a broken pin fails this deterministically — over this many time slices a +/// free-floating task will land on some other core. +fn affinityTest() void { + log("DANOS-TEST-BEGIN: affinity\n", .{}); + affinity_cores = .{false} ** 8; + affinity_running = true; + + if (!sched.spawnOn(affinityWorker, 4, 1)) { + check("worker pinned to core 1 (run with -smp)", false); + result(); + return; + } + + var spins: u64 = 0; + while (spins < 3_000_000_000) spins +%= 1; // many time slices across the cores + affinity_running = false; + var settle: u64 = 0; + while (settle < 200_000_000) settle +%= 1; // let the worker see the flag and exit + + var others: u32 = 0; + for (affinity_cores, 0..) |seen, c| { + if (seen and c != 1) others += 1; + } + log("DANOS-AFFINITY: pinned worker touched core 1={}, other cores={d}\n", .{ affinity_cores[1], others }); + check("pinned task ran on its core (1)", affinity_cores[1]); + check("pinned task never migrated to another core", others == 0); + result(); +} + // --- SMP stress: hammer the big kernel lock across cores ------------------ const stress_pairs = 4; // producer/consumer pairs (8 tasks; fits the 16-task pool) @@ -704,3 +754,44 @@ fn faultDoubleFault() void { ); bad_sp += 0; } + +var ap_reached_fault: bool = false; + +/// A task that faults with a #DF *on whatever core it's pinned to*. Announces the +/// core, then triggers the same double fault as `faultDoubleFault` — which is only +/// survivable on IST1, so it exercises that core's own TSS. +fn apDoubleFaultTask() void { + log("DANOS-AP: task running on core {d}, triggering #DF\n", .{sched.currentCpuIndex()}); + @atomicStore(bool, &ap_reached_fault, true, .release); + arch.disableInterrupts(); + var bad_sp: u64 = 0x5000000000; + asm volatile ( + \\mov %[sp], %%rsp + \\ud2 + : + : [sp] "r" (bad_sp), + : .{ .memory = true } + ); + bad_sp += 0; +} + +/// Fault on an application processor. Pins a double-faulting task to core 1, so the +/// fault is taken and handled by *that core's own* IDT and TSS/IST — not the BSP's. +/// The harness matches "core N: double fault (vector 8)" with N ≥ 1, which can only +/// appear if the AP caught the #DF on its IST1 (a broken per-core TSS would +/// triple-fault and reset instead). We then show the BSP still runs afterwards, so +/// the fault was *contained* to the AP, not fatal to the system. +fn faultApTest() void { + log("DANOS-TEST-BEGIN: fault-ap-df\n", .{}); + if (!sched.spawnOn(apDoubleFaultTask, 6, 1)) { + log("DANOS-AP: could not pin to core 1 (run with -smp) - FAIL\n", .{}); + arch.halt(); + } + // Wait until the AP is about to fault, then keep running to prove containment. + var spins: u64 = 0; + while (!@atomicLoad(bool, &ap_reached_fault, .acquire) and spins < 5_000_000_000) spins +%= 1; + var settle: u64 = 0; + while (settle < 500_000_000) settle +%= 1; // let the AP take + report the fault + log("DANOS-BSP: core {d} still running after the AP fault (contained)\n", .{sched.currentCpuIndex()}); + arch.halt(); +} diff --git a/test/qemu_test.py b/test/qemu_test.py index 938ef03..10f0db3 100644 --- a/test/qemu_test.py +++ b/test/qemu_test.py @@ -119,6 +119,11 @@ CASES = [ "smp": 4, "expect": r"DANOS-TEST-RESULT: PASS", "fail": r"DANOS-TEST-RESULT: FAIL"}, + # Affinity: a pinned task must never migrate off its core. + {"name": "affinity", + "smp": 4, + "expect": r"DANOS-TEST-RESULT: PASS", + "fail": r"DANOS-TEST-RESULT: FAIL"}, # Stress the big kernel lock across cores; heavier, so a longer timeout. {"name": "smp-stress", "smp": 4, @@ -133,6 +138,12 @@ CASES = [ {"name": "fault-ud", "expect": r"invalid opcode \(vector 6\)"}, {"name": "fault-pf", "expect": r"page fault \(vector 14\)"}, {"name": "fault-df", "expect": r"double fault \(vector 8\)"}, + # Fault on an application processor: a #DF pinned to core 1 must be caught by that + # core's own IST (N >= 1), proving per-core TSS works; a broken one triple-faults. + {"name": "fault-ap-df", + "smp": 4, + "expect": r"core [1-9]\d*: double fault \(vector 8\)", + "fail": r"could not pin"}, {"name": "fault-nx", "expect": r"page fault \(vector 14\)", "fail": r"NX not enforced"},