apic: time-bound the TSC warp check — 47 s of AP bring-up becomes ~0.3 s

First real per-process logs off the stick (the logging track paying for
itself): kernel.log showed 16-core bring-up costing 47 s — per-core gaps
of 2-21 s — on a machine whose clocksource is the TSC, so every AP runs
the pairwise warp check. QEMU always picks HPET, so the harness never
executed this path at all.

The check was bounded by ITERATIONS: 1<<20 warp ticks, each a locked
read-modify-write on a cacheline two cores fight over — microseconds
under real contention, not the nanosecond the '~1 ms' comment assumed —
and the 1<<32-PAUSE rendezvous 'bound' is ~2 minutes on modern Intel
(PAUSE ~140 cycles). Both are now bounded by TIME measured on the TSC
itself: ~5 ms of pairwise hammering per core (Linux's check_tsc_warp
budget — ample to catch a lagging TSC) and a ~100 ms rendezvous window.
This commit is contained in:
Daniel Samson
2026-07-21 19:47:53 +01:00
parent 59ba95a315
commit eb6e8edafe
2 changed files with 27 additions and 12 deletions
+1 -1
View File
@@ -6,4 +6,4 @@ zig-out/
.idea/ .idea/
.claude/ .claude/
.github/ .github//var/
+26 -11
View File
@@ -529,8 +529,22 @@ var warp_ap_ready: u32 = 0;
var warp_stop: u32 = 0; var warp_stop: u32 = 0;
var warp_checks: u32 = 0; // completed per-AP rendezvous count (for the tsc-sync test) var warp_checks: u32 = 0; // completed per-AP rendezvous count (for the tsc-sync test)
const warp_rounds: u32 = 1 << 20; // locked reads on the BSP: ~1 ms at GHz rates // The warp check is bounded by TIME, not iterations: a warp tick is a locked
const warp_spin_limit: u64 = 1 << 32; // bound every rendezvous wait so a lost core can't hang boot // read-modify-write on a cacheline two cores are fighting over — microseconds
// under real contention, not the nanosecond an uncontended count assumes (a
// 1<<20-round budget measured 2-21 SECONDS per core on a 16-core machine), and
// a PAUSE costs ~140 cycles on modern Intel, so an iteration-counted await
// mis-measures by two orders of magnitude too. ~5 ms of pairwise hammering per
// core is plenty to catch a lagging TSC (Linux's check_tsc_warp budget), and
// ~100 ms is a generous rendezvous window for a healthy core.
const warp_check_ns: u64 = 5_000_000; // per-AP pairwise check duration
const warp_await_ns: u64 = 100_000_000; // rendezvous wait before giving up
/// TSC ticks for `ns` nanoseconds (valid whenever the warp check runs: the TSC
/// is the clocksource, so tsc_hz is calibrated).
fn warpTicksFor(ns: u64) u64 {
return @intCast(@as(u128, ns) * tsc_hz / 1_000_000_000);
}
fn warpTick() void { fn warpTick() void {
while (@cmpxchgWeak(u32, &warp_lock, 0, 1, .acquire, .monotonic) != null) asm volatile ("pause"); while (@cmpxchgWeak(u32, &warp_lock, 0, 1, .acquire, .monotonic) != null) asm volatile ("pause");
@@ -545,11 +559,11 @@ fn warpTick() void {
@atomicStore(u32, &warp_lock, 0, .release); @atomicStore(u32, &warp_lock, 0, .release);
} }
/// Spin (bounded) until `flag` is nonzero; false on timeout. /// Spin (time-bounded) until `flag` is nonzero; false on timeout.
fn warpAwait(flag: *u32) bool { fn warpAwait(flag: *u32) bool {
var spins: u64 = 0; const deadline = rdtsc() +% warpTicksFor(warp_await_ns);
while (@atomicLoad(u32, flag, .acquire) == 0) : (spins += 1) { while (@atomicLoad(u32, flag, .acquire) == 0) {
if (spins >= warp_spin_limit) return false; if (rdtsc() -% deadline < (1 << 62)) return false; // past the deadline
asm volatile ("pause"); asm volatile ("pause");
} }
return true; return true;
@@ -568,8 +582,8 @@ pub fn checkWarpSource() void {
@atomicStore(u32, &warp_bsp_ready, 0, .release); @atomicStore(u32, &warp_bsp_ready, 0, .release);
return; return;
} }
var i: u32 = 0; const deadline = rdtsc() +% warpTicksFor(warp_check_ns);
while (i < warp_rounds) : (i += 1) warpTick(); while (rdtsc() -% deadline >= (1 << 62)) warpTick(); // until the time budget is spent
@atomicStore(u32, &warp_stop, 1, .release); @atomicStore(u32, &warp_stop, 1, .release);
@atomicStore(u32, &warp_bsp_ready, 0, .release); @atomicStore(u32, &warp_bsp_ready, 0, .release);
warp_checks += 1; warp_checks += 1;
@@ -583,9 +597,10 @@ pub fn checkWarpTarget() void {
if (clock_source != .tsc) return; if (clock_source != .tsc) return;
if (!warpAwait(&warp_bsp_ready)) return; if (!warpAwait(&warp_bsp_ready)) return;
@atomicStore(u32, &warp_ap_ready, 1, .release); @atomicStore(u32, &warp_ap_ready, 1, .release);
var spins: u64 = 0; // The BSP owns the budget; this bound only protects against a lost BSP.
while (@atomicLoad(u32, &warp_stop, .acquire) == 0) : (spins += 1) { const deadline = rdtsc() +% warpTicksFor(2 * warp_check_ns + warp_await_ns);
if (spins >= warp_spin_limit) return; while (@atomicLoad(u32, &warp_stop, .acquire) == 0) {
if (rdtsc() -% deadline < (1 << 62)) return; // past the deadline
warpTick(); warpTick();
} }
} }