Per-process address spaces (AddressSpace = a PML4 with an empty user half and the shared kernel half copied in; create/destroy in paging.zig) with CR3 switched on context switch and TSS.rsp0/kernel_rsp published per switch. The GS base now points at an arch per-CPU block and every ring transition observes the swapgs discipline, so a ring-3 `mov %ax,%gs` can no longer poison per-CPU access. syscall/sysret is the primary user entry (int 0x80 kept as a test path); one handler, installed once at boot, serves both and dispatches on whether the caller is a scheduled process or a borrowed test thread. spawnProcess loads an ELF into a fresh address space and schedules it; exit frees the address space after switching to the kernel tables. New `process` test: init runs twice as a real process (create/exit/recreate) on its own page tables, coexisting with a kernel task under preemption. Suite 28/28. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
928 lines
34 KiB
Zig
928 lines
34 KiB
Zig
//! In-kernel test cases, run at the end of bring-up when the kernel is built with
|
|
//! `-Dtest-case=<name>`. Each case writes structured markers to the serial port
|
|
//! that the QEMU harness (test/qemu_test.py) asserts on:
|
|
//!
|
|
//! [PASS]/[FAIL] <check> per assertion
|
|
//! DANOS-TEST-RESULT: PASS|FAIL overall, for non-faulting cases
|
|
//!
|
|
//! Faulting cases (fault-ud, fault-pf, fault-df) deliberately don't return a
|
|
//! result line — they trigger a CPU exception, and the harness asserts on the
|
|
//! exception report the handler prints (which also reaches serial).
|
|
|
|
const std = @import("std");
|
|
const danos = @import("danos");
|
|
const arch = @import("arch");
|
|
const platform = @import("platform");
|
|
const pmm = @import("pmm.zig");
|
|
const heap = @import("heap.zig");
|
|
const sched = @import("scheduler.zig");
|
|
const ipc = @import("ipc.zig");
|
|
const usermode = @import("usermode.zig");
|
|
|
|
/// Formatted write straight to serial, independent of the framebuffer console.
|
|
fn log(comptime fmt: []const u8, args: anytype) void {
|
|
var buf: [128]u8 = undefined;
|
|
arch.serialWrite(std.fmt.bufPrint(&buf, fmt, args) catch return);
|
|
}
|
|
|
|
var passed: u32 = 0;
|
|
var failed: u32 = 0;
|
|
|
|
fn check(name: []const u8, ok: bool) void {
|
|
if (ok) {
|
|
passed += 1;
|
|
log("[PASS] {s}\n", .{name});
|
|
} else {
|
|
failed += 1;
|
|
log("[FAIL] {s}\n", .{name});
|
|
}
|
|
}
|
|
|
|
/// Emit the overall result line the harness matches, then the done sentinel.
|
|
fn result() void {
|
|
log("DANOS-TEST-RESULT: {s} ({d} passed, {d} failed)\n", .{
|
|
if (failed == 0) "PASS" else "FAIL",
|
|
passed,
|
|
failed,
|
|
});
|
|
log("DANOS-TEST-DONE\n", .{});
|
|
}
|
|
|
|
pub fn run(case: []const u8, boot_info: *const BootInfo) void {
|
|
if (eql(case, "smoke")) {
|
|
smoke(boot_info);
|
|
} else if (eql(case, "discovery")) {
|
|
discoveryTest();
|
|
} else if (eql(case, "wx")) {
|
|
wxTest();
|
|
} else if (eql(case, "timer")) {
|
|
timer();
|
|
} else if (eql(case, "clock")) {
|
|
clock();
|
|
} else if (eql(case, "vmm")) {
|
|
vmm();
|
|
} else if (eql(case, "heap")) {
|
|
heapTest();
|
|
} else if (eql(case, "sched")) {
|
|
schedTest();
|
|
} else if (eql(case, "priority")) {
|
|
priorityTest();
|
|
} else if (eql(case, "sleep")) {
|
|
sleepTest();
|
|
} else if (eql(case, "event")) {
|
|
eventTest();
|
|
} else if (eql(case, "ipc")) {
|
|
ipcTest();
|
|
} else if (eql(case, "smp")) {
|
|
smpTest();
|
|
} else if (eql(case, "affinity")) {
|
|
affinityTest();
|
|
} else if (eql(case, "smp-stress")) {
|
|
stressTest();
|
|
} else if (eql(case, "smp-retry")) {
|
|
smpRetryTest();
|
|
} else if (eql(case, "fault-ud")) {
|
|
faultInvalidOpcode();
|
|
} else if (eql(case, "fault-pf")) {
|
|
faultPageFault();
|
|
} else if (eql(case, "fault-df")) {
|
|
faultDoubleFault();
|
|
} else if (eql(case, "fault-ap-df")) {
|
|
faultApTest();
|
|
} else if (eql(case, "fault-nx")) {
|
|
faultNoExecute();
|
|
} else if (eql(case, "fault-null")) {
|
|
faultNull();
|
|
} else if (eql(case, "user")) {
|
|
userTest();
|
|
} else if (eql(case, "user-pf")) {
|
|
userPfTest();
|
|
} else if (eql(case, "init")) {
|
|
initTest(boot_info);
|
|
} else if (eql(case, "process")) {
|
|
processTest(boot_info);
|
|
} else if (eql(case, "poweroff")) {
|
|
powerTest(.off);
|
|
} else if (eql(case, "reboot")) {
|
|
powerTest(.reboot);
|
|
} else {
|
|
log("DANOS-TEST-RESULT: FAIL (unknown case '{s}')\n", .{case});
|
|
}
|
|
}
|
|
|
|
fn platformHal() platform.Hal {
|
|
return .{
|
|
.mapMmio = arch.mapMmio,
|
|
.pioRead = arch.pioRead,
|
|
.pioWrite = arch.pioWrite,
|
|
};
|
|
}
|
|
|
|
/// Drive an ACPI power transition. On success the machine powers off or resets,
|
|
/// so QEMU exits — the harness observes the process exit. If control returns, the
|
|
/// transition failed and we emit a FAIL result.
|
|
fn powerTest(comptime action: enum { off, reboot }) void {
|
|
const name = if (action == .off) "poweroff" else "reboot";
|
|
log("DANOS-TEST-BEGIN: {s}\n", .{name});
|
|
const hal = platformHal();
|
|
log("DANOS-POWER: attempting {s}\n", .{name});
|
|
switch (action) {
|
|
.off => platform.shutdown(hal),
|
|
.reboot => platform.reboot(hal),
|
|
}
|
|
check("power transition took effect", false);
|
|
result();
|
|
}
|
|
|
|
const BootInfo = danos.BootInfo;
|
|
|
|
fn eql(a: []const u8, b: []const u8) bool {
|
|
return std.mem.eql(u8, a, b);
|
|
}
|
|
|
|
/// Non-destructive checks of the memory map and frame allocator.
|
|
fn smoke(boot_info: *const BootInfo) void {
|
|
log("DANOS-TEST-BEGIN: smoke\n", .{});
|
|
|
|
// The memory map has some usable RAM.
|
|
const mm = boot_info.memory_map;
|
|
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physToVirt(mm.regions)))[0..mm.len];
|
|
var usable: u64 = 0;
|
|
for (regions) |r| {
|
|
if (r.kind == .usable) usable += r.pages;
|
|
}
|
|
check("memory map reports usable RAM", usable > 0);
|
|
|
|
// The frame allocator hands out distinct, page-aligned frames.
|
|
const a = pmm.alloc();
|
|
const b = pmm.alloc();
|
|
check("alloc returns a frame", a != null);
|
|
check("alloc returns distinct frames", a != null and b != null and a.? != b.?);
|
|
check("frames are page-aligned", (a orelse 1) % danos.page_size == 0);
|
|
|
|
// Freeing restores the count.
|
|
const before = pmm.stats().free_frames;
|
|
if (a) |p| pmm.free(p);
|
|
if (b) |p| pmm.free(p);
|
|
check("free returns frames to the pool", pmm.stats().free_frames == before + 2);
|
|
|
|
// Paging is active on our own tables (CR3 is non-zero and page-aligned).
|
|
const cr3 = arch.readCr3();
|
|
check("paging active (CR3 set)", cr3 != 0 and cr3 % danos.page_size == 0);
|
|
|
|
result();
|
|
}
|
|
|
|
/// Verify device interrupts fire and return: the timer tick counter must advance
|
|
/// on its own. Interrupts are already enabled by kmain before tests run.
|
|
fn timer() void {
|
|
log("DANOS-TEST-BEGIN: timer\n", .{});
|
|
const start = arch.ticks();
|
|
// Busy-wait for the counter to advance. arch.ticks() is a volatile load, so
|
|
// the compiler re-reads it each iteration and sees the interrupt's update.
|
|
// The cap is only a safety net; the harness timeout is the real backstop.
|
|
var spins: u64 = 0;
|
|
while (arch.ticks() == start and spins < 5_000_000_000) spins +%= 1;
|
|
check("timer interrupts advance the tick count", arch.ticks() > start);
|
|
|
|
result();
|
|
}
|
|
|
|
/// Verify device discovery populated the platform facts the rest of the kernel
|
|
/// depends on — the results ACPI parsing stashed in globals at boot. These are
|
|
/// stable for the QEMU q35 + OVMF machine the harness runs, and span the tables:
|
|
/// MADT (LAPIC base, CPU count), FADT (PM/reset registers), and the AML parse
|
|
/// (the sleep type, plus the integrity check that every byte was consumed).
|
|
fn discoveryTest() void {
|
|
log("DANOS-TEST-BEGIN: discovery\n", .{});
|
|
const pinfo = platform.platformInfo();
|
|
const pw = platform.powerInfo();
|
|
const am = platform.amlStats();
|
|
|
|
check("LAPIC base discovered (MADT)", pinfo.lapic_base == 0xFEE00000);
|
|
check("ACPI PM timer found (FADT)", pinfo.pm_timer.present());
|
|
check("PM1a control register found (FADT)", pw.pm1a_cnt.present());
|
|
check("reset register supported (FADT)", pw.reset_supported);
|
|
check("S5 sleep type found (AML)", pw.s5 != null);
|
|
check("AML parsed completely (consumed == total)", am.total > 0 and am.consumed == am.total);
|
|
check("at least one CPU enumerated (MADT)", platform.cpus().len >= 1);
|
|
|
|
result();
|
|
}
|
|
|
|
/// Audit the W^X invariant across the memory classes: kernel code must be
|
|
/// executable, everything else must not be. `arch.pageExecutable` reads the leaf
|
|
/// page-table entry's NX bit, so this guards the permission overlay in paging.zig —
|
|
/// a broader check than `fault-nx`, which only exercises one data page.
|
|
fn wxTest() void {
|
|
log("DANOS-TEST-BEGIN: wx\n", .{});
|
|
|
|
check("kernel code is executable (R+X)", arch.pageExecutable(@intFromPtr(&wxTest)));
|
|
|
|
const ro = "danos-wx-probe"; // string literal -> .rodata
|
|
check("rodata is non-executable (NX)", !arch.pageExecutable(@intFromPtr(ro.ptr)));
|
|
|
|
check("kernel data is non-executable (NX)", !arch.pageExecutable(@intFromPtr(&passed)));
|
|
|
|
if (heap.allocator().alloc(u8, 64) catch null) |h| {
|
|
check("heap is non-executable (NX)", !arch.pageExecutable(@intFromPtr(h.ptr)));
|
|
heap.allocator().free(h);
|
|
}
|
|
|
|
var local: u64 = 0;
|
|
_ = &local;
|
|
check("stack is non-executable (NX)", !arch.pageExecutable(@intFromPtr(&local)));
|
|
|
|
result();
|
|
}
|
|
|
|
/// Verify the on-demand VMM: map a fresh frame at an unused virtual address, and
|
|
/// check it's writable and reads back.
|
|
fn vmm() void {
|
|
log("DANOS-TEST-BEGIN: vmm\n", .{});
|
|
const frame = pmm.alloc();
|
|
check("frame available to map", frame != null);
|
|
if (frame) |phys| {
|
|
var virt: u64 = 0x0000_4000_0000_0000; // canonical, well clear of everything mapped
|
|
arch.mapPage(virt, phys, true);
|
|
const p: *volatile u64 = @ptrFromInt(virt);
|
|
p.* = 0xdead_c0de_cafe_babe;
|
|
check("mapped page is writable and reads back", p.* == 0xdead_c0de_cafe_babe);
|
|
arch.unmapPage(virt);
|
|
pmm.free(phys);
|
|
virt += 0;
|
|
}
|
|
result();
|
|
}
|
|
|
|
/// Exercise the kernel heap: basic alloc/write/free, reuse, growth beyond the
|
|
/// initial region, and a std container backed by it.
|
|
fn heapTest() void {
|
|
log("DANOS-TEST-BEGIN: heap\n", .{});
|
|
const a = heap.allocator();
|
|
|
|
// Allocate, write a pattern, read it back, free.
|
|
const buf = a.alloc(u8, 4096) catch null;
|
|
check("alloc 4096 bytes", buf != null);
|
|
if (buf) |b| {
|
|
@memset(b, 0xAB);
|
|
check("heap memory is writable and reads back", b[0] == 0xAB and b[4095] == 0xAB);
|
|
a.free(b);
|
|
}
|
|
|
|
// Freeing then re-allocating the same size should reuse the block.
|
|
const p1 = a.alloc(u64, 8) catch null;
|
|
const addr1 = if (p1) |p| @intFromPtr(p.ptr) else 0;
|
|
if (p1) |p| a.free(p);
|
|
const p2 = a.alloc(u64, 8) catch null;
|
|
const addr2 = if (p2) |p| @intFromPtr(p.ptr) else 0;
|
|
check("freed block is reused", addr1 != 0 and addr1 == addr2);
|
|
if (p2) |p| a.free(p);
|
|
|
|
// Force growth past the initial page and check every block is usable.
|
|
var blocks: [64]?[]u8 = .{null} ** 64;
|
|
var ok = true;
|
|
for (&blocks, 0..) |*slot, i| {
|
|
const b = a.alloc(u8, 4096) catch null;
|
|
slot.* = b;
|
|
if (b) |bb| @memset(bb, @intCast(i & 0xff)) else {
|
|
ok = false;
|
|
}
|
|
}
|
|
for (blocks, 0..) |slot, i| {
|
|
if (slot) |bb| {
|
|
if (bb[0] != @as(u8, @intCast(i & 0xff)) or bb[4095] != @as(u8, @intCast(i & 0xff))) ok = false;
|
|
}
|
|
}
|
|
check("many allocations (heap growth) stay valid", ok);
|
|
for (blocks) |slot| {
|
|
if (slot) |bb| a.free(bb);
|
|
}
|
|
|
|
// A std container backed by the kernel heap.
|
|
var list: std.ArrayList(u32) = .empty;
|
|
var sum: u64 = 0;
|
|
var expected: u64 = 0;
|
|
var i: u32 = 0;
|
|
var list_ok = true;
|
|
while (i < 1000) : (i += 1) {
|
|
list.append(a, i) catch {
|
|
list_ok = false;
|
|
};
|
|
expected += i;
|
|
}
|
|
for (list.items) |v| sum += v;
|
|
list.deinit(a);
|
|
check("std.ArrayList on the kernel heap", list_ok and sum == expected);
|
|
|
|
result();
|
|
}
|
|
|
|
/// Verify the calibrated clocks: sane measured frequencies, monotonic uptime that
|
|
/// advances with real ticks, and — the point of the TSC clock — nanosecond
|
|
/// resolution far finer than the 1 ms tick, with the unit functions consistent.
|
|
fn clock() void {
|
|
log("DANOS-TEST-BEGIN: clock\n", .{});
|
|
|
|
const lapic = arch.lapicHz();
|
|
check("LAPIC frequency measured", lapic > 1_000_000 and lapic < 100_000_000_000);
|
|
const tsc = arch.tscHz();
|
|
check("TSC frequency measured", tsc > 100_000_000 and tsc < 100_000_000_000);
|
|
|
|
// Uptime advances over ~5 real ticks (1000 Hz => 1 tick == 1 ms).
|
|
const start_ticks = arch.ticks();
|
|
const start_ms = arch.millis();
|
|
var spins: u64 = 0;
|
|
while (arch.ticks() < start_ticks + 5 and spins < 5_000_000_000) spins +%= 1;
|
|
const elapsed_ms = arch.millis() - start_ms;
|
|
check("uptime advances with ticks", elapsed_ms >= 5 and elapsed_ms < 100);
|
|
|
|
// Sub-millisecond resolution: spin until nanos() first advances, then confirm
|
|
// that first step happened within a millisecond — so nanos() resolves finer
|
|
// than the 1 ms tick (a tick clock's smallest step *is* 1 ms). Spinning to the
|
|
// first change is robust to QEMU's coarse TSC update granularity.
|
|
const n1 = arch.nanos();
|
|
var s2: u64 = 0;
|
|
while (arch.nanos() == n1 and s2 < 10_000_000) s2 +%= 1;
|
|
const n2 = arch.nanos();
|
|
check("nanos() has sub-millisecond resolution", n2 > n1 and (n2 - n1) < 1_000_000);
|
|
|
|
// The unit functions agree (within rounding).
|
|
const ns = arch.nanos();
|
|
check("nanos/micros/millis are consistent", diffWithin(arch.micros(), ns / 1000, 1000) and diffWithin(arch.millis(), ns / 1_000_000, 2));
|
|
|
|
result();
|
|
}
|
|
|
|
fn diffWithin(a: u64, b: u64, tol: u64) bool {
|
|
return if (a > b) a - b <= tol else b - a <= tol;
|
|
}
|
|
|
|
// --- scheduler tests ------------------------------------------------------
|
|
|
|
var counters = [_]u64{0} ** 3;
|
|
|
|
fn spin0() void {
|
|
const p: *volatile u64 = &counters[0];
|
|
while (true) p.* = p.* +% 1;
|
|
}
|
|
fn spin1() void {
|
|
const p: *volatile u64 = &counters[1];
|
|
while (true) p.* = p.* +% 1;
|
|
}
|
|
fn spin2() void {
|
|
const p: *volatile u64 = &counters[2];
|
|
while (true) p.* = p.* +% 1;
|
|
}
|
|
|
|
/// Preemption: spawn three tasks that busy-loop *without* yielding. If they all
|
|
/// make progress, the timer must be preempting between them (and the context
|
|
/// switch works) — because nothing yields voluntarily.
|
|
fn schedTest() void {
|
|
log("DANOS-TEST-BEGIN: sched\n", .{});
|
|
counters = .{ 0, 0, 0 };
|
|
sched.spawn(spin0, 4);
|
|
sched.spawn(spin1, 4);
|
|
sched.spawn(spin2, 4);
|
|
|
|
const c0: *volatile u64 = &counters[0];
|
|
const c1: *volatile u64 = &counters[1];
|
|
const c2: *volatile u64 = &counters[2];
|
|
var spins: u64 = 0;
|
|
while ((c0.* == 0 or c1.* == 0 or c2.* == 0) and spins < 5_000_000_000) spins +%= 1;
|
|
|
|
check("all three non-yielding tasks made progress (preemption)", c0.* > 0 and c1.* > 0 and c2.* > 0);
|
|
result();
|
|
}
|
|
|
|
var run_order = [_]u8{0} ** 4;
|
|
var run_n: usize = 0;
|
|
|
|
fn recordExit(priority: u8) void {
|
|
run_order[run_n] = priority;
|
|
run_n += 1;
|
|
sched.exit();
|
|
}
|
|
fn taskHigh() void {
|
|
recordExit(6);
|
|
}
|
|
fn taskMid() void {
|
|
recordExit(4);
|
|
}
|
|
fn taskLow() void {
|
|
recordExit(2);
|
|
}
|
|
|
|
/// Fixed priority: with preemption off (deterministic), spawn tasks at three
|
|
/// priorities and let them run cooperatively. They must run highest-first.
|
|
fn priorityTest() void {
|
|
log("DANOS-TEST-BEGIN: priority\n", .{});
|
|
sched.setPreemption(false);
|
|
sched.setPriority(1); // above the idle task (0), below the workers — runs last
|
|
run_n = 0;
|
|
|
|
sched.spawn(taskLow, 2);
|
|
sched.spawn(taskMid, 4);
|
|
sched.spawn(taskHigh, 6);
|
|
|
|
while (run_n < 3) sched.yield(); // regain control only once the workers are done
|
|
|
|
check("tasks ran highest-priority first", run_order[0] == 6 and run_order[1] == 4 and run_order[2] == 2);
|
|
|
|
sched.setPriority(4);
|
|
sched.setPreemption(true);
|
|
result();
|
|
}
|
|
|
|
var event_wq: sched.WaitQueue = .{};
|
|
var event_stage: u32 = 0;
|
|
|
|
fn eventWaiter() void {
|
|
event_stage = 1; // reached the wait
|
|
sched.wait(&event_wq); // block until woken
|
|
event_stage = 3; // woken and resumed
|
|
sched.exit();
|
|
}
|
|
|
|
/// Event-based blocking: a task blocks on a wait queue and is woken. The waiter is
|
|
/// higher priority, so waking it preempts us and it runs to completion at once.
|
|
fn eventTest() void {
|
|
log("DANOS-TEST-BEGIN: event\n", .{});
|
|
event_stage = 0;
|
|
sched.spawn(eventWaiter, 6); // higher priority than this task (4)
|
|
|
|
var spins: u64 = 0;
|
|
while (event_stage != 1 and spins < 1_000_000_000) : (spins += 1) sched.yield();
|
|
check("waiter reached the wait and blocked", event_stage == 1);
|
|
|
|
sched.wake(&event_wq);
|
|
check("wake resumed the blocked waiter (preempting)", event_stage == 3);
|
|
result();
|
|
}
|
|
|
|
var channel: ipc.Channel(u64, 4) = .{};
|
|
var recv_sum: u64 = 0;
|
|
var recv_count: u64 = 0;
|
|
|
|
fn producer() void {
|
|
var i: u64 = 1;
|
|
while (i <= 100) : (i += 1) channel.send(i);
|
|
sched.exit();
|
|
}
|
|
fn consumer() void {
|
|
var n: u64 = 0;
|
|
while (n < 100) : (n += 1) {
|
|
recv_sum += channel.recv();
|
|
recv_count += 1;
|
|
}
|
|
sched.exit();
|
|
}
|
|
|
|
/// IPC: a producer and consumer pass 100 messages through a 4-slot channel. The
|
|
/// small buffer forces the channel full and empty repeatedly, exercising both the
|
|
/// blocking-send and blocking-recv paths. The messages must arrive intact.
|
|
fn ipcTest() void {
|
|
log("DANOS-TEST-BEGIN: ipc\n", .{});
|
|
channel = .{};
|
|
recv_sum = 0;
|
|
recv_count = 0;
|
|
sched.spawn(consumer, 5); // above this task (4) so they run and we observe after
|
|
sched.spawn(producer, 5);
|
|
|
|
var spins: u64 = 0;
|
|
while (recv_count < 100 and spins < 2_000_000_000) : (spins += 1) sched.yield();
|
|
|
|
check("all 100 messages received", recv_count == 100);
|
|
check("messages arrived intact (sum 1..100 == 5050)", recv_sum == 5050);
|
|
result();
|
|
}
|
|
|
|
/// Blocking: sleep(50) should block this task for about 50 ms (measured on the
|
|
/// calibrated clock) — not busy-wait — while the idle task runs.
|
|
fn sleepTest() void {
|
|
log("DANOS-TEST-BEGIN: sleep\n", .{});
|
|
const t0 = arch.millis();
|
|
sched.sleep(50);
|
|
const elapsed = arch.millis() - t0;
|
|
check("sleep(50) blocked for ~50 ms", elapsed >= 50 and elapsed <= 70);
|
|
result();
|
|
}
|
|
|
|
// --- SMP parallelism ------------------------------------------------------
|
|
|
|
var seen_core = [_]bool{false} ** 8;
|
|
var smp_running: bool = true;
|
|
|
|
/// A worker that, while running, records which core it's executing on. Spread across
|
|
/// spawned workers and idle APs, these should land on more than one core.
|
|
fn smpWorker() void {
|
|
const p: *volatile bool = &smp_running;
|
|
while (p.*) {
|
|
const c = sched.currentCpuIndex();
|
|
if (c < seen_core.len) seen_core[c] = true;
|
|
}
|
|
sched.exit();
|
|
}
|
|
|
|
/// Prove tasks run **in parallel** on multiple cores (not just interleaved on one).
|
|
/// Spawn several CPU-bound workers; each stamps the core it runs on into `seen_core`.
|
|
/// With the application processors online, more than one core should show up — which
|
|
/// can only happen if work is genuinely running at the same time on different cores.
|
|
/// (Run with QEMU `-smp N`; on a single core this would see just one and fail.)
|
|
fn smpTest() void {
|
|
log("DANOS-TEST-BEGIN: smp\n", .{});
|
|
seen_core = .{false} ** 8;
|
|
smp_running = true;
|
|
|
|
var i: usize = 0;
|
|
while (i < 4) : (i += 1) sched.spawn(smpWorker, 4);
|
|
|
|
// Let the workers run across cores for a stretch of real time.
|
|
var spins: u64 = 0;
|
|
while (spins < 2_000_000_000) spins +%= 1;
|
|
smp_running = false;
|
|
|
|
var cores_seen: u32 = 0;
|
|
for (seen_core) |s| {
|
|
if (s) cores_seen += 1;
|
|
}
|
|
log("DANOS-SMP: workers ran on {d} distinct core(s)\n", .{cores_seen});
|
|
check("tasks ran on multiple cores in parallel", cores_seen >= 2);
|
|
|
|
// Bring-up is done, so the trampoline frame must be inert: zeroed (no stale code)
|
|
// and non-executable (W^X restored). It's armed only while a core is climbing.
|
|
const tramp = arch.trampolinePage();
|
|
check("trampoline frame reserved", tramp != 0);
|
|
if (tramp != 0) {
|
|
const bytes: [*]const u8 = @ptrFromInt(danos.physToVirt(tramp));
|
|
var zeroed = true;
|
|
for (0..4096) |b| {
|
|
if (bytes[b] != 0) zeroed = false;
|
|
}
|
|
check("trampoline page zeroed when dormant", zeroed);
|
|
check("trampoline page non-executable when dormant", !arch.pageExecutable(tramp));
|
|
}
|
|
result();
|
|
}
|
|
|
|
// --- affinity: a pinned task never migrates -------------------------------
|
|
|
|
var affinity_cores = [_]bool{false} ** 8;
|
|
var affinity_running: bool = true;
|
|
|
|
fn affinityWorker() void {
|
|
const p: *volatile bool = &affinity_running;
|
|
while (p.*) {
|
|
const c = sched.currentCpuIndex();
|
|
if (c < affinity_cores.len) affinity_cores[c] = true;
|
|
}
|
|
sched.exit();
|
|
}
|
|
|
|
/// A task pinned to a core must run **only** on that core. Pin a busy worker to
|
|
/// core 1 and let it run through many preemptions; it must have stamped core 1 and no
|
|
/// other. An *unpinned* task scatters across cores (that's what the smp test shows),
|
|
/// so a broken pin fails this deterministically — over this many time slices a
|
|
/// free-floating task will land on some other core.
|
|
fn affinityTest() void {
|
|
log("DANOS-TEST-BEGIN: affinity\n", .{});
|
|
affinity_cores = .{false} ** 8;
|
|
affinity_running = true;
|
|
|
|
if (!sched.spawnOn(affinityWorker, 4, 1)) {
|
|
check("worker pinned to core 1 (run with -smp)", false);
|
|
result();
|
|
return;
|
|
}
|
|
|
|
var spins: u64 = 0;
|
|
while (spins < 3_000_000_000) spins +%= 1; // many time slices across the cores
|
|
affinity_running = false;
|
|
var settle: u64 = 0;
|
|
while (settle < 200_000_000) settle +%= 1; // let the worker see the flag and exit
|
|
|
|
var others: u32 = 0;
|
|
for (affinity_cores, 0..) |seen, c| {
|
|
if (seen and c != 1) others += 1;
|
|
}
|
|
log("DANOS-AFFINITY: pinned worker touched core 1={}, other cores={d}\n", .{ affinity_cores[1], others });
|
|
check("pinned task ran on its core (1)", affinity_cores[1]);
|
|
check("pinned task never migrated to another core", others == 0);
|
|
result();
|
|
}
|
|
|
|
// --- SMP stress: hammer the big kernel lock across cores ------------------
|
|
|
|
const stress_pairs = 4; // producer/consumer pairs (8 tasks; fits the 16-task pool)
|
|
const stress_msgs = 100_000; // messages per pair
|
|
const stress_cap = 4; // small channel -> constant block/wake, more lock churn
|
|
|
|
var stress_chan = [_]ipc.Channel(u64, stress_cap){.{}} ** stress_pairs;
|
|
var stress_recv = [_]u64{0} ** stress_pairs; // messages received per pair
|
|
var stress_order_ok = [_]bool{true} ** stress_pairs; // FIFO order held per pair
|
|
var stress_cores = [_]bool{false} ** 8; // cores that ran a consumer
|
|
var stress_prod_claim: usize = 0;
|
|
var stress_cons_claim: usize = 0;
|
|
|
|
fn stressProducer() void {
|
|
// Claim a unique pair index (atomic: producers start on different cores).
|
|
const idx = @atomicRmw(usize, &stress_prod_claim, .Add, 1, .monotonic);
|
|
var v: u64 = 1;
|
|
while (v <= stress_msgs) : (v += 1) stress_chan[idx].send(v);
|
|
sched.exit();
|
|
}
|
|
|
|
fn stressConsumer() void {
|
|
const idx = @atomicRmw(usize, &stress_cons_claim, .Add, 1, .monotonic);
|
|
var expected: u64 = 1;
|
|
while (expected <= stress_msgs) : (expected += 1) {
|
|
const got = stress_chan[idx].recv();
|
|
if (got != expected) stress_order_ok[idx] = false; // lost/reordered => lock broke
|
|
const c = sched.currentCpuIndex();
|
|
if (c < stress_cores.len) stress_cores[c] = true;
|
|
stress_recv[idx] = expected;
|
|
}
|
|
sched.exit();
|
|
}
|
|
|
|
/// Stress the big kernel lock under sustained cross-core contention. Each pair drives
|
|
/// `stress_msgs` sequenced messages through a 4-slot channel — every send and recv
|
|
/// takes the lock, and the small buffer forces constant block/wake (so the scheduler
|
|
/// churns too). A single-producer/single-consumer channel must deliver in strict FIFO
|
|
/// order; if the lock let two cores into a critical section at once, the ring buffer
|
|
/// corrupts and the consumer sees a wrong or out-of-order value (or the run hangs /
|
|
/// faults). Passing means ~320k lock acquisitions across the cores stayed consistent.
|
|
fn stressTest() void {
|
|
log("DANOS-TEST-BEGIN: smp-stress\n", .{});
|
|
stress_chan = [_]ipc.Channel(u64, stress_cap){.{}} ** stress_pairs;
|
|
stress_recv = [_]u64{0} ** stress_pairs;
|
|
stress_order_ok = [_]bool{true} ** stress_pairs;
|
|
stress_cores = [_]bool{false} ** 8;
|
|
stress_prod_claim = 0;
|
|
stress_cons_claim = 0;
|
|
|
|
var i: usize = 0;
|
|
while (i < stress_pairs) : (i += 1) sched.spawn(stressConsumer, 4);
|
|
i = 0;
|
|
while (i < stress_pairs) : (i += 1) sched.spawn(stressProducer, 4);
|
|
|
|
// Drop below the workers so they get the cores; wake periodically to check for
|
|
// completion. A broken lock instead hangs here (harness timeout) or faults.
|
|
sched.setPriority(1);
|
|
var spins: u64 = 0;
|
|
while (spins < 40_000_000_000) : (spins += 1) {
|
|
var done = true;
|
|
for (stress_recv) |n| {
|
|
if (n < stress_msgs) done = false;
|
|
}
|
|
if (done) break;
|
|
}
|
|
sched.setPriority(4);
|
|
|
|
var total: u64 = 0;
|
|
for (stress_recv) |n| total += n;
|
|
var order_ok = true;
|
|
for (stress_order_ok) |ok| {
|
|
if (!ok) order_ok = false;
|
|
}
|
|
var cores: u32 = 0;
|
|
for (stress_cores) |s| {
|
|
if (s) cores += 1;
|
|
}
|
|
|
|
log("DANOS-STRESS: {d}/{d} pairs complete on {d} cores\n", .{ total, @as(u64, stress_pairs) * stress_msgs, cores });
|
|
check("every message delivered", total == @as(u64, stress_pairs) * stress_msgs);
|
|
check("strict FIFO order held (no lock corruption)", order_ok);
|
|
check("contention was genuinely cross-core", cores >= 2);
|
|
result();
|
|
}
|
|
|
|
/// Retry: `main` forced the first AP wake attempt to fail (arch.testFailNextWakes),
|
|
/// so a core missed its first INIT-SIPI-SIPI. The boot retry must have brought it back
|
|
/// anyway — every enumerated core should be online. If retry were broken, that core
|
|
/// would be parked and the count would fall short.
|
|
fn smpRetryTest() void {
|
|
log("DANOS-TEST-BEGIN: smp-retry\n", .{});
|
|
const total = platform.cpus().len;
|
|
const online = sched.onlineCount();
|
|
log("DANOS-RETRY: {d}/{d} cores online after a forced first-wake failure\n", .{ online, total });
|
|
check("multiple cores enumerated (run with -smp)", total >= 2);
|
|
check("retry brought every core online despite a failed first wake", online == total);
|
|
result();
|
|
}
|
|
|
|
// --- ring 3 (user mode) -----------------------------------------------------
|
|
|
|
/// The full ring-3 round trip: enter user mode, take syscalls and timer
|
|
/// interrupts from CPL 3, and come back. Preemption is disabled for the run —
|
|
/// enter_user publishes TSS.rsp0 on *this* core, so the task must not migrate
|
|
/// (interrupts still fire and iretq back into ring 3, which is the point).
|
|
fn userTest() void {
|
|
log("DANOS-TEST-BEGIN: user\n", .{});
|
|
sched.setPreemption(false);
|
|
const ran = if (usermode.run(usermode.helloBlob())) true else |err| blk: {
|
|
log("DANOS-USER: run failed: {s}\n", .{@errorName(err)});
|
|
break :blk false;
|
|
};
|
|
sched.setPreemption(true);
|
|
|
|
check("user program ran and exited (ring-3 round trip)", ran);
|
|
check("two ping syscalls received", usermode.ping_count == 2);
|
|
check("syscall args passed in registers (0xC0DE, 0xBEEF)", usermode.pings[0].value == 0xC0DE and usermode.pings[1].value == 0xBEEF);
|
|
check("syscalls came from CPL 3 (CS = user selector | RPL 3)", usermode.pings[0].cs == 0x23 and usermode.pings[1].cs == 0x23);
|
|
check("timer ticks advanced while in ring 3", usermode.pings[1].ticks > usermode.pings[0].ticks);
|
|
result();
|
|
}
|
|
|
|
var proc_worker_run: bool = true;
|
|
var proc_worker_ran: bool = false;
|
|
|
|
/// A kernel task that spins while a process runs, to prove the two coexist under
|
|
/// preemption (a process on its own CR3 does not stall kernel work).
|
|
fn procWorker() void {
|
|
const running: *volatile bool = &proc_worker_run;
|
|
const ran: *volatile bool = &proc_worker_ran;
|
|
while (running.*) ran.* = true;
|
|
sched.exit();
|
|
}
|
|
|
|
/// Real processes: load /sbin/init as a scheduled ring-3 process with its own
|
|
/// address space, twice in succession. The first run proves a process executes
|
|
/// on its own page tables (write from CPL 3) and coexists preemptively with a
|
|
/// kernel task; its exit frees the address space. The second run reuses those
|
|
/// reclaimed frames — succeeding proves create/exit/teardown/recreate is sound.
|
|
fn processTest(boot_info: *const BootInfo) void {
|
|
log("DANOS-TEST-BEGIN: process\n", .{});
|
|
check("bootloader handed over sbin/init", boot_info.init_len != 0);
|
|
if (boot_info.init_len == 0) {
|
|
result();
|
|
return;
|
|
}
|
|
const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len];
|
|
const expected = "init: hello from user space\n";
|
|
|
|
proc_worker_run = true;
|
|
proc_worker_ran = false;
|
|
sched.spawn(procWorker, 4); // kernel task, same priority as the processes
|
|
|
|
var runs: u32 = 0;
|
|
var last_cs: u64 = 0;
|
|
sched.setPriority(1); // drop below the workers so they get the cores
|
|
var round: u32 = 0;
|
|
while (round < 2) : (round += 1) {
|
|
usermode.write_len = 0;
|
|
usermode.write_cs = 0;
|
|
usermode.exit_code = 0xdead;
|
|
usermode.spawnProcess(image, 4) catch {
|
|
log("DANOS-PROC: spawn round {d} failed\n", .{round});
|
|
continue;
|
|
};
|
|
var spins: u64 = 0;
|
|
while (usermode.exit_code == 0xdead and spins < 5_000_000_000) : (spins += 1) sched.yield();
|
|
log("DANOS-PROC: round {d} write_len={d} exit_code=0x{x}\n", .{ round, usermode.write_len, usermode.exit_code });
|
|
if (eql(usermode.write_buf[0..usermode.write_len], expected)) {
|
|
runs += 1;
|
|
last_cs = usermode.write_cs;
|
|
}
|
|
}
|
|
sched.setPriority(4);
|
|
proc_worker_run = false;
|
|
|
|
check("process ran twice on its own address space (create/exit/recreate)", runs == 2);
|
|
check("process wrote from CPL 3 (CS = user selector | RPL 3)", last_cs == 0x23);
|
|
check("a kernel task coexisted with the process (preemption)", proc_worker_ran);
|
|
result();
|
|
}
|
|
|
|
/// Isolation: a ring-3 read of a kernel-only page (the LAPIC page — present,
|
|
/// supervisor) must page-fault with error code 0x5 (present | user) at the user
|
|
/// RIP. The fault report is the pass signal (matched by the harness); if the
|
|
/// read is somehow allowed the blob spins and the harness times out.
|
|
fn userPfTest() void {
|
|
log("DANOS-TEST-BEGIN: user-pf\n", .{});
|
|
sched.setPreemption(false);
|
|
_ = usermode.run(usermode.pfBlob()) catch {};
|
|
log("DANOS-TEST-RESULT: FAIL (user read of kernel memory did not fault)\n", .{});
|
|
}
|
|
|
|
/// The full user-binary path: the bootloader read sbin/init off the boot
|
|
/// volume and handed it over; load it as a user ELF and run it in ring 3. The
|
|
/// same call the normal boot path makes — here with teeth.
|
|
fn initTest(boot_info: *const BootInfo) void {
|
|
log("DANOS-TEST-BEGIN: init\n", .{});
|
|
check("bootloader handed over sbin/init", boot_info.init_len != 0);
|
|
if (boot_info.init_len == 0) {
|
|
result();
|
|
return;
|
|
}
|
|
const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.init_base)))[0..boot_info.init_len];
|
|
sched.setPreemption(false); // see userTest: pins the run to this core's rsp0
|
|
const code = usermode.runInitElf(image);
|
|
sched.setPreemption(true);
|
|
|
|
if (code) |c| {
|
|
check("init loaded, ran, and exited (user ELF path)", true);
|
|
check("init exited cleanly (code 0)", c == 0);
|
|
} else |err| {
|
|
log("DANOS-INIT-ERR: {s}\n", .{@errorName(err)});
|
|
check("init loaded, ran, and exited (user ELF path)", false);
|
|
}
|
|
check("init's write arrived intact", eql(usermode.write_buf[0..usermode.write_len], "init: hello from user space\n"));
|
|
check("write came from CPL 3 (CS = user selector | RPL 3)", usermode.write_cs == 0x23);
|
|
result();
|
|
}
|
|
|
|
fn faultInvalidOpcode() void {
|
|
log("DANOS-TEST-BEGIN: fault-ud\n", .{});
|
|
asm volatile ("ud2");
|
|
}
|
|
|
|
/// Verify NX: fetching an instruction from a data page (mapped no-execute) faults.
|
|
fn faultNoExecute() void {
|
|
log("DANOS-TEST-BEGIN: fault-nx\n", .{});
|
|
var scratch: u64 = 0xC3; // a lone `ret` — harmless if NX somehow let it run
|
|
const f: *const fn () void = @ptrFromInt(@intFromPtr(&scratch));
|
|
f(); // instruction fetch from an NX page -> #PF before it executes
|
|
log("DANOS-TEST-RESULT: FAIL (NX not enforced)\n", .{});
|
|
}
|
|
|
|
/// Verify the null guard: dereferencing address 0 (page 0 left unmapped) faults.
|
|
fn faultNull() void {
|
|
log("DANOS-TEST-BEGIN: fault-null\n", .{});
|
|
// Launder the address through empty asm so the compiler no longer knows it's
|
|
// 0 (otherwise it folds a null-pointer safety panic instead of doing the real
|
|
// access). `allowzero` skips the same null check on the cast. The write then
|
|
// hits the unmapped page 0 and takes a real hardware #PF.
|
|
var addr: u64 = 0;
|
|
addr = asm ("" : [ret] "=r" (-> u64) : [in] "0" (addr));
|
|
const p: *allowzero volatile u64 = @ptrFromInt(addr);
|
|
p.* = 1;
|
|
}
|
|
|
|
fn faultPageFault() void {
|
|
log("DANOS-TEST-BEGIN: fault-pf\n", .{});
|
|
// Runtime address so the backend emits a register store (not a `mov moffs`,
|
|
// which the self-hosted x86_64 backend can't encode).
|
|
var addr: u64 = 0xdeadbeef000; // well above all mapped RAM
|
|
const p: *volatile u64 = @ptrFromInt(addr);
|
|
p.* = 1;
|
|
addr += 0;
|
|
}
|
|
|
|
fn faultDoubleFault() void {
|
|
log("DANOS-TEST-BEGIN: fault-df\n", .{});
|
|
arch.disableInterrupts(); // so only the ud2 delivery (not a timer tick) triggers the #DF
|
|
// Point RSP at unmapped memory, then fault: the CPU can't push the fault
|
|
// frame, which escalates to #DF — survivable only because #DF runs on IST1.
|
|
var bad_sp: u64 = 0x5000000000;
|
|
asm volatile (
|
|
\\mov %[sp], %%rsp
|
|
\\ud2
|
|
:
|
|
: [sp] "r" (bad_sp),
|
|
: .{ .memory = true }
|
|
);
|
|
bad_sp += 0;
|
|
}
|
|
|
|
var ap_reached_fault: bool = false;
|
|
|
|
/// A task that faults with a #DF *on whatever core it's pinned to*. Announces the
|
|
/// core, then triggers the same double fault as `faultDoubleFault` — which is only
|
|
/// survivable on IST1, so it exercises that core's own TSS.
|
|
fn apDoubleFaultTask() void {
|
|
log("DANOS-AP: task running on core {d}, triggering #DF\n", .{sched.currentCpuIndex()});
|
|
@atomicStore(bool, &ap_reached_fault, true, .release);
|
|
arch.disableInterrupts();
|
|
var bad_sp: u64 = 0x5000000000;
|
|
asm volatile (
|
|
\\mov %[sp], %%rsp
|
|
\\ud2
|
|
:
|
|
: [sp] "r" (bad_sp),
|
|
: .{ .memory = true }
|
|
);
|
|
bad_sp += 0;
|
|
}
|
|
|
|
/// Fault on an application processor. Pins a double-faulting task to core 1, so the
|
|
/// fault is taken and handled by *that core's own* IDT and TSS/IST — not the BSP's.
|
|
/// The harness matches "core N: double fault (vector 8)" with N ≥ 1, which can only
|
|
/// appear if the AP caught the #DF on its IST1 (a broken per-core TSS would
|
|
/// triple-fault and reset instead). We then show the BSP still runs afterwards, so
|
|
/// the fault was *contained* to the AP, not fatal to the system.
|
|
fn faultApTest() void {
|
|
log("DANOS-TEST-BEGIN: fault-ap-df\n", .{});
|
|
if (!sched.spawnOn(apDoubleFaultTask, 6, 1)) {
|
|
log("DANOS-AP: could not pin to core 1 (run with -smp) - FAIL\n", .{});
|
|
arch.halt();
|
|
}
|
|
// Wait until the AP is about to fault, then keep running to prove containment.
|
|
var spins: u64 = 0;
|
|
while (!@atomicLoad(bool, &ap_reached_fault, .acquire) and spins < 5_000_000_000) spins +%= 1;
|
|
var settle: u64 = 0;
|
|
while (settle < 500_000_000) settle +%= 1; // let the AP take + report the fault
|
|
log("DANOS-BSP: core {d} still running after the AP fault (contained)\n", .{sched.currentCpuIndex()});
|
|
arch.halt();
|
|
}
|