Add a `system_spawn(name)` system call: the kernel loads a binary bundled in the initial-ramdisk, by name, as a fresh ring-3 process. It's the mechanism a user-space supervisor needs — discovery and policy stay in user space, the kernel only spawns. The kernel already holds the initial-ramdisk image from the boot handoff; it now stashes it (process.setInitialRamdisk) so the handler can resolve names, bounds-checks the name into the user half like debug_write, and returns -1 for an unknown name or a load failure. Ungated for now (any process may spawn any bundled binary); a spawn capability belongs here once the model grows one. The device manager stops logging "would spawn it" and calls runtime.system.spawn on its matched driver. On QEMU it discovers the HPET, matches `hpet`, and spawns it — and the driver comes all the way up (claims the timer, maps its MMIO, binds and services its IRQ, prints "hpet: ok"). The device-manager test now keys on that final marker: since only the manager is spawned, `hpet: ok` appearing proves the whole discover -> match -> system_spawn -> driver-up chain end to end. Transitional: the kernel still auto-spawns the whole initial-ramdisk at boot, so a real boot briefly double-spawns hpet (the second claim fails harmlessly). Increment 3 removes that redundancy so the manager is the sole owner of driver spawning. Suite 36/36 plus host tests.
1468 lines
59 KiB
Zig
1468 lines
59 KiB
Zig
//! In-kernel test cases, run at the end of bring-up when the kernel is built with
|
|
//! `-Dtest-case=<name>`. Each case writes structured markers to the serial port
|
|
//! that the QEMU harness (test/qemu_test.py) asserts on:
|
|
//!
|
|
//! [PASS]/[FAIL] <check> per assertion
|
|
//! DANOS-TEST-RESULT: PASS|FAIL overall, for non-faulting cases
|
|
//!
|
|
//! Faulting cases (fault-ud, fault-pf, fault-df) deliberately don't return a
|
|
//! result line — they trigger a CPU exception, and the harness asserts on the
|
|
//! exception report the handler prints (which also reaches serial).
|
|
|
|
const std = @import("std");
|
|
const boot_handoff = @import("boot-handoff");
|
|
const abi = @import("abi");
|
|
const device_abi = @import("device-abi");
|
|
const architecture = @import("architecture");
|
|
const devices_broker = @import("devices-broker.zig");
|
|
const platform = @import("platform");
|
|
const pmm = @import("pmm.zig");
|
|
const heap = @import("heap.zig");
|
|
const scheduler = @import("scheduler.zig");
|
|
const ipc = @import("ipc.zig");
|
|
const ipcsync = @import("ipc-synchronous.zig");
|
|
const irq = @import("irq.zig");
|
|
const sync = @import("sync.zig");
|
|
const process = @import("process.zig");
|
|
const initial_ramdisk = @import("initial-ramdisk");
|
|
|
|
/// Formatted write straight to serial, independent of the framebuffer console.
|
|
fn log(comptime fmt: []const u8, args: anytype) void {
|
|
var buffer: [128]u8 = undefined;
|
|
architecture.serialWrite(std.fmt.bufPrint(&buffer, fmt, args) catch return);
|
|
}
|
|
|
|
var passed: u32 = 0;
|
|
var failed: u32 = 0;
|
|
|
|
fn check(name: []const u8, ok: bool) void {
|
|
if (ok) {
|
|
passed += 1;
|
|
log("[PASS] {s}\n", .{name});
|
|
} else {
|
|
failed += 1;
|
|
log("[FAIL] {s}\n", .{name});
|
|
}
|
|
}
|
|
|
|
/// Emit the overall result line the harness matches, then the done sentinel.
|
|
fn result() void {
|
|
log("DANOS-TEST-RESULT: {s} ({d} passed, {d} failed)\n", .{
|
|
if (failed == 0) "PASS" else "FAIL",
|
|
passed,
|
|
failed,
|
|
});
|
|
log("DANOS-TEST-DONE\n", .{});
|
|
}
|
|
|
|
pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
|
if (eql(case, "smoke")) {
|
|
smoke(boot_information);
|
|
} else if (eql(case, "discovery")) {
|
|
discoveryTest();
|
|
} else if (eql(case, "wx")) {
|
|
wxTest();
|
|
} else if (eql(case, "timer")) {
|
|
timer();
|
|
} else if (eql(case, "clock")) {
|
|
clock();
|
|
} else if (eql(case, "vmm")) {
|
|
vmm();
|
|
} else if (eql(case, "heap")) {
|
|
heapTest();
|
|
} else if (eql(case, "sched")) {
|
|
schedulerTest();
|
|
} else if (eql(case, "priority")) {
|
|
priorityTest();
|
|
} else if (eql(case, "sleep")) {
|
|
sleepTest();
|
|
} else if (eql(case, "event")) {
|
|
eventTest();
|
|
} else if (eql(case, "ipc")) {
|
|
ipcTest();
|
|
} else if (eql(case, "ipc-call")) {
|
|
ipcCallTest();
|
|
} else if (eql(case, "smp")) {
|
|
smpTest();
|
|
} else if (eql(case, "affinity")) {
|
|
affinityTest();
|
|
} else if (eql(case, "smp-stress")) {
|
|
stressTest();
|
|
} else if (eql(case, "smp-retry")) {
|
|
smpRetryTest();
|
|
} else if (eql(case, "fault-ud")) {
|
|
faultInvalidOpcode();
|
|
} else if (eql(case, "fault-pf")) {
|
|
faultPageFault();
|
|
} else if (eql(case, "fault-df")) {
|
|
faultDoubleFault();
|
|
} else if (eql(case, "fault-ap-df")) {
|
|
faultApTest();
|
|
} else if (eql(case, "fault-nx")) {
|
|
faultNoExecute();
|
|
} else if (eql(case, "fault-null")) {
|
|
faultNull();
|
|
} else if (eql(case, "usermem")) {
|
|
userMemTest();
|
|
} else if (eql(case, "user-pf")) {
|
|
userPfTest();
|
|
} else if (eql(case, "init")) {
|
|
initTest(boot_information);
|
|
} else if (eql(case, "process")) {
|
|
processTest(boot_information);
|
|
} else if (eql(case, "initial-ramdisk")) {
|
|
initialRamdiskTest(boot_information);
|
|
} else if (eql(case, "vfs")) {
|
|
vfsTest(boot_information);
|
|
} else if (eql(case, "hpet")) {
|
|
hpetTest(boot_information);
|
|
} else if (eql(case, "iopass")) {
|
|
ioPassTest();
|
|
} else if (eql(case, "irqfree")) {
|
|
irqFreeTest();
|
|
} else if (eql(case, "bus")) {
|
|
busTest(boot_information);
|
|
} else if (eql(case, "device-manager")) {
|
|
deviceManagerTest(boot_information);
|
|
} else if (eql(case, "poweroff")) {
|
|
powerTest(.off);
|
|
} else if (eql(case, "reboot")) {
|
|
powerTest(.reboot);
|
|
} else {
|
|
log("DANOS-TEST-RESULT: FAIL (unknown case '{s}')\n", .{case});
|
|
}
|
|
}
|
|
|
|
fn platformHal() platform.Hal {
|
|
return .{
|
|
.mapMmio = architecture.mapMmio,
|
|
.pioRead = architecture.pioRead,
|
|
.pioWrite = architecture.pioWrite,
|
|
};
|
|
}
|
|
|
|
/// Drive an ACPI power transition. On success the machine powers off or resets,
|
|
/// so QEMU exits — the harness observes the process exit. If control returns, the
|
|
/// transition failed and we emit a FAIL result.
|
|
fn powerTest(comptime action: enum { off, reboot }) void {
|
|
const name = if (action == .off) "poweroff" else "reboot";
|
|
log("DANOS-TEST-BEGIN: {s}\n", .{name});
|
|
const hal = platformHal();
|
|
log("DANOS-POWER: attempting {s}\n", .{name});
|
|
switch (action) {
|
|
.off => platform.shutdown(hal),
|
|
.reboot => platform.reboot(hal),
|
|
}
|
|
check("power transition took effect", false);
|
|
result();
|
|
}
|
|
|
|
const BootInformation = boot_handoff.BootInformation;
|
|
|
|
fn eql(a: []const u8, b: []const u8) bool {
|
|
return std.mem.eql(u8, a, b);
|
|
}
|
|
|
|
/// Non-destructive checks of the memory map and frame allocator.
|
|
fn smoke(boot_information: *const BootInformation) void {
|
|
log("DANOS-TEST-BEGIN: smoke\n", .{});
|
|
|
|
// The memory map has some usable RAM.
|
|
const mm = boot_information.memory_map;
|
|
const regions = @as([*]const boot_handoff.MemoryRegion, @ptrFromInt(boot_handoff.physicalToVirtual(mm.regions)))[0..mm.len];
|
|
var usable: u64 = 0;
|
|
for (regions) |r| {
|
|
if (r.kind == .usable) usable += r.pages;
|
|
}
|
|
check("memory map reports usable RAM", usable > 0);
|
|
|
|
// The frame allocator hands out distinct, page-aligned frames.
|
|
const a = pmm.alloc();
|
|
const b = pmm.alloc();
|
|
check("alloc returns a frame", a != null);
|
|
check("alloc returns distinct frames", a != null and b != null and a.? != b.?);
|
|
check("frames are page-aligned", (a orelse 1) % abi.page_size == 0);
|
|
|
|
// Freeing restores the count.
|
|
const before = pmm.stats().free_frames;
|
|
if (a) |p| pmm.free(p);
|
|
if (b) |p| pmm.free(p);
|
|
check("free returns frames to the pool", pmm.stats().free_frames == before + 2);
|
|
|
|
// Paging is active on our own tables (the root is non-zero and page-aligned).
|
|
const root = architecture.activePageTable();
|
|
check("paging active (page-table root set)", root != 0 and root % abi.page_size == 0);
|
|
|
|
result();
|
|
}
|
|
|
|
/// Verify device interrupts fire and return: the timer tick counter must advance
|
|
/// on its own. Interrupts are already enabled by kmain before tests run.
|
|
fn timer() void {
|
|
log("DANOS-TEST-BEGIN: timer\n", .{});
|
|
const start = architecture.ticks();
|
|
// Busy-wait for the counter to advance. architecture.ticks() is a volatile load, so
|
|
// the compiler re-reads it each iteration and sees the interrupt's update.
|
|
// The cap is only a safety net; the harness timeout is the real backstop.
|
|
var spins: u64 = 0;
|
|
while (architecture.ticks() == start and spins < 5_000_000_000) spins +%= 1;
|
|
check("timer interrupts advance the tick count", architecture.ticks() > start);
|
|
|
|
result();
|
|
}
|
|
|
|
/// Verify device discovery populated the platform facts the rest of the kernel
|
|
/// depends on — the results ACPI parsing stashed in globals at boot. These are
|
|
/// stable for the QEMU q35 + OVMF machine the harness runs, and span the tables:
|
|
/// MADT (LAPIC base, CPU count), FADT (PM/reset registers), and the AML parse
|
|
/// (the sleep type, plus the integrity check that every byte was consumed).
|
|
fn discoveryTest() void {
|
|
log("DANOS-TEST-BEGIN: discovery\n", .{});
|
|
const pinfo = platform.platformInformation();
|
|
const pw = platform.powerInformation();
|
|
const am = platform.amlStats();
|
|
|
|
check("LAPIC base discovered (MADT)", pinfo.lapic_base == 0xFEE00000);
|
|
check("ACPI PM timer found (FADT)", pinfo.pm_timer.present());
|
|
check("PM1a control register found (FADT)", pw.pm1a_cnt.present());
|
|
check("reset register supported (FADT)", pw.reset_supported);
|
|
check("S5 sleep type found (AML)", pw.s5 != null);
|
|
check("AML parsed completely (consumed == total)", am.total > 0 and am.consumed == am.total);
|
|
check("at least one CPU enumerated (MADT)", platform.cpus().len >= 1);
|
|
|
|
result();
|
|
}
|
|
|
|
/// Audit the W^X invariant across the memory classes: kernel code must be
|
|
/// executable, everything else must not be. `architecture.pageExecutable` reads the leaf
|
|
/// page-table entry's NX bit, so this guards the permission overlay in paging.zig —
|
|
/// a broader check than `fault-nx`, which only exercises one data page.
|
|
fn wxTest() void {
|
|
log("DANOS-TEST-BEGIN: wx\n", .{});
|
|
|
|
check("kernel code is executable (R+X)", architecture.pageExecutable(@intFromPtr(&wxTest)));
|
|
|
|
const ro = "danos-wx-probe"; // string literal -> .rodata
|
|
check("rodata is non-executable (NX)", !architecture.pageExecutable(@intFromPtr(ro.ptr)));
|
|
|
|
check("kernel data is non-executable (NX)", !architecture.pageExecutable(@intFromPtr(&passed)));
|
|
|
|
if (heap.allocator().alloc(u8, 64) catch null) |h| {
|
|
check("heap is non-executable (NX)", !architecture.pageExecutable(@intFromPtr(h.ptr)));
|
|
heap.allocator().free(h);
|
|
}
|
|
|
|
var local: u64 = 0;
|
|
_ = &local;
|
|
check("stack is non-executable (NX)", !architecture.pageExecutable(@intFromPtr(&local)));
|
|
|
|
result();
|
|
}
|
|
|
|
/// Verify the on-demand VMM: map a fresh frame at an unused virtual address, and
|
|
/// check it's writable and reads back.
|
|
fn vmm() void {
|
|
log("DANOS-TEST-BEGIN: vmm\n", .{});
|
|
const frame = pmm.alloc();
|
|
check("frame available to map", frame != null);
|
|
if (frame) |physical| {
|
|
var virtual: u64 = 0x0000_4000_0000_0000; // canonical, well clear of everything mapped
|
|
architecture.mapPage(virtual, physical, true);
|
|
const p: *volatile u64 = @ptrFromInt(virtual);
|
|
p.* = 0xdead_c0de_cafe_babe;
|
|
check("mapped page is writable and reads back", p.* == 0xdead_c0de_cafe_babe);
|
|
architecture.unmapPage(virtual);
|
|
pmm.free(physical);
|
|
virtual += 0;
|
|
}
|
|
result();
|
|
}
|
|
|
|
/// Exercise the kernel heap: basic alloc/write/free, reuse, growth beyond the
|
|
/// initial region, and a std container backed by it.
|
|
fn heapTest() void {
|
|
log("DANOS-TEST-BEGIN: heap\n", .{});
|
|
const a = heap.allocator();
|
|
|
|
// Allocate, write a pattern, read it back, free.
|
|
const buffer = a.alloc(u8, 4096) catch null;
|
|
check("alloc 4096 bytes", buffer != null);
|
|
if (buffer) |b| {
|
|
@memset(b, 0xAB);
|
|
check("heap memory is writable and reads back", b[0] == 0xAB and b[4095] == 0xAB);
|
|
a.free(b);
|
|
}
|
|
|
|
// Freeing then re-allocating the same size should reuse the block.
|
|
const p1 = a.alloc(u64, 8) catch null;
|
|
const address1 = if (p1) |p| @intFromPtr(p.ptr) else 0;
|
|
if (p1) |p| a.free(p);
|
|
const p2 = a.alloc(u64, 8) catch null;
|
|
const address2 = if (p2) |p| @intFromPtr(p.ptr) else 0;
|
|
check("freed block is reused", address1 != 0 and address1 == address2);
|
|
if (p2) |p| a.free(p);
|
|
|
|
// Force growth past the initial page and check every block is usable.
|
|
var blocks: [64]?[]u8 = .{null} ** 64;
|
|
var ok = true;
|
|
for (&blocks, 0..) |*slot, i| {
|
|
const b = a.alloc(u8, 4096) catch null;
|
|
slot.* = b;
|
|
if (b) |bb| @memset(bb, @intCast(i & 0xff)) else {
|
|
ok = false;
|
|
}
|
|
}
|
|
for (blocks, 0..) |slot, i| {
|
|
if (slot) |bb| {
|
|
if (bb[0] != @as(u8, @intCast(i & 0xff)) or bb[4095] != @as(u8, @intCast(i & 0xff))) ok = false;
|
|
}
|
|
}
|
|
check("many allocations (heap growth) stay valid", ok);
|
|
for (blocks) |slot| {
|
|
if (slot) |bb| a.free(bb);
|
|
}
|
|
|
|
// A std container backed by the kernel heap.
|
|
var list: std.ArrayList(u32) = .empty;
|
|
var sum: u64 = 0;
|
|
var expected: u64 = 0;
|
|
var i: u32 = 0;
|
|
var list_ok = true;
|
|
while (i < 1000) : (i += 1) {
|
|
list.append(a, i) catch {
|
|
list_ok = false;
|
|
};
|
|
expected += i;
|
|
}
|
|
for (list.items) |v| sum += v;
|
|
list.deinit(a);
|
|
check("std.ArrayList on the kernel heap", list_ok and sum == expected);
|
|
|
|
result();
|
|
}
|
|
|
|
/// Verify the calibrated clocks: sane measured frequencies, monotonic uptime that
|
|
/// advances with real ticks, and — the point of the TSC clock — nanosecond
|
|
/// resolution far finer than the 1 ms tick, with the unit functions consistent.
|
|
fn clock() void {
|
|
log("DANOS-TEST-BEGIN: clock\n", .{});
|
|
|
|
const timer_clock = architecture.timerClockHz();
|
|
check("timer clock frequency measured", timer_clock > 1_000_000 and timer_clock < 100_000_000_000);
|
|
const clock_hz = architecture.clockHz();
|
|
check("monotonic clock frequency measured", clock_hz > 100_000_000 and clock_hz < 100_000_000_000);
|
|
|
|
// Uptime advances over ~5 real ticks (1000 Hz => 1 tick == 1 ms).
|
|
const start_ticks = architecture.ticks();
|
|
const start_ms = architecture.millis();
|
|
var spins: u64 = 0;
|
|
while (architecture.ticks() < start_ticks + 5 and spins < 5_000_000_000) spins +%= 1;
|
|
const elapsed_ms = architecture.millis() - start_ms;
|
|
check("uptime advances with ticks", elapsed_ms >= 5 and elapsed_ms < 100);
|
|
|
|
// Sub-millisecond resolution: spin until nanos() first advances, then confirm
|
|
// that first step happened within a millisecond — so nanos() resolves finer
|
|
// than the 1 ms tick (a tick clock's smallest step *is* 1 ms). Spinning to the
|
|
// first change is robust to QEMU's coarse TSC update granularity.
|
|
const n1 = architecture.nanos();
|
|
var s2: u64 = 0;
|
|
while (architecture.nanos() == n1 and s2 < 10_000_000) s2 +%= 1;
|
|
const n2 = architecture.nanos();
|
|
check("nanos() has sub-millisecond resolution", n2 > n1 and (n2 - n1) < 1_000_000);
|
|
|
|
// The unit functions agree (within rounding).
|
|
const ns = architecture.nanos();
|
|
check("nanos/micros/millis are consistent", diffWithin(architecture.micros(), ns / 1000, 1000) and diffWithin(architecture.millis(), ns / 1_000_000, 2));
|
|
|
|
result();
|
|
}
|
|
|
|
fn diffWithin(a: u64, b: u64, tol: u64) bool {
|
|
return if (a > b) a - b <= tol else b - a <= tol;
|
|
}
|
|
|
|
// --- scheduler tests ------------------------------------------------------
|
|
|
|
var counters = [_]u64{0} ** 3;
|
|
|
|
fn spin0() void {
|
|
const p: *volatile u64 = &counters[0];
|
|
while (true) p.* = p.* +% 1;
|
|
}
|
|
fn spin1() void {
|
|
const p: *volatile u64 = &counters[1];
|
|
while (true) p.* = p.* +% 1;
|
|
}
|
|
fn spin2() void {
|
|
const p: *volatile u64 = &counters[2];
|
|
while (true) p.* = p.* +% 1;
|
|
}
|
|
|
|
/// Preemption: spawn three tasks that busy-loop *without* yielding. If they all
|
|
/// make progress, the timer must be preempting between them (and the context
|
|
/// switch works) — because nothing yields voluntarily.
|
|
fn schedulerTest() void {
|
|
log("DANOS-TEST-BEGIN: sched\n", .{});
|
|
counters = .{ 0, 0, 0 };
|
|
scheduler.spawn(spin0, 4);
|
|
scheduler.spawn(spin1, 4);
|
|
scheduler.spawn(spin2, 4);
|
|
|
|
const c0: *volatile u64 = &counters[0];
|
|
const c1: *volatile u64 = &counters[1];
|
|
const c2: *volatile u64 = &counters[2];
|
|
var spins: u64 = 0;
|
|
while ((c0.* == 0 or c1.* == 0 or c2.* == 0) and spins < 5_000_000_000) spins +%= 1;
|
|
|
|
check("all three non-yielding tasks made progress (preemption)", c0.* > 0 and c1.* > 0 and c2.* > 0);
|
|
result();
|
|
}
|
|
|
|
var run_order = [_]u8{0} ** 4;
|
|
var run_n: usize = 0;
|
|
|
|
fn recordExit(priority: u8) void {
|
|
run_order[run_n] = priority;
|
|
run_n += 1;
|
|
scheduler.exit();
|
|
}
|
|
fn taskHigh() void {
|
|
recordExit(6);
|
|
}
|
|
fn taskMid() void {
|
|
recordExit(4);
|
|
}
|
|
fn taskLow() void {
|
|
recordExit(2);
|
|
}
|
|
|
|
/// Fixed priority: with preemption off (deterministic), spawn tasks at three
|
|
/// priorities and let them run cooperatively. They must run highest-first.
|
|
fn priorityTest() void {
|
|
log("DANOS-TEST-BEGIN: priority\n", .{});
|
|
scheduler.setPreemption(false);
|
|
scheduler.setPriority(1); // above the idle task (0), below the workers — runs last
|
|
run_n = 0;
|
|
|
|
scheduler.spawn(taskLow, 2);
|
|
scheduler.spawn(taskMid, 4);
|
|
scheduler.spawn(taskHigh, 6);
|
|
|
|
while (run_n < 3) scheduler.yield(); // regain control only once the workers are done
|
|
|
|
check("tasks ran highest-priority first", run_order[0] == 6 and run_order[1] == 4 and run_order[2] == 2);
|
|
|
|
scheduler.setPriority(4);
|
|
scheduler.setPreemption(true);
|
|
result();
|
|
}
|
|
|
|
var event_wait_queue: scheduler.WaitQueue = .{};
|
|
var event_stage: u32 = 0;
|
|
|
|
fn eventWaiter() void {
|
|
event_stage = 1; // reached the wait
|
|
scheduler.wait(&event_wait_queue); // block until woken
|
|
event_stage = 3; // woken and resumed
|
|
scheduler.exit();
|
|
}
|
|
|
|
/// Event-based blocking: a task blocks on a wait queue and is woken. The waiter is
|
|
/// higher priority, so waking it preempts us and it runs to completion at once.
|
|
fn eventTest() void {
|
|
log("DANOS-TEST-BEGIN: event\n", .{});
|
|
event_stage = 0;
|
|
scheduler.spawn(eventWaiter, 6); // higher priority than this task (4)
|
|
|
|
var spins: u64 = 0;
|
|
while (event_stage != 1 and spins < 1_000_000_000) : (spins += 1) scheduler.yield();
|
|
check("waiter reached the wait and blocked", event_stage == 1);
|
|
|
|
scheduler.wake(&event_wait_queue);
|
|
check("wake resumed the blocked waiter (preempting)", event_stage == 3);
|
|
result();
|
|
}
|
|
|
|
var channel: ipc.Channel(u64, 4) = .{};
|
|
var receive_sum: u64 = 0;
|
|
var receive_count: u64 = 0;
|
|
|
|
fn producer() void {
|
|
var i: u64 = 1;
|
|
while (i <= 100) : (i += 1) channel.send(i);
|
|
scheduler.exit();
|
|
}
|
|
fn consumer() void {
|
|
var n: u64 = 0;
|
|
while (n < 100) : (n += 1) {
|
|
receive_sum += channel.receive();
|
|
receive_count += 1;
|
|
}
|
|
scheduler.exit();
|
|
}
|
|
|
|
/// IPC: a producer and consumer pass 100 messages through a 4-slot channel. The
|
|
/// small buffer forces the channel full and empty repeatedly, exercising both the
|
|
/// blocking-send and blocking-receive paths. The messages must arrive intact.
|
|
fn ipcTest() void {
|
|
log("DANOS-TEST-BEGIN: ipc\n", .{});
|
|
channel = .{};
|
|
receive_sum = 0;
|
|
receive_count = 0;
|
|
scheduler.spawn(consumer, 5); // above this task (4) so they run and we observe after
|
|
scheduler.spawn(producer, 5);
|
|
|
|
var spins: u64 = 0;
|
|
while (receive_count < 100 and spins < 2_000_000_000) : (spins += 1) scheduler.yield();
|
|
|
|
check("all 100 messages received", receive_count == 100);
|
|
check("messages arrived intact (sum 1..100 == 5050)", receive_sum == 5050);
|
|
result();
|
|
}
|
|
|
|
/// Blocking: sleep(50) should block this task for about 50 ms (measured on the
|
|
/// calibrated clock) — not busy-wait — while the idle task runs.
|
|
fn sleepTest() void {
|
|
log("DANOS-TEST-BEGIN: sleep\n", .{});
|
|
const t0 = architecture.millis();
|
|
scheduler.sleep(50);
|
|
const elapsed = architecture.millis() - t0;
|
|
check("sleep(50) blocked for ~50 ms", elapsed >= 50 and elapsed <= 70);
|
|
result();
|
|
}
|
|
|
|
// --- SMP parallelism ------------------------------------------------------
|
|
|
|
var seen_core = [_]bool{false} ** 8;
|
|
var smp_running: bool = true;
|
|
|
|
/// A worker that, while running, records which core it's executing on. Spread across
|
|
/// spawned workers and idle APs, these should land on more than one core.
|
|
fn smpWorker() void {
|
|
const p: *volatile bool = &smp_running;
|
|
while (p.*) {
|
|
const c = scheduler.currentCpuIndex();
|
|
if (c < seen_core.len) seen_core[c] = true;
|
|
}
|
|
scheduler.exit();
|
|
}
|
|
|
|
/// Prove tasks run **in parallel** on multiple cores (not just interleaved on one).
|
|
/// Spawn several CPU-bound workers; each stamps the core it runs on into `seen_core`.
|
|
/// With the application processors online, more than one core should show up — which
|
|
/// can only happen if work is genuinely running at the same time on different cores.
|
|
/// (Run with QEMU `-smp N`; on a single core this would see just one and fail.)
|
|
fn smpTest() void {
|
|
log("DANOS-TEST-BEGIN: smp\n", .{});
|
|
seen_core = .{false} ** 8;
|
|
smp_running = true;
|
|
|
|
var i: usize = 0;
|
|
while (i < 4) : (i += 1) scheduler.spawn(smpWorker, 4);
|
|
|
|
// Let the workers run across cores for a stretch of real time.
|
|
var spins: u64 = 0;
|
|
while (spins < 2_000_000_000) spins +%= 1;
|
|
smp_running = false;
|
|
|
|
var cores_seen: u32 = 0;
|
|
for (seen_core) |s| {
|
|
if (s) cores_seen += 1;
|
|
}
|
|
log("DANOS-SMP: workers ran on {d} distinct core(s)\n", .{cores_seen});
|
|
check("tasks ran on multiple cores in parallel", cores_seen >= 2);
|
|
|
|
// Bring-up is done, so the trampoline frame must be inert: zeroed (no stale code)
|
|
// and non-executable (W^X restored). It's armed only while a core is climbing.
|
|
const tramp = architecture.trampolinePage();
|
|
check("trampoline frame reserved", tramp != 0);
|
|
if (tramp != 0) {
|
|
const bytes: [*]const u8 = @ptrFromInt(boot_handoff.physicalToVirtual(tramp));
|
|
var zeroed = true;
|
|
for (0..4096) |b| {
|
|
if (bytes[b] != 0) zeroed = false;
|
|
}
|
|
check("trampoline page zeroed when dormant", zeroed);
|
|
check("trampoline page non-executable when dormant", !architecture.pageExecutable(tramp));
|
|
}
|
|
result();
|
|
}
|
|
|
|
// --- affinity: a pinned task never migrates -------------------------------
|
|
|
|
var affinity_cores = [_]bool{false} ** 8;
|
|
var affinity_running: bool = true;
|
|
|
|
fn affinityWorker() void {
|
|
const p: *volatile bool = &affinity_running;
|
|
while (p.*) {
|
|
const c = scheduler.currentCpuIndex();
|
|
if (c < affinity_cores.len) affinity_cores[c] = true;
|
|
}
|
|
scheduler.exit();
|
|
}
|
|
|
|
/// A task pinned to a core must run **only** on that core. Pin a busy worker to
|
|
/// core 1 and let it run through many preemptions; it must have stamped core 1 and no
|
|
/// other. An *unpinned* task scatters across cores (that's what the smp test shows),
|
|
/// so a broken pin fails this deterministically — over this many time slices a
|
|
/// free-floating task will land on some other core.
|
|
fn affinityTest() void {
|
|
log("DANOS-TEST-BEGIN: affinity\n", .{});
|
|
affinity_cores = .{false} ** 8;
|
|
affinity_running = true;
|
|
|
|
if (!scheduler.spawnOn(affinityWorker, 4, 1)) {
|
|
check("worker pinned to core 1 (run with -smp)", false);
|
|
result();
|
|
return;
|
|
}
|
|
|
|
var spins: u64 = 0;
|
|
while (spins < 3_000_000_000) spins +%= 1; // many time slices across the cores
|
|
affinity_running = false;
|
|
var settle: u64 = 0;
|
|
while (settle < 200_000_000) settle +%= 1; // let the worker see the flag and exit
|
|
|
|
var others: u32 = 0;
|
|
for (affinity_cores, 0..) |seen, c| {
|
|
if (seen and c != 1) others += 1;
|
|
}
|
|
log("DANOS-AFFINITY: pinned worker touched core 1={}, other cores={d}\n", .{ affinity_cores[1], others });
|
|
check("pinned task ran on its core (1)", affinity_cores[1]);
|
|
check("pinned task never migrated to another core", others == 0);
|
|
result();
|
|
}
|
|
|
|
// --- SMP stress: hammer the big kernel lock across cores ------------------
|
|
|
|
const stress_pairs = 4; // producer/consumer pairs (8 tasks; fits the 16-task pool)
|
|
const stress_msgs = 100_000; // messages per pair
|
|
const stress_cap = 4; // small channel -> constant block/wake, more lock churn
|
|
|
|
var stress_chan = [_]ipc.Channel(u64, stress_cap){.{}} ** stress_pairs;
|
|
var stress_receive = [_]u64{0} ** stress_pairs; // messages received per pair
|
|
var stress_order_ok = [_]bool{true} ** stress_pairs; // FIFO order held per pair
|
|
var stress_cores = [_]bool{false} ** 8; // cores that ran a consumer
|
|
var stress_prod_claim: usize = 0;
|
|
var stress_cons_claim: usize = 0;
|
|
|
|
fn stressProducer() void {
|
|
// Claim a unique pair index (atomic: producers start on different cores).
|
|
const index = @atomicRmw(usize, &stress_prod_claim, .Add, 1, .monotonic);
|
|
var v: u64 = 1;
|
|
while (v <= stress_msgs) : (v += 1) stress_chan[index].send(v);
|
|
scheduler.exit();
|
|
}
|
|
|
|
fn stressConsumer() void {
|
|
const index = @atomicRmw(usize, &stress_cons_claim, .Add, 1, .monotonic);
|
|
var expected: u64 = 1;
|
|
while (expected <= stress_msgs) : (expected += 1) {
|
|
const got = stress_chan[index].receive();
|
|
if (got != expected) stress_order_ok[index] = false; // lost/reordered => lock broke
|
|
const c = scheduler.currentCpuIndex();
|
|
if (c < stress_cores.len) stress_cores[c] = true;
|
|
stress_receive[index] = expected;
|
|
}
|
|
scheduler.exit();
|
|
}
|
|
|
|
/// Stress the big kernel lock under sustained cross-core contention. Each pair drives
|
|
/// `stress_msgs` sequenced messages through a 4-slot channel — every send and receive
|
|
/// takes the lock, and the small buffer forces constant block/wake (so the scheduler
|
|
/// churns too). A single-producer/single-consumer channel must deliver in strict FIFO
|
|
/// order; if the lock let two cores into a critical section at once, the ring buffer
|
|
/// corrupts and the consumer sees a wrong or out-of-order value (or the run hangs /
|
|
/// faults). Passing means ~320k lock acquisitions across the cores stayed consistent.
|
|
fn stressTest() void {
|
|
log("DANOS-TEST-BEGIN: smp-stress\n", .{});
|
|
stress_chan = [_]ipc.Channel(u64, stress_cap){.{}} ** stress_pairs;
|
|
stress_receive = [_]u64{0} ** stress_pairs;
|
|
stress_order_ok = [_]bool{true} ** stress_pairs;
|
|
stress_cores = [_]bool{false} ** 8;
|
|
stress_prod_claim = 0;
|
|
stress_cons_claim = 0;
|
|
|
|
var i: usize = 0;
|
|
while (i < stress_pairs) : (i += 1) scheduler.spawn(stressConsumer, 4);
|
|
i = 0;
|
|
while (i < stress_pairs) : (i += 1) scheduler.spawn(stressProducer, 4);
|
|
|
|
// Drop below the workers so they get the cores; wake periodically to check for
|
|
// completion. A broken lock instead hangs here (harness timeout) or faults.
|
|
scheduler.setPriority(1);
|
|
var spins: u64 = 0;
|
|
while (spins < 40_000_000_000) : (spins += 1) {
|
|
var done = true;
|
|
for (stress_receive) |n| {
|
|
if (n < stress_msgs) done = false;
|
|
}
|
|
if (done) break;
|
|
}
|
|
scheduler.setPriority(4);
|
|
|
|
var total: u64 = 0;
|
|
for (stress_receive) |n| total += n;
|
|
var order_ok = true;
|
|
for (stress_order_ok) |ok| {
|
|
if (!ok) order_ok = false;
|
|
}
|
|
var cores: u32 = 0;
|
|
for (stress_cores) |s| {
|
|
if (s) cores += 1;
|
|
}
|
|
|
|
log("DANOS-STRESS: {d}/{d} pairs complete on {d} cores\n", .{ total, @as(u64, stress_pairs) * stress_msgs, cores });
|
|
check("every message delivered", total == @as(u64, stress_pairs) * stress_msgs);
|
|
check("strict FIFO order held (no lock corruption)", order_ok);
|
|
check("contention was genuinely cross-core", cores >= 2);
|
|
result();
|
|
}
|
|
|
|
/// Retry: `main` forced the first AP wake attempt to fail (architecture.testFailNextWakes),
|
|
/// so a core missed its first INIT-SIPI-SIPI. The boot retry must have brought it back
|
|
/// anyway — every enumerated core should be online. If retry were broken, that core
|
|
/// would be parked and the count would fall short.
|
|
fn smpRetryTest() void {
|
|
log("DANOS-TEST-BEGIN: smp-retry\n", .{});
|
|
const total = platform.cpus().len;
|
|
const online = scheduler.onlineCount();
|
|
log("DANOS-RETRY: {d}/{d} cores online after a forced first-wake failure\n", .{ online, total });
|
|
check("multiple cores enumerated (run with -smp)", total >= 2);
|
|
check("retry brought every core online despite a failed first wake", online == total);
|
|
result();
|
|
}
|
|
|
|
// --- ring 3 (user mode) -----------------------------------------------------
|
|
|
|
/// The mmap/munmap grant path: create a fresh address space, hand out pages into
|
|
/// its mmap arena the way the `mmap` system_call does, prove they are real (write and
|
|
/// read them back through the physmap), then release them via the `munmap` path
|
|
/// (translate -> unmap -> free) and tear the address space down. The frame count
|
|
/// must return exactly to where it started — a leak or a double-free would show
|
|
/// as drift. Exercises the new `translate`/`unmapUserPageInto` primitives that
|
|
/// back munmap, without needing a user binary that calls the syscalls.
|
|
fn userMemTest() void {
|
|
log("DANOS-TEST-BEGIN: usermem\n", .{});
|
|
const base_free = pmm.stats().free_frames;
|
|
|
|
const aspace = architecture.createAddressSpace() orelse {
|
|
check("created a fresh address space", false);
|
|
result();
|
|
return;
|
|
};
|
|
check("created a fresh address space", aspace != 0);
|
|
|
|
// Grant three pages into the arena, mapped RW + NX (the mmap contract).
|
|
const npages = 3;
|
|
const arena = process.heap_arena_base;
|
|
var frames: [npages]u64 = undefined;
|
|
var mapped: usize = 0;
|
|
while (mapped < npages) : (mapped += 1) {
|
|
frames[mapped] = pmm.alloc() orelse break;
|
|
architecture.mapUserPageInto(aspace, arena + mapped * abi.page_size, frames[mapped], true, false);
|
|
}
|
|
check("granted three user pages", mapped == npages);
|
|
|
|
// Each page resolves back to the frame we mapped, and is writable RAM.
|
|
var translate_ok = true;
|
|
var rw_ok = true;
|
|
for (0..npages) |i| {
|
|
const va = arena + i * abi.page_size;
|
|
const physical = architecture.translate(aspace, va) orelse {
|
|
translate_ok = false;
|
|
continue;
|
|
};
|
|
if (physical != frames[i]) translate_ok = false;
|
|
const p: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
|
p[0] = 0xA5;
|
|
if (p[0] != 0xA5) rw_ok = false;
|
|
}
|
|
check("translate resolves each grant to its frame", translate_ok);
|
|
check("granted pages are writable RAM", rw_ok);
|
|
|
|
// Release them the way munmap does, then tear down the address space.
|
|
for (0..npages) |i| {
|
|
const va = arena + i * abi.page_size;
|
|
if (architecture.translate(aspace, va)) |physical| {
|
|
architecture.unmapUserPageInto(aspace, va);
|
|
pmm.free(physical);
|
|
}
|
|
}
|
|
check("munmap unmapped every grant", architecture.translate(aspace, arena) == null);
|
|
architecture.destroyAddressSpace(aspace);
|
|
|
|
check("no frames leaked (free count restored)", pmm.stats().free_frames == base_free);
|
|
result();
|
|
}
|
|
|
|
// --- synchronous IPC --------------------------------------------------------
|
|
|
|
var ipc_endpoint: *ipcsync.Endpoint = undefined;
|
|
var ipc_replies_ok: bool = false;
|
|
var ipc_done: bool = false;
|
|
|
|
/// Echo-increment server: reply to each request with request+1, forever.
|
|
fn ipcServer() void {
|
|
var reply_buffer: [8]u8 = undefined;
|
|
var reply_len: u64 = 0;
|
|
var badge: u64 = 0;
|
|
while (true) {
|
|
var receive: [8]u8 = undefined;
|
|
const n = ipcsync.replyWait(ipc_endpoint, @intFromPtr(&reply_buffer), reply_len, @intFromPtr(&receive), receive.len, &badge);
|
|
if (n < 0) scheduler.exit();
|
|
const v = std.mem.readInt(u64, receive[0..8], .little);
|
|
std.mem.writeInt(u64, reply_buffer[0..8], v + 1, .little);
|
|
reply_len = 8;
|
|
}
|
|
}
|
|
|
|
/// Client: make 100 synchronous calls, checking every reply is request+1.
|
|
fn ipcClient() void {
|
|
var ok = true;
|
|
var i: u64 = 0;
|
|
while (i < 100) : (i += 1) {
|
|
var message: [8]u8 = undefined;
|
|
std.mem.writeInt(u64, message[0..8], i, .little);
|
|
var reply: [8]u8 = undefined;
|
|
const n = ipcsync.call(ipc_endpoint, @intFromPtr(&message), 8, @intFromPtr(&reply), reply.len);
|
|
if (n != 8 or std.mem.readInt(u64, reply[0..8], .little) != i + 1) ok = false;
|
|
}
|
|
ipc_replies_ok = ok;
|
|
ipc_done = true;
|
|
scheduler.exit();
|
|
}
|
|
|
|
/// Synchronous IPC: a client and a server (two kernel tasks) ping-pong 100 calls
|
|
/// through one Endpoint. Each round exercises the full rendezvous — the client
|
|
/// blocks in `call`, the server blocks in `replyWait`, the reply is routed back to
|
|
/// the exact caller, and `copyAcross` moves the bytes — so 100 correct replies
|
|
/// prove the block/wake and reply-routing paths. (Kernel tasks, so no user ELF.)
|
|
fn ipcCallTest() void {
|
|
log("DANOS-TEST-BEGIN: ipc-call\n", .{});
|
|
ipc_endpoint = ipcsync.createEndpoint().?;
|
|
ipc_replies_ok = false;
|
|
ipc_done = false;
|
|
scheduler.spawn(ipcServer, 5); // above this task, so the workers run
|
|
scheduler.spawn(ipcClient, 5);
|
|
|
|
const done: *volatile bool = &ipc_done;
|
|
var spins: u64 = 0;
|
|
while (!done.* and spins < 100_000_000) : (spins += 1) scheduler.yield();
|
|
|
|
check("client completed 100 synchronous calls", ipc_done);
|
|
check("every reply was request+1 (rendezvous + reply routing intact)", ipc_replies_ok);
|
|
result();
|
|
}
|
|
|
|
var proc_worker_run: bool = true;
|
|
var proc_worker_ran: bool = false;
|
|
|
|
/// A kernel task that spins while a process runs, to prove the two coexist under
|
|
/// preemption (a process on its own CR3 does not stall kernel work).
|
|
fn procWorker() void {
|
|
const running: *volatile bool = &proc_worker_run;
|
|
const ran: *volatile bool = &proc_worker_ran;
|
|
while (running.*) ran.* = true;
|
|
scheduler.exit();
|
|
}
|
|
|
|
/// Real processes: load /system/services/init as TWO scheduled ring-3 processes, each with
|
|
/// its own address space at the same virtual addresses, running concurrently
|
|
/// with a kernel task. Both must make heartbeat syscalls from CPL 3 — which can
|
|
/// only happen if each runs on its own page tables (CR3 switched correctly per
|
|
/// process) and preemption interleaves them with the kernel worker. This is the
|
|
/// strongest cheap proof of address-space isolation.
|
|
fn processTest(boot_information: *const BootInformation) void {
|
|
log("DANOS-TEST-BEGIN: process\n", .{});
|
|
check("bootloader handed over /system/services/init", boot_information.init_len != 0);
|
|
if (boot_information.init_len == 0) {
|
|
result();
|
|
return;
|
|
}
|
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
|
|
|
process.write_count = 0;
|
|
process.write_from_user = false;
|
|
proc_worker_run = true;
|
|
proc_worker_ran = false;
|
|
scheduler.spawn(procWorker, 4); // kernel task at the processes' priority
|
|
|
|
var spawned: u32 = 0;
|
|
if (process.spawnProcess(image, 4)) spawned += 1 else |_| {}
|
|
if (process.spawnProcess(image, 4)) spawned += 1 else |_| {}
|
|
|
|
// Wait (real time) for several heartbeats across the two processes. Each
|
|
// process sleeps ~1 s between beats, so a few seconds yields several.
|
|
scheduler.setPriority(1); // drop below the workers so they get the cores
|
|
const deadline = architecture.millis() + 8000;
|
|
while (process.write_count < 4 and architecture.millis() < deadline) scheduler.yield();
|
|
scheduler.setPriority(4);
|
|
proc_worker_run = false;
|
|
|
|
log("DANOS-PROC: spawned {d} processes, {d} heartbeats\n", .{ spawned, process.write_count });
|
|
check("both init processes spawned on their own address spaces", spawned == 2);
|
|
check("processes made repeated heartbeat syscalls (>=4)", process.write_count >= 4);
|
|
check("heartbeats came from user mode (CPL 3)", process.write_from_user);
|
|
check("a kernel task coexisted with the processes (preemption)", proc_worker_ran);
|
|
result();
|
|
}
|
|
|
|
/// Isolation: a ring-3 read of a kernel-only page (the LAPIC page — present,
|
|
/// supervisor) must page-fault with error code 0x5 (present | user) at the user
|
|
/// RIP. The fault report is the pass signal (matched by the harness); if the
|
|
/// read is somehow allowed the blob spins and the harness times out.
|
|
fn userPfTest() void {
|
|
log("DANOS-TEST-BEGIN: user-pf\n", .{});
|
|
scheduler.setPreemption(false);
|
|
_ = process.run(process.pfBlob()) catch {};
|
|
log("DANOS-TEST-RESULT: FAIL (user read of kernel memory did not fault)\n", .{});
|
|
}
|
|
|
|
/// The full PID-1 path: the bootloader read /system/services/init off the boot volume and
|
|
/// handed it over; load it as a user ELF and spawn it as a real ring-3 process
|
|
/// — the same call the normal boot path makes — then confirm it beats. init
|
|
/// heartbeats forever, so this proves it reaches ring 3, makes repeated syscalls
|
|
/// (write + sleep), and stays alive rather than exiting.
|
|
fn initTest(boot_information: *const BootInformation) void {
|
|
log("DANOS-TEST-BEGIN: init\n", .{});
|
|
check("bootloader handed over /system/services/init", boot_information.init_len != 0);
|
|
if (boot_information.init_len == 0) {
|
|
result();
|
|
return;
|
|
}
|
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
|
process.write_count = 0;
|
|
const spawned = if (process.spawnProcess(image, 4)) true else |err| blk: {
|
|
log("DANOS-INIT-ERR: {s}\n", .{@errorName(err)});
|
|
break :blk false;
|
|
};
|
|
check("init loaded and spawned as a process", spawned);
|
|
|
|
// Wait (real time) for at least two heartbeats — proving it runs, writes,
|
|
// and sleeps repeatedly (init sleeps ~1 s between beats).
|
|
scheduler.setPriority(1);
|
|
const deadline = architecture.millis() + 8000;
|
|
while (process.write_count < 2 and architecture.millis() < deadline) scheduler.yield();
|
|
scheduler.setPriority(4);
|
|
|
|
const prefix = "init: heartbeat";
|
|
const beat_ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix);
|
|
check("init produced repeated heartbeats (>=2)", process.write_count >= 2);
|
|
check("heartbeat text arrived intact", beat_ok);
|
|
check("heartbeats came from user mode (CPL 3)", process.write_from_user);
|
|
check("init is still alive (did not exit)", process.exit_code == 0);
|
|
result();
|
|
}
|
|
|
|
/// The initial_ramdisk path: the bootloader handed over an image bundling extra user
|
|
/// binaries; parse it, spawn every program, and confirm one (the vfs stub)
|
|
/// reaches ring 3 and heartbeats — proving the whole ferry-parse-spawn pipeline.
|
|
fn initialRamdiskTest(boot_information: *const BootInformation) void {
|
|
log("DANOS-TEST-BEGIN: initial_ramdisk\n", .{});
|
|
check("bootloader handed over an initial_ramdisk", boot_information.initial_ramdisk_len != 0);
|
|
if (boot_information.initial_ramdisk_len == 0) {
|
|
result();
|
|
return;
|
|
}
|
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
|
check("initial_ramdisk image is valid", false);
|
|
result();
|
|
return;
|
|
};
|
|
check("initial_ramdisk image is valid", true);
|
|
check("initial_ramdisk contains at least one binary", rd.count >= 1);
|
|
|
|
process.write_count = 0;
|
|
process.write_from_user = false;
|
|
var spawned: u32 = 0;
|
|
var i: u32 = 0;
|
|
while (i < rd.count) : (i += 1) {
|
|
const item = rd.entry(i) orelse continue;
|
|
if (process.spawnProcess(item.blob, 4)) spawned += 1 else |err| {
|
|
log("DANOS-INITRD-ERR: {s}: {s}\n", .{ item.name, @errorName(err) });
|
|
}
|
|
}
|
|
check("every initial_ramdisk binary spawned", spawned == rd.count);
|
|
|
|
// Wait for the spawned programs to run and make syscalls (they write + sleep).
|
|
scheduler.setPriority(1);
|
|
const deadline = architecture.millis() + 8000;
|
|
while (process.write_count < 2 and architecture.millis() < deadline) scheduler.yield();
|
|
scheduler.setPriority(4);
|
|
|
|
check("initial_ramdisk processes ran and made syscalls (>=2)", process.write_count >= 2);
|
|
check("syscalls came from user mode (CPL 3)", process.write_from_user);
|
|
result();
|
|
}
|
|
|
|
/// The full VFS path: spawn the user-space VFS server and a client from the
|
|
/// initial_ramdisk. The client opens a file through the runtime file API, writes, seeks, reads
|
|
/// it back, and — only if the round trip matched — heartbeats "vfstest: ok". So
|
|
/// seeing that marker proves client open/write/read reached the server over IPC
|
|
/// and came back correct. (The client retries until the server registers.)
|
|
fn vfsTest(boot_information: *const BootInformation) void {
|
|
log("DANOS-TEST-BEGIN: vfs\n", .{});
|
|
if (boot_information.initial_ramdisk_len == 0) {
|
|
check("bootloader handed over an initial_ramdisk", false);
|
|
result();
|
|
return;
|
|
}
|
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
|
check("initial_ramdisk image is valid", false);
|
|
result();
|
|
return;
|
|
};
|
|
|
|
process.write_count = 0;
|
|
process.write_from_user = false;
|
|
// Spawn just the server and its client (other initial_ramdisk binaries would write to
|
|
// the shared evidence buffer and confuse the marker check).
|
|
_ = spawnNamed(rd, "vfs");
|
|
_ = spawnNamed(rd, "vfs-test");
|
|
|
|
// Wait for the client's success heartbeat (it round-trips, then beats ~1/s).
|
|
const prefix = "vfstest: ok";
|
|
scheduler.setPriority(1);
|
|
const deadline = architecture.millis() + 10000;
|
|
while (architecture.millis() < deadline) {
|
|
if (process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix) and process.write_count >= 2) break;
|
|
scheduler.yield();
|
|
}
|
|
scheduler.setPriority(4);
|
|
|
|
const ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix);
|
|
check("client completed the VFS round trip (open/write/read matched)", ok);
|
|
check("the round trip ran repeatedly (server stays up)", process.write_count >= 2);
|
|
check("client syscalls came from user mode (CPL 3)", process.write_from_user);
|
|
result();
|
|
}
|
|
|
|
/// Spawn the initial_ramdisk binary named `name` as a ring-3 process. Returns false if it
|
|
/// isn't in the image or fails to load.
|
|
fn spawnNamed(rd: initial_ramdisk.Reader, name: []const u8) bool {
|
|
var i: u32 = 0;
|
|
while (i < rd.count) : (i += 1) {
|
|
const item = rd.entry(i) orelse continue;
|
|
if (eql(item.name, name)) {
|
|
return if (process.spawnProcess(item.blob, 4)) true else |_| false;
|
|
}
|
|
}
|
|
return false;
|
|
}
|
|
|
|
/// IO passthrough + IRQ-as-IPC: a user-space driver drives real hardware and is
|
|
/// *woken by it*. Spawn hpet, which claims the HPET, maps its registers into its
|
|
/// own ring-3 address space, arms a level-triggered comparator, binds the interrupt
|
|
/// to an IPC endpoint, and then blocks. It prints "hpet: ok" only after being woken
|
|
/// `target_ticks` times — it cannot reach that line by polling, because the loop's
|
|
/// only exit is through `replyWait` returning a notification badge.
|
|
///
|
|
/// The interesting assertion is the last one, which doesn't trust hpet at all: it
|
|
/// reads the I/O APIC's redirection entry back and checks the kernel really routed
|
|
/// the line (our vector, level-triggered) and really left it unmasked after the
|
|
/// driver's final `irq_ack`. hpet disables its comparator on the last interrupt, so
|
|
/// that state is quiescent and not a race.
|
|
fn hpetTest(boot_information: *const BootInformation) void {
|
|
log("DANOS-TEST-BEGIN: hpet\n", .{});
|
|
if (boot_information.initial_ramdisk_len == 0) {
|
|
check("bootloader handed over an initial_ramdisk", false);
|
|
result();
|
|
return;
|
|
}
|
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
|
check("initial_ramdisk image is valid", false);
|
|
result();
|
|
return;
|
|
};
|
|
|
|
process.write_count = 0;
|
|
process.write_from_user = false;
|
|
check("hpet spawned from the initial_ramdisk", spawnNamed(rd, "hpet"));
|
|
|
|
const prefix = "hpet: ok";
|
|
scheduler.setPriority(1);
|
|
const deadline = architecture.millis() + 10000;
|
|
while (architecture.millis() < deadline) {
|
|
if (process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix) and process.write_count >= 2) break;
|
|
scheduler.yield();
|
|
}
|
|
scheduler.setPriority(4);
|
|
|
|
const ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix);
|
|
check("user driver mapped HPET MMIO and was woken by its interrupt", ok);
|
|
check("driver syscalls came from user mode (CPL 3)", process.write_from_user);
|
|
check("kernel routed and re-armed the HPET's line at the I/O APIC", hpetRouteOk());
|
|
result();
|
|
}
|
|
|
|
/// Read back the I/O APIC redirection entry for the HPET's GSI and confirm the
|
|
/// kernel programmed it: a vector in the device window, level-triggered, unmasked.
|
|
/// Independent of anything the driver reported about itself.
|
|
fn hpetRouteOk() bool {
|
|
const gsi = hpetGsi() orelse return false;
|
|
if (gsi >= architecture.irqRouteCount()) return false;
|
|
const low = architecture.irqRouteRaw(gsi); // entry index == GSI (this I/O APIC's gsi_base is 0)
|
|
const vector: u8 = @truncate(low & 0xFF);
|
|
const masked = low & (1 << 16) != 0;
|
|
const level = low & (1 << 15) != 0;
|
|
return vector >= architecture.irq_vector_base and
|
|
vector < architecture.irq_vector_base + architecture.irq_vector_count and
|
|
level and !masked;
|
|
}
|
|
|
|
/// The GSI discovery recorded for the HPET, from the same device table the driver saw.
|
|
fn hpetGsi() ?u32 {
|
|
var buffer: [16]device_abi.DeviceDescriptor = undefined;
|
|
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
|
for (buffer[0..n]) |d| {
|
|
if (d.class != @intFromEnum(device_abi.DeviceClass.timer)) continue;
|
|
if (d.parent != device_abi.no_parent) continue; // the block, not a comparator child
|
|
for (0..d.resource_count) |j| {
|
|
const r = d.resources[j];
|
|
if (r.kind == @intFromEnum(device_abi.ResourceKind.irq)) return @intCast(r.start);
|
|
}
|
|
}
|
|
return null;
|
|
}
|
|
|
|
/// Bus driver: a user process claims a device that contains other devices, enumerates
|
|
/// them from the hardware, and publishes each as a child via `device_register` — the
|
|
/// primitive a PCI bridge or USB hub driver is built from.
|
|
///
|
|
/// `bus` treats the HPET's register block as a bus and its comparators as children,
|
|
/// giving each a 0x20 sub-window. It checks its own work (children come back from the
|
|
/// table with the right parent and a strictly narrower window) and, importantly, that
|
|
/// the kernel **refuses** a child whose window escapes the parent's — without that,
|
|
/// `device_register` would be a system_call for mapping arbitrary physical memory. It prints
|
|
/// "bus: ok" only if all of that holds.
|
|
///
|
|
/// The kernel-side check here is the one bus can't make: that the children really did
|
|
/// land in the device table with the containment invariant intact.
|
|
fn busTest(boot_information: *const BootInformation) void {
|
|
log("DANOS-TEST-BEGIN: bus\n", .{});
|
|
if (boot_information.initial_ramdisk_len == 0) {
|
|
check("bootloader handed over an initial_ramdisk", false);
|
|
result();
|
|
return;
|
|
}
|
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
|
check("initial_ramdisk image is valid", false);
|
|
result();
|
|
return;
|
|
};
|
|
|
|
process.write_count = 0;
|
|
process.write_from_user = false;
|
|
check("bus spawned from the initial_ramdisk", spawnNamed(rd, "bus"));
|
|
|
|
const prefix = "bus: ok";
|
|
scheduler.setPriority(1);
|
|
const deadline = architecture.millis() + 10000;
|
|
while (architecture.millis() < deadline) {
|
|
if (process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix)) break;
|
|
scheduler.yield();
|
|
}
|
|
scheduler.setPriority(4);
|
|
|
|
const ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix);
|
|
check("bus driver published children and the kernel refused an out-of-window one", ok);
|
|
check("driver syscalls came from user mode (CPL 3)", process.write_from_user);
|
|
check("every registered child is contained in its parent", childrenContained());
|
|
result();
|
|
}
|
|
|
|
/// The device manager (a ring-3 service) enumerates /system/devices, matches each
|
|
/// device to a driver, and — eventually — spawns it. This increment only checks the
|
|
/// discovery+matching half: it must find the HPET (a timer) and decide `hpet` serves
|
|
/// it, printing "device-manager: ok". It uses no special privilege — the same
|
|
/// `device_enumerate` any process could call. (Spawning is the next increment.)
|
|
fn deviceManagerTest(boot_information: *const BootInformation) void {
|
|
log("DANOS-TEST-BEGIN: device-manager\n", .{});
|
|
if (boot_information.initial_ramdisk_len == 0) {
|
|
check("bootloader handed over an initial_ramdisk", false);
|
|
result();
|
|
return;
|
|
}
|
|
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
|
const rd = initial_ramdisk.Reader.init(image) orelse {
|
|
check("initial_ramdisk image is valid", false);
|
|
result();
|
|
return;
|
|
};
|
|
|
|
// Let `system_spawn` find bundled binaries by name (the normal boot path does
|
|
// this too). Only the device-manager is spawned here — so if `hpet` runs at all,
|
|
// it's because the manager discovered the timer, matched, and spawned it.
|
|
process.setInitialRamdisk(image);
|
|
|
|
process.write_count = 0;
|
|
process.write_from_user = false;
|
|
check("device-manager spawned from the initial_ramdisk", spawnNamed(rd, "device-manager"));
|
|
|
|
// End-to-end proof: the driver the manager spawned reaches its own live marker.
|
|
// `hpet: ok` is hpet's final, stable message (it claims the timer, maps its MMIO,
|
|
// binds its IRQ, services one, then sleeps) — nothing overwrites the buffer after,
|
|
// so it's race-free to poll for. Its arrival means the whole
|
|
// discover -> match -> system_spawn -> driver-up chain worked.
|
|
const prefix = "hpet: ok";
|
|
scheduler.setPriority(1);
|
|
const deadline = architecture.millis() + 10000;
|
|
while (architecture.millis() < deadline) {
|
|
if (process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix)) break;
|
|
scheduler.yield();
|
|
}
|
|
scheduler.setPriority(4);
|
|
|
|
const ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix);
|
|
check("device manager matched the timer and system_spawn'd hpet, which came up", ok);
|
|
check("its syscalls came from user mode (CPL 3)", process.write_from_user);
|
|
result();
|
|
}
|
|
|
|
/// Every child `bus` registered must have each of its resources inside a parent
|
|
/// resource of the same kind — the invariant `device_register` exists to maintain,
|
|
/// checked from the kernel's own table rather than the driver's word for it.
|
|
///
|
|
/// Only *registered* children are checked, not the whole tree. Firmware topology is
|
|
/// trusted and doesn't obey containment: a PCI function's BAR is not inside its host
|
|
/// bridge's `bus_range`, because a bus-number range isn't an address window.
|
|
fn childrenContained() bool {
|
|
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
|
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
|
|
|
const bus_id = hpetDeviceId() orelse return false;
|
|
const p = buffer[@intCast(bus_id)];
|
|
|
|
var children: usize = 0;
|
|
for (buffer[0..n]) |d| {
|
|
if (d.parent != bus_id) continue;
|
|
children += 1;
|
|
for (0..d.resource_count) |i| {
|
|
const r = d.resources[i];
|
|
var ok = false;
|
|
for (0..p.resource_count) |j| {
|
|
const pr = p.resources[j];
|
|
if (pr.kind != r.kind) continue;
|
|
if (r.kind == @intFromEnum(device_abi.ResourceKind.irq)) {
|
|
if (pr.start == r.start) ok = true;
|
|
} else if (r.len != 0 and r.start >= pr.start and
|
|
r.start + r.len <= pr.start + pr.len) ok = true;
|
|
}
|
|
if (!ok) return false;
|
|
}
|
|
}
|
|
return children > 0; // bus must have published at least one
|
|
}
|
|
|
|
/// Device id of the HPET (the bus bus claims), from the same table drivers see.
|
|
fn hpetDeviceId() ?u64 {
|
|
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
|
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
|
for (buffer[0..n]) |d| {
|
|
if (d.class != @intFromEnum(device_abi.DeviceClass.timer)) continue;
|
|
if (d.parent != device_abi.no_parent) continue; // a comparator child, not the block
|
|
for (0..d.resource_count) |j| {
|
|
if (d.resources[j].kind == @intFromEnum(device_abi.ResourceKind.memory)) return d.id;
|
|
}
|
|
}
|
|
return null;
|
|
}
|
|
|
|
/// IRQ teardown. When a driver exits, its bindings must be released: the line masked
|
|
/// (so a dead driver's device goes quiet instead of storming) and the slot cleared
|
|
/// (so an ISR never posts a notification into the endpoint that is about to be freed).
|
|
///
|
|
/// This is the path `hpet` never takes — it runs forever — so it gets its own test.
|
|
/// Two properties, both read back from the hardware rather than from our own state:
|
|
///
|
|
/// 1. A bound GSI is routed and unmasked.
|
|
/// 2. After `releaseOwner` for the binding's owner, that same entry is masked again.
|
|
///
|
|
/// And one property that can only be checked from kernel state: a *different* owner's
|
|
/// binding on the same endpoint survives. Endpoints are shared (ipc_register hands out
|
|
/// references), so teardown keyed on the endpoint pointer rather than the owning task
|
|
/// would mask a live sibling driver's device line.
|
|
fn irqFreeTest() void {
|
|
log("DANOS-TEST-BEGIN: irqfree\n", .{});
|
|
|
|
const gsi = hpetGsi() orelse {
|
|
check("discovery recorded an IRQ resource for the HPET", false);
|
|
result();
|
|
return;
|
|
};
|
|
|
|
const endpoint = ipcsync.createEndpoint() orelse {
|
|
check("allocated an endpoint", false);
|
|
result();
|
|
return;
|
|
};
|
|
|
|
// Two owners, one shared endpoint. `other` binds a second line if the I/O APIC has
|
|
// one spare; if not, the sharing half of the test is skipped rather than faked.
|
|
const owner: u32 = 4242;
|
|
const other: u32 = 4343;
|
|
const spare: ?u32 = if (gsi + 1 < irq.maximum_gsi and architecture.irqOwnsGsi(gsi + 1)) gsi + 1 else null;
|
|
|
|
{
|
|
const flags = sync.enter();
|
|
defer sync.leave(flags);
|
|
irq.bind(gsi, endpoint, owner) catch {};
|
|
if (spare) |s| irq.bind(s, endpoint, other) catch {};
|
|
}
|
|
check("bound GSI is routed and unmasked", !entryMasked(gsi));
|
|
if (spare) |s| check("second owner's GSI is routed and unmasked", !entryMasked(s));
|
|
|
|
{
|
|
const flags = sync.enter();
|
|
defer sync.leave(flags);
|
|
irq.releaseOwner(owner);
|
|
}
|
|
check("exiting owner's line is masked again", entryMasked(gsi));
|
|
if (spare) |s| {
|
|
check("a sibling owner's binding on the same endpoint survives", !entryMasked(s));
|
|
const flags = sync.enter();
|
|
defer sync.leave(flags);
|
|
irq.releaseOwner(other);
|
|
}
|
|
|
|
result();
|
|
}
|
|
|
|
/// Is redirection entry `gsi` masked? (bit 16 of the low dword; entry index == GSI
|
|
/// because this I/O APIC's gsi_base is 0.)
|
|
fn entryMasked(gsi: u32) bool {
|
|
return architecture.irqRouteRaw(gsi) & (1 << 16) != 0;
|
|
}
|
|
|
|
/// The MMIO-grant teardown fix: a device-granted leaf must NOT be returned to the
|
|
/// RAM allocator when its address space is destroyed. Map a real RAM frame as a
|
|
/// device grant, tear the address space down, and confirm the frame is still held
|
|
/// (only the page tables came back) — then free it explicitly. A regression guard
|
|
/// for the freeSubtree device_grant skip that keeps IO passthrough from corrupting
|
|
/// the frame pool.
|
|
fn ioPassTest() void {
|
|
log("DANOS-TEST-BEGIN: iopass\n", .{});
|
|
const base_free = pmm.stats().free_frames;
|
|
|
|
const aspace = architecture.createAddressSpace() orelse {
|
|
check("created a fresh address space", false);
|
|
result();
|
|
return;
|
|
};
|
|
const frame = pmm.alloc() orelse {
|
|
architecture.destroyAddressSpace(aspace);
|
|
check("allocated a frame to grant", false);
|
|
result();
|
|
return;
|
|
};
|
|
// Map it the way mmio_map does (device grant), then tear the space down.
|
|
architecture.mapUserDeviceInto(aspace, process.device_arena_base, frame, abi.page_size);
|
|
architecture.destroyAddressSpace(aspace);
|
|
|
|
// The page tables were reclaimed; the device-granted frame must not have been.
|
|
check("device-granted frame survived teardown (not reclaimed as RAM)", pmm.stats().free_frames == base_free - 1);
|
|
pmm.free(frame);
|
|
check("no leak once the frame is explicitly freed", pmm.stats().free_frames == base_free);
|
|
result();
|
|
}
|
|
|
|
fn faultInvalidOpcode() void {
|
|
log("DANOS-TEST-BEGIN: fault-ud\n", .{});
|
|
asm volatile ("ud2");
|
|
}
|
|
|
|
/// Verify NX: fetching an instruction from a data page (mapped no-execute) faults.
|
|
fn faultNoExecute() void {
|
|
log("DANOS-TEST-BEGIN: fault-nx\n", .{});
|
|
var scratch: u64 = 0xC3; // a lone `ret` — harmless if NX somehow let it run
|
|
const f: *const fn () void = @ptrFromInt(@intFromPtr(&scratch));
|
|
f(); // instruction fetch from an NX page -> #PF before it executes
|
|
log("DANOS-TEST-RESULT: FAIL (NX not enforced)\n", .{});
|
|
}
|
|
|
|
/// Verify the null guard: dereferencing address 0 (page 0 left unmapped) faults.
|
|
fn faultNull() void {
|
|
log("DANOS-TEST-BEGIN: fault-null\n", .{});
|
|
// Launder the address through empty asm so the compiler no longer knows it's
|
|
// 0 (otherwise it folds a null-pointer safety panic instead of doing the real
|
|
// access). `allowzero` skips the same null check on the cast. The write then
|
|
// hits the unmapped page 0 and takes a real hardware #PF.
|
|
var address: u64 = 0;
|
|
address = asm ("" : [ret] "=r" (-> u64) : [in] "0" (address));
|
|
const p: *allowzero volatile u64 = @ptrFromInt(address);
|
|
p.* = 1;
|
|
}
|
|
|
|
fn faultPageFault() void {
|
|
log("DANOS-TEST-BEGIN: fault-pf\n", .{});
|
|
// Runtime address so the backend emits a register store (not a `mov moffs`,
|
|
// which the self-hosted x86_64 backend can't encode).
|
|
var address: u64 = 0xdeadbeef000; // well above all mapped RAM
|
|
const p: *volatile u64 = @ptrFromInt(address);
|
|
p.* = 1;
|
|
address += 0;
|
|
}
|
|
|
|
fn faultDoubleFault() void {
|
|
log("DANOS-TEST-BEGIN: fault-df\n", .{});
|
|
architecture.disableInterrupts(); // so only the ud2 delivery (not a timer tick) triggers the #DF
|
|
// Point RSP at unmapped memory, then fault: the CPU can't push the fault
|
|
// frame, which escalates to #DF — survivable only because #DF runs on IST1.
|
|
var bad_sp: u64 = 0x5000000000;
|
|
asm volatile (
|
|
\\mov %[sp], %%rsp
|
|
\\ud2
|
|
:
|
|
: [sp] "r" (bad_sp),
|
|
: .{ .memory = true }
|
|
);
|
|
bad_sp += 0;
|
|
}
|
|
|
|
var ap_reached_fault: bool = false;
|
|
|
|
/// A task that faults with a #DF *on whatever core it's pinned to*. Announces the
|
|
/// core, then triggers the same double fault as `faultDoubleFault` — which is only
|
|
/// survivable on IST1, so it exercises that core's own TSS.
|
|
fn apDoubleFaultTask() void {
|
|
log("DANOS-AP: task running on core {d}, triggering #DF\n", .{scheduler.currentCpuIndex()});
|
|
@atomicStore(bool, &ap_reached_fault, true, .release);
|
|
architecture.disableInterrupts();
|
|
var bad_sp: u64 = 0x5000000000;
|
|
asm volatile (
|
|
\\mov %[sp], %%rsp
|
|
\\ud2
|
|
:
|
|
: [sp] "r" (bad_sp),
|
|
: .{ .memory = true }
|
|
);
|
|
bad_sp += 0;
|
|
}
|
|
|
|
/// Fault on an application processor. Pins a double-faulting task to core 1, so the
|
|
/// fault is taken and handled by *that core's own* IDT and TSS/IST — not the BSP's.
|
|
/// The harness matches "core N: double fault (vector 8)" with N ≥ 1, which can only
|
|
/// appear if the AP caught the #DF on its IST1 (a broken per-core TSS would
|
|
/// triple-fault and reset instead). We then show the BSP still runs afterwards, so
|
|
/// the fault was *contained* to the AP, not fatal to the system.
|
|
fn faultApTest() void {
|
|
log("DANOS-TEST-BEGIN: fault-ap-df\n", .{});
|
|
if (!scheduler.spawnOn(apDoubleFaultTask, 6, 1)) {
|
|
log("DANOS-AP: could not pin to core 1 (run with -smp) - FAIL\n", .{});
|
|
architecture.halt();
|
|
}
|
|
// Wait until the AP is about to fault, then keep running to prove containment.
|
|
var spins: u64 = 0;
|
|
while (!@atomicLoad(bool, &ap_reached_fault, .acquire) and spins < 5_000_000_000) spins +%= 1;
|
|
var settle: u64 = 0;
|
|
while (settle < 500_000_000) settle +%= 1; // let the AP take + report the fault
|
|
log("DANOS-BSP: core {d} still running after the AP fault (contained)\n", .{scheduler.currentCpuIndex()});
|
|
architecture.halt();
|
|
}
|