danos/test/qemu_test.py

1183 lines
63 KiB
Python

#!/usr/bin/env python3
"""Reproducible QEMU integration tests for danos.
For each test case it builds the kernel with `-Dtest-case=<name>`, boots it
headless in QEMU with the serial port captured to a file, and asserts that the
expected marker appears in that output before a timeout. The kernel's serial log
is machine-readable (see src/kernel/tests.zig and src/kernel/arch/*/serial.zig), so no
screen-scraping is involved.
Structured per-architecture so a second CPU (e.g. an AArch64 Raspberry Pi) is a
matter of adding an ARCHES entry and its serial/boot support — the test cases and
the runner stay the same.
Usage:
python3 test/qemu_test.py # run all cases on the default arch
python3 test/qemu_test.py --arch x86_64 # pick an architecture
python3 test/qemu_test.py smoke fault-pf # run only named cases
"""
import argparse
import json
import os
import re
import shutil
import socket
import subprocess
import sys
import tempfile
import time
REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
WORK = os.path.join(REPO, "zig-out", "qemu-test")
# --- Architecture configurations -------------------------------------------
# Each describes how to build and boot that architecture. Only the fields that
# actually differ between architectures live here.
ARCHES = {
"x86_64": {
"zig_flags": [], # build.zig currently pins the kernel to x86_64
"qemu": "qemu-system-x86_64",
# Firmware lives in different places per OS/distro; resolve_firmware()
# picks the first that exists so the harness runs unconfigured on Arch,
# Debian/Ubuntu, Fedora, and macOS (Homebrew). Override by reordering or
# dropping an absolute path at the front.
"ovmf_code": [
"/usr/share/edk2/x64/OVMF_CODE.4m.fd", # Arch
"/usr/share/OVMF/OVMF_CODE_4M.fd", # Debian/Ubuntu
"/usr/share/OVMF/OVMF_CODE.fd", # older Debian/Ubuntu
"/usr/share/edk2-ovmf/x64/OVMF_CODE.fd", # Fedora
"/opt/homebrew/share/qemu/edk2-x86_64-code.fd", # macOS Homebrew (Apple Silicon)
"/usr/local/share/qemu/edk2-x86_64-code.fd", # macOS Homebrew (Intel)
],
"ovmf_vars": [
"/usr/share/edk2/x64/OVMF_VARS.4m.fd", # Arch
"/usr/share/OVMF/OVMF_VARS_4M.fd", # Debian/Ubuntu
"/usr/share/OVMF/OVMF_VARS.fd", # older Debian/Ubuntu
"/usr/share/edk2-ovmf/x64/OVMF_VARS.fd", # Fedora
"/opt/homebrew/share/qemu/edk2-i386-vars.fd", # macOS Homebrew (Apple Silicon)
"/usr/local/share/qemu/edk2-i386-vars.fd", # macOS Homebrew (Intel)
],
# zig-out is itself the FHS-shaped boot volume (docs/efi.md): the build
# installs BOOTX64.efi, the kernel, init, and the initial-ramdisk at their
# boot paths. The harness presents zig-out to the guest directly — exactly
# as `zig build run-x86-64` does — so there is no separate ESP to assemble.
# Built as a function so we can splice in per-run paths.
"qemu_args": lambda a, boot_volume, vars_fd, serial: [
"-machine", "q35", "-m", "128M",
# The default TCG model (qemu64) predates the supervisor-hardening bits:
# no SMEP, no SMAP, so the kernel's probes would find nothing and the
# enabled paths would never run in CI. `max` is "everything this
# accelerator can emulate", which under TCG includes both.
"-cpu", "max",
"-drive", f"if=pflash,format=raw,readonly=on,file={a['ovmf_code']}",
"-drive", f"if=pflash,format=raw,file={vars_fd}",
# Boot off a FAT USB device: the boot volume is a mass-storage device on
# the xHCI bus (usb-kbd/usb-mouse ride the same controller). `boot_volume`
# is the FAT image the build produces. bootindex=0 steers OVMF to it.
"-device", "qemu-xhci,id=xhci",
"-device", "usb-kbd,bus=xhci.0",
"-device", "usb-mouse,bus=xhci.0",
"-drive", f"if=none,id=bootusb,format=raw,file={boot_volume}",
"-device", "usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
"-net", "none",
"-vga", "none", "-device", "VGA,edid=on,xres=1280,yres=720",
"-display", "none",
"-serial", f"file:{serial}",
"-no-reboot", # exit instead of rebooting on a triple fault
],
},
# To add an architecture, e.g. "aarch64": provide its qemu binary, firmware,
# boot method, and the AArch64 kernel/serial support in src/kernel/arch/aarch64/.
}
# --- Test cases ------------------------------------------------------------
# `expect`: a regex that must appear in serial output => pass.
# `fail`: optional regex whose appearance => immediate fail.
CASES = [
# smoke also proves the QMP channel: the harmless query must be delivered
# (handshake + command) before the case may pass — see run_case.
{"name": "smoke",
"qmp_after": {"delay": 2, "command": "query-status"},
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
{"name": "discovery",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
{"name": "wx",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
{"name": "timer",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
{"name": "clock",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# Wall-clock: the CMOS RTC read at boot gives a plausible current epoch (the
# foundation for filesystem mtime).
{"name": "wall-clock",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
{"name": "vmm",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
{"name": "heap",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
{"name": "sched",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
{"name": "priority",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
{"name": "sleep",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
{"name": "event",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
{"name": "ipc",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# Synchronous IPC: a client and server ping-pong 100 calls through an
# endpoint (rendezvous, reply routing, cross-address-space copy).
{"name": "ipc-call",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# IPC capability passing (M13): a client hands the server an endpoint in a call,
# the server hands one back in its reply; each is verified same-object + shared.
{"name": "ipc-cap",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# DMA memory (M14): contiguous frame allocation, below-4G cap, coherent mapping,
# and reclaim on teardown.
# Boots with no IOMMU device, so it also asserts the explicit fail-open boot line
# (the DMA-isolation posture must be stated, never silent).
{"name": "dma",
"expect": r"(?s)(?=.*iommu : none present - DMA fail-open \(unisolated\))(?=.*DANOS-TEST-RESULT: PASS)",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# MSI (M15): allocate a per-device vector and deliver it as a notification (a
# self-IPI stands in for the device's MSI write, since the HPET has no MSI).
{"name": "msi",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# IOMMU: boot with an emulated VT-d unit; danos parses the DMAR table, enables
# translation, and proves the domain walker (map/resolve/unmap on a scratch domain)
# with no spurious faults. The `enabled` line is a lookahead so a silently-dead unit
# cannot fake a pass.
{"name": "iommu",
"qemu_extra": ["-device", "intel-iommu,intremap=off"],
"expect": r"(?s)(?=.*DANOS-IOMMU: enabled base=0x[0-9a-f]+ domains active)(?=.*DANOS-TEST-RESULT: PASS)",
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
# DMA under translation: the full USB storage stack (xHC ring DMA + BOT/SCSI + fat's
# cross-process bounce buffer) runs with VT-d enabled. Every device is identity-
# mapped in the blanket domain, so DMA works, but through real second-level walks.
{"name": "iommu-usb-storage",
"build_case": "fat-mount",
"smp": 4,
"timeout": 150,
"qemu_extra": ["-device", "intel-iommu,intremap=off"],
"expect": r"(?s)(?=.*/system/kernel: iommu online)(?=.*fat: mounted /volumes/usb)(?=.*fat-test: ok)",
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
# DMA + MSI under translation: interrupt-IN reports arrive through translated DMA and
# the xHC's MSI/MSI-X still delivers (the 0xFEE00000 interrupt window bypasses second-
# level translation with interrupt remapping off).
{"name": "iommu-usb-hid",
"build_case": "usb-hid",
"smp": 4,
"timeout": 150,
"qemu_extra": ["-device", "intel-iommu,intremap=off"],
"expect": r"(?s)(?=.*/system/kernel: iommu online)(?=.*usb-hid-keyboard: ok)(?=.*usb-hid-mouse: ok)",
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
# Enforcement, the negative proof: a claimed e1000e fires a DMA at an unmapped page;
# VT-d must fault it (logged) and the system must stay alive. The fault line is the
# point here, so unlike the positive cases it appears in `expect`, not `fail`.
{"name": "iommu-fault",
"smp": 4,
"timeout": 120,
"qemu_extra": ["-device", "intel-iommu,intremap=off", "-device", "e1000e"],
"expect": r"(?s)(?=.*DANOS-IOMMU-FAULT: bdf=)(?=.*iommu-fault-test: system alive)(?=.*DANOS-TEST-RESULT: PASS)",
"fail": r"iommu-fault-test: FAIL|DANOS-TEST-RESULT: FAIL|CPU EXCEPTION|KERNEL PANIC"},
# AMD-Vi: the same detection + scratch-domain walker proof as the `iommu` case, but on
# the AMD backend (IVRS parse, device table, command buffer). QEMU's amd-iommu needs
# dma-remap=on (default off = translation silently ignored). UNTESTED on real AMD.
{"name": "amd-iommu",
"build_case": "iommu",
"qemu_extra": ["-device", "amd-iommu,dma-remap=on,intremap=off"],
"expect": r"(?s)(?=.*iommu online \(AMD-Vi\))(?=.*DANOS-IOMMU: enabled base=0x[0-9a-f]+ domains active)(?=.*DANOS-TEST-RESULT: PASS)",
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
# DMA under AMD-Vi translation: the full USB storage stack through AMD device-table
# translation. Tentative — land depends on QEMU amd-iommu behaving. UNTESTED on real AMD.
{"name": "amd-iommu-usb-storage",
"build_case": "fat-mount",
"smp": 4,
"timeout": 150,
"qemu_extra": ["-device", "amd-iommu,dma-remap=on,intremap=off"],
"expect": r"(?s)(?=.*iommu online \(AMD-Vi\))(?=.*fat: mounted /volumes/usb)(?=.*fat-test: ok)",
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
# Port I/O grants: a claimed device's io_port resource lets a driver read/write its
# ports (PS/2 status 0x64), gated by the claim; out-of-range/unclaimed is refused.
{"name": "ioport",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# Display handoff (D1): the kernel seeds the loader's framebuffer as a claimable
# `display` device with a write-combining memory resource; the claim + mmio_map path
# maps it, and the leaf is genuinely write-combining (PAT entry 4), not the UC default.
{"name": "display",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# Display service (D2/D3): the user-space compositor claims the framebuffer, allocates
# a cacheable back buffer, clears it, and presents that composed frame (double-buffer
# path); then a startup self-check composites two overlapping layers and confirms the
# overlap shows the top layer (D3). Matched on the service's own heartbeats.
{"name": "display-service",
"expect": r"display: online \d+x\d+ pitch \d+[\s\S]*display: presented frame 0[\s\S]*display: compositor self-check ok",
"fail": r"display: could not|self-check FAILED|CPU EXCEPTION|KERNEL PANIC"},
# Display demo (D4): a separate process (display-demo) drives the compositor over the
# layer client API — wallpaper + a moving rectangle + a cursor, presented in a loop.
# `display-demo: ok` is printed only after it drove a run of frames of motion through
# the service (the visible motion is a screenshot via `zig build run-x86-64`).
{"name": "display-demo",
"expect": r"display-demo: scene up[\s\S]*display-demo: ok",
"fail": r"display-demo: (no display|create failed)|display: could not|CPU EXCEPTION|KERNEL PANIC"},
# Threaded compositor tracks a mouse (docs/threading.md, docs/display.md): the display
# runs a mouse-listener thread alongside its compositor loop. `input-source mouse`
# publishes pure motion -> the input service fans it to the display's listener -> the
# listener accumulates it into a cursor position handed to the render loop over a
# single-slot channel. `display: cursor tracking mouse ok` latches once the cursor has
# tracked a run of that motion end to end.
{"name": "display-cursor",
"smp": 4,
"expect": r"display: online \d+x\d+[\s\S]*display: cursor tracking mouse ok",
"fail": r"display: (could not|mouse subscribe failed)|CPU EXCEPTION|KERNEL PANIC"},
# Shared memory (v2 V2): shared-memory-client creates a region, writes a pattern, and passes its
# capability to shared-memory-server, which maps it and confirms the same bytes — proving
# cross-process shared pages over the extended capability passing.
{"name": "shared-memory",
"expect": r"shared-memory: shared 4096 bytes ok",
"fail": r"shared-memory: (shared FAILED|create failed|no server|call failed|map)|CPU EXCEPTION|KERNEL PANIC"},
# virtio-gpu driver (v2 V3): boot with an emulated virtio-gpu. The device-manager stack
# discovers the PCI function and spawns the driver, which brings up the control virtqueue,
# creates a 2D scanout resource backed by DMA memory, set_scanouts it, paints a test
# pattern, transfers + flushes it, and waits for the device's used-ring ack, then reads
# the backing back. `scanout WxH online` + `flush acked, pixel check ok` are the markers.
{"name": "virtio-gpu",
"qemu_extra": ["-device", "virtio-gpu-pci"],
"expect": r"virtio-gpu: scanout \d+x\d+ online[\s\S]*virtio-gpu: flush acked, pixel check ok",
"fail": r"virtio-gpu:.*(failed|not acked|mismatch|unable to claim|not a virtio-gpu|too small|no PCI capability|does not offer|rejected|missing common-config|not a mapped resource|could not spawn)|CPU EXCEPTION|KERNEL PANIC"},
# Native backend + hot-attach (v2 V4): boot the compositor + display-demo with an emulated
# virtio-gpu. The driver announces its shared scanout surface to the compositor, which maps
# it, upgrades off the GOP floor, and drives frames through the native backend — reading a
# pixel back to confirm the composited frame reached the shared surface, while the demo runs.
{"name": "display-native",
"qemu_extra": ["-device", "virtio-gpu-pci"],
"mem": "512M", # boots the compositor + demo + the whole device-manager driver stack at once
# Order-independent: the demo's `ok` may print before or after the driver announces, so
# require all three markers to appear somewhere rather than in a fixed order.
"expect": r"(?s)(?=.*display: scanout upgraded to virtio-gpu)(?=.*display: native present verified)(?=.*display-demo: ok)",
"fail": r"display: native present FAILED|display: could not|display-demo: (no display|create failed)|CPU EXCEPTION|KERNEL PANIC"},
# Mode-set + EDID + fenced presents (v2 V5): same boot as display-native. After upgrading,
# the compositor queries the driver's modes, switches to a different resolution, and confirms
# the backend now reports it; each present is fenced — completion-acknowledged and tear-free,
# not vblank-paced (docs/display-v2.md, "Fenced is not vsync"). (The driver also logs the
# EDID preferred mode during bring-up.) Reuses the display-native kernel scenario.
{"name": "display-modeset",
"build_case": "display-native",
"qemu_extra": ["-device", "virtio-gpu-pci"],
"mem": "512M",
"expect": r"(?s)(?=.*display: mode set to \d+x\d+, verified)(?=.*display: fenced present ok)",
"fail": r"display: mode set FAILED|display: mode-set self-check: |display: native present FAILED|CPU EXCEPTION|KERNEL PANIC"},
# Resilience: driver restart + re-attach (v2 V6). device-manager (in test-scanout-restart
# mode) kills the virtio-gpu driver once after it hellos; the restart policy respawns it, it
# re-announces, and the compositor re-attaches — surviving the loss. Expect the initial
# upgrade AND the re-attach; any CPU exception / panic (the compositor crashing) is a fail.
{"name": "display-reattach",
"qemu_extra": ["-device", "virtio-gpu-pci"],
"mem": "512M",
"expect": r"(?s)(?=.*display: scanout upgraded to virtio-gpu)(?=.*display: scanout re-attached)",
"fail": r"CPU EXCEPTION|KERNEL PANIC|display: could not"},
# Monotonic clock (clock() syscall source): calibrated, advancing, never backwards.
{"name": "clock",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# Parallelism: needs more than one core, so this case boots with -smp 4.
{"name": "smp",
"smp": 4,
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# Affinity: a pinned task must never migrate off its core.
{"name": "affinity",
"smp": 4,
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# Stress the big kernel lock across cores; heavier, so a longer timeout.
{"name": "smp-stress",
"smp": 4,
"timeout": 150,
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# Retry: a forced first-wake failure must still bring every core online.
{"name": "smp-retry",
"smp": 4,
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# TSC clocksource + cross-core warp check (real Intel/AMD / KVM path). TCG won't
# advertise an invariant TSC, so the kernel forces the TSC clocksource on for this
# case (gated in kernel.zig) and runs the per-AP warp check across the 4 cores;
# their TSCs are synchronized, so it stays on the TSC (no HPET fallback). The rest
# of the suite exercises the HPET fallback instead. See docs/timers.md.
{"name": "tsc-sync",
"smp": 4,
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
{"name": "fault-ud", "expect": r"invalid opcode \(vector 6\)"},
{"name": "fault-pf", "expect": r"page fault \(vector 14\)"},
{"name": "fault-df", "expect": r"double fault \(vector 8\)"},
# Fault on an application processor: a #DF pinned to core 1 must be caught by that
# core's own IST (N >= 1), proving per-core TSS works; a broken one triple-faults.
{"name": "fault-ap-df",
"smp": 4,
"expect": r"core [1-9]\d*: double fault \(vector 8\)",
"fail": r"could not pin"},
{"name": "fault-nx",
"expect": r"page fault \(vector 14\)",
"fail": r"NX not enforced"},
{"name": "fault-null", "expect": r"page fault \(vector 14\)"},
# SMEP: ring 0 calls into a page that is present, executable and *user*-accessible
# — the ret2usr shape. CR4.SMEP must refuse the instruction fetch: #PF with error
# code 0x11 (present | instruction-fetch; the U/S bit reports the *access* ring, so
# it is 0 for a ring-0 fetch), taken in ring 0, reported at the page it tried to
# fetch from (a fetch #PF is raised at the un-fetchable instruction, so the IP is
# the user address, not the kernel-half call site). The boot line is asserted too:
# on a CPU without SMEP nothing would fault and the case would prove nothing.
{"name": "fault-smep",
"expect": r"(?s)(?=.*smep\s+: enabled)(?=.*page fault \(vector 14\))"
r"(?=.*ring 0 \(kernel\))(?=.*error code : 0x11)"
r"(?=.*fault addr : 0x0000700000000000)",
"fail": r"SMEP not enforced|SMEP not enabled|no frame for the probe page"},
# SMAP: ring 0 reads a page that is present and *user*-accessible, the way a syscall
# that dereferenced its caller's pointer would. CR4.SMAP (plus the `clac` patched
# into the interrupt entry, which keeps ring 3 from suspending it by taking an
# interrupt with AC set) must refuse the load: #PF with error code 0x1 — present,
# and nothing else. A SMAP violation has no bit of its own; bit 1 is clear because
# it is a read and bit 2 (U/S) reports the ring of the *access*, which was 0. Taken
# in ring 0, at the user address it read. The boot line is asserted too: without
# SMAP nothing would fault and the case would prove nothing.
{"name": "fault-smap",
"expect": r"(?s)(?=.*smap\s+: enabled)(?=.*page fault \(vector 14\))"
r"(?=.*ring 0 \(kernel\))(?=.*error code : 0x1(?![0-9a-f]))"
r"(?=.*fault addr : 0x0000700000000000)",
"fail": r"SMAP not enforced|SMAP not enabled|no frame for the probe page"},
# SYSRET canonical-RIP guard: a ring-3 probe whose `syscall` sits on the last two
# bytes of executable address space, so the return RIP the kernel must hand back
# is 0x0000800000000000 — non-canonical, and a ring-0 #GP if it reached SYSRETQ.
# The kernel must return through IRETQ instead, which faults the *process* in
# ring 3: the kill line names a general protection fault at exactly that address,
# and the probe's own result line reports the guard counter. Any "ring 0 (kernel)"
# report, panic or reset here is the hazard itself, so they are failures.
# TCG does not model the ring-0 #GP (measured: with the guard removed these two
# lines are unchanged), so the case's teeth are the counter assertion inside
# DANOS-TEST-RESULT, not the kill line.
{"name": "sysret-canonical",
"expect": r"(?s)(?=.*killed by general protection fault \(vector 13\))"
r"(?=.*IP\s+: 0x0000800000000000)"
r"(?=.*DANOS-TEST-RESULT: PASS)",
"fail": r"DANOS-TEST-RESULT: FAIL|CPU EXCEPTION|KERNEL PANIC"},
# Memory grants: the mmap/munmap path hands out user pages into a process's
# arena, translate resolves them, munmap frees them, and no frames leak.
{"name": "usermem",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# The checked copy layer (system/kernel/user-memory.zig): kernel-side unit
# checks for the bound, presence, and leaf U/S + writable refusals, then a
# fixture aiming unmapped-but-in-range pointers at klog_read/klog_status/
# process_enumerate/device_enumerate/fs_resolve/debug_write. Each must come
# back as a wrapped -errno with the machine still running — before H1 every
# one of them dereferenced the bad page in ring 0 and halted it.
{"name": "user-memory",
"timeout": 60,
"expect": r"(?s)(?=.*DANOS-TEST-RESULT: PASS)(?=.*user-memory-test: ok)",
"fail": r"user-memory-test: FAIL|DANOS-TEST-RESULT: FAIL|CPU EXCEPTION|KERNEL PANIC"},
# Isolation: a ring-3 read of a kernel-only page must #PF with error code
# 0x5 (present|user) at the user IP. ([\s\S] spans lines; `.` doesn't.)
{"name": "user-pf",
"expect": r"page fault \(vector 14\)[\s\S]*error code : 0x5[\s\S]*IP\s*: 0x00007000000000",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# Fault recovery: a scheduled ring-3 process that faults is killed — resources
# reclaimed, core kept — while init's heartbeat proves the OS survived.
{"name": "fault-recovery",
"timeout": 60,
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# docs/threading-plan.md M1: address-space refcount — spaces destroyed exactly
# once per process, no leak/double-free (the foundation shared-address-space threads need).
{"name": "address-space-refcount",
"timeout": 60,
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# docs/threading-plan.md M2: runtime.Thread.spawn — a worker thread runs in the
# caller's address space (a shared-memory write, observed by the main thread).
{"name": "thread-spawn",
"timeout": 60,
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# docs/threading-plan.md M3: join + parallelism — N workers each do K atomic
# increments (total exactly N*K after join) and run on >1 core; plus detach.
{"name": "thread-join",
"smp": 4,
"timeout": 60,
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# docs/shared-fate-plan.md M4: a worker's CPU fault kills the whole group —
# one notification (badged with the leader), the leader's reason is the fault
# class, and every counter returns to base.
{"name": "thread-fault-group",
"smp": 4,
"timeout": 60,
"expect": r"thread-test: worker faulting[\s\S]*DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# docs/shared-fate-plan.md M4: process_kill on a threaded group — the
# spinning worker dies by the deferred (condemned) path, both members'
# device claims are free before the single leader-badged notification.
{"name": "kill-threaded-group",
"smp": 4,
"timeout": 60,
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# docs/shared-fate-plan.md M4: process_kill aimed at a WORKER tid kills the
# whole group (the capability is per-process), still badged with the leader.
{"name": "kill-via-worker-tid",
"smp": 4,
"timeout": 60,
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# docs/shared-fate-plan.md M4: two members fault near-simultaneously on
# different cores; the dying latch makes them one group death
# (fault_kill_count == 1) with a deterministic leader reason.
{"name": "racing-triggers",
"smp": 4,
"timeout": 60,
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# docs/shared-fate-plan.md M4: exit(3) from a WORKER is group death with the
# leader's reason reading .aborted (the exit_group lesson).
{"name": "exit-group",
"smp": 4,
"timeout": 60,
"expect": r"thread-test: worker exiting the process[\s\S]*DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# docs/shared-fate-plan.md M4: the LEADER's thread_exit is refused (-EPERM);
# the worker is unaffected and the process then exits clean.
{"name": "leader-thread-exit",
"timeout": 60,
"expect": r"thread-test: leader thread_exit refused ok[\s\S]*DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# docs/shared-fate-plan.md M4: regression — a WORKER's thread_exit stays
# per-thread; the sibling survives it.
{"name": "thread-exit-solo",
"timeout": 60,
"expect": r"thread-test: solo sibling survived ok[\s\S]*DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# docs/shared-fate-plan.md M3/M4: a shared-memory region whose only HANDLE
# died with its creator thread survives on the MAPPING reference — the
# sibling's view stays intact through allocator churn.
{"name": "shm-mapping-ref",
"timeout": 60,
"expect": r"thread-shm: mapping survives creator ok[\s\S]*DANOS-TEST-RESULT: PASS",
"fail": r"thread-shm: FAIL|DANOS-TEST-RESULT: FAIL"},
# docs/threading-plan.md M4: futex — a thread parks in futex_wait and is woken by
# futex_wake (serial order waiting/waking/woke), and timedWait reports a timeout.
{"name": "thread-futex",
"smp": 4,
"timeout": 60,
"expect": r"thread-futex: waiting[\s\S]*thread-futex: waking[\s\S]*thread-futex: woke[\s\S]*DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# docs/threading-plan.md M5: Mutex + Condition — a bounded producer/consumer moves
# N unique items across cores; the consumed checksum matches exactly (no loss).
{"name": "thread-mutex",
"smp": 4,
"timeout": 60,
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# docs/threading-plan.md M6: thread identity — getCurrentId is distinct and non-zero
# for the main thread and two workers.
{"name": "thread-id",
"smp": 4,
"timeout": 60,
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# docs/threading-plan.md M7: thread-safe allocation — N threads hammer the shared heap
# (per-aspace mmap arena + locked free list) with no cross-block corruption.
{"name": "thread-alloc",
"smp": 4,
"timeout": 60,
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# docs/threading-plan.md M8: the task reaper — spawn+kill many processes; total kernel
# stack bytes return to baseline (every dead task's stack reclaimed, no leak).
{"name": "task-reap",
"smp": 4,
"timeout": 60,
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# docs/threading-plan.md M10: per-thread fs.base — two threads keep private %fs:8 TLS
# slots across context switches (no cross-talk).
{"name": "thread-tls",
"smp": 4,
"timeout": 60,
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# docs/threading-plan.md M11: RwLock — readers/writers across cores; a reader never
# observes a half-written value (writers hold it exclusively).
{"name": "thread-rwlock",
"smp": 4,
"timeout": 60,
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# Process arguments: argv arrives on the SysV entry stack (argv[0] = the spawned
# name, argv[1..] = the system_spawn argument blob) and echoes back intact.
{"name": "args",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# The real user binary: the bootloader ships /system/services/init off the ESP, the
# kernel loads the ELF and runs it in ring 3, and it writes + exits cleanly.
{"name": "init",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# Real processes: /system/services/init loaded as a scheduled ring-3 process with its
# own address space, run twice (create/exit/teardown/recreate), coexisting
# with a kernel task under preemption.
{"name": "process",
"smp": 4,
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# process_enumerate: a task-table snapshot lists spawned processes by name and
# id alongside the kernel tasks, and a too-small buffer still reports the total.
{"name": "process-list",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# process_kill + the exit notification: only the supervisor may kill; a blocked
# victim is reaped in place and a spinning one dies by the deferred (tick) path;
# each death posts one exit badge to the endpoint given at spawn.
{"name": "process-kill",
"smp": 4,
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# The user-side whole: process-test supervises two children from ring 3 —
# spawn with an exit endpoint, enumerate, kill, notification, gone.
{"name": "supervision",
"smp": 4,
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# M17.1: a dead process's device claims are released by the reap — kill a child
# holding a claim, the device must be claimable again (process-lifecycle.md).
{"name": "claim-release",
"smp": 4,
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# M17.3: published exit events — the VFS subscribes, a client dies holding an
# open handle, and the VFS releases it (process-lifecycle.md "Who learns of a death").
{"name": "vfs-client-death",
"smp": 4,
"timeout": 90, # the park client waits out the whole USB->block->fat chain
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# M17.4: signals over IPC — ping, reload, terminate (clean exit), the one-shot
# timer, and the stop sequence's two endings, all driven from ring 3.
{"name": "signals",
"smp": 4,
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# M18.2: bus tree reports — the xHCI driver scans its root-hub ports and
# reports both QEMU devices; the manager mirrors, prunes on the reporter's
# death, and the respawned driver re-reports (docs/device-manager.md).
{"name": "usb-report",
"smp": 4,
"timeout": 150,
# The xHCI bus + usb-kbd/usb-mouse come from the default boot config now
# (every case boots off a usb-storage device on that bus).
"expect": r"device-manager: child added[\s\S]*"
r"device-manager: child added[\s\S]*"
r"device-manager: test mode: killing the reporter[\s\S]*"
r"device-manager: child removed[\s\S]*"
r"device-manager: restarting \S*usb-xhci-bus[\s\S]*"
r"device-manager: child added",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# USB HID end to end: boot the full tree, enumerate the xHCI, and let the
# manager spawn the USB keyboard driver, which opens its device over the
# transfer protocol, asks for boot protocol, subscribes to its interrupt
# endpoint, and comes up — proof the class-driver <-> controller path works.
{"name": "usb-hid",
"smp": 4,
"timeout": 150,
# usb-kbd/usb-mouse ride the default boot xHCI bus (see qemu_args).
"expect": r"(?=[\s\S]*usb-hid-keyboard: ok)(?=[\s\S]*usb-hid-mouse: ok)",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# Keyboard echo: inject a known phrase via QMP send-key; the usb-hid-keyboard
# driver decodes it and echoes each character to the log (the simple
# real-hardware keyboard check — type a phrase, read it back off the stick,
# or watch it live on screen in a -Ddiagnose boot). Proves the whole path:
# HID report -> decode -> layout -> character.
{"name": "usb-key-echo",
"build_case": "usb-hid",
"smp": 4,
"timeout": 150,
"qmp_sequence": [
{"delay": 8, "command": "send-key", "arguments": {"keys": [{"type": "qcode", "data": c}]}}
for c in ["k", "e", "y", "t", "e", "s", "t"]
],
"expect": r"keytest",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# USB hub (B4a/B4b, docs/usb-hub.md): a USB2 hub on the xHCI bus with a
# keyboard behind it (a STATIC boot topology — no hot-plug event needed).
# B4a: the hub enumerates and powers its downstream ports. B4b extends the
# expect to the downstream keyboard binding usb-hid-keyboard.
{"name": "usb-hub",
"build_case": "usb-hid",
"smp": 4,
"timeout": 150,
# A SECOND xhci controller carries the hub topology, isolated from the boot
# controller's auto-assigned devices (whose ports the hub would collide
# with). danos spawns a second usb-xhci-bus for it. The hub sits on port 1,
# the keyboard on the hub's downstream port 1 (port=1.1).
"qemu_extra": ["-device", "qemu-xhci,id=xhci2",
"-device", "usb-hub,bus=xhci2.0,port=1",
"-device", "usb-kbd,bus=xhci2.0,port=1.1"],
# B4a: the hub powers its ports. B4b: the keyboard behind it enumerates
# (route string + TT) and binds usb-hid-keyboard.
"expect": r"(?s)(?=.*hub slot \d+: \d+ downstream ports powered)"
r"(?=.*hub slot \d+ port \d+ device:.*0x0627)"
r"(?=.*usb-hid-keyboard: ok \(device 3)",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# USB mass storage end to end: the boot usb-storage device (the FAT32 image,
# which has a real 0x55AA boot sector) is enough — the manager spawns
# usb-storage, which opens the device, runs the Bulk-Only / SCSI bring-up,
# reads its capacity, and reads block 0 (the 0x55AA boot sig). Proof of the
# bulk transfer path + BOT + SCSI end to end.
{"name": "usb-storage",
"smp": 4,
"timeout": 150,
"expect": r"usb-storage: ready[\s\S]*usb-storage: block 0 signature 0x55aa",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# FAT mount end to end: the fat server mounts the boot usb-storage device (the
# FAT32 image) into the VFS at /volumes/usb. A fat-test client then lists and reads
# through the mount — proof of the whole stack: block device -> FAT parse ->
# VFS routing -> file read.
{"name": "fat-mount",
"smp": 4,
"timeout": 150,
"expect": r"fat: mounted /volumes/usb[\s\S]*fat-test: ok",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# Phase 2b: mkdir/unlink through the mount. Reuses the fat-mount build — the
# fat-test client, after listing, makes a directory, writes+reads a file inside
# it, then removes the file, exercising the whole VFS -> fat mutation path.
{"name": "fat-mutations",
"build_case": "fat-mount",
"smp": 4,
"timeout": 150,
"expect": r"fat-test: mutations ok",
"fail": r"fat-test: mutations FAILED|fat-test: mkdir .* failed|DANOS-TEST-RESULT: FAIL"},
# Phase 2c: rename through the mount — fat-test renames the file it created
# before removing it, and confirms the old name is gone.
{"name": "fat-rename",
"build_case": "fat-mount",
"smp": 4,
"timeout": 150,
"expect": r"fat-test: rename ok",
"fail": r"fat-test: mutations FAILED|DANOS-TEST-RESULT: FAIL"},
# P4c: badge-scoped per-client ids. Two processes of one fixture, on the same
# boot: the owner holds a FAT node id and a compositor layer id, the intruder
# names both and must be refused — while its own node and layer, and the
# owner's after the attempt, keep working. Every refusal is paired with a
# control, so a provider that refused everything (or honoured everything)
# fails this case rather than passing it. Reuses the fat-mount build.
{"name": "badge-scope",
"build_case": "fat-mount",
"smp": 4,
"timeout": 150,
"expect": r"(?s)(?=.*badge-scope-test: ok)(?=.*badge-scope-test: owner intact)",
"fail": r"badge-scope-test: FAILED|DANOS-TEST-RESULT: FAIL"},
# Phase 2d: filesystem timestamps — a freshly-created file's mtime is a real
# current wall-clock time (stamped from the RTC), read back through stat.
{"name": "fat-mtime",
"build_case": "fat-mount",
"smp": 4,
"timeout": 150,
"expect": r"fat-test: mtime ok",
"fail": r"fat-test: mutations FAILED|DANOS-TEST-RESULT: FAIL"},
# Boot-from-USB smoke: the whole system now boots off the FAT32 image on a
# usb-storage device (OVMF -> \EFI\BOOT\BOOTX64.efi -> kernel), so the kernel
# reaching its PASS marker at all proves the USB boot path end to end. Reuses
# the smoke kernel build; the value is the explicit, named regression guard.
{"name": "usb-boot",
"build_case": "smoke",
"qmp_after": {"delay": 2, "command": "query-status"},
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# M20.1: the ring-3 AML parse (the acpi service maps the blobs and parses
# them) finds exactly the Device count the kernel's own parse produced.
{"name": "acpi-parse",
"smp": 4,
"timeout": 60,
"expect": r"acpi-parse: ok",
"fail": r"acpi-parse: too few|DANOS-TEST-RESULT: FAIL"},
# M20.3: the flip — ps2-bus now comes up from the acpi service's report, not
# a kernel-built node. Ordered: report -> spawn -> the driver attaches its
# keyboard, proving discovery runs entirely in ring 3 (docs/discovery.md).
{"name": "acpi-ps2",
"smp": 4,
"timeout": 150,
"expect": r"discovery: device \d+\s+bus=acpi hid=PNP0303[\s\S]*"
r"device-manager: spawned \S*ps2-bus[\s\S]*"
r"ps2-bus: keyboard driver attached",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# M21.1: the SCI + power button. Boot the manager (which spawns the acpi
# service); ~4s in, QMP system_powerdown raises the ACPI power-button fixed
# event; the service's SCI handler must log the press (docs/acpi.md).
{"name": "power-button",
"smp": 4,
"timeout": 60,
"qmp_after": {"delay": 4, "command": "system_powerdown"},
"expect": r"power: button pressed",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# M21.3 capstone: orderly shutdown. Boot init (the full tree comes up);
# ~5s in, QMP system_powerdown raises the power button; the acpi service
# publishes it, init stops its children then requests S5, and QEMU exits.
# The ordered regex proves button -> shutting-down -> entering-S5; the case
# passes on QEMU's self-exit through S5 (docs/power.md).
{"name": "orderly-shutdown",
"smp": 4,
"timeout": 90,
"qmp_after": {"delay": 5, "command": "system_powerdown"},
"expect": r"power: button pressed[\s\S]*"
r"init: shutting down[\s\S]*"
r"power: entering S5",
"fail": r"power: S5 write did not take|DANOS-TEST-RESULT: FAIL"},
# M8: the boot log is persisted to the USB FAT volume. Reuses the orderly-
# shutdown build (full tree + power button): the logger service announces its
# per-boot directory once storage mounts (first marker), then the power
# button drives the orderly stop — the logger, stopped first, final-drains
# and reports the flush (second marker) before S5.
{"name": "logger",
"build_case": "orderly-shutdown",
"smp": 4,
"timeout": 150,
"qmp_after": {"delay": 8, "command": "system_powerdown"},
"expect": r"logger: logging to /system/logs/\d{4}-\d{2}-\d{2}T\d{6}Z[\s\S]*"
r"init: shutting down[\s\S]*"
r"logger: flushed through sequence \d+[\s\S]*"
r"power: entering S5",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# M20.2: the acpi service evaluates _CRS/_STA in ring 3 and registers +
# reports its _HID devices — the two PS/2 nodes must appear with resources
# (keyboard: io 0x60/0x64 + IRQ = 3; mouse: IRQ = 1) (docs/discovery.md).
{"name": "acpi-report",
"smp": 4,
"timeout": 150,
"expect": r"discovery: device \d+\s+bus=acpi hid=PNP0303[^\n]*\(3 resources\)[\s\S]*"
r"discovery: device \d+\s+bus=acpi hid=PNP0F13[^\n]*\(1 resources\)",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# M19.1/M19.3: the ring-3 PCI scan. pci-bus walks the ECAM through its mmio_map
# grant and registers every function it finds; the kernel's own walk retired, so
# the broker starts empty and the driver populates it. The manager then runs the
# restart drill: ~1 s after the scan it kills pci-bus, prunes its child tree, and
# respawns it to re-claim, re-scan, and re-register the same functions. The kernel
# test asserts the broker equivalence (empty before, populated after, no
# duplicates); this ordered regex asserts the drill itself over the whole serial
# log — the backreference requires the respawn to re-scan the same count, and the
# full-capture match is immune to the transient-line races an in-kernel poll hits.
# The driver-side PCI library (library/device/pci) against a real function: an extra
# e1000e NIC — PM + MSI + PCIe + MSI-X capabilities, claimed by no danos driver — is
# claimed by the pci-cap-test fixture, which exercises the capability walks, MSI
# programming (the first driver-side msi_bind use), the MSI-X table, power state,
# and FLR, printing a marker per check.
{"name": "pci-caps",
"smp": 4,
"timeout": 120,
"qemu_extra": ["-device", "e1000e"],
"expect": r"(?s)(?=.*DANOS-TEST-RESULT: PASS)(?=.*pci-cap-test: all checks passed)",
"fail": r"pci-cap-test: FAIL|DANOS-TEST-RESULT: FAIL"},
{"name": "pci-scan",
"smp": 4,
"timeout": 60,
"expect": r"pci-bus: (\d+) functions found[\s\S]*"
r"device-manager: test mode: killing the reporter[\s\S]*"
r"device-manager: restarting \S*pci-bus[\s\S]*"
r"pci-bus: \1 functions found[\s\S]*"
r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# M18.3: the application surface — device-list enumerates the tree over IPC,
# subscribes (endpoint as capability), and observes the removed/added events
# the reporter's test-kill produces (docs/device-manager.md).
{"name": "device-list",
"smp": 4,
"timeout": 150,
"expect": r"device-list: \d+ devices[\s\S]*"
r"device-list: subscribed[\s\S]*"
r"device-manager: test mode: killing the reporter[\s\S]*"
r"device-list: removed \(device[\s\S]*"
r"device-list: added \(device",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# M18.1: the device manager's hello + restart policy — xHCI hellos clean and
# stays; crash-test faults, is restarted with backoff (re-claiming its device
# each time), and hits the crash-loop cap (docs/device-manager.md).
{"name": "driver-restart",
"smp": 4,
"timeout": 150,
"expect": r"usb-xhci-bus: hello acknowledged[\s\S]*"
r"device-manager: restarting crash-test[\s\S]*"
r"device-manager: crash-test is failing repeatedly",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# The initial_ramdisk: the loader ferries a bundle of user binaries; the kernel parses
# it and spawns each as a ring-3 process (here the VFS-server stub heartbeats).
{"name": "initial-ramdisk",
"timeout": 60, # the acpi service's boot-time SCI setup can push the marker past 30s under load
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# The user-space VFS: a client opens/writes/reads a file through the rt file
# API, which IPCs the VFS server process; the round trip must match.
# (A usb-hotplug case was prototyped here, but QEMU's qemu-xhci does not
# raise a runtime port-change event to a polling driver on device_add, so it
# cannot exercise the path. The hot-plug code — port-change queue, teardown
# via Disable Slot, ChildRemoved reporting — is validated on real hardware,
# flagged for the user. The qmp_sequence harness support it added remains.)
# Hub-behind-hub (B4c): route strings compose across tiers — a keyboard two
# hubs deep enumerates and binds. Static nested topology on a 2nd controller.
{"name": "usb-hub-nested",
"build_case": "usb-hid",
"smp": 4,
"timeout": 150,
"qemu_extra": ["-device", "qemu-xhci,id=xhci2",
"-device", "usb-hub,bus=xhci2.0,port=1",
"-device", "usb-hub,bus=xhci2.0,port=1.1",
"-device", "usb-kbd,bus=xhci2.0,port=1.1.1"],
"expect": r"(?s)(?=.*hub slot \d+ port \d+ device:.*0x0409)"
r"(?=.*usb-hid-keyboard: ok \(device 3)",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# Hub-downstream disconnect (B4c): device_del the keyboard behind the hub;
# the hub's status-change endpoint reports it, the device is torn down
# (ChildRemoved + Disable Slot). QEMU's hub DOES raise downstream changes
# (unlike root-port hot-plug).
{"name": "usb-hub-unplug",
"build_case": "usb-hid",
"smp": 4,
"timeout": 150,
"qemu_extra": ["-device", "qemu-xhci,id=xhci2",
"-device", "usb-hub,bus=xhci2.0,port=1",
"-device", "usb-kbd,bus=xhci2.0,port=1.1,id=dkbd"],
"qmp_after": {"delay": 8, "command": "device_del", "arguments": {"id": "dkbd"}},
"expect": r"(?s)(?=.*usb-hid-keyboard: ok \(device 3)"
r"(?=.*slot \d+ disconnected)",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# The kernel VFS root (M-F): the mount table serves the initrd at /system —
# path resolution, node status/read (an ELF magic), and directory listing,
# asserted kernel-side.
{"name": "kvfs",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
{"name": "vfs",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# The input service: a synthetic keyboard source publishes events, the service
# broadcasts them over the async ipc_send primitive, and a subscriber (which joined by
# passing its endpoint as a capability) receives them — source -> service -> subscriber.
{"name": "input",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# The protocol registry (docs/os-development/protocol-namespace.md): init serves
# /protocol, and the fixture drives the registrar's whole contract — an ungranted
# bind refused (-EPERM), the kernel's reserved prefix holding against mount and
# unmount, a live owner's name refused (-EBUSY), and a killed provider's channel
# failing while a re-resolve reaches the restarted instance. Each step prints its
# own line, so a failure says which rule broke, not merely that one did.
# Three of the steps are the registrar's security contract, and each fails
# catastrophically rather than quietly if it regresses: a forged power event
# posted to PID 1's mailbox must not shut the machine down (a regression ends
# the boot), capability-carrying zero-length pings must not consume PID 1's
# handle table (a regression makes every later bind impossible), and a granted
# binary spawned by an unauthorized task must be refused while the same binary
# spawned by an authorized one is not (the laundering deputy).
{"name": "protocol-registry",
"expect": r"(?s)(?=.*protocol-registry: ungranted bind refused)"
r"(?=.*protocol-registry: /protocol reserved)"
r"(?=.*protocol-registry: forged power event ignored)"
r"(?=.*protocol-registry: capability-carrying pings did not exhaust the registrar)"
r"(?=.*protocol-registry: collision refused)"
r"(?=.*protocol-registry: dead channel refused)"
r"(?=.*protocol-registry: restarted provider reached)"
r"(?=.*protocol-registry: laundering deputy refused)"
r"(?=.*protocol-registry: foreign signal binding refused)"
r"(?=.*protocol-registry: foreign timer and exit binding refused)(?=.*protocol-registry: foreign receive refused)"
r"(?=.*protocol-registry: capability-carrying pings did not exhaust the harness)"
r"(?=.*DANOS-TEST-RESULT: PASS)",
"fail": r"DANOS-TEST-RESULT: FAIL|protocol-registry: FAIL"},
# Restriction stage one (docs/os-development/protocol-namespace.md): the
# registrar checks `open` against /system/configuration/protocol.csv, and a
# caller with no grant gets the same answer as a caller naming a contract
# nobody bound. The scenario boots /protocol plus the input service, so the
# forbidden name is genuinely BOUND — the fixture reads the namespace listing
# to prove it — and then compares the refusal with an unbound name field by
# field: status, node, payload length, the whole reply packet, and the
# presence of a capability. All three failure shapes (refused-and-bound,
# granted-and-unbound, neither) must collapse into one answer.
{"name": "protocol-denied",
"expect": r"(?s)(?=.*protocol-denied: granted open succeeded)"
r"(?=.*protocol-denied: ungranted open refused as absent)"
r"(?=.*protocol-denied: refusal is indistinguishable from absence)"
r"(?=.*protocol-denied: ok)"
r"(?=.*DANOS-TEST-RESULT: PASS)",
"fail": r"DANOS-TEST-RESULT: FAIL|protocol-denied: FAIL"},
# The reserved verbs, asked of live providers (P4a). Every protocol built on
# envelope.Define answers `describe` out of its specification and `-ENOSYS`
# for a verb it does not define, without its provider implementing either —
# so a fixture that walks /protocol's own listing and asks both of whatever
# it finds is the proof that `Define` hands those verbs to everyone alike.
# The scenario boots the registry, the input service and the compositor: two
# protocols of different sizes and verb counts, both reached through the real
# registry under the manifest's own grants. The other six the fixture knows —
# vfs, block, scanout, device-manager, power and usb-transfer — need provider
# chains this case does not boot (the last three all arrive with the device
# manager, i.e. with the whole driver tree, whose timing would make this
# fixture's one namespace snapshot a boot race). It names them as unchecked
# rather than skipping them quietly, and the fat-mount / usb-storage /
# virtio-gpu / device-list / driver-restart / pci-scan / usb-* /
# orderly-shutdown scenarios are their proof.
{"name": "protocol-conformance",
"expect": r"(?s)(?=.*protocol-conformance: input v\d+ describes itself)"
r"(?=.*protocol-conformance: display v\d+ describes itself)"
r"(?=.*-> -ENOSYS)"
r"(?=.*protocol-conformance: 2 provider\(s\) answered the reserved verbs identically)"
r"(?=.*protocol-conformance: ok)"
r"(?=.*DANOS-TEST-RESULT: PASS)",
"fail": r"DANOS-TEST-RESULT: FAIL|protocol-conformance: FAIL"},
# Device manager: a ring-3 service enumerates /system/devices, matches the PCI host
# bridge to pci-bus, and spawns it — end-to-end proof of discover -> match -> spawn
# -> driver-up (the spawned pci-bus logs "<N> functions found").
{"name": "device-manager",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# device_register containment (in-kernel): registering a child whose MMIO window
# escapes its parent's grant is refused (NotContained) — else dev_register would map
# arbitrary physical memory — while an identical re-register stays idempotent.
{"name": "containment",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# IRQ teardown: an exiting driver's line is masked and its slot cleared (so no
# ISR notifies a freed endpoint), and a sibling owner sharing that endpoint
# keeps its own binding. A long-running driver never reaches this teardown path.
{"name": "irqfree",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# The MMIO-grant teardown fix: a device-granted frame must not be reclaimed
# as RAM when its address space is destroyed.
{"name": "iopass",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# The ACPI power path succeeds by QEMU *exiting* (S5 off / reset), so match the
# pre-transition marker; the FAIL line only appears if the transition didn't take.
# Soft-off (S5) is owned by the ring-3 acpi service now (see orderly-shutdown);
# the kernel keeps only reboot (FADT reset register, no AML).
{"name": "reboot",
"expect": r"DANOS-POWER: attempting reboot",
"fail": r"DANOS-TEST-RESULT: FAIL"},
]
TIMEOUT = 30 # seconds per case
def build(arch, case):
# -Dserial: the harness asserts on markers the kernel writes to serial0, so the
# serial log sink must be compiled in. It is off by default (a flashed real-
# hardware image keeps its log in RAM instead; see build.zig / serial.zig).
cmd = ["zig", "build", f"-Dtest-case={case}", "-Dserial=true"] + arch["zig_flags"]
r = subprocess.run(cmd, cwd=REPO, capture_output=True, text=True)
if r.returncode != 0:
return r.stderr.strip() or r.stdout.strip()
return None
def resolve_firmware(arch):
"""Collapse the ovmf_code/ovmf_vars candidate lists to the first path that
exists on this machine. Mutates `arch` in place; idempotent (a resolved
string is left untouched). Exits with a clear message if none is found."""
for key, label in (("ovmf_code", "OVMF_CODE"), ("ovmf_vars", "OVMF_VARS")):
val = arch[key]
if isinstance(val, str):
continue # already resolved on a previous call
for cand in val:
if os.path.exists(cand):
arch[key] = cand
break
else:
sys.exit(f"error: no {label} firmware image found; looked in:\n "
+ "\n ".join(val)
+ "\nInstall OVMF (edk2-ovmf / ovmf) or add its path above.")
def qmp_send(path, command, arguments=None):
"""One QMP command: connect, capabilities handshake, execute. Raises on any
failure — the caller retries until the guest's socket is ready. This is how
a case injects a host-side event into the running guest: system_powerdown
(the ACPI power button, docs/power.md) or device_add/device_del (USB
hot-plug, docs/driver-model.md)."""
sock = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM)
sock.settimeout(5)
try:
sock.connect(path)
stream = sock.makefile("rw")
stream.readline() # the QMP greeting
stream.write(json.dumps({"execute": "qmp_capabilities"}) + "\n")
stream.flush()
stream.readline() # {"return": {}}
message = {"execute": command}
if arguments:
message["arguments"] = arguments
stream.write(json.dumps(message) + "\n")
stream.flush()
stream.readline()
finally:
sock.close()
def run_case(arch, case):
# A case's kernel build defaults to its name; `build_case` decouples the two
# so a case can reuse another's kernel (e.g. usb-boot reuses smoke's).
err = build(arch, case.get("build_case", case["name"]))
if err:
return False, "build failed:\n" + err
# The bootable FAT32 USB image the build produced (tools/make-fat-image.py),
# presented to the guest as a usb-storage device (see qemu_args).
# Boot a per-run COPY of the image: the guest MUTATES its boot volume (the
# fat tests create/delete files; the logger writes /system/logs), and QEMU is
# hard-killed after a match — booting the build artifact in place let one
# run's leftovers fail the next (a stale TESTDIR trips the mkdir-duplicate
# refusal) and dirtied the build cache's own output.
built_volume = os.path.join(REPO, "zig-out", "danos-usb.img")
boot_volume = os.path.join(WORK, "boot-volume.img")
shutil.copy(built_volume, boot_volume)
vars_fd = os.path.join(WORK, "vars.fd")
shutil.copy(arch["ovmf_vars"], vars_fd)
serial = os.path.join(WORK, "serial.log")
if os.path.exists(serial):
os.remove(serial)
expect = re.compile(case["expect"])
fail = re.compile(case["fail"]) if case.get("fail") else None
cmd = [arch["qemu"]] + arch["qemu_args"](arch, boot_volume, vars_fd, serial)
if case.get("smp"): # some cases need more than one core (e.g. parallelism)
cmd += ["-smp", str(case["smp"])]
if case.get("mem"): # a case that boots the whole system at once needs more than the 128M floor
cmd[cmd.index("-m") + 1] = case["mem"]
if case.get("qemu_extra"): # extra qemu args, e.g. -device intel-iommu for the IOMMU case
cmd += case["qemu_extra"]
# A QMP control socket, always present (additive): how a case's `qmp_after`
# hook injects host-side events into the guest mid-run. Kept under a short temp
# dir, not WORK: a unix socket path is capped at ~104 bytes (sun_path), and a
# deep worktree path (e.g. .claude/worktrees/<name>/zig-out/qemu-test/qmp.sock)
# blows that limit on macOS, so QEMU fails to bind and exits before booting.
qmp_path = os.path.join(tempfile.gettempdir(), f"danos-qmp-{os.getpid()}.sock")
if os.path.exists(qmp_path):
os.remove(qmp_path)
cmd += ["-qmp", f"unix:{qmp_path},server,nowait"]
# Hooks: a single qmp_after {"delay","command"} or a qmp_sequence list of
# {"delay","command","arguments"} — every hook must deliver before a pass.
qmp_hooks = case.get("qmp_sequence") or ([case["qmp_after"]] if case.get("qmp_after") else [])
qmp_pending = [dict(hook, sent=False) for hook in qmp_hooks]
started = time.monotonic()
qemu = subprocess.Popen(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
try:
timeout = case.get("timeout", TIMEOUT)
deadline = time.monotonic() + timeout
while time.monotonic() < deadline:
time.sleep(0.2)
for hook in qmp_pending:
if not hook["sent"] and time.monotonic() - started >= hook["delay"]:
try:
qmp_send(qmp_path, hook["command"], hook.get("arguments"))
hook["sent"] = True
except OSError:
pass # socket not up yet; retry next tick
text = ""
if os.path.exists(serial):
with open(serial, "r", errors="replace") as f:
text = f.read()
if fail and fail.search(text):
return False, "hit failure marker"
if expect.search(text):
if any(not hook["sent"] for hook in qmp_pending):
continue # every hook must deliver before the case may pass
return True, "matched " + repr(case["expect"])
if qemu.poll() is not None: # QEMU exited on its own
if expect.search(text):
return True, "matched " + repr(case["expect"])
return False, "QEMU exited before matching (triple fault?)"
return False, f"timed out after {timeout}s without matching {case['expect']!r}"
finally:
if qemu.poll() is None:
qemu.terminate()
try:
qemu.wait(timeout=5)
except subprocess.TimeoutExpired:
qemu.kill()
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--arch", default="x86_64", choices=sorted(ARCHES))
ap.add_argument("cases", nargs="*", help="case names to run (default: all)")
args = ap.parse_args()
arch = ARCHES[args.arch]
if not shutil.which(arch["qemu"]):
print(f"error: {arch['qemu']} not found on PATH", file=sys.stderr)
return 2
resolve_firmware(arch)
selected = [c for c in CASES if not args.cases or c["name"] in args.cases]
os.makedirs(WORK, exist_ok=True)
print(f"danos qemu tests arch={args.arch} cases={len(selected)}\n")
failures = 0
for case in selected:
print(f" {case['name']:<12} ... ", end="", flush=True)
ok, detail = run_case(arch, case)
if not ok:
# Keep the evidence: serial.log is otherwise overwritten by the
# next case, and an intermittent failure's log is unrecoverable.
source = os.path.join(WORK, "serial.log")
if os.path.exists(source):
shutil.copy(source, os.path.join(WORK, f"{case['name']}-failed-serial.log"))
print(("PASS" if ok else "FAIL") + f" ({detail})")
if not ok:
failures += 1
print()
total = len(selected)
print(f"{total - failures}/{total} passed")
return 1 if failures else 0
if __name__ == "__main__":
sys.exit(main())