M10: IO passthrough (MMIO grants) + first real driver (hpetd)

A user-space process can now touch real hardware directly, capability-gated by
the device tree — the microkernel driver model.

- src/kernel/devsvc.zig: flattens the discovered device tree into an
  id-indexed snapshot + a claim table at boot (devsvc.init from main.zig).
- Syscalls 11-13: dev_enumerate (snapshot the table), dev_claim (take
  exclusive ownership), mmio_map (map a claimed device's MMIO window into the
  caller's AS and return the register base). The claim is the capability:
  mmio_map refuses any device the caller doesn't own.
- paging.mapUserDeviceInto: maps device MMIO strong-uncacheable (PCD|PWT) and
  marks each leaf with a device_grant PTE bit; freeSubtree skips pmm.free on
  those leaves, so tearing down a driver never returns MMIO frames to the RAM
  pool (the teardown hazard). MMIO grants live in a distinct arena, PML4[226]
  (Task.dev_map_next), so device pages widen no kernel mapping.
- lib/dev.zig: user enumerate/claim/mmioMap wrappers; shared DeviceDesc/ResDesc
  in danos (root.zig). sbin/hpetd.zig: finds the HPET, claims it, maps its
  registers, enables the counter (an MMIO write) and reads it (0xF0) — proving
  read+write passthrough to real hardware.
- Tests: `hpet` (driver reads the counter advancing from ring 3) and `iopass`
  (device-granted frame survives address-space teardown). Suite 33/33.

irq_bind/irq_ack (IRQ-as-message) are stubbed (-1) pending; notifyFromIsr (M7)
is the hook they'll use.
This commit is contained in:
Daniel Samson
2026-07-09 08:03:21 +01:00
parent b0f894f50c
commit 83881641ca
14 changed files with 449 additions and 9 deletions
+3
View File
@@ -219,6 +219,7 @@ pub fn build(b: *std.Build) void {
// which unpacks it and spawns each program (src/user/proto/initrd.zig). // which unpacks it and spawns each program (src/user/proto/initrd.zig).
const vfs_exe = addUserBinary(b, kernel_target, rt_mod, "vfs", "sbin/vfs.zig"); const vfs_exe = addUserBinary(b, kernel_target, rt_mod, "vfs", "sbin/vfs.zig");
const vfstest_exe = addUserBinary(b, kernel_target, rt_mod, "vfstest", "sbin/vfstest.zig"); const vfstest_exe = addUserBinary(b, kernel_target, rt_mod, "vfstest", "sbin/vfstest.zig");
const hpetd_exe = addUserBinary(b, kernel_target, rt_mod, "hpetd", "sbin/hpetd.zig");
// Pack the user binaries into the initrd image with the host-side Python tool // Pack the user binaries into the initrd image with the host-side Python tool
// (the container format is trivial, and Python sidesteps std API churn). Args: // (the container format is trivial, and Python sidesteps std API churn). Args:
@@ -230,6 +231,8 @@ pub fn build(b: *std.Build) void {
mk_run.addFileArg(vfs_exe.getEmittedBin()); mk_run.addFileArg(vfs_exe.getEmittedBin());
mk_run.addArg("vfstest"); mk_run.addArg("vfstest");
mk_run.addFileArg(vfstest_exe.getEmittedBin()); mk_run.addFileArg(vfstest_exe.getEmittedBin());
mk_run.addArg("hpetd");
mk_run.addFileArg(hpetd_exe.getEmittedBin());
// Install the image to zig-out/bin (so the QEMU test harness picks it up like // Install the image to zig-out/bin (so the QEMU test harness picks it up like
// the other binaries). The run-x86-64 ESP install is added below. // the other binaries). The run-x86-64 ESP install is added below.
+32
View File
@@ -0,0 +1,32 @@
//! User-space device access: enumerate the kernel's device table, claim a
//! device, and map its MMIO. A driver uses these to find and take ownership of
//! its hardware; the claim is the capability the kernel checks before mapping.
const danos = @import("danos");
const sc = @import("syscall.zig");
pub const DeviceDesc = danos.DeviceDesc;
pub const ResDesc = danos.ResDesc;
pub const DeviceClass = danos.DeviceClass;
pub const ResourceKind = danos.ResourceKind;
inline fn failed(r: usize) bool {
return r > ~@as(usize, 0) - 4095;
}
/// Copy up to `buf.len` device descriptors into `buf`; returns the total count.
pub fn enumerate(buf: []DeviceDesc) usize {
return sc.syscall2(.dev_enumerate, @intFromPtr(buf.ptr), buf.len);
}
/// Take exclusive ownership of device `id`. Returns false if taken or invalid.
pub fn claim(id: u64) bool {
return !failed(sc.syscall1(.dev_claim, id));
}
/// Map resource `res_idx` (which must be an MMIO window) of claimed device
/// `dev_id` into this address space; returns the register base virtual address.
pub fn mmioMap(dev_id: u64, res_idx: u64) ?usize {
const r = sc.syscall2(.mmio_map, dev_id, res_idx);
return if (failed(r)) null else r;
}
+2
View File
@@ -20,6 +20,8 @@ pub const vfsproto = @import("vfs_proto.zig");
pub const unistd = @import("unistd.zig"); pub const unistd = @import("unistd.zig");
/// C stdio: fopen/fread/fwrite/fseek/ftell/fclose over unistd. /// C stdio: fopen/fread/fwrite/fseek/ftell/fclose over unistd.
pub const stdio = @import("stdio.zig"); pub const stdio = @import("stdio.zig");
/// Device access for drivers: enumerate/claim/mmioMap.
pub const dev = @import("dev.zig");
/// Re-exported so a user binary can `pub const panic = rt.panic;`. /// Re-exported so a user binary can `pub const panic = rt.panic;`.
pub const panic = start.panic; pub const panic = start.panic;
+75
View File
@@ -0,0 +1,75 @@
//! /sbin/hpetd — a user-space HPET driver, the first real device driver. It
//! proves IO passthrough end to end: enumerate the device table, find the HPET
//! (a timer with an MMIO window), claim it, map its registers directly into this
//! ring-3 address space (strong-uncacheable), then drive the hardware — enable
//! the main counter and read it. If the counter advances, a user process is
//! touching real hardware through a kernel-granted MMIO mapping.
//!
//! Register offsets (HPET spec): general config = 0x10 (bit 0 = ENABLE),
//! main counter = 0xF0.
const rt = @import("rt");
const dev = rt.dev;
pub fn main() void {
// Enumerate into a heap buffer (too big for the one-page user stack).
const buf = rt.allocator().alloc(dev.DeviceDesc, 32) catch {
_ = rt.sys.write("hpetd: out of memory\n");
return;
};
const total = dev.enumerate(buf);
const n = @min(total, buf.len);
// Find a timer-class device with an MMIO resource (the HPET).
var dev_id: u64 = 0;
var res_idx: u64 = 0;
var found = false;
var i: usize = 0;
outer: while (i < n) : (i += 1) {
const d = buf[i];
if (d.class != @intFromEnum(dev.DeviceClass.timer)) continue;
var j: usize = 0;
while (j < d.resource_count) : (j += 1) {
if (d.resources[j].kind == @intFromEnum(dev.ResourceKind.memory)) {
dev_id = d.id;
res_idx = j;
found = true;
break :outer;
}
}
}
if (!found) {
_ = rt.sys.write("hpetd: no HPET found\n");
return;
}
if (!dev.claim(dev_id)) {
_ = rt.sys.write("hpetd: claim failed\n");
return;
}
const base = dev.mmioMap(dev_id, res_idx) orelse {
_ = rt.sys.write("hpetd: mmio_map failed\n");
return;
};
// Drive the hardware: enable the counter (an MMIO write), then read it twice.
const config: *volatile u64 = @ptrFromInt(base + 0x10);
config.* |= 1; // ENABLE
const counter: *volatile u64 = @ptrFromInt(base + 0xF0);
const a = counter.*;
rt.sys.sleep(50);
const b = counter.*;
if (b > a) {
while (true) {
_ = rt.sys.write("hpetd: ok\n");
rt.sys.sleep(1000);
}
}
_ = rt.sys.write("hpetd: counter stuck\n");
}
pub const panic = rt.panic;
comptime {
_ = &rt.start._start;
}
+2
View File
@@ -18,6 +18,8 @@ const devicetree = @import("devicetree.zig");
pub const DeviceTree = device.DeviceTree; pub const DeviceTree = device.DeviceTree;
pub const Device = device.Device; pub const Device = device.Device;
pub const DeviceClass = device.DeviceClass; pub const DeviceClass = device.DeviceClass;
pub const Resource = device.Resource;
pub const ResourceKind = device.ResourceKind;
pub const Hal = device.Hal; pub const Hal = device.Hal;
pub const PowerInfo = acpi.PowerInfo; pub const PowerInfo = acpi.PowerInfo;
pub const AmlStats = acpi.AmlStats; pub const AmlStats = acpi.AmlStats;
+6
View File
@@ -154,6 +154,12 @@ pub fn mapUserPageInto(root: u64, virt: u64, phys: u64, writable: bool, executab
paging.mapUserInto(root, virt, phys, writable, executable); paging.mapUserInto(root, virt, phys, writable, executable);
} }
/// Map a device MMIO window into address space `root`: strong-uncacheable, RW+NX,
/// and marked so teardown won't free the MMIO frames as RAM. For IO passthrough.
pub fn mapUserDeviceInto(root: u64, virt: u64, phys: u64, len: u64) void {
paging.mapUserDeviceInto(root, virt, phys, len);
}
/// Map a page into the kernel address space (non-executable). For the heap, etc. /// Map a page into the kernel address space (non-executable). For the heap, etc.
pub fn mapPage(virt: u64, phys: u64, writable: bool) void { pub fn mapPage(virt: u64, phys: u64, writable: bool) void {
paging.map(virt, phys, writable); paging.map(virt, phys, writable);
+36 -2
View File
@@ -19,6 +19,9 @@ const page_size = danos.page_size;
const present: u64 = 1 << 0; const present: u64 = 1 << 0;
const writable: u64 = 1 << 1; const writable: u64 = 1 << 1;
const user: u64 = 1 << 2; // U/S: accessible from ring 3 (must be set at every level) const user: u64 = 1 << 2; // U/S: accessible from ring 3 (must be set at every level)
const pwt: u64 = 1 << 3; // page write-through
const pcd: u64 = 1 << 4; // page cache disable (with PWT: strong-uncacheable under the default PAT)
const device_grant: u64 = 1 << 9; // available bit: this leaf maps device MMIO, not RAM — do not reclaim
const no_execute: u64 = 1 << 63; const no_execute: u64 = 1 << 63;
const addr_mask: u64 = 0x000F_FFFF_FFFF_F000; const addr_mask: u64 = 0x000F_FFFF_FFFF_F000;
@@ -244,6 +247,31 @@ pub fn mapUserInto(pml4: u64, virt: u64, phys: u64, writable_page: bool, executa
invalidate(virt); invalidate(virt);
} }
/// Map a device MMIO window `[phys, phys+len)` into the user (low) half of the
/// address space rooted at `pml4`, page by page. Unlike `mapUserInto` these pages
/// are **strong-uncacheable** (PCD|PWT — device registers must not be cached) and
/// carry the `device_grant` bit so teardown does not return the MMIO frames to the
/// RAM allocator (`freeSubtree`). RW + NX; the caller places `virt` in a
/// user-exclusive range (PML4[225]). Both `virt` and `phys` are page-aligned by
/// the caller; a sub-page `phys` offset is the caller's to re-apply.
pub fn mapUserDeviceInto(pml4: u64, virt: u64, phys: u64, len: u64) void {
const flags: u64 = present | user | writable | no_execute | pcd | pwt | device_grant;
const first = phys & ~@as(u64, page_size - 1);
const last = (phys + (if (len == 0) 1 else len) - 1) & ~@as(u64, page_size - 1);
var off: u64 = 0;
while (first + off <= last) : (off += page_size) {
const v = virt + off;
const pml4e = &tableAt(pml4)[(v >> 39) & 0x1FF];
const pdpt = descendUser(pml4e);
const pdpte = &tableAt(pdpt)[(v >> 30) & 0x1FF];
const pd = descendUser(pdpte);
const pde = &tableAt(pd)[(v >> 21) & 0x1FF];
const pt = descendUser(pde);
tableAt(pt)[(v >> 12) & 0x1FF] = ((first + off) & addr_mask) | flags;
invalidate(v);
}
}
/// Create a new address space: a fresh PML4 with an empty user half and the /// Create a new address space: a fresh PML4 with an empty user half and the
/// kernel's higher half shared in (copying PML4[256..512), whose entries point /// kernel's higher half shared in (copying PML4[256..512), whose entries point
/// at the kernel's PDPTs — pre-created at init and never restaled, so growth in /// at the kernel's PDPTs — pre-created at init and never restaled, so growth in
@@ -274,9 +302,15 @@ fn freeSubtree(phys: u64, level: u32) void {
const t = tableAt(phys); const t = tableAt(phys);
for (t) |e| { for (t) |e| {
if (e & present == 0) continue; if (e & present == 0) continue;
if (level > 1) freeSubtree(e & addr_mask, level - 1) else free_frame(e & addr_mask); if (level > 1) {
freeSubtree(e & addr_mask, level - 1);
} else if (e & device_grant == 0) {
// A device-grant leaf points at MMIO, not RAM — returning it to the
// frame allocator would corrupt the pool. Only reclaim real RAM.
free_frame(e & addr_mask);
}
} }
free_frame(phys); free_frame(phys); // page-table frames are always real RAM
} }
/// Whether `virt` is currently mapped **executable** — present with the NX bit /// Whether `virt` is currently mapped **executable** — present with the NX bit
+78
View File
@@ -0,0 +1,78 @@
//! Device service: the kernel side of user-space driver access. At boot it
//! flattens the discovered device tree (src/device) into a stable, id-indexed
//! snapshot and a per-device claim table. User drivers enumerate the snapshot,
//! claim the device they own, and map its MMIO — the claim is the capability that
//! gates `mmio_map`/`irq_bind`, so a process can only ever touch hardware the
//! firmware-neutral device tree says it owns.
const std = @import("std");
const platform = @import("platform");
const danos = @import("danos");
const max_devices = 32;
var devices: [max_devices]danos.DeviceDesc = undefined;
var claimed: [max_devices]?u32 = .{null} ** max_devices; // owner task id, or null
var count: usize = 0;
/// Snapshot the device tree into the flat table. Run once, right after discovery.
pub fn init(dt: *const platform.DeviceTree) void {
count = 0;
for (&claimed) |*c| c.* = null;
walk(dt.root);
}
fn walk(node: *platform.Device) void {
if (node.class != .root) record(node);
var child = node.first_child;
while (child) |c| : (child = c.next_sibling) walk(c);
}
fn record(node: *platform.Device) void {
if (count >= max_devices) return;
var d = std.mem.zeroes(danos.DeviceDesc);
d.id = count;
d.class = @intFromEnum(node.class);
const h = node.hid();
d.hid_len = @min(h.len, d.hid.len);
@memcpy(d.hid[0..d.hid_len], h[0..d.hid_len]);
const rc = @min(node.resource_count, danos.max_dev_resources);
d.resource_count = rc;
for (0..rc) |i| {
const r = node.resources[i];
d.resources[i] = .{ .kind = @intFromEnum(r.kind), .start = r.start, .len = r.len };
}
devices[count] = d;
count += 1;
}
/// Copy up to `out.len` device descriptors into `out`; returns the total count
/// available (which may exceed `out.len`).
pub fn enumerate(out: []danos.DeviceDesc) usize {
const n = @min(count, out.len);
@memcpy(out[0..n], devices[0..n]);
return count;
}
/// Take exclusive ownership of device `id` for task `owner`. Fails if the id is
/// out of range or already claimed.
pub fn claim(id: u64, owner: u32) bool {
if (id >= count) return false;
if (claimed[@intCast(id)] != null) return false;
claimed[@intCast(id)] = owner;
return true;
}
/// The task that owns device `id`, or null.
pub fn ownerOf(id: u64) ?u32 {
if (id >= count) return null;
return claimed[@intCast(id)];
}
/// Resource `idx` of device `id`, or null if out of range.
pub fn resourceOf(id: u64, idx: u64) ?danos.ResDesc {
if (id >= count) return null;
const d = &devices[@intCast(id)];
if (idx >= d.resource_count) return null;
return d.resources[@intCast(idx)];
}
+5
View File
@@ -8,6 +8,7 @@ const pmm = @import("pmm.zig");
const heap = @import("heap.zig"); const heap = @import("heap.zig");
const scheduler = @import("scheduler.zig"); const scheduler = @import("scheduler.zig");
const process = @import("process.zig"); const process = @import("process.zig");
const devsvc = @import("devsvc.zig");
const initrd = @import("initrd"); const initrd = @import("initrd");
const platform = @import("platform"); const platform = @import("platform");
const tests = @import("tests.zig"); const tests = @import("tests.zig");
@@ -157,6 +158,10 @@ fn kmain(boot_info: *const BootInfo) noreturn {
log.write("\ndanos: device discovery online\n"); log.write("\ndanos: device discovery online\n");
dt.dump(log.write); dt.dump(log.write);
// Snapshot the device tree for user-space drivers (dev_enumerate/claim/
// mmio_map operate on this flat, id-indexed table + claim map).
devsvc.init(&dt);
// Power register map extracted from the FADT + AML, for confidence it parsed. // Power register map extracted from the FADT + AML, for confidence it parsed.
const pw = platform.powerInfo(); const pw = platform.powerInfo();
log.write("danos: power\n"); log.write("danos: power\n");
+59
View File
@@ -27,6 +27,7 @@ const pmm = @import("pmm.zig");
const sched = @import("scheduler.zig"); const sched = @import("scheduler.zig");
const sync = @import("sync.zig"); const sync = @import("sync.zig");
const ipc = @import("ipc_sync.zig"); const ipc = @import("ipc_sync.zig");
const devsvc = @import("devsvc.zig");
const log = @import("log.zig"); const log = @import("log.zig");
const page_size = danos.page_size; const page_size = danos.page_size;
@@ -50,6 +51,13 @@ pub const heap_arena_end: u64 = heap_arena_base + (1 << 30);
/// used to bound the addresses a syscall will dereference on the caller's behalf. /// used to bound the addresses a syscall will dereference on the caller's behalf.
pub const user_half_end: u64 = 0x0000_8000_0000_0000; pub const user_half_end: u64 = 0x0000_8000_0000_0000;
/// The MMIO-grant arena: where `mmio_map` places device windows, in PML4[226] —
/// a user-exclusive region distinct from code/stack/heap (PML4[224]), so mapping
/// device pages user-accessible widens no kernel mapping. Per-process cursor in
/// `Task.dev_map_next`.
pub const dev_arena_base: u64 = 0x0000_7100_0000_0000;
pub const dev_arena_end: u64 = dev_arena_base + (4 << 30);
/// Largest single `mmap` grant, in pages (1 MiB). The user heap grows in small /// Largest single `mmap` grant, in pages (1 MiB). The user heap grows in small
/// chunks, so this bound is generous; it also caps the frame scratch array below. /// chunks, so this bound is generous; it also caps the frame scratch array below.
const max_mmap_pages = 256; const max_mmap_pages = 256;
@@ -114,6 +122,11 @@ fn syscall(state: *arch.CpuState) void {
.ipc_lookup => sysIpcLookup(state), .ipc_lookup => sysIpcLookup(state),
.ipc_call => sysIpcCall(state), .ipc_call => sysIpcCall(state),
.ipc_reply_wait => sysIpcReplyWait(state), .ipc_reply_wait => sysIpcReplyWait(state),
.dev_enumerate => sysDevEnumerate(state),
.dev_claim => sysDevClaim(state),
.mmio_map => sysMmioMap(state),
// irq_bind/irq_ack (IRQ-as-message) land with the driver that needs them.
.irq_bind, .irq_ack => fail(state),
_ => fail(state), _ => fail(state),
} }
} }
@@ -175,6 +188,52 @@ fn sysIpcReplyWait(state: *arch.CpuState) void {
arch.setSyscallResult2(state, badge); arch.setSyscallResult2(state, badge);
} }
/// dev_enumerate(buf, max) -> total: snapshot the device table into the caller's
/// buffer (up to `max` entries), returning the total device count.
fn sysDevEnumerate(state: *arch.CpuState) void {
const buf_ptr = arch.syscallArg(state, 0);
const max = arch.syscallArg(state, 1);
const t = sched.cur();
if (t.aspace == 0 or buf_ptr >= user_half_end) return fail(state);
const sz = @sizeOf(danos.DeviceDesc);
const cap = @min(max, (user_half_end - buf_ptr) / sz); // clamp to the user half
const out: [*]danos.DeviceDesc = @ptrFromInt(buf_ptr);
arch.setSyscallResult(state, devsvc.enumerate(out[0..@intCast(cap)]));
}
/// dev_claim(id) -> 0/-1: take exclusive ownership of a device for this process.
fn sysDevClaim(state: *arch.CpuState) void {
if (devsvc.claim(arch.syscallArg(state, 0), sched.cur().id))
arch.setSyscallResult(state, 0)
else
fail(state);
}
/// mmio_map(dev_id, res_idx) -> vaddr: map a claimed device's MMIO window into
/// this address space (strong-uncacheable) and return the register base address.
/// The claim is the capability — a process can only map hardware it owns.
fn sysMmioMap(state: *arch.CpuState) void {
const dev_id = arch.syscallArg(state, 0);
const res_idx = arch.syscallArg(state, 1);
const t = sched.cur();
if (t.aspace == 0) return fail(state);
const owner = devsvc.ownerOf(dev_id) orelse return fail(state);
if (owner != t.id) return fail(state); // not claimed by this process
const r = devsvc.resourceOf(dev_id, res_idx) orelse return fail(state);
if (r.kind != @intFromEnum(danos.ResourceKind.memory)) return fail(state);
if (t.dev_map_next == 0) t.dev_map_next = dev_arena_base;
const first = r.start & ~@as(u64, page_size - 1);
const last = (r.start + r.len - 1) & ~@as(u64, page_size - 1);
const pages = (last - first) / page_size + 1;
const base_v = t.dev_map_next;
if (base_v + pages * page_size > dev_arena_end) return fail(state);
arch.mapUserDeviceInto(t.aspace, base_v, r.start, r.len);
t.dev_map_next = base_v + pages * page_size;
arch.setSyscallResult(state, base_v + (r.start & (page_size - 1))); // register base
}
/// debug_write(ptr, len): copy bytes from user memory into the kernel log. /// debug_write(ptr, len): copy bytes from user memory into the kernel log.
/// A bring-up diagnostic — real output goes through the VFS/console later. /// A bring-up diagnostic — real output goes through the VFS/console later.
/// ///
+3
View File
@@ -50,6 +50,9 @@ pub const Task = struct {
// process.zig lazily seeds it to the arena base on the first mmap). Bumped up // process.zig lazily seeds it to the arena base on the first mmap). Bumped up
// as the user heap grows; user task only. // as the user heap grows; user task only.
heap_next: u64 = 0, heap_next: u64 = 0,
// Next free virtual address in this task's MMIO-grant arena (PML4[226]; 0 =
// uninitialised, process.zig seeds it on the first mmio_map). User task only.
dev_map_next: u64 = 0,
// --- synchronous IPC (ipc_sync.zig) --- // --- synchronous IPC (ipc_sync.zig) ---
// Per-process handle table: small-int handle -> *ipc_sync.Endpoint, kept // Per-process handle table: small-int handle -> *ipc_sync.Endpoint, kept
// opaque here so the scheduler and IPC modules don't import each other. // opaque here so the scheduler and IPC modules don't import each other.
+91 -7
View File
@@ -109,6 +109,10 @@ pub fn run(case: []const u8, boot_info: *const BootInfo) void {
initrdTest(boot_info); initrdTest(boot_info);
} else if (eql(case, "vfs")) { } else if (eql(case, "vfs")) {
vfsTest(boot_info); vfsTest(boot_info);
} else if (eql(case, "hpet")) {
hpetTest(boot_info);
} else if (eql(case, "iopass")) {
ioPassTest();
} else if (eql(case, "poweroff")) { } else if (eql(case, "poweroff")) {
powerTest(.off); powerTest(.off);
} else if (eql(case, "reboot")) { } else if (eql(case, "reboot")) {
@@ -1004,13 +1008,10 @@ fn vfsTest(boot_info: *const BootInfo) void {
process.write_count = 0; process.write_count = 0;
process.write_from_user = false; process.write_from_user = false;
var i: u32 = 0; // Spawn just the server and its client (other initrd binaries would write to
while (i < rd.count) : (i += 1) { // the shared evidence buffer and confuse the marker check).
const item = rd.entry(i) orelse continue; _ = spawnNamed(rd, "vfs");
process.spawnProcess(item.blob, 4) catch |err| { _ = spawnNamed(rd, "vfstest");
log("DANOS-VFS-ERR: {s}: {s}\n", .{ item.name, @errorName(err) });
};
}
// Wait for the client's success heartbeat (it round-trips, then beats ~1/s). // Wait for the client's success heartbeat (it round-trips, then beats ~1/s).
const prefix = "vfstest: ok"; const prefix = "vfstest: ok";
@@ -1029,6 +1030,89 @@ fn vfsTest(boot_info: *const BootInfo) void {
result(); result();
} }
/// Spawn the initrd binary named `name` as a ring-3 process. Returns false if it
/// isn't in the image or fails to load.
fn spawnNamed(rd: initrd.Reader, name: []const u8) bool {
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (eql(item.name, name)) {
return if (process.spawnProcess(item.blob, 4)) true else |_| false;
}
}
return false;
}
/// IO passthrough: a user-space driver reads real hardware. Spawn hpetd, which
/// enumerates the device table, claims the HPET, maps its MMIO registers into its
/// own ring-3 address space, enables the counter, and reads it — heartbeating
/// "hpetd: ok" only if the counter advanced. Seeing that proves a user process
/// drove real hardware through a kernel-granted MMIO mapping.
fn hpetTest(boot_info: *const BootInfo) void {
log("DANOS-TEST-BEGIN: hpet\n", .{});
if (boot_info.initrd_len == 0) {
check("bootloader handed over an initrd", false);
result();
return;
}
const image = @as([*]const u8, @ptrFromInt(danos.physToVirt(boot_info.initrd_base)))[0..boot_info.initrd_len];
const rd = initrd.Reader.init(image) orelse {
check("initrd image is valid", false);
result();
return;
};
process.write_count = 0;
process.write_from_user = false;
check("hpetd spawned from the initrd", spawnNamed(rd, "hpetd"));
const prefix = "hpetd: ok";
sched.setPriority(1);
const deadline = arch.millis() + 10000;
while (arch.millis() < deadline) {
if (process.write_len >= prefix.len and eql(process.write_buf[0..prefix.len], prefix) and process.write_count >= 2) break;
sched.yield();
}
sched.setPriority(4);
const ok = process.write_len >= prefix.len and eql(process.write_buf[0..prefix.len], prefix);
check("user driver mapped HPET MMIO and read the counter advancing", ok);
check("driver syscalls came from user mode (CPL 3)", process.write_from_user);
result();
}
/// The MMIO-grant teardown fix: a device-granted leaf must NOT be returned to the
/// RAM allocator when its address space is destroyed. Map a real RAM frame as a
/// device grant, tear the address space down, and confirm the frame is still held
/// (only the page tables came back) — then free it explicitly. A regression guard
/// for the freeSubtree device_grant skip that keeps IO passthrough from corrupting
/// the frame pool.
fn ioPassTest() void {
log("DANOS-TEST-BEGIN: iopass\n", .{});
const base_free = pmm.stats().free_frames;
const aspace = arch.createAddressSpace() orelse {
check("created a fresh address space", false);
result();
return;
};
const frame = pmm.alloc() orelse {
arch.destroyAddressSpace(aspace);
check("allocated a frame to grant", false);
result();
return;
};
// Map it the way mmio_map does (device grant), then tear the space down.
arch.mapUserDeviceInto(aspace, process.dev_arena_base, frame, danos.page_size);
arch.destroyAddressSpace(aspace);
// The page tables were reclaimed; the device-granted frame must not have been.
check("device-granted frame survived teardown (not reclaimed as RAM)", pmm.stats().free_frames == base_free - 1);
pmm.free(frame);
check("no leak once the frame is explicitly freed", pmm.stats().free_frames == base_free);
result();
}
fn faultInvalidOpcode() void { fn faultInvalidOpcode() void {
log("DANOS-TEST-BEGIN: fault-ud\n", .{}); log("DANOS-TEST-BEGIN: fault-ud\n", .{});
asm volatile ("ud2"); asm volatile ("ud2");
+47
View File
@@ -75,9 +75,56 @@ pub const Syscall = enum(u64) {
ipc_lookup = 8, // ipc_lookup(service_id) -> handle: find a published endpoint ipc_lookup = 8, // ipc_lookup(service_id) -> handle: find a published endpoint
ipc_call = 9, // ipc_call(h, msg, len, reply, cap) -> reply_len: send + block for reply ipc_call = 9, // ipc_call(h, msg, len, reply, cap) -> reply_len: send + block for reply
ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, recv, cap) -> recv_len (+badge in rdx) ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, recv, cap) -> recv_len (+badge in rdx)
dev_enumerate = 11, // dev_enumerate(buf, max) -> count: snapshot the device table
dev_claim = 12, // dev_claim(id) -> ok: take exclusive ownership of a device
mmio_map = 13, // mmio_map(id, res_idx) -> vaddr: map a claimed device's MMIO into this AS
irq_bind = 14, // irq_bind(id, res_idx, endpoint): deliver a device IRQ as an IPC notification
irq_ack = 15, // irq_ack(id, res_idx): re-arm a bound IRQ after servicing it
_, _,
}; };
/// A device class, mirroring src/device/device.zig's `DeviceClass` **in order**
/// (its `@intFromEnum` values cross the syscall boundary in `DeviceDesc.class`).
/// Keep the two in sync.
pub const DeviceClass = enum(u32) {
root,
processor,
interrupt_controller,
timer,
pci_host_bridge,
pci_device,
acpi_device,
unknown,
};
/// A resource kind, mirroring src/device/device.zig's `ResourceKind` in order.
pub const ResourceKind = enum(u32) {
memory,
io_port,
irq,
bus_range,
};
/// One device resource, as handed to a user-space driver (flat, extern).
pub const ResDesc = extern struct {
kind: u64, // a ResourceKind value
start: u64,
len: u64,
};
pub const max_dev_resources = 8;
/// A device, as snapshotted for user space by `dev_enumerate`. A driver scans
/// these to find the hardware it owns, claims it, and maps its MMIO.
pub const DeviceDesc = extern struct {
id: u64,
class: u64, // a DeviceClass value
hid_len: u64,
resource_count: u64,
hid: [8]u8,
resources: [max_dev_resources]ResDesc,
};
/// Well-known IPC service ids for the bootstrap name registry (create_endpoint + /// Well-known IPC service ids for the bootstrap name registry (create_endpoint +
/// ipc_register/ipc_lookup). Small integers, so no string interning is needed /// ipc_register/ipc_lookup). Small integers, so no string interning is needed
/// during bring-up. The VFS server registers under `vfs`; clients look it up. /// during bring-up. The VFS server registers under `vfs`; clients look it up.
+10
View File
@@ -188,6 +188,16 @@ CASES = [
{"name": "vfs", {"name": "vfs",
"expect": r"DANOS-TEST-RESULT: PASS", "expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"}, "fail": r"DANOS-TEST-RESULT: FAIL"},
# IO passthrough: a user-space HPET driver maps device MMIO into its own
# address space and reads the counter advancing.
{"name": "hpet",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# The MMIO-grant teardown fix: a device-granted frame must not be reclaimed
# as RAM when its address space is destroyed.
{"name": "iopass",
"expect": r"DANOS-TEST-RESULT: PASS",
"fail": r"DANOS-TEST-RESULT: FAIL"},
# The ACPI power path succeeds by QEMU *exiting* (S5 off / reset), so match the # The ACPI power path succeeds by QEMU *exiting* (S5 off / reset), so match the
# pre-transition marker; the FAIL line only appears if the transition didn't take. # pre-transition marker; the FAIL line only appears if the transition didn't take.
{"name": "poweroff", {"name": "poweroff",