Re-organize the source tree as a monorepo mirroring the FHS
The source layout now mirrors the runtime filesystem hierarchy
(docs/danos-file-system-hierarchy-FSH.md): what lives under system/ in the
source is what a running danos represents under /system. Each service and
driver is a sub-project directory that is its own Zig module — cross-project
references go by module name, never by a path into another project's files.
Moves (all git mv, history preserved):
- src/ -> system/ (danos internals; the self-representation)
root.zig -> danos.zig (the kernel<->user contract module)
kernel/arch/ -> kernel/architecture/ (arch -> architecture)
device/ -> devices/ (what /system/devices reflects)
boot/ -> /boot (the loaders, top level)
- sbin/ -> split by role:
init, vfs -> system/services/<name>/<name>.zig
hpetd, busd -> system/drivers/<name>/<name>.zig
vfs-test -> system/services/vfs/vfs-test.zig (inside the vfs project)
- lib/ -> library/runtime/ (room for other libraries beside runtime)
The VFS wire protocol becomes its own module, system/services/vfs/protocol.zig
("vfs-protocol"): the vfs sub-project exposes its interface, and the runtime's
file layer imports it by name. First instance of the "protocol module" pattern
(docs/driver-model.md); usb/block will expose theirs the same way.
Also: fix a naming-standard violation in the protocol — Op -> Operation (and
req -> request, _pad -> _padding). Docs updated: /system/services added to the
FHS doc, a repository-layout section added to the docs index, and stale source
paths swept across comments and docs.
Runtime boot paths are unchanged (the bootloader still loads /sbin/init);
aligning the runtime filesystem to the FHS is a separate follow-up. Suite 35/35
plus host tests green.
This commit is contained in:
@@ -0,0 +1,210 @@
|
||||
//! /sbin/busd — a user-space **bus driver**, and the smallest honest example of one.
|
||||
//!
|
||||
//! A bus driver owns a device that *contains other devices*, enumerates them by some
|
||||
//! bus-specific protocol, and publishes each one into the kernel's device table so a
|
||||
//! class driver can claim it. PCI walks configuration space; USB walks hub descriptors. Here
|
||||
//! the "bus" is the HPET's register block and the "devices" are its comparators, each
|
||||
//! a 0x20-byte window at 0x100 + 0x20*n that can be driven independently.
|
||||
//!
|
||||
//! It's a toy bus, but nothing about the mechanism is: `busd` reads how many children
|
||||
//! exist from the hardware (GENERAL_CAP bits [12:8]), publishes one `DeviceDescriptor` per
|
||||
//! child with a sub-window of its own MMIO plus the shared IRQ, and the kernel checks
|
||||
//! every one of those resources is contained in what `busd` was granted. A comparator
|
||||
//! driver then claims a child and maps only *its* registers — not the whole block.
|
||||
//!
|
||||
//! It also proves the negative: registering a child whose window escapes the parent's
|
||||
//! is refused. Without that check, `device_register` would be a system_call for mapping
|
||||
//! arbitrary physical memory.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
const device = runtime.device;
|
||||
|
||||
const register_general_cap = 0x000;
|
||||
|
||||
/// Comparator n's registers: configuration+comparator+FSB route, 0x20 bytes.
|
||||
fn timerWindow(hpet_base: u64, n: u64) device.ResourceDescriptor {
|
||||
return .{
|
||||
.kind = @intFromEnum(device.ResourceKind.memory),
|
||||
.start = hpet_base + 0x100 + 0x20 * n,
|
||||
.len = 0x20,
|
||||
};
|
||||
}
|
||||
|
||||
fn findHpet(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor {
|
||||
const total = device.enumerate(buffer);
|
||||
const n = @min(total, buffer.len);
|
||||
for (buffer[0..n]) |d| {
|
||||
if (d.class != @intFromEnum(device.DeviceClass.timer)) continue;
|
||||
if (d.parent != device.no_parent) continue; // the block, not a comparator child
|
||||
for (0..d.resource_count) |j| {
|
||||
if (d.resources[j].kind == @intFromEnum(device.ResourceKind.memory)) return d;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// The parent's MMIO resource, and its IRQ if it has one.
|
||||
fn resourcesOf(d: device.DeviceDescriptor) struct { mmio: device.ResourceDescriptor, irq: ?device.ResourceDescriptor } {
|
||||
var mmio: device.ResourceDescriptor = undefined;
|
||||
var irq: ?device.ResourceDescriptor = null;
|
||||
for (0..d.resource_count) |j| {
|
||||
const r = d.resources[j];
|
||||
if (r.kind == @intFromEnum(device.ResourceKind.memory)) mmio = r;
|
||||
if (r.kind == @intFromEnum(device.ResourceKind.irq)) irq = r;
|
||||
}
|
||||
return .{ .mmio = mmio, .irq = irq };
|
||||
}
|
||||
|
||||
fn firstChildOf(buffer: []device.DeviceDescriptor, total: usize, parent_id: u64) ?u64 {
|
||||
for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||
if (d.parent == parent_id) return d.id;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("busd: out of memory\n");
|
||||
return;
|
||||
};
|
||||
|
||||
const parent = findHpet(buffer) orelse {
|
||||
_ = runtime.system.write("busd: no HPET\n");
|
||||
return;
|
||||
};
|
||||
const resource = resourcesOf(parent);
|
||||
|
||||
// Claim the bus. Everything below is subdivision of what this claim granted.
|
||||
//
|
||||
// Claims are exclusive, and at a normal boot the kernel spawns every initrd
|
||||
// binary — so hpetd may own the HPET already. That's not an error, it's the
|
||||
// capability model working: exit quietly and leave the device to its owner. The
|
||||
// `bus` test spawns busd alone, so there it wins the claim.
|
||||
if (!device.claim(parent.id)) {
|
||||
_ = runtime.system.write("busd: HPET already claimed by another driver, nothing to do\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Enumerate the bus: ask the hardware how many children it has.
|
||||
const base = device.mmioMap(parent.id, 0) orelse {
|
||||
_ = runtime.system.write("busd: mmio_map failed\n");
|
||||
return;
|
||||
};
|
||||
const cap: *volatile u64 = @ptrFromInt(base + register_general_cap);
|
||||
const n_children = ((cap.* >> 8) & 0x1F) + 1;
|
||||
|
||||
// Publish one child per comparator, each owning only its own window.
|
||||
var published: u64 = 0;
|
||||
var n: u64 = 0;
|
||||
while (n < n_children) : (n += 1) {
|
||||
var child = std.mem.zeroes(device.DeviceDescriptor);
|
||||
child.class = @intFromEnum(device.DeviceClass.timer);
|
||||
child.hid_len = 6;
|
||||
child.hid[0..6].* = "hpet-t".*;
|
||||
child.resource_count = 1;
|
||||
child.resources[0] = timerWindow(resource.mmio.start, n);
|
||||
// Comparators share the block's interrupt line; only one child can bind it,
|
||||
// but all of them may legitimately name it.
|
||||
if (resource.irq) |i| {
|
||||
child.resources[child.resource_count] = i;
|
||||
child.resource_count += 1;
|
||||
}
|
||||
|
||||
if (device.register(parent.id, &child) == null) {
|
||||
_ = runtime.system.write("busd: register failed\n");
|
||||
return;
|
||||
}
|
||||
published += 1;
|
||||
}
|
||||
|
||||
// The negative case. A window one byte past the end of the parent's must be
|
||||
// refused — otherwise device_register would be "map any physical page you like".
|
||||
// Confirm the table did not grow, not merely that the call returned null: null
|
||||
// also means NoSpace/BadParent, so a size check is what actually proves the
|
||||
// *containment* rule fired.
|
||||
const before = device.enumerate(buffer);
|
||||
var rogue = std.mem.zeroes(device.DeviceDescriptor);
|
||||
rogue.class = @intFromEnum(device.DeviceClass.unknown);
|
||||
rogue.resource_count = 1;
|
||||
rogue.resources[0] = .{
|
||||
.kind = @intFromEnum(device.ResourceKind.memory),
|
||||
.start = resource.mmio.start + resource.mmio.len,
|
||||
.len = 0x1000,
|
||||
};
|
||||
if (device.register(parent.id, &rogue) != null) {
|
||||
_ = runtime.system.write("busd: FAIL out-of-window child was accepted\n");
|
||||
return;
|
||||
}
|
||||
if (device.enumerate(buffer) != before) {
|
||||
_ = runtime.system.write("busd: FAIL rogue child leaked into the table\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// And confirm the children came back with the right parent and a *narrower*
|
||||
// window than the bus — read from the table, not from our own memory.
|
||||
const total = device.enumerate(buffer);
|
||||
var seen: u64 = 0;
|
||||
for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||
if (d.parent != parent.id) continue;
|
||||
const w = d.resources[0];
|
||||
if (w.start < resource.mmio.start or w.len >= resource.mmio.len) {
|
||||
_ = runtime.system.write("busd: FAIL child window is not inside the bus\n");
|
||||
return;
|
||||
}
|
||||
seen += 1;
|
||||
}
|
||||
if (seen != published) {
|
||||
_ = runtime.system.write("busd: FAIL child count mismatch\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Delegation, end to end: claim a child and map *it*. A real class driver would be
|
||||
// a different process; here busd plays both parts, which exercises the same path.
|
||||
// The child's window is 0x20 bytes at parent+0x100, so the register it sees at
|
||||
// offset 0 must be the same timer-0 configuration register the bus sees at 0x100.
|
||||
//
|
||||
// (mmio_map rounds to a page, so the child's mapping physically covers the whole
|
||||
// 4 KiB the HPET lives in — the granularity limit documented in docs/drivers.md.
|
||||
// The *resource* is narrow even though the page isn't.)
|
||||
const child_id = firstChildOf(buffer, device.enumerate(buffer), parent.id) orelse {
|
||||
_ = runtime.system.write("busd: FAIL no child to claim\n");
|
||||
return;
|
||||
};
|
||||
if (!device.claim(child_id)) {
|
||||
_ = runtime.system.write("busd: FAIL could not claim own child\n");
|
||||
return;
|
||||
}
|
||||
const child_base = device.mmioMap(child_id, 0) orelse {
|
||||
_ = runtime.system.write("busd: FAIL child mmio_map refused\n");
|
||||
return;
|
||||
};
|
||||
const via_child: *volatile u64 = @ptrFromInt(child_base);
|
||||
const via_bus: *volatile u64 = @ptrFromInt(base + 0x100);
|
||||
if (via_child.* != via_bus.*) {
|
||||
_ = runtime.system.write("busd: FAIL child window does not alias the bus register\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// A descriptor pointer into an unmapped page must fail the call, not fault the
|
||||
// kernel. Grab a page, free it, and register through the stale address: if the
|
||||
// kernel dereferenced it raw (rather than copying in through the page tables) this
|
||||
// would triple-fault QEMU and the test would time out instead of printing ok.
|
||||
const scratch = runtime.system.mmap(0x1000, runtime.system.PROT_READ | runtime.system.PROT_WRITE);
|
||||
if (!runtime.system.mmapFailed(scratch)) {
|
||||
_ = runtime.system.munmap(scratch, 0x1000);
|
||||
const descriptor: *const device.DeviceDescriptor = @ptrFromInt(scratch);
|
||||
if (device.register(parent.id, descriptor) != null) {
|
||||
_ = runtime.system.write("busd: FAIL register accepted an unmapped descriptor\n");
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
_ = runtime.system.write("busd: ok\n");
|
||||
while (true) runtime.system.sleep(1000);
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
@@ -0,0 +1,187 @@
|
||||
//! /sbin/hpetd — a user-space HPET driver. It proves the whole driver model end to
|
||||
//! end: enumerate the device table, find the HPET, claim it, map its registers into
|
||||
//! this ring-3 address space (strong-uncacheable), **bind its interrupt to an IPC
|
||||
//! endpoint**, then sit blocked in `replyWait` until the hardware wakes it.
|
||||
//!
|
||||
//! Nothing here polls. Between interrupts the process is `.blocked` and off every
|
||||
//! scheduler queue; the core runs other work or idles. That is the point of the
|
||||
//! exercise — a driver is a process that sleeps until its device has something to
|
||||
//! say (see docs/drivers.md).
|
||||
//!
|
||||
//! The comparator is configured **level-triggered** on purpose. Edge would be
|
||||
//! simpler, but level is the discipline every real device line needs, and it forces
|
||||
//! the full cycle to be correct:
|
||||
//!
|
||||
//! kernel ISR mask the GSI -> EOI -> notify this endpoint
|
||||
//! hpetd wake, clear GENERAL_INT_STATUS (deasserts the line), re-arm
|
||||
//! hpetd irq_ack -> kernel unmasks the GSI
|
||||
//!
|
||||
//! Clear the status bit *before* acking, or the line is still asserted when the
|
||||
//! kernel unmasks and the I/O APIC redelivers forever.
|
||||
//!
|
||||
//! Register map (HPET spec 1.0a):
|
||||
//! 0x000 GENERAL_CAP [63:32] fs per tick, [12:8] number timers - 1
|
||||
//! 0x010 GENERAL_CONFIGURATION bit0 ENABLE_CNF, bit1 LEG_RT_CNF
|
||||
//! 0x020 GENERAL_INT_STATUS bit n = timer n asserted (write 1 to clear)
|
||||
//! 0x0F0 MAIN_COUNTER
|
||||
//! 0x100 TIMER0_CONFIGURATION bit1 INT_TYPE(1=level) bit2 INT_ENB bit3 TYPE(periodic)
|
||||
//! bits[13:9] INT_ROUTE, [63:32] INT_ROUTE_CAP
|
||||
//! 0x108 TIMER0_COMPARATOR
|
||||
|
||||
const runtime = @import("runtime");
|
||||
const device = runtime.device;
|
||||
const ipc = runtime.ipc;
|
||||
|
||||
const register_general_cap = 0x000;
|
||||
const register_general_configuration = 0x010;
|
||||
const register_int_status = 0x020;
|
||||
const register_main_counter = 0x0F0;
|
||||
const register_timer0_configuration = 0x100;
|
||||
const register_timer0_comparator = 0x108;
|
||||
|
||||
const configuration_enable: u64 = 1 << 0; // GENERAL_CONFIGURATION.ENABLE_CNF
|
||||
const configuration_leg_rt: u64 = 1 << 1; // GENERAL_CONFIGURATION.LEG_RT_CNF
|
||||
const tn_int_type_level: u64 = 1 << 1;
|
||||
const tn_int_enb: u64 = 1 << 2;
|
||||
const tn_type_periodic: u64 = 1 << 3;
|
||||
const tn_route_shift = 9;
|
||||
const tn_route_mask: u64 = 0x1F << tn_route_shift;
|
||||
|
||||
/// Interrupts to observe before declaring victory.
|
||||
const target_ticks = 5;
|
||||
|
||||
fn register(base: usize, off: usize) *volatile u64 {
|
||||
return @ptrFromInt(base + off);
|
||||
}
|
||||
|
||||
/// A timer-class device exposing both an MMIO window and an IRQ: its id, the two
|
||||
/// resource indices, and the GSI discovery chose out of `Tn_INT_ROUTE_CAP`.
|
||||
const Found = struct { device_id: u64, mmio: u64, irq: u64, gsi: u64 };
|
||||
|
||||
fn findHpet(buffer: []device.DeviceDescriptor) ?Found {
|
||||
const total = device.enumerate(buffer);
|
||||
const n = @min(total, buffer.len);
|
||||
for (buffer[0..n]) |d| {
|
||||
if (d.class != @intFromEnum(device.DeviceClass.timer)) continue;
|
||||
// Skip comparator children a bus driver may have published below the block
|
||||
// (see system/drivers/busd/busd.zig) — we want the register block itself.
|
||||
if (d.parent != device.no_parent) continue;
|
||||
var mmio: ?u64 = null;
|
||||
var irq: ?u64 = null;
|
||||
for (0..d.resource_count) |j| {
|
||||
switch (d.resources[j].kind) {
|
||||
@intFromEnum(device.ResourceKind.memory) => mmio = mmio orelse j,
|
||||
@intFromEnum(device.ResourceKind.irq) => irq = irq orelse j,
|
||||
else => {},
|
||||
}
|
||||
}
|
||||
if (mmio) |m| if (irq) |i| {
|
||||
return .{ .device_id = d.id, .mmio = m, .irq = i, .gsi = d.resources[i].start };
|
||||
};
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
// Enumerate into a heap buffer (too big for the one-page user stack).
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 32) catch {
|
||||
_ = runtime.system.write("hpetd: out of memory\n");
|
||||
return;
|
||||
};
|
||||
|
||||
const hpet = findHpet(buffer) orelse {
|
||||
_ = runtime.system.write("hpetd: no HPET with an IRQ\n");
|
||||
return;
|
||||
};
|
||||
|
||||
if (!device.claim(hpet.device_id)) {
|
||||
_ = runtime.system.write("hpetd: claim failed\n");
|
||||
return;
|
||||
}
|
||||
const base = device.mmioMap(hpet.device_id, hpet.mmio) orelse {
|
||||
_ = runtime.system.write("hpetd: mmio_map failed\n");
|
||||
return;
|
||||
};
|
||||
|
||||
// The GSI discovery picked for us out of Tn_INT_ROUTE_CAP. Program the comparator
|
||||
// to raise exactly this line — the kernel will only bind the one it recorded.
|
||||
const gsi = hpet.gsi;
|
||||
|
||||
const endpoint = ipc.createEndpoint() orelse {
|
||||
_ = runtime.system.write("hpetd: create_endpoint failed\n");
|
||||
return;
|
||||
};
|
||||
|
||||
// --- program the hardware ------------------------------------------------
|
||||
// Counter period, so we can arm the comparator a fixed wall-clock distance out.
|
||||
const femtos_per_tick = register(base, register_general_cap).* >> 32;
|
||||
if (femtos_per_tick == 0) {
|
||||
_ = runtime.system.write("hpetd: bad HPET period\n");
|
||||
return;
|
||||
}
|
||||
const ticks_per_ms = 1_000_000_000_000 / femtos_per_tick;
|
||||
|
||||
// Stop the counter and take the legacy route off while we reconfigure.
|
||||
register(base, register_general_configuration).* &= ~(configuration_enable | configuration_leg_rt);
|
||||
|
||||
// Timer 0: one-shot, level-triggered, routed to our GSI, interrupt enabled.
|
||||
// One-shot (not periodic) sidesteps the HPET's Tn_value_SET accumulator quirk —
|
||||
// we simply re-arm from the driver on each interrupt, which is what a tickless
|
||||
// timer driver does anyway.
|
||||
var t0 = register(base, register_timer0_configuration).*;
|
||||
t0 &= ~(tn_route_mask | tn_type_periodic);
|
||||
t0 |= tn_int_type_level | tn_int_enb | (gsi << tn_route_shift);
|
||||
register(base, register_timer0_configuration).* = t0;
|
||||
|
||||
// Clear any stale assertion, then arm ~100 ms out and start the counter.
|
||||
register(base, register_int_status).* = 1;
|
||||
register(base, register_timer0_comparator).* = register(base, register_main_counter).* + ticks_per_ms * 100;
|
||||
register(base, register_general_configuration).* |= configuration_enable;
|
||||
|
||||
if (!device.irqBind(hpet.device_id, hpet.irq, endpoint)) {
|
||||
_ = runtime.system.write("hpetd: irq_bind failed\n");
|
||||
return;
|
||||
}
|
||||
_ = runtime.system.write("hpetd: bound, sleeping until the hardware speaks\n");
|
||||
|
||||
// --- the driver loop -----------------------------------------------------
|
||||
// Blocked in replyWait. No polling, no spinning: the next line of this function
|
||||
// runs only because an interrupt fired.
|
||||
var receive: [64]u8 = undefined;
|
||||
var count: usize = 0;
|
||||
while (count < target_ticks) {
|
||||
// Blocked here. The task is `.blocked` and off every scheduler queue; the
|
||||
// next line runs only because the HPET raised its line.
|
||||
const r = ipc.replyWait(endpoint, &.{}, &receive);
|
||||
if (!r.isNotification()) continue; // a client request, not our IRQ
|
||||
|
||||
// Quiet the device: write 1 to timer 0's status bit. Until this lands, the
|
||||
// line is still asserted and unmasking would refire immediately.
|
||||
register(base, register_int_status).* = 1;
|
||||
count += 1;
|
||||
|
||||
if (count < target_ticks) {
|
||||
register(base, register_timer0_comparator).* = register(base, register_main_counter).* + ticks_per_ms * 100;
|
||||
} else {
|
||||
// Last one: stop the source rather than re-arming, so the line is left
|
||||
// both quiet *and* unmasked by the ack below. Re-arming here would leave
|
||||
// a pending interrupt that nobody is waiting for, and the ISR would mask
|
||||
// the line again a moment later.
|
||||
register(base, register_timer0_configuration).* &= ~tn_int_enb;
|
||||
}
|
||||
|
||||
_ = runtime.system.write("hpetd: irq\n");
|
||||
if (!device.irqAck(hpet.device_id, hpet.irq)) {
|
||||
_ = runtime.system.write("hpetd: irq_ack failed\n");
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
_ = runtime.system.write("hpetd: ok\n");
|
||||
while (true) runtime.system.sleep(1000);
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start;
|
||||
}
|
||||
Reference in New Issue
Block a user