M11–M12: IRQ-as-IPC and bus drivers; expand names tree-wide

Two driver-model milestones plus a tree-wide naming pass. Suite 35/35
(QEMU) + host tests green.

M11 — IRQ-as-IPC. A ring-3 driver now sleeps until its device interrupts
it. New src/kernel/irq.zig: per-GSI endpoint bindings, comptime per-vector
trampolines, dispatch = mask GSI -> LAPIC EOI -> notifyLocked, all under one
lock region. irq_bind/irq_ack syscalls, gated by the device claim like
mmio_map. interruptDispatch no longer EOIs — each handler owns its EOI,
because a level line must be masked before it is acknowledged (irq_ack is
the unmask). Bindings are keyed on the owning task and released on exit
(a shared endpoint's siblings survive). hpetd rewritten interrupt-driven.
Tests: hpet (rewritten, reads back the I/O APIC routing) and irqfree.

M12 — bus drivers. DeviceDesc gains a parent, making the device table a
tree. dev_register (device_register) lets a process publish children below
a device it claimed; the kernel enforces resource containment (a child's
resources must nest in its parent's), so a descriptor can't fabricate a
window over kernel RAM. Descriptor copied in via copyFromUser (physmap
walk — an unmapped user pointer fails the call instead of faulting the
kernel). Per-parent child cap bounds table exhaustion. sbin/busd.zig is a
worked bus driver. Test: bus.

Naming — per docs/coding-standards.md: non-acronym abbreviations spelled
out (message, descriptor, device_service, scheduler, runtime, physical,
interpreter, ...); acronyms kept (IPC, MMIO, DMA, HCD, ...); files are
kebab-case (ipc-synchronous.zig, device-service.zig, vfs-protocol.zig, ...).
Exceptions: POSIX/C ABI names and Zig idioms (init/len/ptr) kept. Module
collisions resolved by specific naming (config -> parameters, device.zig
alias -> device_model). AML op/Op disambiguated: op = opcode, Op =
operation; per-opcode parse handlers renamed opX -> parseX.

New driver docs: drivers.md, driver-model.md (bus/class/HCD shapes + the
proposed M13–M16 ABI), coding-standards.md.
This commit is contained in:
Daniel Samson
2026-07-10 11:39:56 +01:00
parent 83881641ca
commit 15b70856c9
63 changed files with 4722 additions and 2690 deletions
+210
View File
@@ -0,0 +1,210 @@
//! /sbin/busd — a user-space **bus driver**, and the smallest honest example of one.
//!
//! A bus driver owns a device that *contains other devices*, enumerates them by some
//! bus-specific protocol, and publishes each one into the kernel's device table so a
//! class driver can claim it. PCI walks configuration space; USB walks hub descriptors. Here
//! the "bus" is the HPET's register block and the "devices" are its comparators, each
//! a 0x20-byte window at 0x100 + 0x20*n that can be driven independently.
//!
//! It's a toy bus, but nothing about the mechanism is: `busd` reads how many children
//! exist from the hardware (GENERAL_CAP bits [12:8]), publishes one `DeviceDescriptor` per
//! child with a sub-window of its own MMIO plus the shared IRQ, and the kernel checks
//! every one of those resources is contained in what `busd` was granted. A comparator
//! driver then claims a child and maps only *its* registers — not the whole block.
//!
//! It also proves the negative: registering a child whose window escapes the parent's
//! is refused. Without that check, `device_register` would be a system_call for mapping
//! arbitrary physical memory.
const std = @import("std");
const runtime = @import("runtime");
const device = runtime.device;
const register_general_cap = 0x000;
/// Comparator n's registers: configuration+comparator+FSB route, 0x20 bytes.
fn timerWindow(hpet_base: u64, n: u64) device.ResourceDescriptor {
return .{
.kind = @intFromEnum(device.ResourceKind.memory),
.start = hpet_base + 0x100 + 0x20 * n,
.len = 0x20,
};
}
fn findHpet(buffer: []device.DeviceDescriptor) ?device.DeviceDescriptor {
const total = device.enumerate(buffer);
const n = @min(total, buffer.len);
for (buffer[0..n]) |d| {
if (d.class != @intFromEnum(device.DeviceClass.timer)) continue;
if (d.parent != device.no_parent) continue; // the block, not a comparator child
for (0..d.resource_count) |j| {
if (d.resources[j].kind == @intFromEnum(device.ResourceKind.memory)) return d;
}
}
return null;
}
/// The parent's MMIO resource, and its IRQ if it has one.
fn resourcesOf(d: device.DeviceDescriptor) struct { mmio: device.ResourceDescriptor, irq: ?device.ResourceDescriptor } {
var mmio: device.ResourceDescriptor = undefined;
var irq: ?device.ResourceDescriptor = null;
for (0..d.resource_count) |j| {
const r = d.resources[j];
if (r.kind == @intFromEnum(device.ResourceKind.memory)) mmio = r;
if (r.kind == @intFromEnum(device.ResourceKind.irq)) irq = r;
}
return .{ .mmio = mmio, .irq = irq };
}
fn firstChildOf(buffer: []device.DeviceDescriptor, total: usize, parent_id: u64) ?u64 {
for (buffer[0..@min(total, buffer.len)]) |d| {
if (d.parent == parent_id) return d.id;
}
return null;
}
pub fn main() void {
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
_ = runtime.system.write("busd: out of memory\n");
return;
};
const parent = findHpet(buffer) orelse {
_ = runtime.system.write("busd: no HPET\n");
return;
};
const resource = resourcesOf(parent);
// Claim the bus. Everything below is subdivision of what this claim granted.
//
// Claims are exclusive, and at a normal boot the kernel spawns every initrd
// binary — so hpetd may own the HPET already. That's not an error, it's the
// capability model working: exit quietly and leave the device to its owner. The
// `bus` test spawns busd alone, so there it wins the claim.
if (!device.claim(parent.id)) {
_ = runtime.system.write("busd: HPET already claimed by another driver, nothing to do\n");
return;
}
// Enumerate the bus: ask the hardware how many children it has.
const base = device.mmioMap(parent.id, 0) orelse {
_ = runtime.system.write("busd: mmio_map failed\n");
return;
};
const cap: *volatile u64 = @ptrFromInt(base + register_general_cap);
const n_children = ((cap.* >> 8) & 0x1F) + 1;
// Publish one child per comparator, each owning only its own window.
var published: u64 = 0;
var n: u64 = 0;
while (n < n_children) : (n += 1) {
var child = std.mem.zeroes(device.DeviceDescriptor);
child.class = @intFromEnum(device.DeviceClass.timer);
child.hid_len = 6;
child.hid[0..6].* = "hpet-t".*;
child.resource_count = 1;
child.resources[0] = timerWindow(resource.mmio.start, n);
// Comparators share the block's interrupt line; only one child can bind it,
// but all of them may legitimately name it.
if (resource.irq) |i| {
child.resources[child.resource_count] = i;
child.resource_count += 1;
}
if (device.register(parent.id, &child) == null) {
_ = runtime.system.write("busd: register failed\n");
return;
}
published += 1;
}
// The negative case. A window one byte past the end of the parent's must be
// refused — otherwise device_register would be "map any physical page you like".
// Confirm the table did not grow, not merely that the call returned null: null
// also means NoSpace/BadParent, so a size check is what actually proves the
// *containment* rule fired.
const before = device.enumerate(buffer);
var rogue = std.mem.zeroes(device.DeviceDescriptor);
rogue.class = @intFromEnum(device.DeviceClass.unknown);
rogue.resource_count = 1;
rogue.resources[0] = .{
.kind = @intFromEnum(device.ResourceKind.memory),
.start = resource.mmio.start + resource.mmio.len,
.len = 0x1000,
};
if (device.register(parent.id, &rogue) != null) {
_ = runtime.system.write("busd: FAIL out-of-window child was accepted\n");
return;
}
if (device.enumerate(buffer) != before) {
_ = runtime.system.write("busd: FAIL rogue child leaked into the table\n");
return;
}
// And confirm the children came back with the right parent and a *narrower*
// window than the bus — read from the table, not from our own memory.
const total = device.enumerate(buffer);
var seen: u64 = 0;
for (buffer[0..@min(total, buffer.len)]) |d| {
if (d.parent != parent.id) continue;
const w = d.resources[0];
if (w.start < resource.mmio.start or w.len >= resource.mmio.len) {
_ = runtime.system.write("busd: FAIL child window is not inside the bus\n");
return;
}
seen += 1;
}
if (seen != published) {
_ = runtime.system.write("busd: FAIL child count mismatch\n");
return;
}
// Delegation, end to end: claim a child and map *it*. A real class driver would be
// a different process; here busd plays both parts, which exercises the same path.
// The child's window is 0x20 bytes at parent+0x100, so the register it sees at
// offset 0 must be the same timer-0 configuration register the bus sees at 0x100.
//
// (mmio_map rounds to a page, so the child's mapping physically covers the whole
// 4 KiB the HPET lives in — the granularity limit documented in docs/drivers.md.
// The *resource* is narrow even though the page isn't.)
const child_id = firstChildOf(buffer, device.enumerate(buffer), parent.id) orelse {
_ = runtime.system.write("busd: FAIL no child to claim\n");
return;
};
if (!device.claim(child_id)) {
_ = runtime.system.write("busd: FAIL could not claim own child\n");
return;
}
const child_base = device.mmioMap(child_id, 0) orelse {
_ = runtime.system.write("busd: FAIL child mmio_map refused\n");
return;
};
const via_child: *volatile u64 = @ptrFromInt(child_base);
const via_bus: *volatile u64 = @ptrFromInt(base + 0x100);
if (via_child.* != via_bus.*) {
_ = runtime.system.write("busd: FAIL child window does not alias the bus register\n");
return;
}
// A descriptor pointer into an unmapped page must fail the call, not fault the
// kernel. Grab a page, free it, and register through the stale address: if the
// kernel dereferenced it raw (rather than copying in through the page tables) this
// would triple-fault QEMU and the test would time out instead of printing ok.
const scratch = runtime.system.mmap(0x1000, runtime.system.PROT_READ | runtime.system.PROT_WRITE);
if (!runtime.system.mmapFailed(scratch)) {
_ = runtime.system.munmap(scratch, 0x1000);
const descriptor: *const device.DeviceDescriptor = @ptrFromInt(scratch);
if (device.register(parent.id, descriptor) != null) {
_ = runtime.system.write("busd: FAIL register accepted an unmapped descriptor\n");
return;
}
}
_ = runtime.system.write("busd: ok\n");
while (true) runtime.system.sleep(1000);
}
pub const panic = runtime.panic;
comptime {
_ = &runtime.start._start;
}
+172 -60
View File
@@ -1,75 +1,187 @@
//! /sbin/hpetd — a user-space HPET driver, the first real device driver. It
//! proves IO passthrough end to end: enumerate the device table, find the HPET
//! (a timer with an MMIO window), claim it, map its registers directly into this
//! ring-3 address space (strong-uncacheable), then drive the hardware — enable
//! the main counter and read it. If the counter advances, a user process is
//! touching real hardware through a kernel-granted MMIO mapping.
//! /sbin/hpetd — a user-space HPET driver. It proves the whole driver model end to
//! end: enumerate the device table, find the HPET, claim it, map its registers into
//! this ring-3 address space (strong-uncacheable), **bind its interrupt to an IPC
//! endpoint**, then sit blocked in `replyWait` until the hardware wakes it.
//!
//! Register offsets (HPET spec): general config = 0x10 (bit 0 = ENABLE),
//! main counter = 0xF0.
//! Nothing here polls. Between interrupts the process is `.blocked` and off every
//! scheduler queue; the core runs other work or idles. That is the point of the
//! exercise — a driver is a process that sleeps until its device has something to
//! say (see docs/drivers.md).
//!
//! The comparator is configured **level-triggered** on purpose. Edge would be
//! simpler, but level is the discipline every real device line needs, and it forces
//! the full cycle to be correct:
//!
//! kernel ISR mask the GSI -> EOI -> notify this endpoint
//! hpetd wake, clear GENERAL_INT_STATUS (deasserts the line), re-arm
//! hpetd irq_ack -> kernel unmasks the GSI
//!
//! Clear the status bit *before* acking, or the line is still asserted when the
//! kernel unmasks and the I/O APIC redelivers forever.
//!
//! Register map (HPET spec 1.0a):
//! 0x000 GENERAL_CAP [63:32] fs per tick, [12:8] number timers - 1
//! 0x010 GENERAL_CONFIGURATION bit0 ENABLE_CNF, bit1 LEG_RT_CNF
//! 0x020 GENERAL_INT_STATUS bit n = timer n asserted (write 1 to clear)
//! 0x0F0 MAIN_COUNTER
//! 0x100 TIMER0_CONFIGURATION bit1 INT_TYPE(1=level) bit2 INT_ENB bit3 TYPE(periodic)
//! bits[13:9] INT_ROUTE, [63:32] INT_ROUTE_CAP
//! 0x108 TIMER0_COMPARATOR
const rt = @import("rt");
const dev = rt.dev;
const runtime = @import("runtime");
const device = runtime.device;
const ipc = runtime.ipc;
const register_general_cap = 0x000;
const register_general_configuration = 0x010;
const register_int_status = 0x020;
const register_main_counter = 0x0F0;
const register_timer0_configuration = 0x100;
const register_timer0_comparator = 0x108;
const configuration_enable: u64 = 1 << 0; // GENERAL_CONFIGURATION.ENABLE_CNF
const configuration_leg_rt: u64 = 1 << 1; // GENERAL_CONFIGURATION.LEG_RT_CNF
const tn_int_type_level: u64 = 1 << 1;
const tn_int_enb: u64 = 1 << 2;
const tn_type_periodic: u64 = 1 << 3;
const tn_route_shift = 9;
const tn_route_mask: u64 = 0x1F << tn_route_shift;
/// Interrupts to observe before declaring victory.
const target_ticks = 5;
fn register(base: usize, off: usize) *volatile u64 {
return @ptrFromInt(base + off);
}
/// A timer-class device exposing both an MMIO window and an IRQ: its id, the two
/// resource indices, and the GSI discovery chose out of `Tn_INT_ROUTE_CAP`.
const Found = struct { device_id: u64, mmio: u64, irq: u64, gsi: u64 };
fn findHpet(buffer: []device.DeviceDescriptor) ?Found {
const total = device.enumerate(buffer);
const n = @min(total, buffer.len);
for (buffer[0..n]) |d| {
if (d.class != @intFromEnum(device.DeviceClass.timer)) continue;
// Skip comparator children a bus driver may have published below the block
// (see sbin/busd.zig) — we want the register block itself.
if (d.parent != device.no_parent) continue;
var mmio: ?u64 = null;
var irq: ?u64 = null;
for (0..d.resource_count) |j| {
switch (d.resources[j].kind) {
@intFromEnum(device.ResourceKind.memory) => mmio = mmio orelse j,
@intFromEnum(device.ResourceKind.irq) => irq = irq orelse j,
else => {},
}
}
if (mmio) |m| if (irq) |i| {
return .{ .device_id = d.id, .mmio = m, .irq = i, .gsi = d.resources[i].start };
};
}
return null;
}
pub fn main() void {
// Enumerate into a heap buffer (too big for the one-page user stack).
const buf = rt.allocator().alloc(dev.DeviceDesc, 32) catch {
_ = rt.sys.write("hpetd: out of memory\n");
return;
};
const total = dev.enumerate(buf);
const n = @min(total, buf.len);
// Find a timer-class device with an MMIO resource (the HPET).
var dev_id: u64 = 0;
var res_idx: u64 = 0;
var found = false;
var i: usize = 0;
outer: while (i < n) : (i += 1) {
const d = buf[i];
if (d.class != @intFromEnum(dev.DeviceClass.timer)) continue;
var j: usize = 0;
while (j < d.resource_count) : (j += 1) {
if (d.resources[j].kind == @intFromEnum(dev.ResourceKind.memory)) {
dev_id = d.id;
res_idx = j;
found = true;
break :outer;
}
}
}
if (!found) {
_ = rt.sys.write("hpetd: no HPET found\n");
return;
}
if (!dev.claim(dev_id)) {
_ = rt.sys.write("hpetd: claim failed\n");
return;
}
const base = dev.mmioMap(dev_id, res_idx) orelse {
_ = rt.sys.write("hpetd: mmio_map failed\n");
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 32) catch {
_ = runtime.system.write("hpetd: out of memory\n");
return;
};
// Drive the hardware: enable the counter (an MMIO write), then read it twice.
const config: *volatile u64 = @ptrFromInt(base + 0x10);
config.* |= 1; // ENABLE
const counter: *volatile u64 = @ptrFromInt(base + 0xF0);
const a = counter.*;
rt.sys.sleep(50);
const b = counter.*;
const hpet = findHpet(buffer) orelse {
_ = runtime.system.write("hpetd: no HPET with an IRQ\n");
return;
};
if (b > a) {
while (true) {
_ = rt.sys.write("hpetd: ok\n");
rt.sys.sleep(1000);
if (!device.claim(hpet.device_id)) {
_ = runtime.system.write("hpetd: claim failed\n");
return;
}
const base = device.mmioMap(hpet.device_id, hpet.mmio) orelse {
_ = runtime.system.write("hpetd: mmio_map failed\n");
return;
};
// The GSI discovery picked for us out of Tn_INT_ROUTE_CAP. Program the comparator
// to raise exactly this line — the kernel will only bind the one it recorded.
const gsi = hpet.gsi;
const endpoint = ipc.createEndpoint() orelse {
_ = runtime.system.write("hpetd: create_endpoint failed\n");
return;
};
// --- program the hardware ------------------------------------------------
// Counter period, so we can arm the comparator a fixed wall-clock distance out.
const femtos_per_tick = register(base, register_general_cap).* >> 32;
if (femtos_per_tick == 0) {
_ = runtime.system.write("hpetd: bad HPET period\n");
return;
}
const ticks_per_ms = 1_000_000_000_000 / femtos_per_tick;
// Stop the counter and take the legacy route off while we reconfigure.
register(base, register_general_configuration).* &= ~(configuration_enable | configuration_leg_rt);
// Timer 0: one-shot, level-triggered, routed to our GSI, interrupt enabled.
// One-shot (not periodic) sidesteps the HPET's Tn_value_SET accumulator quirk —
// we simply re-arm from the driver on each interrupt, which is what a tickless
// timer driver does anyway.
var t0 = register(base, register_timer0_configuration).*;
t0 &= ~(tn_route_mask | tn_type_periodic);
t0 |= tn_int_type_level | tn_int_enb | (gsi << tn_route_shift);
register(base, register_timer0_configuration).* = t0;
// Clear any stale assertion, then arm ~100 ms out and start the counter.
register(base, register_int_status).* = 1;
register(base, register_timer0_comparator).* = register(base, register_main_counter).* + ticks_per_ms * 100;
register(base, register_general_configuration).* |= configuration_enable;
if (!device.irqBind(hpet.device_id, hpet.irq, endpoint)) {
_ = runtime.system.write("hpetd: irq_bind failed\n");
return;
}
_ = runtime.system.write("hpetd: bound, sleeping until the hardware speaks\n");
// --- the driver loop -----------------------------------------------------
// Blocked in replyWait. No polling, no spinning: the next line of this function
// runs only because an interrupt fired.
var receive: [64]u8 = undefined;
var count: usize = 0;
while (count < target_ticks) {
// Blocked here. The task is `.blocked` and off every scheduler queue; the
// next line runs only because the HPET raised its line.
const r = ipc.replyWait(endpoint, &.{}, &receive);
if (!r.isNotification()) continue; // a client request, not our IRQ
// Quiet the device: write 1 to timer 0's status bit. Until this lands, the
// line is still asserted and unmasking would refire immediately.
register(base, register_int_status).* = 1;
count += 1;
if (count < target_ticks) {
register(base, register_timer0_comparator).* = register(base, register_main_counter).* + ticks_per_ms * 100;
} else {
// Last one: stop the source rather than re-arming, so the line is left
// both quiet *and* unmasked by the ack below. Re-arming here would leave
// a pending interrupt that nobody is waiting for, and the ISR would mask
// the line again a moment later.
register(base, register_timer0_configuration).* &= ~tn_int_enb;
}
_ = runtime.system.write("hpetd: irq\n");
if (!device.irqAck(hpet.device_id, hpet.irq)) {
_ = runtime.system.write("hpetd: irq_ack failed\n");
return;
}
}
_ = rt.sys.write("hpetd: counter stuck\n");
_ = runtime.system.write("hpetd: ok\n");
while (true) runtime.system.sleep(1000);
}
pub const panic = rt.panic;
pub const panic = runtime.panic;
comptime {
_ = &rt.start._start;
_ = &runtime.start._start;
}
+12 -12
View File
@@ -2,7 +2,7 @@
//! freestanding binary (see build.zig), shipped on the boot volume at sbin/init,
//! loaded by the bootloader, and started in ring 3 as a scheduled process by the
//! kernel (src/kernel/process.zig). It links against the shared user runtime
//! library `rt` and talks to the kernel only through `rt`'s syscall wrappers.
//! library `runtime` and talks to the kernel only through `runtime`'s system_call wrappers.
//!
//! Today it proves the C-convention heap works, then settles into a heartbeat:
//! it prints a line and sleeps, forever — enough to show the system reaches user
@@ -10,7 +10,7 @@
//! idle loop. It grows into the real init (service supervision) once there are
//! other user programs to supervise.
const rt = @import("rt");
const runtime = @import("runtime");
pub fn main() void {
// Prove the heap end to end: allocate through the runtime allocator (which
@@ -19,21 +19,21 @@ pub fn main() void {
// free it. A fault here would kill init before it heartbeats — so the init
// test doubles as the heap regression test. (C code links the same heap via
// the extern malloc/free symbols; Zig code uses this allocator.)
const gpa = rt.allocator();
if (gpa.alloc(u8, 64)) |buf| {
const msg = "init: heap ok\n";
@memcpy(buf[0..msg.len], msg);
_ = rt.sys.write(buf[0..msg.len]);
gpa.free(buf);
const gpa = runtime.allocator();
if (gpa.alloc(u8, 64)) |buffer| {
const message = "init: heap ok\n";
@memcpy(buffer[0..message.len], message);
_ = runtime.system.write(buffer[0..message.len]);
gpa.free(buffer);
} else |_| {}
while (true) {
_ = rt.sys.write("init: heartbeat\n");
rt.sys.sleep(1000);
_ = runtime.system.write("init: heartbeat\n");
runtime.system.sleep(1000);
}
}
pub const panic = rt.panic;
pub const panic = runtime.panic;
comptime {
_ = &rt.start._start; // pull the runtime entry shim into the image
_ = &runtime.start._start; // pull the runtime entry shim into the image
}
+14 -14
View File
@@ -1,13 +1,13 @@
//! /sbin/vfstest — a client that proves the VFS round trip end to end: open a
//! file through the `rt` file API, write to it, seek back, read it, and compare.
//! file through the `runtime` file API, write to it, seek back, read it, and compare.
//! On success it heartbeats "vfstest: ok" so the kernel test can observe it;
//! on failure it reports what went wrong. Shipped in the initrd alongside vfs.
const std = @import("std");
const rt = @import("rt");
const runtime = @import("runtime");
pub fn main() void {
const u = rt.unistd;
const u = runtime.unistd;
const payload = "hello-vfs";
// The VFS server may not have registered yet — retry open until it's up.
@@ -15,33 +15,33 @@ pub fn main() void {
var tries: u32 = 0;
while (fd < 0 and tries < 200) : (tries += 1) {
fd = u.open("greeting", u.O_CREAT);
if (fd < 0) rt.sys.sleep(20);
if (fd < 0) runtime.system.sleep(20);
}
if (fd < 0) {
_ = rt.sys.write("vfstest: open failed\n");
_ = runtime.system.write("vfstest: open failed\n");
return;
}
if (u.write(fd, payload) != @as(isize, payload.len)) {
_ = rt.sys.write("vfstest: write failed\n");
_ = runtime.system.write("vfstest: write failed\n");
return;
}
_ = u.lseek(fd, 0, u.SEEK_SET);
var buf: [32]u8 = undefined;
const n = u.read(fd, &buf);
var buffer: [32]u8 = undefined;
const n = u.read(fd, &buffer);
u.close(fd);
if (n == @as(isize, payload.len) and std.mem.eql(u8, buf[0..@intCast(n)], payload)) {
if (n == @as(isize, payload.len) and std.mem.eql(u8, buffer[0..@intCast(n)], payload)) {
while (true) {
_ = rt.sys.write("vfstest: ok\n");
rt.sys.sleep(1000);
_ = runtime.system.write("vfstest: ok\n");
runtime.system.sleep(1000);
}
}
_ = rt.sys.write("vfstest: mismatch\n");
_ = runtime.system.write("vfstest: mismatch\n");
}
pub const panic = rt.panic;
pub const panic = runtime.panic;
comptime {
_ = &rt.start._start;
_ = &runtime.start._start;
}
+28 -28
View File
@@ -1,16 +1,16 @@
//! /sbin/vfs — the user-space VFS server. Shipped in the initrd, spawned as a
//! ring-3 process, and reached by every other process through IPC (the `rt`
//! ring-3 process, and reached by every other process through IPC (the `runtime`
//! file API marshals open/read/write/stat/close into calls to this server's
//! endpoint, published under the well-known `vfs` service id).
//!
//! For now the namespace is a small in-memory ramfs (opening a name creates it):
//! enough to prove the whole path — client file API -> IPC -> server dispatch ->
//! reply. Device nodes backed by user-space drivers (/dev) layer on top in M10,
//! where `open` on a /dev name forwards to the owning driver's endpoint.
//! reply. Device nodes backed by user-space drivers (/device) layer on top in M10,
//! where `open` on a /device name forwards to the owning driver's endpoint.
const std = @import("std");
const rt = @import("rt");
const proto = rt.vfsproto;
const runtime = @import("runtime");
const protocol = runtime.vfs_protocol;
const Node = struct {
used: bool = false,
@@ -54,11 +54,11 @@ fn openAt(id: u64) ?*OpenFile {
}
/// Serialise a reply header + payload into `out`; returns the total length.
fn writeReply(out: []u8, reply: proto.Reply, payload: []const u8) usize {
@memcpy(out[0..proto.reply_size], std.mem.asBytes(&reply));
const n = @min(payload.len, out.len - proto.reply_size);
@memcpy(out[proto.reply_size..][0..n], payload[0..n]);
return proto.reply_size + n;
fn writeReply(out: []u8, reply: protocol.Reply, payload: []const u8) usize {
@memcpy(out[0..protocol.reply_size], std.mem.asBytes(&reply));
const n = @min(payload.len, out.len - protocol.reply_size);
@memcpy(out[protocol.reply_size..][0..n], payload[0..n]);
return protocol.reply_size + n;
}
fn fail(out: []u8) usize {
@@ -66,10 +66,10 @@ fn fail(out: []u8) usize {
}
/// Handle one request; write the reply into `out`, return its length.
fn handle(msg: []const u8, out: []u8) usize {
if (msg.len < proto.req_size) return fail(out);
const req = std.mem.bytesToValue(proto.Request, msg[0..proto.req_size]);
const payload = msg[proto.req_size..];
fn handle(message: []const u8, out: []u8) usize {
if (message.len < protocol.req_size) return fail(out);
const req = std.mem.bytesToValue(protocol.Request, message[0..protocol.req_size]);
const payload = message[protocol.req_size..];
switch (req.op) {
.open => {
@@ -88,7 +88,7 @@ fn handle(msg: []const u8, out: []u8) usize {
const nd = &nodes[of.node];
const off: usize = @intCast(req.offset);
if (off >= nd.size) return writeReply(out, .{ .status = 0, .len = 0 }, &.{}); // EOF
const n = @min(@min(nd.size - off, req.len), proto.max_payload);
const n = @min(@min(nd.size - off, req.len), protocol.maximum_payload);
return writeReply(out, .{ .status = 0, .len = @intCast(n) }, nd.data[off .. off + n]);
},
.write => {
@@ -103,8 +103,8 @@ fn handle(msg: []const u8, out: []u8) usize {
},
.stat => {
const of = openAt(req.node) orelse return fail(out);
const st = proto.Stat{ .size = nodes[of.node].size, .kind = 0 };
return writeReply(out, .{ .status = 0, .len = @sizeOf(proto.Stat) }, std.mem.asBytes(&st));
const st = protocol.Stat{ .size = nodes[of.node].size, .kind = 0 };
return writeReply(out, .{ .status = 0, .len = @sizeOf(protocol.Stat) }, std.mem.asBytes(&st));
},
.close => {
if (req.node < opens.len) opens[@intCast(req.node)].used = false;
@@ -114,27 +114,27 @@ fn handle(msg: []const u8, out: []u8) usize {
}
pub fn main() void {
const ep = rt.ipc.createEndpoint() orelse {
_ = rt.sys.write("vfs: no endpoint\n");
const endpoint = runtime.ipc.createEndpoint() orelse {
_ = runtime.system.write("vfs: no endpoint\n");
return;
};
if (!rt.ipc.register(.vfs, ep)) {
_ = rt.sys.write("vfs: register failed\n");
if (!runtime.ipc.register(.vfs, endpoint)) {
_ = runtime.system.write("vfs: register failed\n");
return;
}
_ = rt.sys.write("vfs: ready\n");
_ = runtime.system.write("vfs: ready\n");
var reply_buf: [proto.msg_max]u8 = undefined;
var reply_buffer: [protocol.message_maximum]u8 = undefined;
var reply_len: usize = 0;
var recv: [proto.msg_max]u8 = undefined;
var receive: [protocol.message_maximum]u8 = undefined;
while (true) {
const got = rt.ipc.replyWait(ep, reply_buf[0..reply_len], &recv);
const got = runtime.ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive);
// Ignore notifications (none expected here); handle a request.
reply_len = handle(recv[0..got.len], &reply_buf);
reply_len = handle(receive[0..got.len], &reply_buffer);
}
}
pub const panic = rt.panic;
pub const panic = runtime.panic;
comptime {
_ = &rt.start._start;
_ = &runtime.start._start;
}