Device manager (increment 2): system_spawn — actually start the driver

Add a `system_spawn(name)` system call: the kernel loads a binary bundled in the
initial-ramdisk, by name, as a fresh ring-3 process. It's the mechanism a user-space
supervisor needs — discovery and policy stay in user space, the kernel only spawns.
The kernel already holds the initial-ramdisk image from the boot handoff; it now
stashes it (process.setInitialRamdisk) so the handler can resolve names, bounds-checks
the name into the user half like debug_write, and returns -1 for an unknown name or a
load failure. Ungated for now (any process may spawn any bundled binary); a spawn
capability belongs here once the model grows one.

The device manager stops logging "would spawn it" and calls runtime.system.spawn on
its matched driver. On QEMU it discovers the HPET, matches `hpet`, and spawns it — and
the driver comes all the way up (claims the timer, maps its MMIO, binds and services
its IRQ, prints "hpet: ok"). The device-manager test now keys on that final marker:
since only the manager is spawned, `hpet: ok` appearing proves the whole
discover -> match -> system_spawn -> driver-up chain end to end.

Transitional: the kernel still auto-spawns the whole initial-ramdisk at boot, so a
real boot briefly double-spawns hpet (the second claim fails harmlessly). Increment 3
removes that redundancy so the manager is the sole owner of driver spawning. Suite
36/36 plus host tests.
This commit is contained in:
Daniel Samson
2026-07-10 18:24:46 +01:00
parent b61b7775b9
commit afbf10f7fc
6 changed files with 79 additions and 12 deletions
+2
View File
@@ -304,6 +304,8 @@ fn kmain(boot_information: *const BootInformation) noreturn {
fn startInitialRamdiskBinaries(boot_information: *const boot_handoff.BootInformation) void {
if (boot_information.initial_ramdisk_len == 0) return;
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
// Hand the image to the process layer so user space can `system_spawn` from it.
process.setInitialRamdisk(image);
const rd = initial_ramdisk.Reader.init(image) orelse {
status("initial_ramdisk: bad image, skipping\n");
return;
+41
View File
@@ -31,6 +31,7 @@ const sync = @import("sync.zig");
const ipc = @import("ipc-synchronous.zig");
const devices_broker = @import("devices-broker.zig");
const irq = @import("irq.zig");
const initial_ramdisk = @import("initial-ramdisk");
const log = @import("log.zig");
const page_size = abi.page_size;
@@ -81,6 +82,17 @@ pub var write_from_user: bool = false;
pub var write_count: u64 = 0; // total write syscalls served (for the heartbeat tests)
pub var exit_code: u64 = 0;
/// The initial-ramdisk image, recorded at boot so `system_spawn` can find bundled
/// binaries by name. Null until `setInitialRamdisk` runs; `system_spawn` then fails
/// cleanly rather than reaching into unset memory.
var ramdisk_image: ?[]const u8 = null;
/// Record the initial-ramdisk image (the kernel already holds it from the boot
/// handoff) so a user-space supervisor can `system_spawn` binaries out of it.
pub fn setInitialRamdisk(image: []const u8) void {
ramdisk_image = image;
}
/// The system_call surface, dispatched on the saved system_call number (`abi.SystemCall`).
/// This is the microkernel-minimal set — memory + scheduling only; file/device
/// I/O will arrive as IPC to user-space servers (docs/syscall.md). The result is
@@ -137,6 +149,7 @@ fn system_call(state: *architecture.CpuState) void {
.irq_bind => systemIrqBind(state),
.irq_ack => systemIrqAck(state),
.device_register => systemDeviceRegister(state),
.system_spawn => systemSpawn(state),
_ => fail(state),
}
}
@@ -269,6 +282,34 @@ fn systemDeviceRegister(state: *architecture.CpuState) void {
architecture.setSystemCallResult(state, id);
}
/// system_spawn(name_ptr, name_len) -> 0 on success, -1 on failure. Load the binary
/// bundled in the initial-ramdisk under `name` as a fresh ring-3 process. This is the
/// mechanism a user-space supervisor (the device manager) uses to start a driver it
/// matched: discovery and policy stay in user space, the kernel only spawns.
///
/// Ungated for now — any process may spawn any bundled binary. A capability (only a
/// supervisor holds the right to spawn) belongs here once the model grows one; see
/// docs/driver-model.md. The name is bounds-checked into the user half exactly like
/// `debug_write`, and an unknown name or a load failure returns -1.
fn systemSpawn(state: *architecture.CpuState) void {
const ptr = architecture.systemCallArg(state, 0);
const len = architecture.systemCallArg(state, 1);
if (len == 0 or len > 64 or ptr >= user_half_end or ptr + len > user_half_end) return fail(state);
const image = ramdisk_image orelse return fail(state);
const rd = initial_ramdisk.Reader.init(image) orelse return fail(state);
const name = @as([*]const u8, @ptrFromInt(ptr))[0..len];
var i: u32 = 0;
while (i < rd.count) : (i += 1) {
const item = rd.entry(i) orelse continue;
if (!std.mem.eql(u8, item.name, name)) continue;
spawnProcess(item.blob, 4) catch return fail(state);
architecture.setSystemCallResult(state, 0);
return;
}
fail(state); // no bundled binary by that name
}
/// Drop every IRQ binding `t` made. Called on exit, before the handle table is closed
/// (which is what frees the endpoints an ISR would otherwise notify into).
fn releaseIrqs(t: *scheduler.Task) void {
+12 -2
View File
@@ -1196,11 +1196,21 @@ fn deviceManagerTest(boot_information: *const BootInformation) void {
return;
};
// Let `system_spawn` find bundled binaries by name (the normal boot path does
// this too). Only the device-manager is spawned here — so if `hpet` runs at all,
// it's because the manager discovered the timer, matched, and spawned it.
process.setInitialRamdisk(image);
process.write_count = 0;
process.write_from_user = false;
check("device-manager spawned from the initial_ramdisk", spawnNamed(rd, "device-manager"));
const prefix = "device-manager: ok";
// End-to-end proof: the driver the manager spawned reaches its own live marker.
// `hpet: ok` is hpet's final, stable message (it claims the timer, maps its MMIO,
// binds its IRQ, services one, then sleeps) — nothing overwrites the buffer after,
// so it's race-free to poll for. Its arrival means the whole
// discover -> match -> system_spawn -> driver-up chain worked.
const prefix = "hpet: ok";
scheduler.setPriority(1);
const deadline = architecture.millis() + 10000;
while (architecture.millis() < deadline) {
@@ -1210,7 +1220,7 @@ fn deviceManagerTest(boot_information: *const BootInformation) void {
scheduler.setPriority(4);
const ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix);
check("device manager enumerated the tree and matched a driver to a device", ok);
check("device manager matched the timer and system_spawn'd hpet, which came up", ok);
check("its syscalls came from user mode (CPL 3)", process.write_from_user);
result();
}