Split the system contract into boot-handoff / abi / device-abi

The `system` module (formerly `danos`) had become a grab-bag: it held the
loader<->kernel handoff *and* the kernel<->user ABI *and* the device wire types, in
one module three different audiences imported. Usage proved the seam — the
bootloader never touched the syscall/device ABI, and user space never touched the
boot handoff — so split it by audience, one module per contract:

  system/boot-handoff.zig       loader <-> kernel: BootInformation, Framebuffer,
                                MemoryMap, the VM layout + physicalToVirtual, kernel_abi
  system/abi.zig                kernel <-> user, core: SystemCall, mmap prot flags,
                                page_size, notify_badge_bit, ServiceId
  system/devices/device-abi.zig kernel <-> user, devices: DeviceDescriptor,
                                DeviceClass, ResourceDescriptor, ResourceKind, ...

device-abi is the devices sub-project's public interface, exposed as its own module
the way vfs exposes vfs-protocol — importable by user space, unlike the
kernel-internal device model it also feeds. That collapses a real duplication:
DeviceClass and ResourceKind were defined twice (device-model.zig and the contract,
kept "in sync by hand"); device-model now re-exports them from device-abi, so the
enum a driver matches on and the one the kernel classifies with are one type.

Each import now declares which contract it speaks: the bootloader imports only
boot-handoff; a driver only abi + device-abi (via the runtime); the kernel all
three. This also retires the `system` / `runtime.system` name overlap. page_size
lands in abi (it's part of the mmap contract user space aligns to); the bootloader
keeps its own local 4 KiB constant so it depends on nothing but the handoff.

All 21 importers rewired, docs updated to keep /system mapping to source. Build,
host tests, and the QEMU suite (36/36) all green.
This commit is contained in:
Daniel Samson
2026-07-10 18:08:51 +01:00
parent 47610e8ee2
commit be81394be3
37 changed files with 395 additions and 337 deletions
+14 -12
View File
@@ -21,7 +21,9 @@
const std = @import("std");
const elf = std.elf;
const system = @import("system");
const boot_handoff = @import("boot-handoff");
const abi = @import("abi");
const device_abi = @import("device-abi");
const architecture = @import("architecture");
const pmm = @import("pmm.zig");
const scheduler = @import("scheduler.zig");
@@ -31,8 +33,8 @@ const devices_broker = @import("devices-broker.zig");
const irq = @import("irq.zig");
const log = @import("log.zig");
const page_size = system.page_size;
const SystemCall = system.SystemCall;
const page_size = abi.page_size;
const SystemCall = abi.SystemCall;
/// User virtual addresses. PML4 index 224 — a user-exclusive region, far from
/// the identity map (low indices) and the vmm test address (index 128), so
@@ -79,7 +81,7 @@ pub var write_from_user: bool = false;
pub var write_count: u64 = 0; // total write syscalls served (for the heartbeat tests)
pub var exit_code: u64 = 0;
/// The system_call surface, dispatched on the saved system_call number (`system.SystemCall`).
/// The system_call surface, dispatched on the saved system_call number (`abi.SystemCall`).
/// This is the microkernel-minimal set — memory + scheduling only; file/device
/// I/O will arrive as IPC to user-space servers (docs/syscall.md). The result is
/// written back into the trap frame, since the entry paths restore user registers
@@ -203,9 +205,9 @@ fn systemDeviceEnumerate(state: *architecture.CpuState) void {
const maximum = architecture.systemCallArg(state, 1);
const t = scheduler.current();
if (t.aspace == 0 or buffer_ptr >= user_half_end) return fail(state);
const sz = @sizeOf(system.DeviceDescriptor);
const sz = @sizeOf(device_abi.DeviceDescriptor);
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
const out: [*]system.DeviceDescriptor = @ptrFromInt(buffer_ptr);
const out: [*]device_abi.DeviceDescriptor = @ptrFromInt(buffer_ptr);
architecture.setSystemCallResult(state, devices_broker.enumerate(out[0..@intCast(cap)]));
}
@@ -228,7 +230,7 @@ fn systemMmioMap(state: *architecture.CpuState) void {
const owner = devices_broker.ownerOf(device_id) orelse return fail(state);
if (owner != t.id) return fail(state); // not claimed by this process
const r = devices_broker.resourceOf(device_id, resource_index) orelse return fail(state);
if (r.kind != @intFromEnum(system.ResourceKind.memory)) return fail(state);
if (r.kind != @intFromEnum(device_abi.ResourceKind.memory)) return fail(state);
if (t.device_map_next == 0) t.device_map_next = device_arena_base;
const first = r.start & ~@as(u64, page_size - 1);
@@ -260,7 +262,7 @@ fn systemDeviceRegister(state: *architecture.CpuState) void {
const t = scheduler.current();
if (t.aspace == 0) return fail(state);
var descriptor: system.DeviceDescriptor = undefined;
var descriptor: device_abi.DeviceDescriptor = undefined;
if (!ipc.copyFromUser(t.aspace, descriptor_ptr, std.mem.asBytes(&descriptor))) return fail(state);
const id = devices_broker.register(parent_id, t.id, &descriptor) catch return fail(state);
@@ -284,7 +286,7 @@ fn ownedGsi(t: *scheduler.Task, device_id: u64, resource_index: u64) ?u32 {
const owner = devices_broker.ownerOf(device_id) orelse return null;
if (owner != t.id) return null;
const r = devices_broker.resourceOf(device_id, resource_index) orelse return null;
if (r.kind != @intFromEnum(system.ResourceKind.irq)) return null;
if (r.kind != @intFromEnum(device_abi.ResourceKind.irq)) return null;
if (r.start >= irq.maximum_gsi) return null;
return @intCast(r.start);
}
@@ -373,7 +375,7 @@ fn systemMmap(state: *architecture.CpuState) void {
}
for (frames[0..pages], 0..) |frame, i| {
const destination: [*]u8 = @ptrFromInt(system.physicalToVirtual(frame));
const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(frame));
@memset(destination[0..page_size], 0); // hand out zeroed memory
architecture.mapUserPageInto(t.aspace, base + i * page_size, frame, true, false); // RW + NX
}
@@ -428,7 +430,7 @@ pub fn run(blob: []const u8) RunError!void {
// Fill the code frame through the physmap (supervisor RW): the user-facing
// mapping is read-only, and this also sidesteps CR0.WP/SMAP. The tail is
// padded with int3 so a stray jump traps instead of sliding.
const code: [*]u8 = @ptrFromInt(system.physicalToVirtual(code_frame));
const code: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(code_frame));
@memcpy(code[0..blob.len], blob);
@memset(code[blob.len..page_size], 0xCC);
@@ -542,7 +544,7 @@ fn parseSegments(image: []const u8, segs: *[maximum_segments]Segment) InitError!
/// frame mapped into it — so no per-page rollback list is needed here.
fn loadPageInto(aspace: u64, image: []const u8, seg: Segment, page_index: u64) InitError!void {
const frame = pmm.alloc() orelse return error.OutOfMemory;
const destination: [*]u8 = @ptrFromInt(system.physicalToVirtual(frame));
const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(frame));
@memset(destination[0..page_size], 0);
const page_off = page_index * page_size;
if (page_off < seg.filesz) {