Split the system contract into boot-handoff / abi / device-abi

The `system` module (formerly `danos`) had become a grab-bag: it held the
loader<->kernel handoff *and* the kernel<->user ABI *and* the device wire types, in
one module three different audiences imported. Usage proved the seam — the
bootloader never touched the syscall/device ABI, and user space never touched the
boot handoff — so split it by audience, one module per contract:

  system/boot-handoff.zig       loader <-> kernel: BootInformation, Framebuffer,
                                MemoryMap, the VM layout + physicalToVirtual, kernel_abi
  system/abi.zig                kernel <-> user, core: SystemCall, mmap prot flags,
                                page_size, notify_badge_bit, ServiceId
  system/devices/device-abi.zig kernel <-> user, devices: DeviceDescriptor,
                                DeviceClass, ResourceDescriptor, ResourceKind, ...

device-abi is the devices sub-project's public interface, exposed as its own module
the way vfs exposes vfs-protocol — importable by user space, unlike the
kernel-internal device model it also feeds. That collapses a real duplication:
DeviceClass and ResourceKind were defined twice (device-model.zig and the contract,
kept "in sync by hand"); device-model now re-exports them from device-abi, so the
enum a driver matches on and the one the kernel classifies with are one type.

Each import now declares which contract it speaks: the bootloader imports only
boot-handoff; a driver only abi + device-abi (via the runtime); the kernel all
three. This also retires the `system` / `runtime.system` name overlap. page_size
lands in abi (it's part of the mmap contract user space aligns to); the bootloader
keeps its own local 4 KiB constant so it depends on nothing but the handoff.

All 21 importers rewired, docs updated to keep /system mapping to source. Build,
host tests, and the QEMU suite (36/36) all green.
This commit is contained in:
Daniel Samson
2026-07-10 18:08:51 +01:00
parent 47610e8ee2
commit be81394be3
37 changed files with 395 additions and 337 deletions
+16 -16
View File
@@ -1,8 +1,8 @@
const std = @import("std");
const uefi = std.os.uefi;
const elf = std.elf;
const system = @import("system");
const BootInformation = system.BootInformation;
const boot_handoff = @import("boot-handoff");
const BootInformation = boot_handoff.BootInformation;
const GraphicsOutput = uefi.protocol.GraphicsOutput;
const EdidActive = uefi.protocol.edid.Active;
const MemoryMapSlice = uefi.tables.MemoryMapSlice;
@@ -45,7 +45,7 @@ fn boot() !noreturn {
var boot_information: BootInformation = .{
// A missing GOP (a headless machine) is not fatal — hand the kernel a
// "no framebuffer" descriptor (base 0) and let it log to serial instead.
.framebuffer = queryFramebuffer(bs) catch system.Framebuffer{
.framebuffer = queryFramebuffer(bs) catch boot_handoff.Framebuffer{
.base = 0,
.width = 0,
.height = 0,
@@ -100,7 +100,7 @@ const Resolution = struct { width: u32, height: u32 };
/// Switch the GPU to the monitor's native resolution (when we can determine it)
/// and read the resulting graphics mode into our own framebuffer description.
fn queryFramebuffer(bs: *uefi.tables.BootServices) !system.Framebuffer {
fn queryFramebuffer(bs: *uefi.tables.BootServices) !boot_handoff.Framebuffer {
// Enumerate the handles carrying the Graphics Output Protocol. We go through
// handles (rather than locateProtocol) so we can also ask them for their EDID,
// which is what tells us the panel's native resolution.
@@ -132,7 +132,7 @@ fn queryFramebuffer(bs: *uefi.tables.BootServices) !system.Framebuffer {
/// Map a GOP pixel format to ours. bit_mask / blt_only have no linear 32bpp
/// layout we can paint into, so they're rejected.
fn pixelFormat(fmt: GraphicsOutput.PixelFormat) !system.PixelFormat {
fn pixelFormat(fmt: GraphicsOutput.PixelFormat) !boot_handoff.PixelFormat {
return switch (fmt) {
.red_green_blue_reserved_8_bit_per_color => .rgbx,
.blue_green_red_reserved_8_bit_per_color => .bgrx,
@@ -235,7 +235,7 @@ fn loadKernel(bs: *uefi.tables.BootServices, boot_information: *BootInformation)
// the first set of real page tables and switches CR3 before jumping in. They
// carry: an identity map of low RAM (so the loader's own code/stack executing
// the switch stays valid, and the low-linked kernel keeps working during the
// staged move), a physmap at system.physmap_base (the kernel's permanent way to
// staged move), a physmap at boot_handoff.physmap_base (the kernel's permanent way to
// reach physical memory), and 4 KiB mappings of any higher-half kernel segment.
// The kernel later builds its own precise tables (paging.init) and abandons
// these; they leak as reserved LoaderData (~a handful of frames).
@@ -308,7 +308,7 @@ fn buildBootstrapTables(bs: *uefi.tables.BootServices, boot_information: *const
var address: u64 = 0;
while (address < 4 * gib) : (address += 2 << 20) {
try pool.map2M(pml4, address, address); // identity
try pool.map2M(pml4, system.physicalToVirtual(address), address); // physmap
try pool.map2M(pml4, boot_handoff.physicalToVirtual(address), address); // physmap
}
// A framebuffer above the 4 GiB window needs its own identity + physmap
@@ -319,7 +319,7 @@ fn buildBootstrapTables(bs: *uefi.tables.BootServices, boot_information: *const
const fb_end = fb.base + @as(u64, fb.pitch) * fb.height;
while (p < fb_end) : (p += 2 << 20) {
try pool.map2M(pml4, p, p);
try pool.map2M(pml4, system.physicalToVirtual(p), p);
try pool.map2M(pml4, boot_handoff.physicalToVirtual(p), p);
}
}
@@ -328,7 +328,7 @@ fn buildBootstrapTables(bs: *uefi.tables.BootServices, boot_information: *const
// (and 4 KiB-mapping them would collide with the 2 MiB identity leaves), so
// only map segments that actually live in the higher half.
for (boot_information.kernel_segments[0..boot_information.kernel_segment_count]) |seg| {
if (seg.virtual < system.kernel_virt_base) continue;
if (seg.virtual < boot_handoff.kernel_virt_base) continue;
var off: u64 = 0;
while (off < seg.pages * page_size) : (off += page_size) {
try pool.map4K(pml4, seg.virtual + off, seg.physical + off);
@@ -463,14 +463,14 @@ fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_information: *BootIn
/// neutral form. Allocating the buffers can itself change the map (invalidating
/// the key), so retry until it takes. Both buffers are LoaderData, which survives
/// the exit, so the returned map stays valid for the kernel.
fn exitBootServices(bs: *uefi.tables.BootServices) !system.MemoryMap {
fn exitBootServices(bs: *uefi.tables.BootServices) !boot_handoff.MemoryMap {
var attempts: usize = 0;
while (attempts < 8) : (attempts += 1) {
const info = try bs.getMemoryMapInfo();
// Spare descriptors to absorb the growth from the allocations below.
const cap = info.len + 8;
const map_buffer = try bs.allocatePool(.loader_data, cap * info.descriptor_size);
const regions_buffer = try bs.allocatePool(.loader_data, cap * @sizeOf(system.MemoryRegion));
const regions_buffer = try bs.allocatePool(.loader_data, cap * @sizeOf(boot_handoff.MemoryRegion));
const map = bs.getMemoryMap(map_buffer) catch {
_ = bs.freePool(map_buffer.ptr) catch {};
_ = bs.freePool(regions_buffer.ptr) catch {};
@@ -492,8 +492,8 @@ fn exitBootServices(bs: *uefi.tables.BootServices) !system.MemoryMap {
/// into `out` (sized for at least `map.info.len` regions). Adjacent regions of
/// the same kind are coalesced. This is the loader's job precisely so the kernel
/// never sees UEFI's vocabulary — the same seam the framebuffer already uses.
fn convertMemoryMap(map: MemoryMapSlice, out: []u8) system.MemoryMap {
const regions: [*]system.MemoryRegion = @ptrCast(@alignCast(out.ptr));
fn convertMemoryMap(map: MemoryMapSlice, out: []u8) boot_handoff.MemoryMap {
const regions: [*]boot_handoff.MemoryRegion = @ptrCast(@alignCast(out.ptr));
// We're about to call boot-services memory `usable`, but our own stack lives
// in it and the kernel starts out running on it. Keep the region holding the
// current stack pointer reserved so it's never handed out.
@@ -510,14 +510,14 @@ fn convertMemoryMap(map: MemoryMapSlice, out: []u8) system.MemoryMap {
if (d.number_of_pages == 0) continue;
var kind = classify(d);
// The descriptor we're executing on stays reserved (see rsp above).
const region_end = d.physical_start + d.number_of_pages * system.page_size;
const region_end = d.physical_start + d.number_of_pages * page_size;
if (kind == .usable and rsp >= d.physical_start and rsp < region_end) kind = .reserved;
// Coalesce with the previous region if it's the same kind and contiguous.
if (count > 0) {
const previous = &regions[count - 1];
if (previous.kind == kind and
previous.base + previous.pages * system.page_size == d.physical_start)
previous.base + previous.pages * page_size == d.physical_start)
{
previous.pages += d.number_of_pages;
continue;
@@ -544,7 +544,7 @@ fn convertMemoryMap(map: MemoryMapSlice, out: []u8) system.MemoryMap {
/// ever the firmware's (the one live piece, our stack, is reserved by the caller).
/// Anything unrecognised is `reserved` — the safe default; our own LoaderData (the
/// kernel image and these buffers) lands there and stays reserved.
fn classify(d: *const uefi.tables.MemoryDescriptor) system.MemoryKind {
fn classify(d: *const uefi.tables.MemoryDescriptor) boot_handoff.MemoryKind {
if (!d.attribute.wb) return .mmio;
return switch (d.type) {
.conventional_memory, .boot_services_code, .boot_services_data => .usable,