Compare commits
10
Commits
8754d4e46a
...
56110b0019
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
56110b0019 | ||
|
|
afbf10f7fc | ||
|
|
b61b7775b9 | ||
|
|
be81394be3 | ||
|
|
47610e8ee2 | ||
|
|
193fd71a50 | ||
|
|
1c2b3ae64d | ||
|
|
d19a0ae38d | ||
|
|
3d1de37d0e | ||
|
|
ceacc6b514 |
@@ -4,3 +4,6 @@ zig-out/
|
||||
|
||||
# JetBrains IDE
|
||||
.idea/
|
||||
|
||||
.claude/
|
||||
.github/
|
||||
@@ -3,7 +3,7 @@ Codename: Shodan
|
||||
Version: 1
|
||||
|
||||
A small operating system, written from scratch in Zig — a bootloader (`boot/`)
|
||||
and a microkernel (`system/kernel/`), sharing a neutral handoff contract (`system/danos.zig`).
|
||||
and a microkernel (`system/kernel/`), sharing a neutral handoff contract (`system/boot-handoff.zig`).
|
||||
It boots x86-64 via UEFI, and so far has a framebuffer console, a physical frame
|
||||
allocator, its own paging with W^X permissions, interrupt/exception handling, a
|
||||
LAPIC timer, a kernel heap, a fixed-priority preemptive scheduler, and in-kernel IPC
|
||||
@@ -30,8 +30,10 @@ channels. See [`docs/`](docs/README.md) for how each piece works.
|
||||
zig build
|
||||
```
|
||||
|
||||
Produces the UEFI bootloader (`zig-out/bin/BOOTX64.efi`) and the kernel ELF
|
||||
(`zig-out/bin/kernel`).
|
||||
Produces a FHS-shaped `zig-out/` that *is* the danos filesystem and the boot volume:
|
||||
the UEFI bootloader at `zig-out/EFI/BOOT/BOOTX64.efi`, the kernel at
|
||||
`zig-out/system/kernel`, init at `zig-out/system/services/init`, drivers under
|
||||
`zig-out/system/drivers/`, and the initial-ramdisk at `zig-out/boot/`.
|
||||
|
||||
## Run
|
||||
|
||||
|
||||
+39
-37
@@ -1,22 +1,24 @@
|
||||
const std = @import("std");
|
||||
const uefi = std.os.uefi;
|
||||
const elf = std.elf;
|
||||
const danos = @import("danos");
|
||||
const BootInformation = danos.BootInformation;
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const BootInformation = boot_handoff.BootInformation;
|
||||
const GraphicsOutput = uefi.protocol.GraphicsOutput;
|
||||
const EdidActive = uefi.protocol.edid.Active;
|
||||
const MemoryMapSlice = uefi.tables.MemoryMapSlice;
|
||||
|
||||
/// Name of the kernel ELF on the boot volume (installed to the ESP root by
|
||||
/// build.zig). UEFI wants a UTF-16, null-terminated path.
|
||||
const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("kernel");
|
||||
// The boot volume is the FHS-shaped zig-out (see build.zig / docs/README.md), so the
|
||||
// loader reads each artifact from its addressed FHS path. UEFI paths use backslashes;
|
||||
// the FAT driver walks the components itself, so no per-directory dance is needed.
|
||||
|
||||
/// Path of the init program on the boot volume (UEFI paths use backslashes;
|
||||
/// the FAT driver walks the components itself, so no directory dance needed).
|
||||
const init_file_name = std.unicode.utf8ToUtf16LeStringLiteral("sbin\\init");
|
||||
/// The kernel image: /system/kernel.
|
||||
const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("system\\kernel");
|
||||
|
||||
/// Path of the initrd image on the boot volume (the VFS server + drivers).
|
||||
const initrd_file_name = std.unicode.utf8ToUtf16LeStringLiteral("initrd.img");
|
||||
/// The init program: /system/services/init.
|
||||
const init_file_name = std.unicode.utf8ToUtf16LeStringLiteral("system\\services\\init");
|
||||
|
||||
/// The initial-ramdisk (the VFS server + drivers), in /boot.
|
||||
const initial_ramdisk_file_name = std.unicode.utf8ToUtf16LeStringLiteral("boot\\initial-ramdisk.img");
|
||||
|
||||
/// Physical page size, and the sentinel UEFI uses to seek to end-of-file.
|
||||
const page_size = 4096;
|
||||
@@ -43,7 +45,7 @@ fn boot() !noreturn {
|
||||
var boot_information: BootInformation = .{
|
||||
// A missing GOP (a headless machine) is not fatal — hand the kernel a
|
||||
// "no framebuffer" descriptor (base 0) and let it log to serial instead.
|
||||
.framebuffer = queryFramebuffer(bs) catch danos.Framebuffer{
|
||||
.framebuffer = queryFramebuffer(bs) catch boot_handoff.Framebuffer{
|
||||
.base = 0,
|
||||
.width = 0,
|
||||
.height = 0,
|
||||
@@ -61,16 +63,16 @@ fn boot() !noreturn {
|
||||
|
||||
const entry = try loadKernel(bs, &boot_information);
|
||||
|
||||
// Best effort: a volume without sbin/init still boots (kernel-only).
|
||||
// Best effort: a volume without /system/services/init still boots (kernel-only).
|
||||
loadInit(bs, &boot_information) catch |err| {
|
||||
log("danos: no sbin/init (");
|
||||
log("danos: no /system/services/init (");
|
||||
logBytes(@errorName(err));
|
||||
log(") - booting without user space\r\n");
|
||||
};
|
||||
|
||||
// Best effort: the initrd (VFS server + drivers) is optional too.
|
||||
loadInitrd(bs, &boot_information) catch |err| {
|
||||
log("danos: no initrd (");
|
||||
// Best effort: the initial_ramdisk (VFS server + drivers) is optional too.
|
||||
loadInitialRamdisk(bs, &boot_information) catch |err| {
|
||||
log("danos: no initial_ramdisk (");
|
||||
logBytes(@errorName(err));
|
||||
log(")\r\n");
|
||||
};
|
||||
@@ -98,7 +100,7 @@ const Resolution = struct { width: u32, height: u32 };
|
||||
|
||||
/// Switch the GPU to the monitor's native resolution (when we can determine it)
|
||||
/// and read the resulting graphics mode into our own framebuffer description.
|
||||
fn queryFramebuffer(bs: *uefi.tables.BootServices) !danos.Framebuffer {
|
||||
fn queryFramebuffer(bs: *uefi.tables.BootServices) !boot_handoff.Framebuffer {
|
||||
// Enumerate the handles carrying the Graphics Output Protocol. We go through
|
||||
// handles (rather than locateProtocol) so we can also ask them for their EDID,
|
||||
// which is what tells us the panel's native resolution.
|
||||
@@ -130,7 +132,7 @@ fn queryFramebuffer(bs: *uefi.tables.BootServices) !danos.Framebuffer {
|
||||
|
||||
/// Map a GOP pixel format to ours. bit_mask / blt_only have no linear 32bpp
|
||||
/// layout we can paint into, so they're rejected.
|
||||
fn pixelFormat(fmt: GraphicsOutput.PixelFormat) !danos.PixelFormat {
|
||||
fn pixelFormat(fmt: GraphicsOutput.PixelFormat) !boot_handoff.PixelFormat {
|
||||
return switch (fmt) {
|
||||
.red_green_blue_reserved_8_bit_per_color => .rgbx,
|
||||
.blue_green_red_reserved_8_bit_per_color => .bgrx,
|
||||
@@ -233,7 +235,7 @@ fn loadKernel(bs: *uefi.tables.BootServices, boot_information: *BootInformation)
|
||||
// the first set of real page tables and switches CR3 before jumping in. They
|
||||
// carry: an identity map of low RAM (so the loader's own code/stack executing
|
||||
// the switch stays valid, and the low-linked kernel keeps working during the
|
||||
// staged move), a physmap at danos.physmap_base (the kernel's permanent way to
|
||||
// staged move), a physmap at boot_handoff.physmap_base (the kernel's permanent way to
|
||||
// reach physical memory), and 4 KiB mappings of any higher-half kernel segment.
|
||||
// The kernel later builds its own precise tables (paging.init) and abandons
|
||||
// these; they leak as reserved LoaderData (~a handful of frames).
|
||||
@@ -306,7 +308,7 @@ fn buildBootstrapTables(bs: *uefi.tables.BootServices, boot_information: *const
|
||||
var address: u64 = 0;
|
||||
while (address < 4 * gib) : (address += 2 << 20) {
|
||||
try pool.map2M(pml4, address, address); // identity
|
||||
try pool.map2M(pml4, danos.physicalToVirtual(address), address); // physmap
|
||||
try pool.map2M(pml4, boot_handoff.physicalToVirtual(address), address); // physmap
|
||||
}
|
||||
|
||||
// A framebuffer above the 4 GiB window needs its own identity + physmap
|
||||
@@ -317,7 +319,7 @@ fn buildBootstrapTables(bs: *uefi.tables.BootServices, boot_information: *const
|
||||
const fb_end = fb.base + @as(u64, fb.pitch) * fb.height;
|
||||
while (p < fb_end) : (p += 2 << 20) {
|
||||
try pool.map2M(pml4, p, p);
|
||||
try pool.map2M(pml4, danos.physicalToVirtual(p), p);
|
||||
try pool.map2M(pml4, boot_handoff.physicalToVirtual(p), p);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -326,7 +328,7 @@ fn buildBootstrapTables(bs: *uefi.tables.BootServices, boot_information: *const
|
||||
// (and 4 KiB-mapping them would collide with the 2 MiB identity leaves), so
|
||||
// only map segments that actually live in the higher half.
|
||||
for (boot_information.kernel_segments[0..boot_information.kernel_segment_count]) |seg| {
|
||||
if (seg.virtual < danos.kernel_virt_base) continue;
|
||||
if (seg.virtual < boot_handoff.kernel_virt_base) continue;
|
||||
var off: u64 = 0;
|
||||
while (off < seg.pages * page_size) : (off += page_size) {
|
||||
try pool.map4K(pml4, seg.virtual + off, seg.physical + off);
|
||||
@@ -387,21 +389,21 @@ fn loadFile(bs: *uefi.tables.BootServices, name: [*:0]const u16) ![]u8 {
|
||||
return image[0..size];
|
||||
}
|
||||
|
||||
/// Ferry the init program (sbin/init) to the kernel. The kernel does the ELF
|
||||
/// Ferry the init program (/system/services/init) to the kernel. The kernel does the ELF
|
||||
/// loading itself (into ring-3 mappings) — the loader just carries the bytes.
|
||||
fn loadInit(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||
const image = try loadFile(bs, init_file_name);
|
||||
boot_information.init_base = @intFromPtr(image.ptr);
|
||||
boot_information.init_len = image.len;
|
||||
log("danos: sbin/init loaded\r\n");
|
||||
log("danos: /system/services/init loaded\r\n");
|
||||
}
|
||||
|
||||
/// Ferry the initrd (the VFS server + drivers) to the kernel, same as init.
|
||||
fn loadInitrd(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||
const image = try loadFile(bs, initrd_file_name);
|
||||
boot_information.initrd_base = @intFromPtr(image.ptr);
|
||||
boot_information.initrd_len = image.len;
|
||||
log("danos: initrd loaded\r\n");
|
||||
/// Ferry the initial_ramdisk (the VFS server + drivers) to the kernel, same as init.
|
||||
fn loadInitialRamdisk(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||
const image = try loadFile(bs, initial_ramdisk_file_name);
|
||||
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
||||
boot_information.initial_ramdisk_len = image.len;
|
||||
log("danos: initial_ramdisk loaded\r\n");
|
||||
}
|
||||
|
||||
/// Validate the ELF, copy every PT_LOAD segment to its physical address, and
|
||||
@@ -461,14 +463,14 @@ fn loadElf(bs: *uefi.tables.BootServices, image: []u8, boot_information: *BootIn
|
||||
/// neutral form. Allocating the buffers can itself change the map (invalidating
|
||||
/// the key), so retry until it takes. Both buffers are LoaderData, which survives
|
||||
/// the exit, so the returned map stays valid for the kernel.
|
||||
fn exitBootServices(bs: *uefi.tables.BootServices) !danos.MemoryMap {
|
||||
fn exitBootServices(bs: *uefi.tables.BootServices) !boot_handoff.MemoryMap {
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 8) : (attempts += 1) {
|
||||
const info = try bs.getMemoryMapInfo();
|
||||
// Spare descriptors to absorb the growth from the allocations below.
|
||||
const cap = info.len + 8;
|
||||
const map_buffer = try bs.allocatePool(.loader_data, cap * info.descriptor_size);
|
||||
const regions_buffer = try bs.allocatePool(.loader_data, cap * @sizeOf(danos.MemoryRegion));
|
||||
const regions_buffer = try bs.allocatePool(.loader_data, cap * @sizeOf(boot_handoff.MemoryRegion));
|
||||
const map = bs.getMemoryMap(map_buffer) catch {
|
||||
_ = bs.freePool(map_buffer.ptr) catch {};
|
||||
_ = bs.freePool(regions_buffer.ptr) catch {};
|
||||
@@ -490,8 +492,8 @@ fn exitBootServices(bs: *uefi.tables.BootServices) !danos.MemoryMap {
|
||||
/// into `out` (sized for at least `map.info.len` regions). Adjacent regions of
|
||||
/// the same kind are coalesced. This is the loader's job precisely so the kernel
|
||||
/// never sees UEFI's vocabulary — the same seam the framebuffer already uses.
|
||||
fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
||||
const regions: [*]danos.MemoryRegion = @ptrCast(@alignCast(out.ptr));
|
||||
fn convertMemoryMap(map: MemoryMapSlice, out: []u8) boot_handoff.MemoryMap {
|
||||
const regions: [*]boot_handoff.MemoryRegion = @ptrCast(@alignCast(out.ptr));
|
||||
// We're about to call boot-services memory `usable`, but our own stack lives
|
||||
// in it and the kernel starts out running on it. Keep the region holding the
|
||||
// current stack pointer reserved so it's never handed out.
|
||||
@@ -508,14 +510,14 @@ fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
||||
if (d.number_of_pages == 0) continue;
|
||||
var kind = classify(d);
|
||||
// The descriptor we're executing on stays reserved (see rsp above).
|
||||
const region_end = d.physical_start + d.number_of_pages * danos.page_size;
|
||||
const region_end = d.physical_start + d.number_of_pages * page_size;
|
||||
if (kind == .usable and rsp >= d.physical_start and rsp < region_end) kind = .reserved;
|
||||
|
||||
// Coalesce with the previous region if it's the same kind and contiguous.
|
||||
if (count > 0) {
|
||||
const previous = ®ions[count - 1];
|
||||
if (previous.kind == kind and
|
||||
previous.base + previous.pages * danos.page_size == d.physical_start)
|
||||
previous.base + previous.pages * page_size == d.physical_start)
|
||||
{
|
||||
previous.pages += d.number_of_pages;
|
||||
continue;
|
||||
@@ -542,7 +544,7 @@ fn convertMemoryMap(map: MemoryMapSlice, out: []u8) danos.MemoryMap {
|
||||
/// ever the firmware's (the one live piece, our stack, is reserved by the caller).
|
||||
/// Anything unrecognised is `reserved` — the safe default; our own LoaderData (the
|
||||
/// kernel image and these buffers) lands there and stays reserved.
|
||||
fn classify(d: *const uefi.tables.MemoryDescriptor) danos.MemoryKind {
|
||||
fn classify(d: *const uefi.tables.MemoryDescriptor) boot_handoff.MemoryKind {
|
||||
if (!d.attribute.wb) return .mmio;
|
||||
return switch (d.type) {
|
||||
.conventional_memory, .boot_services_code, .boot_services_data => .usable,
|
||||
|
||||
@@ -58,6 +58,7 @@ fn addUserBinary(
|
||||
b: *std.Build,
|
||||
target: std.Build.ResolvedTarget,
|
||||
runtime_module: *std.Build.Module,
|
||||
posix_module: *std.Build.Module,
|
||||
name: []const u8,
|
||||
root: []const u8,
|
||||
) *std.Build.Step.Compile {
|
||||
@@ -74,6 +75,9 @@ fn addUserBinary(
|
||||
.stack_protector = false,
|
||||
.imports = &.{
|
||||
.{ .name = "runtime", .module = runtime_module },
|
||||
// POSIX/C compatibility layer, available to any program that wants it
|
||||
// (danos-native code uses `runtime` directly). See library/posix/.
|
||||
.{ .name = "posix", .module = posix_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
@@ -91,11 +95,23 @@ pub fn build(b: *std.Build) void {
|
||||
const target = b.standardTargetOptions(.{});
|
||||
const optimize = b.standardOptimizeOption(.{});
|
||||
|
||||
// Shared handoff definitions (BootInformation, Framebuffer, ...). No target is set,
|
||||
// so the module inherits the target of whichever binary imports it — the
|
||||
// freestanding kernel or the UEFI bootloader.
|
||||
const danos_module = b.addModule("danos", .{
|
||||
.root_source_file = b.path("system/danos.zig"),
|
||||
// The three shared contracts, each with its own audience so every import
|
||||
// declares which one it speaks (no target is set, so each inherits the target of
|
||||
// whichever binary imports it). See docs/coding-standards.md.
|
||||
// boot-handoff : loader <-> kernel (BootInformation, framebuffer, VM layout)
|
||||
// abi : kernel <-> user, core (SystemCall, mmap prot flags, page_size)
|
||||
// device-abi : kernel <-> user, devices (DeviceDescriptor, DeviceClass, ...)
|
||||
const boot_handoff_module = b.addModule("boot-handoff", .{
|
||||
.root_source_file = b.path("system/boot-handoff.zig"),
|
||||
});
|
||||
const abi_module = b.addModule("abi", .{
|
||||
.root_source_file = b.path("system/abi.zig"),
|
||||
});
|
||||
// The devices sub-project's public interface (the flat wire types), exposed as
|
||||
// its own module like vfs-protocol — importable by user space, unlike the
|
||||
// kernel-internal device model it also feeds (system/devices/device-model.zig).
|
||||
const device_abi_module = b.addModule("device-abi", .{
|
||||
.root_source_file = b.path("system/devices/device-abi.zig"),
|
||||
});
|
||||
|
||||
// Kernel tunables (maximum_cpus, stack sizes, tick rate). A dependency-free module of
|
||||
@@ -111,7 +127,8 @@ pub fn build(b: *std.Build) void {
|
||||
const architecture_module = b.addModule("architecture", .{
|
||||
.root_source_file = b.path("system/kernel/architecture/x86_64/cpu.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "danos", .module = danos_module }, // paging uses the shared BootInformation/memory-map types
|
||||
.{ .name = "boot-handoff", .module = boot_handoff_module }, // paging uses BootInformation/memory-map + physicalToVirtual
|
||||
.{ .name = "abi", .module = abi_module }, // paging works in page_size units
|
||||
.{ .name = "parameters", .module = parameters_module }, // maximum_cpus, ist_stack_size, timer_hz
|
||||
},
|
||||
});
|
||||
@@ -130,7 +147,9 @@ pub fn build(b: *std.Build) void {
|
||||
const platform_module = b.addModule("platform", .{
|
||||
.root_source_file = b.path("system/devices/platform.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "danos", .module = danos_module }, // BootInformation (carries the ACPI RSDP)
|
||||
.{ .name = "boot-handoff", .module = boot_handoff_module }, // BootInformation (carries the ACPI RSDP), physicalToVirtual
|
||||
.{ .name = "abi", .module = abi_module }, // acpi.zig works in page_size units
|
||||
.{ .name = "device-abi", .module = device_abi_module }, // device-model's DeviceClass/ResourceKind live here
|
||||
.{ .name = "parameters", .module = parameters_module }, // maximum_cpus (the discovery pool)
|
||||
},
|
||||
});
|
||||
@@ -144,23 +163,40 @@ pub fn build(b: *std.Build) void {
|
||||
.root_source_file = b.path("system/services/vfs/protocol.zig"),
|
||||
});
|
||||
|
||||
// The user-space runtime library (a nascent libc): system_call wrappers, the
|
||||
// C-convention heap, IPC helpers, the process start shim. Compiled into every
|
||||
// user binary (see addUserBinary), so it inherits each exe's `.large` code
|
||||
// model — do NOT set a target/code_model here. It imports `danos` for the
|
||||
// shared SystemCall numbers and `vfs-protocol` for the file API.
|
||||
// The danos-native user-space runtime: system_call wrappers, the C-convention
|
||||
// heap, IPC helpers, the process start shim, device access. This is the stable
|
||||
// application ABI; POSIX compatibility is a separate library on top (see below).
|
||||
// Compiled into every user binary (see addUserBinary), so it inherits each exe's
|
||||
// `.large` code model — do NOT set a target/code_model here. It imports `abi`
|
||||
// for the shared SystemCall numbers / mmap flags, `device-abi` for the device
|
||||
// types its `device` helper wraps, and re-exports `vfs-protocol` for the VFS
|
||||
// server. It never touches `boot-handoff` — user space has no business with the
|
||||
// loader↔kernel handoff.
|
||||
const runtime_module = b.addModule("runtime", .{
|
||||
.root_source_file = b.path("library/runtime/runtime.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "danos", .module = danos_module },
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
.{ .name = "device-abi", .module = device_abi_module },
|
||||
.{ .name = "vfs-protocol", .module = vfs_protocol_module },
|
||||
},
|
||||
});
|
||||
|
||||
// The initrd container format, shared by the kernel (unpacks it) and the
|
||||
// build-time packer tools/mkinitrd.zig (produces it). No dependencies.
|
||||
const initrd_module = b.addModule("initrd", .{
|
||||
.root_source_file = b.path("system/initrd.zig"),
|
||||
// The POSIX / C compatibility layer, a separate library layered strictly over the
|
||||
// runtime (it calls the runtime's IPC/heap, never system calls directly). This is
|
||||
// the one place POSIX/C spellings are allowed verbatim — see docs/coding-standards.md
|
||||
// and library/posix/posix.zig.
|
||||
const posix_module = b.addModule("posix", .{
|
||||
.root_source_file = b.path("library/posix/posix.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "runtime", .module = runtime_module },
|
||||
.{ .name = "vfs-protocol", .module = vfs_protocol_module },
|
||||
},
|
||||
});
|
||||
|
||||
// The initial_ramdisk container format, shared by the kernel (unpacks it) and the
|
||||
// build-time packer tools/make-initial-ramdisk.py (produces it). No dependencies.
|
||||
const initial_ramdisk_module = b.addModule("initial-ramdisk", .{
|
||||
.root_source_file = b.path("system/initial-ramdisk.zig"),
|
||||
});
|
||||
|
||||
// Compile-time configuration the kernel reads as `@import("build_options")`. The
|
||||
@@ -183,7 +219,7 @@ pub fn build(b: *std.Build) void {
|
||||
const exe = b.addExecutable(.{
|
||||
.name = "kernel",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("system/kernel/main.zig"),
|
||||
.root_source_file = b.path("system/kernel/kernel.zig"),
|
||||
.target = kernel_target,
|
||||
.optimize = optimize,
|
||||
.code_model = .kernel, // kernel runs in the top 2 GiB (higher half)
|
||||
@@ -193,12 +229,14 @@ pub fn build(b: *std.Build) void {
|
||||
.stack_check = false, // stack-probe calls have no runtime to land in
|
||||
.stack_protector = false,
|
||||
.imports = &.{
|
||||
.{ .name = "danos", .module = danos_module },
|
||||
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
.{ .name = "device-abi", .module = device_abi_module },
|
||||
.{ .name = "architecture", .module = architecture_module },
|
||||
.{ .name = "platform", .module = platform_module },
|
||||
.{ .name = "parameters", .module = parameters_module },
|
||||
.{ .name = "build_options", .module = build_options_module },
|
||||
.{ .name = "initrd", .module = initrd_module },
|
||||
.{ .name = "initial-ramdisk", .module = initial_ramdisk_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
@@ -214,43 +252,64 @@ pub fn build(b: *std.Build) void {
|
||||
// (.text at 1 MiB), which the loader allocates and copies into.
|
||||
exe.image_base = 0xFFFFFFFF80100000;
|
||||
|
||||
b.installArtifact(exe);
|
||||
// Everything installs into a FHS-shaped zig-out: it IS the danos filesystem *and*
|
||||
// the boot volume. Each binary lands at its addressed, leaf-collapsed path — the
|
||||
// kernel at zig-out/system/kernel (from system/kernel/kernel.zig), init at
|
||||
// zig-out/system/services/init, and so on (see docs/README.md). The bootloader
|
||||
// then loads these FHS paths off the volume.
|
||||
const kernel_install = b.addInstallArtifact(exe, .{ .dest_dir = .{ .override = .{ .custom = "system" } } });
|
||||
b.getInstallStep().dependOn(&kernel_install.step);
|
||||
|
||||
// --- /sbin/init: the first user-space program ---
|
||||
// --- init: the first user-space program (a system service) ---
|
||||
// Built by the shared user-binary recipe (see addUserBinary): freestanding,
|
||||
// linked into the kernel's user region against the `runtime` runtime library, and
|
||||
// started in ring 3 by the kernel's user-ELF loader.
|
||||
const init_exe = addUserBinary(b, kernel_target, runtime_module, "init", "system/services/init/init.zig");
|
||||
b.installArtifact(init_exe);
|
||||
const init_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, "init", "system/services/init/init.zig");
|
||||
const init_install = b.addInstallArtifact(init_exe, .{ .dest_dir = .{ .override = .{ .custom = "system/services" } } });
|
||||
b.getInstallStep().dependOn(&init_install.step);
|
||||
|
||||
// --- initrd: a bundle of extra user binaries (VFS server + drivers) ---
|
||||
// --- initial_ramdisk: a bundle of extra user binaries (VFS server + drivers) ---
|
||||
// Each is built by the same user-binary recipe, then packed into one image by
|
||||
// the host-side mkinitrd tool. The bootloader ferries the image to the kernel,
|
||||
// which unpacks it and spawns each program (system/initrd.zig).
|
||||
const vfs_exe = addUserBinary(b, kernel_target, runtime_module, "vfs", "system/services/vfs/vfs.zig");
|
||||
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, "vfs-test", "system/services/vfs/vfs-test.zig");
|
||||
const hpetd_exe = addUserBinary(b, kernel_target, runtime_module, "hpetd", "system/drivers/hpetd/hpetd.zig");
|
||||
const busd_exe = addUserBinary(b, kernel_target, runtime_module, "busd", "system/drivers/busd/busd.zig");
|
||||
// the host-side make-initial-ramdisk tool. The bootloader ferries the image to the kernel,
|
||||
// which unpacks it and spawns each program (system/initial-ramdisk.zig).
|
||||
const vfs_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, "vfs", "system/services/vfs/vfs.zig");
|
||||
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, "vfs-test", "system/services/vfs/vfs-test.zig");
|
||||
const hpet_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, "hpet", "system/drivers/hpet/hpet.zig");
|
||||
const bus_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, "bus", "system/drivers/bus/bus.zig");
|
||||
const device_manager_exe = addUserBinary(b, kernel_target, runtime_module, posix_module, "device-manager", "system/services/device-manager/device-manager.zig");
|
||||
|
||||
// Pack the user binaries into the initrd image with the host-side Python tool
|
||||
// Pack the user binaries into the initial_ramdisk image with the host-side Python tool
|
||||
// (the container format is trivial, and Python sidesteps std API churn). Args:
|
||||
// mkinitrd.py <out> [<name> <file>]... — one name/file pair per binary.
|
||||
// make-initial-ramdisk.py <out> [<name> <file>]... — one name/file pair per binary.
|
||||
const mk_run = b.addSystemCommand(&.{"python3"});
|
||||
mk_run.addFileArg(b.path("tools/mkinitrd.py"));
|
||||
const initrd_img = mk_run.addOutputFileArg("initrd.img");
|
||||
mk_run.addFileArg(b.path("tools/make-initial-ramdisk.py"));
|
||||
const initial_ramdisk_img = mk_run.addOutputFileArg("initial-ramdisk.img");
|
||||
mk_run.addArg("vfs");
|
||||
mk_run.addFileArg(vfs_exe.getEmittedBin());
|
||||
mk_run.addArg("vfs-test");
|
||||
mk_run.addFileArg(vfstest_exe.getEmittedBin());
|
||||
mk_run.addArg("hpetd");
|
||||
mk_run.addFileArg(hpetd_exe.getEmittedBin());
|
||||
mk_run.addArg("busd");
|
||||
mk_run.addFileArg(busd_exe.getEmittedBin());
|
||||
mk_run.addArg("hpet");
|
||||
mk_run.addFileArg(hpet_exe.getEmittedBin());
|
||||
mk_run.addArg("bus");
|
||||
mk_run.addFileArg(bus_exe.getEmittedBin());
|
||||
mk_run.addArg("device-manager");
|
||||
mk_run.addFileArg(device_manager_exe.getEmittedBin());
|
||||
|
||||
// Install the image to zig-out/bin (so the QEMU test harness picks it up like
|
||||
// the other binaries). The run-x86-64 ESP install is added below.
|
||||
const initrd_install = b.addInstallFile(initrd_img, "bin/initrd.img");
|
||||
b.getInstallStep().dependOn(&initrd_install.step);
|
||||
// Also install the packed binaries to their FHS homes, so zig-out is a true image
|
||||
// of the filesystem — even though at boot they arrive inside the initial-ramdisk.
|
||||
for ([_]struct { *std.Build.Step.Compile, []const u8 }{
|
||||
.{ vfs_exe, "system/services" },
|
||||
.{ device_manager_exe, "system/services" },
|
||||
.{ hpet_exe, "system/drivers" },
|
||||
.{ bus_exe, "system/drivers" },
|
||||
}) |entry| {
|
||||
const step = b.addInstallArtifact(entry[0], .{ .dest_dir = .{ .override = .{ .custom = entry[1] } } });
|
||||
b.getInstallStep().dependOn(&step.step);
|
||||
}
|
||||
|
||||
// The initial-ramdisk itself installs to /boot (with the loaders).
|
||||
const initial_ramdisk_install = b.addInstallFile(initial_ramdisk_img, "boot/initial-ramdisk.img");
|
||||
b.getInstallStep().dependOn(&initial_ramdisk_install.step);
|
||||
|
||||
// Boot methods live in boot/, one per way of getting the kernel running.
|
||||
// Each is its own binary/entry (a loader is built for its own target); today
|
||||
@@ -265,12 +324,16 @@ pub fn build(b: *std.Build) void {
|
||||
}),
|
||||
.optimize = optimize,
|
||||
.imports = &.{
|
||||
.{ .name = "danos", .module = danos_module },
|
||||
// The bootloader speaks only the handoff contract — never the user ABI.
|
||||
.{ .name = "boot-handoff", .module = boot_handoff_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
|
||||
b.installArtifact(efiexe);
|
||||
// UEFI firmware requires the removable-media loader at exactly \EFI\BOOT\BOOTX64.efi,
|
||||
// so that path is fixed by the firmware (it is /boot's EFI stub, conceptually).
|
||||
const efi_install = b.addInstallArtifact(efiexe, .{ .dest_dir = .{ .override = .{ .custom = "EFI/BOOT" } } });
|
||||
b.getInstallStep().dependOn(&efi_install.step);
|
||||
|
||||
// --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF ---
|
||||
// Firmware lives in different places per OS/distro, so probe the known
|
||||
@@ -301,21 +364,8 @@ pub fn build(b: *std.Build) void {
|
||||
"/usr/local/share/qemu/edk2-i386-vars.fd", // macOS Homebrew (Intel)
|
||||
});
|
||||
|
||||
// Assemble an EFI System Partition layout: esp/EFI/BOOT/BOOTX64.efi
|
||||
const efi_install = b.addInstallArtifact(efiexe, .{
|
||||
.dest_dir = .{ .override = .{ .custom = "esp/EFI/BOOT" } },
|
||||
});
|
||||
// The bootloader loads the kernel by name from the volume root, so drop the
|
||||
// kernel ELF at esp/kernel.
|
||||
const kernel_install = b.addInstallArtifact(exe, .{
|
||||
.dest_dir = .{ .override = .{ .custom = "esp" } },
|
||||
});
|
||||
// The bootloader loads init from sbin/init on the same volume.
|
||||
const init_install = b.addInstallArtifact(init_exe, .{
|
||||
.dest_dir = .{ .override = .{ .custom = "esp/sbin" } },
|
||||
});
|
||||
// ...and the initrd (VFS server + drivers) from the volume root.
|
||||
const initrd_esp_install = b.addInstallFile(initrd_img, "esp/initrd.img");
|
||||
// The FHS zig-out (installed above) *is* the boot volume — no separate ESP to
|
||||
// assemble. QEMU presents it to the guest as a FAT drive below.
|
||||
|
||||
// The firmware needs to write NVRAM, so give it a writable copy of the vars.
|
||||
const vars_copy = b.addSystemCommand(&.{ "cp", "-f", ovmf_vars });
|
||||
@@ -332,10 +382,10 @@ pub fn build(b: *std.Build) void {
|
||||
});
|
||||
run_efi.addArg("-drive");
|
||||
run_efi.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
||||
// Present the ESP directory to the guest as a FAT drive.
|
||||
// Present the FHS zig-out to the guest as a FAT drive — it is the boot volume.
|
||||
run_efi.addArgs(&.{
|
||||
"-drive",
|
||||
b.fmt("format=raw,file=fat:rw:{s}/esp", .{b.install_path}),
|
||||
b.fmt("format=raw,file=fat:rw:{s}", .{b.install_path}),
|
||||
"-net",
|
||||
"none",
|
||||
// Emulated display advertising 1280x720 as its native (EDID preferred)
|
||||
@@ -346,16 +396,19 @@ pub fn build(b: *std.Build) void {
|
||||
"-device",
|
||||
"VGA,edid=on,xres=1280,yres=720",
|
||||
});
|
||||
// Always capture the guest's serial0 (the kernel's machine-readable log) to a
|
||||
// timestamped file under zig-out, so each run leaves its own log behind.
|
||||
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ b.install_path, timestamp(b) });
|
||||
// Capture the guest's serial0 (danos's machine-readable log) to the qemu-test
|
||||
// scratch area — a dev/host artifact, kept out of the FHS boot volume we mount.
|
||||
// (/var/log/system is reserved for the kernel's own logging system later.) One
|
||||
// timestamped file per run.
|
||||
const log_dir = b.fmt("{s}/qemu-test", .{b.install_path});
|
||||
const make_log_dir = b.addSystemCommand(&.{ "mkdir", "-p", log_dir });
|
||||
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
||||
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
|
||||
run_efi.step.dependOn(&efi_install.step);
|
||||
run_efi.step.dependOn(&kernel_install.step);
|
||||
run_efi.step.dependOn(&init_install.step);
|
||||
run_efi.step.dependOn(&initrd_esp_install.step);
|
||||
// The whole FHS zig-out must be installed (and the scratch dir created) before we mount it.
|
||||
run_efi.step.dependOn(b.getInstallStep());
|
||||
run_efi.step.dependOn(&make_log_dir.step);
|
||||
|
||||
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/run-x86-64-serial0-<timestamp>.log");
|
||||
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/qemu-test/run-x86-64-serial0-<timestamp>.log");
|
||||
run_efi_step.dependOn(&run_efi.step);
|
||||
|
||||
// const run_cmd = b.addRunArtifact(exe);
|
||||
@@ -368,18 +421,22 @@ pub fn build(b: *std.Build) void {
|
||||
// }
|
||||
|
||||
// Tests run on the host. The kernel and bootloader target freestanding/UEFI
|
||||
// and can't be executed natively, so only the shared module is unit-tested
|
||||
// here (compiled for the host rather than inheriting a freestanding target).
|
||||
const mod_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("system/danos.zig"),
|
||||
.target = target,
|
||||
.optimize = optimize,
|
||||
}),
|
||||
});
|
||||
|
||||
const run_mod_tests = b.addRunArtifact(mod_tests);
|
||||
|
||||
// and can't be executed natively, so only the shared contracts are unit-tested
|
||||
// here (compiled for the host rather than inheriting a freestanding target) —
|
||||
// which also compile-checks that the three-way split stays self-consistent.
|
||||
const test_step = b.step("test", "Run tests");
|
||||
test_step.dependOn(&run_mod_tests.step);
|
||||
for ([_][]const u8{
|
||||
"system/boot-handoff.zig",
|
||||
"system/abi.zig",
|
||||
"system/devices/device-abi.zig",
|
||||
}) |root| {
|
||||
const mod_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(root),
|
||||
.target = target,
|
||||
.optimize = optimize,
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(mod_tests).step);
|
||||
}
|
||||
}
|
||||
|
||||
+45
-14
@@ -125,33 +125,63 @@ reach it *by module name*, never by a path into its files. The source tree delib
|
||||
what you see under `system/` in the source is what a running danos represents under
|
||||
`/system`.
|
||||
|
||||
**A sub-project is addressed by its directory; its entry point repeats the directory's
|
||||
name.** `system/services/init/` contains `init.zig` (its root), and produces a binary
|
||||
addressed as **`system/services/init`** — the repeated leaf resolves away:
|
||||
|
||||
| Source (root file) | Addressed as (module / binary / FHS path) |
|
||||
|----------------------------------------|--------------------------------------------|
|
||||
| `system/services/init/init.zig` | `system/services/init` → `/system/services/init` |
|
||||
| `system/drivers/hpet/hpet.zig` | `system/drivers/hpet` → `/system/drivers/hpet` |
|
||||
| `library/runtime/runtime.zig` | `library/runtime` (the `runtime` module) |
|
||||
|
||||
In **source**, a sub-project is a directory so it can hold many files — the entry is
|
||||
`init/init.zig`, beside it `vfs/vfs-test.zig`, `vfs/protocol.zig`, and so on. When
|
||||
**addressed or installed**, that collapses to the single canonical path: the `init`
|
||||
binary installs to `/system/services/init` (a file at that path), not
|
||||
`/system/services/init/init`. The repeated leaf exists only in source; the directory is
|
||||
the identity, the entry file is its implementation. (Same idea as a Go package being its
|
||||
directory, or a macOS `.app` bundle addressed by the bundle, not the executable within.)
|
||||
A sub-project's extra files are reached through the module, never as separate paths.
|
||||
|
||||
```
|
||||
system/ → /system danos's own internals (the self-representation)
|
||||
danos.zig the kernel↔user ABI contract (the `danos` module)
|
||||
parameters.zig initrd.zig shared contracts
|
||||
boot-handoff.zig the loader↔kernel contract (the `boot-handoff` module)
|
||||
abi.zig the core kernel↔user ABI (the `abi` module)
|
||||
parameters.zig initial-ramdisk.zig shared contracts
|
||||
kernel/ IPC, memory, scheduling, the private syscall dispatch
|
||||
architecture/x86_64/ the `architecture` module (never named by generic code)
|
||||
devices/ the device model /system/devices reflects (+ aml/)
|
||||
drivers/ hpetd/ busd/ one sub-project per driver → /system/drivers
|
||||
services/ init/ vfs/ system servers → /system/services (vfs/ holds
|
||||
device-abi.zig the device wire types (the `device-abi` module)
|
||||
drivers/ hpet/ bus/ one sub-project per driver → /system/drivers
|
||||
services/ init/ vfs/ device-manager/ system servers → /system/services (vfs/ holds
|
||||
vfs.zig, vfs-test.zig, protocol.zig)
|
||||
library/ → /lib the runtime library (the stable application ABI)
|
||||
library/ → /lib libraries, one sub-directory each
|
||||
runtime/ the danos-native runtime — the stable application ABI
|
||||
posix/ POSIX/C compatibility, layered over runtime
|
||||
boot/ → /boot the loaders
|
||||
tools/ test/ host-side build + QEMU test harness
|
||||
```
|
||||
|
||||
A sub-project exposes its **public interface as a module**: `system/services/vfs/` owns
|
||||
the VFS wire protocol (`protocol.zig`, the `vfs-protocol` module), which the runtime's
|
||||
file layer imports by name. `usb`/`block` drivers will expose their protocols the same
|
||||
way.
|
||||
the VFS wire protocol (`protocol.zig`, the `vfs-protocol` module), which the POSIX
|
||||
layer imports by name. `usb`/`block` drivers will expose their protocols the same way.
|
||||
|
||||
`library/posix/` is special: it is the **one place** POSIX/C spellings are allowed
|
||||
verbatim (`stat`, `O_CREAT`, `fopen`, `errno`). Everywhere else follows the danos
|
||||
naming rule with no exception — see [coding-standards.md](coding-standards.md). The
|
||||
POSIX layer calls the runtime, never the kernel's system calls directly, so it never
|
||||
appears in the private-ABI path.
|
||||
|
||||
## Source map
|
||||
|
||||
| Area | Code |
|
||||
|------|------|
|
||||
| Boot methods (one per way of booting the kernel) | `boot/` — `efi.zig` (UEFI) → `BOOTX64.efi` |
|
||||
| Kernel entry, panic, bring-up | `system/kernel/main.zig` |
|
||||
| Shared loader↔kernel contract (`BootInfo`, `Framebuffer`, `MemoryMap`, `Syscall`, ABI) | `system/danos.zig` |
|
||||
| Kernel entry, panic, bring-up | `system/kernel/kernel.zig` |
|
||||
| Loader↔kernel handoff (`BootInfo`, `Framebuffer`, `MemoryMap`, VM layout) | `system/boot-handoff.zig` |
|
||||
| Core kernel↔user ABI (`SystemCall`, mmap prot flags, `page_size`) | `system/abi.zig` |
|
||||
| Device wire types (`DeviceDescriptor`, `DeviceClass`, …) | `system/devices/device-abi.zig` |
|
||||
| Physical frame allocator | `system/kernel/pmm.zig` |
|
||||
| Kernel heap (`std.mem.Allocator`) | `system/kernel/heap.zig` |
|
||||
| Scheduler (fixed-priority preemptive; blocking, wait queues) | `system/kernel/scheduler.zig` |
|
||||
@@ -159,14 +189,15 @@ way.
|
||||
| IPC channels between kernel threads (message passing) | `system/kernel/ipc.zig` |
|
||||
| IPC endpoints: cross-address-space call/reply, handles, notifications | `system/kernel/ipc-synchronous.zig` |
|
||||
| User processes: ELF loading, address spaces, the syscall table | `system/kernel/process.zig` |
|
||||
| Device tree + claim capability + `device_register` containment | `system/kernel/device-service.zig` |
|
||||
| Device tree + claim capability + `device_register` containment | `system/kernel/devices-broker.zig` |
|
||||
| IRQ-as-IPC: routing a device interrupt to a driver's endpoint | `system/kernel/irq.zig` |
|
||||
| Hardware discovery (ACPI/device tree) behind one neutral device model | `system/devices/` |
|
||||
| Framebuffer text console (mirrors to serial) | `system/kernel/console.zig` |
|
||||
| In-kernel test cases | `system/kernel/tests.zig` |
|
||||
| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/IO-APIC/timer, serial, linker script) | `system/kernel/architecture/x86_64/` |
|
||||
| Runtime library (`runtime`): syscall wrappers, heap, stdio, IPC, device access — the stable application ABI | `library/runtime/` |
|
||||
| System services (init, the VFS server + its `protocol` module) | `system/services/` |
|
||||
| Device drivers, one sub-project each (`hpetd` leaf driver, `busd` bus driver) | `system/drivers/` |
|
||||
| danos-native runtime (`runtime`): syscall wrappers, heap, IPC, device access — the stable application ABI | `library/runtime/` |
|
||||
| POSIX/C compatibility (`posix`): unistd, stdio — the one place POSIX names are allowed | `library/posix/` |
|
||||
| System services (init, the VFS server + `protocol`, the device-manager) | `system/services/` |
|
||||
| Device drivers, one sub-project each (`hpet` leaf driver, `bus` bus driver) | `system/drivers/` |
|
||||
| Build + `run-x86-64` (QEMU/OVMF) | `build.zig` |
|
||||
| QEMU integration test harness | `test/qemu_test.py` |
|
||||
|
||||
+2
-2
@@ -16,7 +16,7 @@ the RSDT's address is a field *inside* the RSDP. The platform follows that point
|
||||
UEFI configuration table
|
||||
│ the loader reads the RSDP's physical address
|
||||
▼
|
||||
BootInfo.acpi_rsdp (u64, in the shared `danos` module) system/danos.zig
|
||||
BootInfo.acpi_rsdp (u64, in the loader↔kernel handoff) system/boot-handoff.zig
|
||||
│ the kernel forwards the whole BootInfo
|
||||
▼
|
||||
platform.discover(boot_info, …) system/devices/platform.zig
|
||||
@@ -44,7 +44,7 @@ the [memory map](memory-map.md).
|
||||
|
||||
The loader can't just call the device module: the bootloader binary and the kernel
|
||||
binary are compiled separately, and **the loader isn't linked against the `platform`
|
||||
module at all** (it imports only the shared `danos` module). So instead of a call, it
|
||||
module at all** (it imports only the `boot-handoff` contract). So instead of a call, it
|
||||
deposits a value in the handoff struct:
|
||||
|
||||
```zig
|
||||
|
||||
+1
-1
@@ -83,7 +83,7 @@ There are really two independent questions, and it's worth not conflating them:
|
||||
|
||||
The kernel entry point `_start` currently still lives in the generic `main.zig` as
|
||||
a thin trampoline into `kmain`. It's arch-adjacent (its calling convention is
|
||||
x86_64 [SysV](sysv.md), via the shared `danos.kernel_abi`), but it's three lines
|
||||
x86_64 [SysV](sysv.md), via the shared `system.kernel_abi`), but it's three lines
|
||||
and mostly generic, so it stays put for now. When AArch64 arrives — where entry means setting
|
||||
up a stack and reading a device-tree pointer from a register — the entry work will
|
||||
be substantial and per-arch, and *that* is when we extract an entry interface into
|
||||
|
||||
+30
-15
@@ -6,7 +6,7 @@ Conventions for danos source. The overriding one, from which most of the rest fo
|
||||
> abbreviation is an acronym.**
|
||||
|
||||
`interruptDispatch`, not `intDisp`. `message_len`, not `message_len` (`msg` expands, `len`
|
||||
is a Zig idiom — see the exceptions). `device_service`, not `device_service`. `scheduler`, not
|
||||
is a Zig idiom — see the exceptions). `devices_broker`, not `devices_broker`. `scheduler`, not
|
||||
`sched`. The cost of a longer name is paid once, at the keyboard; the cost of a
|
||||
cryptic one is paid every time the code is read, by everyone who reads it. In a
|
||||
microkernel whose whole argument is that a human can hold each piece in their head,
|
||||
@@ -58,13 +58,22 @@ abbreviation, expand it.
|
||||
|
||||
Three, and only three.
|
||||
|
||||
1. **Foreign ABI names are spelled exactly as the ABI spells them.** A function that
|
||||
*is* the C or POSIX interface keeps its name: `fopen`, `fwrite`, `fread`, `malloc`,
|
||||
`calloc`, `realloc`, `free`, `memcpy`, `mmap`, `munmap`, `open`, `read`, `write`,
|
||||
`close`, `lseek`, `stat`, `errno`. We don't get to rename `fwrite` to
|
||||
`fileWrite` — it wouldn't be `fwrite` any more. This also covers the syscall
|
||||
*wrappers* that exist to match those names. It does **not** license inventing new
|
||||
abbreviated names in that style.
|
||||
1. **Foreign ABI names are spelled exactly as the ABI spells them — but only inside
|
||||
the layer that *is* that ABI.** A function that *is* the C or POSIX interface keeps
|
||||
its name: `fopen`, `fwrite`, `fread`, `malloc`, `calloc`, `realloc`, `free`,
|
||||
`memcpy`, `mmap`, `munmap`, `open`, `read`, `write`, `close`, `lseek`, `stat`,
|
||||
`errno`, `O_CREAT`. We don't get to rename `fwrite` to `fileWrite` — it wouldn't be
|
||||
`fwrite` any more.
|
||||
|
||||
**This exception is scoped to one place: `library/posix/`.** A file under
|
||||
`library/posix/` *is* the foreign ABI, so it keeps the ABI's spellings — that is the
|
||||
whole rule for that directory. **Everywhere else, Zig/danos naming applies with no
|
||||
POSIX exception**, so there is nothing to get wrong: if you're not in
|
||||
`library/posix/`, expand it. A concept POSIX also has gets a danos name outside that
|
||||
layer — the VFS wire protocol carries a `FileStatus`, not a `Stat`, and a `create`
|
||||
flag, not `O_CREAT`; `library/posix/` is what maps `stat`→`status` and
|
||||
`O_CREAT`→`create` at the boundary. (The `syscall` *wrappers* elsewhere are not an
|
||||
exception to this — they wrap the private danos ABI, so they use danos names.)
|
||||
|
||||
2. **Zig idioms are spelled the way Zig spells them.** Three names are the language's,
|
||||
not ours, and are left alone:
|
||||
@@ -83,11 +92,12 @@ Three, and only three.
|
||||
keep `i`; a coordinate may be `x`, `y`. The moment the scope is big enough that the
|
||||
letter's meaning isn't obvious on sight, give it a real name. When in doubt, name it.
|
||||
|
||||
4. **Established Unix filesystem and program conventions.** Top-level directories keep
|
||||
their conventional names — `src`, `lib`, `sbin`, `bin`, `docs` — as do daemon
|
||||
programs by their `d` suffix (`hpetd`, `busd`, following `sshd`/`httpd`). These are
|
||||
names a Unix reader already knows; expanding them fights the convention rather than
|
||||
serving it.
|
||||
That's all — no Unix-abbreviation exception. The source directories are full words
|
||||
(`system`, `library`, not `src`/`lib`), and there is no daemon `d` suffix: a driver
|
||||
lives in `system/drivers/` and a service in `system/services/`, so the *location*
|
||||
already says what it is. Encoding the role in the name too (`busd`, `vfsd`) is
|
||||
redundant — the program is just `bus`, `vfs`. Don't put in a name what its directory
|
||||
already tells you.
|
||||
|
||||
## A note on collisions
|
||||
|
||||
@@ -118,15 +128,20 @@ Within those spelling rules, follow Zig's own conventions:
|
||||
|
||||
- **Types** — `PascalCase`: `DeviceDescriptor`, `Endpoint`, `WaitQueue`.
|
||||
- **Functions** — `camelCase`: `mapUserDeviceInto`, `notifyFromIsr`.
|
||||
- **Variables, fields, constants** — `snake_case`: `message_length`, `device_service`,
|
||||
- **Variables, fields, constants** — `snake_case`: `message_length`, `devices_broker`,
|
||||
`notify_badge_bit`.
|
||||
|
||||
**File names are `kebab-case`.** A file named for a multi-word thing hyphenates it:
|
||||
`device-tree.zig`, `ipc-synchronous.zig`, `vfs-protocol.zig`, `device-service.zig`. A
|
||||
`device-tree.zig`, `ipc-synchronous.zig`, `vfs-protocol.zig`, `devices-broker.zig`. A
|
||||
single word or acronym needs no hyphen: `scheduler.zig`, `paging.zig`, `apic.zig`,
|
||||
`idt.zig`. (The module *alias* a file is imported under still follows the code
|
||||
conventions above — `snake_case` — because it's an identifier, not a filename.)
|
||||
|
||||
**A sub-project's entry point repeats its directory's name** — `init/init.zig`,
|
||||
`runtime/runtime.zig`, `hpet/hpet.zig` — and the sub-project is addressed by the
|
||||
*directory* (`system/services/init`, `library/runtime`), with the repeated leaf
|
||||
resolving away. See the repository-layout section of [README.md](README.md).
|
||||
|
||||
## Why acronyms are the line
|
||||
|
||||
Because an acronym has no letters to restore. `MMIO` doesn't become "memory mapped
|
||||
|
||||
@@ -8,7 +8,7 @@ Most modern Unix and Unix-like operating systems follow the FHS. DanOS has its o
|
||||
|-----------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| / | Primary hierarchy root and root directory of the entire file system hierarchy. |
|
||||
| /bin | Essential command binaries that need to be available in single-user mode, including to bring up the system or repair it, for all users (e.g., cat, ls, cp). |
|
||||
| /boot | Boot loader files (e.g., EFI, initrd.img ). |
|
||||
| /boot | Boot loader files (e.g., EFI, initial-ramdisk.img ). |
|
||||
| /dev | POSIX Device files (e.g., /dev/null, /dev/disk0, /dev/tty, /dev/random). |
|
||||
| /etc | Host-specific system-wide configuration files. |
|
||||
| /home | Users' home directories, containing saved files, personal settings, etc. |
|
||||
@@ -17,7 +17,7 @@ Most modern Unix and Unix-like operating systems follow the FHS. DanOS has its o
|
||||
| /srv | Site-specific data served by this system, such as data and scripts for web servers, data offered by FTP servers, and repositories for version control systems |
|
||||
| /system | DanOS operating system files (similar idea to C:\Windows). A true representation of danos — its layout mirrors the source tree, so `/system` is what danos *is*. |
|
||||
| /system/devices | danos virtual device tree e.g. similar to /sys on linux but with danos device tree conventions (the structures in the devices module) |
|
||||
| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/hpetd) |
|
||||
| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/hpet) |
|
||||
| /system/services | system-service binaries — the VFS server, init, and other user-mode servers (e.g. /system/services/vfs, /system/services/init) |
|
||||
| /system/kernel | the kernel image |
|
||||
| /tmp | Directory for temporary files (see also /var/tmp). Often not preserved between system reboots and may be severely size-restricted. |
|
||||
@@ -61,7 +61,7 @@ to the driver in the order written, and a read consumes what is there. Terminals
|
||||
serial lines, keyboards and mice are all of this shape. These are the natural first
|
||||
device nodes in danos, because a character driver needs nothing the kernel doesn't
|
||||
already provide — it claims its device, maps its registers with `mmio_map`, and blocks
|
||||
on `replyWait` for either an interrupt or a client request. `system/drivers/hpetd/hpetd.zig` is already
|
||||
on `replyWait` for either an interrupt or a client request. `system/drivers/hpet/hpet.zig` is already
|
||||
that program, minus the client half.
|
||||
|
||||
The obstacle is not the file type, it is which hardware a ring-3 driver can actually
|
||||
@@ -91,7 +91,7 @@ device to a driver process, with no IOMMU programmed, is equivalent to granting
|
||||
which would forfeit the isolation that motivates user-space drivers in the first place.
|
||||
|
||||
Block devices therefore wait on DMA-capable memory, memory barriers, and VT-d/DMAR —
|
||||
the M14–M16 work in [driver-model.md](driver-model.md). A ramdisk over the initrd is
|
||||
the M14–M16 work in [driver-model.md](driver-model.md). A ramdisk over the initial ramdisk is
|
||||
the one block-shaped thing implementable now, and it needs no driver process.
|
||||
|
||||
### Pseudo-devices
|
||||
|
||||
+20
-20
@@ -32,19 +32,19 @@ plain bus driver with no controller — a USB hub — is also a real thing.
|
||||
|
||||
## The device table is the spine
|
||||
|
||||
danos already has the right central structure. `system/kernel/device-service.zig` holds a table of
|
||||
danos already has the right central structure. `system/kernel/devices-broker.zig` holds a table of
|
||||
`DeviceDesc`, each with a parent, a class, and a set of resources. Firmware discovery
|
||||
seeds it ([discovery.md](discovery.md)); `device_register` grows it.
|
||||
|
||||
Three invariants make it a capability system rather than a directory:
|
||||
|
||||
1. **A claim is exclusive.** `device_claim(id)` succeeds once. Everything downstream —
|
||||
`mmio_map`, `irq_bind`, `device_register` — checks `device_service.ownerOf(id) == me`.
|
||||
`mmio_map`, `irq_bind`, `device_register` — checks `devices_broker.ownerOf(id) == me`.
|
||||
2. **A descriptor is a licence to map physical memory.** Whoever claims a device may
|
||||
map its `.memory` resources and bind its `.irq` resources. This is why
|
||||
`device_register` cannot be a free-for-all.
|
||||
3. **Therefore: containment.** Every resource of a registered child must lie inside a
|
||||
resource of the same kind on its parent (`device_service.contains`). A bus driver can only
|
||||
resource of the same kind on its parent (`devices_broker.contains`). A bus driver can only
|
||||
ever *subdivide* what it already holds. Without this, `device_register` would be a
|
||||
syscall named "map any physical page you like."
|
||||
|
||||
@@ -57,7 +57,7 @@ is not an address window. Discovery is trusted; user space is not.
|
||||
|
||||
### What a bus driver looks like
|
||||
|
||||
`system/drivers/busd/busd.zig` is the smallest honest one. Its "bus" is the HPET's register block and
|
||||
`system/drivers/bus/bus.zig` is the smallest honest one. Its "bus" is the HPET's register block and
|
||||
its "devices" are the block's comparators:
|
||||
|
||||
```zig
|
||||
@@ -78,7 +78,7 @@ for (0..n) |i| { // 3. publish each child
|
||||
|
||||
Each child is left **unclaimed**, which is the handoff: a comparator driver can now
|
||||
`device_claim` one and `mmio_map` it, and will see only its own 0x20-byte window. A child
|
||||
whose window escapes the bus is refused — `busd` asserts that, and the `bus` test
|
||||
whose window escapes the bus is refused — `bus` asserts that, and the `bus` test
|
||||
asserts the kernel's table upholds it.
|
||||
|
||||
A USB device has *no* resources at all: `resource_count = 0`, because it's addressed
|
||||
@@ -99,21 +99,21 @@ danos already has one of each: `library/runtime/device.zig` is a logic module,
|
||||
and its clients. The pattern generalises directly:
|
||||
|
||||
```
|
||||
lib/
|
||||
rt.zig module "rt" — syscalls, heap, ipc, dev, stdio
|
||||
mmio.zig module "mmio" — volatile register access + barriers [M14]
|
||||
library/
|
||||
runtime/ module "runtime" — syscalls, heap, ipc, device, stdio
|
||||
mmio/ module "mmio" — volatile register access + barriers [M14]
|
||||
bus/
|
||||
pci.zig module "pci" — ECAM, BAR decode, capability walk
|
||||
usb.zig module "usb" — descriptors, control transfers, hubs
|
||||
pci/ module "pci" — ECAM, BAR decode, capability walk
|
||||
usb/ module "usb" — descriptors, control transfers, hubs
|
||||
proto/
|
||||
vfs.zig module "proto.vfs" (today: system/services/vfs/protocol.zig)
|
||||
block.zig module "proto.block"
|
||||
hid.zig module "proto.hid"
|
||||
vfs/ module "vfs-protocol" (today: system/services/vfs/protocol.zig)
|
||||
block/ module "block-protocol"
|
||||
hid/ module "hid-protocol"
|
||||
|
||||
sbin/
|
||||
xhcid.zig HCD + bus driver imports rt, pci, usb, mmio
|
||||
usbhid.zig class driver imports rt, usb, proto.hid
|
||||
blockd.zig class driver imports rt, proto.block
|
||||
system/drivers/ one sub-project each → /system/drivers (no `d` suffix)
|
||||
xhci/ HCD + bus driver imports runtime, pci, usb, mmio
|
||||
usb-hid/ class driver imports runtime, usb, hid-protocol
|
||||
block/ class driver imports runtime, block-protocol
|
||||
```
|
||||
|
||||
The only build change needed: [`addUserBinary`](build.zig) currently takes exactly one
|
||||
@@ -247,7 +247,7 @@ barrier, or per-arch inline asm — which is what `library/mmio.zig` should hide
|
||||
|
||||
**The blocker, and it's a hard one.** No PCI device can take an interrupt today.
|
||||
[`addBars`](system/devices/acpi.zig) records `.memory` and `.io_port` BARs and never an
|
||||
`.irq`; there is no `_PRT` parsing anywhere in the tree. `hpetd` only works because the
|
||||
`.irq`; there is no `_PRT` parsing anywhere in the tree. `hpet` only works because the
|
||||
HPET advertises its own routing options in its own registers — a privilege no ordinary
|
||||
device has.
|
||||
|
||||
@@ -271,7 +271,7 @@ which means **discovery should give each `pci_device` a `.memory` resource for i
|
||||
4 KiB ECAM slot**. That's a small change to `parseMcfg` and it unblocks the whole
|
||||
capability walk (MSI, MSI-X, PCIe extended caps) without any new syscall.
|
||||
|
||||
Note QEMU's HPET reports `Tn_FSB_INT_DEL_CAP = 0` — no MSI — so `hpetd` can never
|
||||
Note QEMU's HPET reports `Tn_FSB_INT_DEL_CAP = 0` — no MSI — so `hpet` can never
|
||||
exercise this path. The first MSI driver will be the first PCI driver.
|
||||
|
||||
## M16 — the IOMMU, and the honest caveat
|
||||
@@ -291,7 +291,7 @@ gap should be named rather than implied.
|
||||
|
||||
`M13` (capability passing) is independent of `M14`/`M15` and is the cheapest. It
|
||||
unlocks class drivers, which are the shape with no hardware requirements at all — you
|
||||
could write a real one against `busd`'s comparators tomorrow.
|
||||
could write a real one against `bus`'s comparators tomorrow.
|
||||
|
||||
`M14` and `M15` together unlock the first HCD. `M14`'s barrier layer is worth landing
|
||||
on its own regardless: it's small, obviously correct, and stops every future driver
|
||||
|
||||
+9
-8
@@ -22,7 +22,8 @@ say.*
|
||||
|
||||
## The capability: claim before touch
|
||||
|
||||
The five driver syscalls (`system/danos.zig`, dispatched in `system/kernel/process.zig`):
|
||||
The driver syscall numbers (`system/abi.zig`) with the device types they carry
|
||||
(`system/devices/device-abi.zig`), dispatched in `system/kernel/process.zig`:
|
||||
|
||||
| # | Call | Meaning |
|
||||
|---|------|---------|
|
||||
@@ -40,7 +41,7 @@ memory; if `irq_bind` took a GSI, any process could bind the keyboard's line and
|
||||
silently intercept it. Instead the kernel checks two things (`process.ownedGsi`, and
|
||||
the same check at the top of `sysMmioMap`):
|
||||
|
||||
- `device_service.ownerOf(dev_id) == me` — you claimed it, and claims are exclusive
|
||||
- `devices_broker.ownerOf(dev_id) == me` — you claimed it, and claims are exclusive
|
||||
- the resource at `res_idx` is of the right *kind* — `memory` for `mmio_map`, `irq`
|
||||
for `irq_bind`
|
||||
|
||||
@@ -138,7 +139,7 @@ Two properties worth knowing:
|
||||
|
||||
## A whole driver
|
||||
|
||||
`system/drivers/hpetd/hpetd.zig` is ~150 lines and does all of it. The shape:
|
||||
`system/drivers/hpet/hpet.zig` is ~150 lines and does all of it. The shape:
|
||||
|
||||
```zig
|
||||
const hpet = findHpet(buf) orelse return; // device_enumerate, look for
|
||||
@@ -206,7 +207,7 @@ bus driver may only ever subdivide what it already owns.
|
||||
A device with **no resources** is legal and common. A USB device is reached through its
|
||||
controller, not by MMIO, so it gets `resource_count = 0`.
|
||||
|
||||
See [`system/drivers/busd/busd.zig`](../system/drivers/busd/busd.zig) for a complete one, and
|
||||
See [`system/drivers/bus/bus.zig`](../system/drivers/bus/bus.zig) for a complete one, and
|
||||
[driver-model.md](driver-model.md) for how bus drivers, class drivers and host
|
||||
controller drivers fit together.
|
||||
|
||||
@@ -269,8 +270,8 @@ Worth knowing before you write the second driver:
|
||||
|
||||
## Verifying it
|
||||
|
||||
The `hpet` test spawns `hpetd` from the initrd and watches the serial log. The driver
|
||||
prints `hpetd: ok` only after being woken five times, and its loop's only exit is
|
||||
The `hpet` test spawns `hpet` from the initial ramdisk and watches the serial log. The driver
|
||||
prints `hpet: ok` only after being woken five times, and its loop's only exit is
|
||||
through `replyWait` returning a notification — it cannot reach that line by polling.
|
||||
|
||||
The last check doesn't trust the driver's self-report at all: the kernel reads the I/O
|
||||
@@ -285,7 +286,7 @@ $ python3 test/qemu_test.py hpet irqfree iopass
|
||||
iopass ... PASS (matched 'DANOS-TEST-RESULT: PASS')
|
||||
```
|
||||
|
||||
Two companions cover what `hpetd` can't, because it never exits:
|
||||
Two companions cover what `hpet` can't, because it never exits:
|
||||
|
||||
- **`irqfree`** — the teardown path. Binds two owners to one shared endpoint, releases
|
||||
one, and reads the I/O APIC back: the departing owner's line is masked, the sibling's
|
||||
@@ -305,7 +306,7 @@ controller drivers), and the IOMMU — have proposed signatures in
|
||||
I/O permission bitmap swapped on context switch, or `io_in`/`io_out` syscalls gated
|
||||
by the same claim. The legacy devices that need it are all low-rate, so the syscall
|
||||
is likely fast enough.
|
||||
- **Releasing a claim.** There is no `dev_release`, and `device_service` never drops a claim on
|
||||
- **Releasing a claim.** There is no `dev_release`, and `devices_broker` never drops a claim on
|
||||
exit — only IRQ bindings are released. A dead driver's device stays owned forever,
|
||||
which blocks restart.
|
||||
- **Unregistering children.** `device_register` only appends. A USB device that is
|
||||
|
||||
+23
-18
@@ -20,14 +20,17 @@ UEFI boots by looking for a FAT-formatted partition called the **EFI System
|
||||
Partition (ESP)** and running a file at a well-known fallback path:
|
||||
|
||||
```
|
||||
esp/EFI/BOOT/BOOTX64.efi <- the "removable media" default for x86-64
|
||||
EFI/BOOT/BOOTX64.efi <- the "removable media" default for x86-64
|
||||
```
|
||||
|
||||
That's exactly the layout `build.zig` assembles. It builds `boot/efi.zig` for the
|
||||
`uefi` target, installs it to `esp/EFI/BOOT/BOOTX64.efi`, and drops the kernel ELF
|
||||
at `esp/kernel`. The `run-x86-64` step then points QEMU at OVMF (UEFI firmware for
|
||||
virtual machines) and presents that `esp/` directory to the guest as a FAT drive.
|
||||
The firmware finds `BOOTX64.efi` and runs it — that's our `main()`.
|
||||
The boot volume is the **FHS-shaped `zig-out`** itself (see the repository-layout note
|
||||
in [README.md](README.md)): `build.zig` installs `boot/efi.zig` (built for the `uefi`
|
||||
target) to `zig-out/EFI/BOOT/BOOTX64.efi` — the one path UEFI firmware fixes — and lays
|
||||
the rest out by FHS path: the kernel at `zig-out/system/kernel`, init at
|
||||
`zig-out/system/services/init`, the initial-ramdisk at `zig-out/boot/`. The
|
||||
`run-x86-64` step points QEMU at OVMF (UEFI firmware for virtual machines) and presents
|
||||
`zig-out` to the guest as a FAT drive. The firmware finds `BOOTX64.efi` and runs it —
|
||||
that's our `main()`, which then loads the kernel and init from their FHS paths.
|
||||
|
||||
## Boot services: the firmware's API
|
||||
|
||||
@@ -83,9 +86,9 @@ All of this *must* happen now, because after exit there's no GOP to ask. (See
|
||||
|
||||
- Use the **LoadedImage** protocol to discover which device we booted from, then
|
||||
**SimpleFileSystem** to open that volume.
|
||||
- Open the file named `danos`, seek to the end to learn its size, rewind, and read
|
||||
the whole ELF into a firmware-allocated pool buffer. (`read` may return short, so
|
||||
we loop.)
|
||||
- Open the kernel ELF at its FHS path (`system\kernel`), seek to the end to learn its
|
||||
size, rewind, and read the whole ELF into a firmware-allocated pool buffer. (`read`
|
||||
may return short, so we loop.)
|
||||
- Parse the ELF: validate the `\x7fELF` magic and the `x86_64` machine type, then
|
||||
walk the program headers. For every `PT_LOAD` segment we:
|
||||
- reserve the exact physical pages it's linked at (`p_paddr`) via
|
||||
@@ -121,7 +124,7 @@ entirely ours.
|
||||
### 4. Jump to the kernel
|
||||
|
||||
```zig
|
||||
const kernel: *const fn (*const BootInfo) callconv(danos.kernel_abi) noreturn =
|
||||
const kernel: *const fn (*const BootInfo) callconv(boot_handoff.kernel_abi) noreturn =
|
||||
@ptrFromInt(entry);
|
||||
kernel(&boot_info);
|
||||
```
|
||||
@@ -139,17 +142,19 @@ kernel is freestanding and uses the **SysV AMD64** convention (first argument in
|
||||
read garbage.
|
||||
|
||||
So both sides pin the convention explicitly to SysV via the shared
|
||||
`danos.kernel_abi` (defined in `system/danos.zig`). The loader's function-pointer type
|
||||
and the kernel's `_start` both reference it, so the pointer lands in the register
|
||||
the kernel expects. This is the whole reason `kernel_abi` lives in the shared
|
||||
`danos` module: it's a contract both binaries must agree on. See
|
||||
`boot_handoff.kernel_abi` (defined in `system/boot-handoff.zig`). The loader's
|
||||
function-pointer type and the kernel's `_start` both reference it, so the pointer lands
|
||||
in the register the kernel expects. This is the whole reason `kernel_abi` lives in the
|
||||
shared `boot-handoff` module: it's a contract both binaries must agree on. See
|
||||
[sysv.md](sysv.md) for what "SysV" means and where else it shows up.
|
||||
|
||||
## The handoff contract
|
||||
|
||||
The loader and kernel are two *separate* binaries built for two different targets,
|
||||
so everything they exchange must have an identically-defined memory layout. That's
|
||||
what `system/danos.zig` provides — imported by both as the `danos` module:
|
||||
what `system/boot-handoff.zig` provides — imported by both as the `boot-handoff` module.
|
||||
It is *only* the handoff: the kernel↔user ABI (`system/abi.zig`) and the device types
|
||||
(`system/devices/device-abi.zig`) are separate contracts the bootloader never sees.
|
||||
|
||||
- `BootInfo` — the top-level struct passed to the kernel (currently just the
|
||||
framebuffer; this is where future handoff data like the memory map will go).
|
||||
@@ -164,13 +169,13 @@ the loader writes are the bytes the kernel reads.
|
||||
```
|
||||
power on
|
||||
-> UEFI firmware initialises hardware
|
||||
-> finds esp/EFI/BOOT/BOOTX64.efi, runs it (our efi.zig main)
|
||||
-> finds EFI/BOOT/BOOTX64.efi on the FHS volume, runs it (our efi.zig main)
|
||||
-> grab boot services
|
||||
-> queryFramebuffer (via GOP: EDID native res, setMode, describe fb)
|
||||
-> loadKernel (read danos ELF, load PT_LOAD segments to 0x100000)
|
||||
-> loadKernel (read system/kernel ELF, load PT_LOAD segments to 0x100000)
|
||||
-> exitBootServices (retry until the memory-map key holds)
|
||||
-> jump to e_entry, boot_info pointer in RDI
|
||||
-> kernel _start (system/kernel/main.zig: framebuffer console, then halt)
|
||||
-> kernel _start (system/kernel/kernel.zig: framebuffer console, then halt)
|
||||
```
|
||||
|
||||
Bottom line: **UEFI's job is to give us a CPU, memory, and a framebuffer, then
|
||||
|
||||
@@ -8,7 +8,7 @@ natural unit because that's the granularity the CPU's paging hardware maps — a
|
||||
it is the primitive everything above it stands on: page tables, the kernel heap,
|
||||
per-process memory all ultimately ask the frame allocator for pages.
|
||||
|
||||
It's **generic kernel code**: it operates on the neutral `danos.MemoryRegion`
|
||||
It's **generic kernel code**: it operates on the neutral `system.MemoryRegion`
|
||||
array, so there's no UEFI in it and nothing architecture-specific beyond the 4 KiB
|
||||
page. (Contrast [arch.md](arch.md), which is where CPU-specific code lives.)
|
||||
|
||||
|
||||
+1
-1
@@ -12,7 +12,7 @@ exactly what `Console.pixel` does:
|
||||
self.rowPtr(y)[x] = color; // system/kernel/console.zig
|
||||
```
|
||||
|
||||
Our `Framebuffer` struct (`system/danos.zig`) is the four facts you need to
|
||||
Our `Framebuffer` struct (`system/boot-handoff.zig`) is the four facts you need to
|
||||
address it:
|
||||
|
||||
| Field | Meaning |
|
||||
|
||||
+1
-1
@@ -83,7 +83,7 @@ treats the call:
|
||||
signature for a kernel entry point — the bootloader jumps in and nothing ever
|
||||
jumps back out.
|
||||
|
||||
You can see the chain in `system/kernel/main.zig`: `_start` is `noreturn`, it calls
|
||||
You can see the chain in `system/kernel/kernel.zig`: `_start` is `noreturn`, it calls
|
||||
`kmain` which is `noreturn`, which ends by calling `arch.halt()` which is
|
||||
`noreturn`. The "never returns" property is threaded all the way down.
|
||||
|
||||
|
||||
+1
-1
@@ -58,7 +58,7 @@ screen. `console.write` is a no-op when the firmware gave us no framebuffer.
|
||||
A framebuffer is not guaranteed — a headless server exposes no UEFI Graphics Output
|
||||
Protocol. That used to be *fatal* (the loader failed the boot). Now the loader hands
|
||||
over a "no framebuffer" descriptor (`base == 0`) rather than failing, and
|
||||
`Framebuffer.present()` (in `system/danos.zig`) gates every on-screen path. A headless,
|
||||
`Framebuffer.present()` (in `system/boot-handoff.zig`) gates every on-screen path. A headless,
|
||||
serial-less machine boots and runs correctly — it just goes quiet.
|
||||
|
||||
## Last-resort channels (no text output at all)
|
||||
|
||||
+2
-2
@@ -27,7 +27,7 @@ danos's own neutral format, and the kernel only ever sees that.**
|
||||
|
||||
## The neutral format
|
||||
|
||||
Defined in `system/danos.zig`, the shared loader↔kernel contract:
|
||||
Defined in `system/boot-handoff.zig`, the shared loader↔kernel contract:
|
||||
|
||||
```zig
|
||||
pub const MemoryKind = enum(u32) {
|
||||
@@ -121,7 +121,7 @@ The kernel receives a plain array and reads it with zero UEFI knowledge:
|
||||
|
||||
```zig
|
||||
const mm = boot_info.memory_map;
|
||||
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
||||
const regions = @as([*]const system.MemoryRegion, @ptrFromInt(mm.regions))[0..mm.len];
|
||||
for (regions) |r| {
|
||||
if (r.kind == .usable) usable_pages += r.pages;
|
||||
}
|
||||
|
||||
+1
-1
@@ -27,7 +27,7 @@ address to the low load address in its bootstrap tables and jumps in). The entir
|
||||
alongside a **physmap** — a straight window onto all of physical memory at
|
||||
`physmap_base + phys`. Wherever the kernel needs to touch a physical address (a
|
||||
page-table frame, an ACPI table, a device register), it adds that constant:
|
||||
`danos.physToVirt(phys)`. The layout constants live in `system/danos.zig`:
|
||||
`system.physToVirt(phys)`. The layout constants live in `system/boot-handoff.zig`:
|
||||
|
||||
| region | virtual base | PML4 slot |
|
||||
|--------|--------------|-----------|
|
||||
|
||||
+1
-1
@@ -203,7 +203,7 @@ next lands.
|
||||
[scheduling.md](scheduling.md#affinity-pinning-a-task-to-a-core)). The `affinity`
|
||||
test confirms a pinned task never migrates. This is the mechanism the fault-on-AP
|
||||
test rides on, and the *explicit-affinity* real-time-predictable model.
|
||||
- **Right-sized footprint** — the per-CPU ceiling (`danos.max_cpus`, one constant
|
||||
- **Right-sized footprint** — the per-CPU ceiling (`system.max_cpus`, one constant
|
||||
shared by discovery, the scheduler, and the per-core GDT/TSS) is generous (128), but
|
||||
the *large* per-core resources — the kernel and IST (double-fault) stacks — are
|
||||
**heap-allocated at bring-up**, only for cores that actually come online. Only the
|
||||
|
||||
+3
-3
@@ -1,6 +1,6 @@
|
||||
# SysV: the kernel's calling convention
|
||||
|
||||
Several places in danos say "the kernel is SysV" — most visibly `system/danos.zig`:
|
||||
Several places in danos say "the kernel is SysV" — most visibly `system/boot-handoff.zig`:
|
||||
|
||||
```zig
|
||||
pub const kernel_abi: std.builtin.CallingConvention = .{ .x86_64_sysv = .{} };
|
||||
@@ -59,9 +59,9 @@ danos's two binaries default to different conventions:
|
||||
When the loader jumps to the kernel passing the `BootInfo` pointer, both sides have
|
||||
to agree *which register that pointer lands in*. Left to their defaults, the loader
|
||||
would place it in RCX while the kernel looked in RDI — and the kernel would read
|
||||
garbage. So both sides reference the same `danos.kernel_abi` (SysV): the loader's
|
||||
garbage. So both sides reference the same `system.kernel_abi` (SysV): the loader's
|
||||
function-pointer type and the kernel's `_start` both carry
|
||||
`callconv(danos.kernel_abi)`, and the pointer reliably arrives in RDI. That is the
|
||||
`callconv(system.kernel_abi)`, and the pointer reliably arrives in RDI. That is the
|
||||
whole reason `kernel_abi` lives in the shared contract — see [efi.md](efi.md) for
|
||||
the handoff it governs.
|
||||
|
||||
|
||||
+3
-2
@@ -8,8 +8,9 @@ without a human staring at the screen.
|
||||
There are two layers:
|
||||
|
||||
- **Host unit tests** (`zig build test`) — for pure, platform-independent logic in
|
||||
the shared `danos` module (the handoff layout in `system/danos.zig`). These compile
|
||||
for the host and run natively.
|
||||
the shared contracts (`system/boot-handoff.zig`, `system/abi.zig`,
|
||||
`system/devices/device-abi.zig`), which also compile-checks the three-way split
|
||||
stays self-consistent. These compile for the host and run natively.
|
||||
- **QEMU integration tests** (`python3 test/qemu_test.py`) — boot the real kernel
|
||||
and check its behaviour. This is the interesting part.
|
||||
|
||||
|
||||
+2
-2
@@ -84,13 +84,13 @@ interrupts](interrupts.md), a [calibrated timer + ns clock](device-interrupts.md
|
||||
in-kernel [IPC channels](ipc.md), SMP (all cores scheduling, with affinity), a
|
||||
**higher-half kernel** with a physmap, and **user space**: per-process address
|
||||
spaces, `syscall`/`sysret` with the `swapgs` discipline, a user-ELF loader, and
|
||||
`/sbin/init` — a real user ELF built from `sbin/`, running at CPL 3 as PID 1 on its
|
||||
`/system/services/init` — a real user ELF built from `system/services/init/`, running at CPL 3 as PID 1 on its
|
||||
own page tables — plus a [test harness](testing.md).
|
||||
|
||||
- **Isolation track** — **user mode + address-space isolation**. *Done: a
|
||||
higher-half kernel with a physmap (the low half is user space), per-process
|
||||
address spaces with CR3 switched on context switch, the `swapgs` discipline,
|
||||
`syscall`/`sysret`, a user-ELF loader, and `/sbin/init` running as a real
|
||||
`syscall`/`sysret`, a user-ELF loader, and `/system/services/init` running as a real
|
||||
preemptive ring-3 process (PID 1). Remaining polish: an address-space/stack
|
||||
reaper for exited tasks, SMAP + fault-recovering copy-in/out, the real IPC
|
||||
syscalls (IPC_Call/IPC_ReplyWait — they arrive with the second user server),
|
||||
|
||||
@@ -0,0 +1,13 @@
|
||||
//! DanOS's POSIX / C compatibility layer — `unistd`, `stdio`, and (later) the C
|
||||
//! `errno` / `struct stat` / `extern "C"` surface. This is the *one* place POSIX and
|
||||
//! C spellings are allowed to appear verbatim (see docs/coding-standards.md): a file
|
||||
//! under library/posix/ *is* the foreign ABI, so it keeps the ABI's names. Everything
|
||||
//! it touches on the danos side (the VFS protocol, the runtime) uses danos names,
|
||||
//! which this layer translates to at the boundary.
|
||||
//!
|
||||
//! It is layered strictly *over* the runtime: it calls the runtime's IPC and heap,
|
||||
//! never the kernel's system calls directly. danos-native applications use the
|
||||
//! runtime; this exists so *POSIX* software can too.
|
||||
|
||||
pub const unistd = @import("unistd.zig");
|
||||
pub const stdio = @import("stdio.zig");
|
||||
@@ -5,7 +5,7 @@
|
||||
|
||||
const std = @import("std");
|
||||
const unistd = @import("unistd.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const heap = @import("runtime").heap;
|
||||
|
||||
pub const SEEK_SET = unistd.SEEK_SET;
|
||||
pub const SEEK_CURRENT = unistd.SEEK_CURRENT;
|
||||
@@ -5,10 +5,9 @@
|
||||
|
||||
const std = @import("std");
|
||||
const protocol = @import("vfs-protocol");
|
||||
const ipc = @import("ipc.zig");
|
||||
const danos = @import("danos");
|
||||
const ipc = @import("runtime").ipc;
|
||||
|
||||
pub const O_CREAT = protocol.O_CREAT;
|
||||
pub const O_CREAT = protocol.create;
|
||||
pub const SEEK_SET: u32 = 0;
|
||||
pub const SEEK_CURRENT: u32 = 1;
|
||||
pub const SEEK_END: u32 = 2;
|
||||
@@ -110,11 +109,11 @@ pub fn lseek(fd: i32, off: i64, whence: u32) i64 {
|
||||
SEEK_SET => 0,
|
||||
SEEK_CURRENT => @intCast(f.offset),
|
||||
SEEK_END => blk: {
|
||||
const request = protocol.Request{ .operation = .stat, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
var sbuf: [@sizeOf(protocol.Stat)]u8 = undefined;
|
||||
const request = protocol.Request{ .operation = .status, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
var sbuf: [@sizeOf(protocol.FileStatus)]u8 = undefined;
|
||||
const r = transact(request, &.{}, &sbuf) orelse return -1;
|
||||
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.Stat)) return -1;
|
||||
const st = std.mem.bytesToValue(protocol.Stat, sbuf[0..@sizeOf(protocol.Stat)]);
|
||||
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.FileStatus)) return -1;
|
||||
const st = std.mem.bytesToValue(protocol.FileStatus, sbuf[0..@sizeOf(protocol.FileStatus)]);
|
||||
break :blk @intCast(st.size);
|
||||
},
|
||||
else => return -1,
|
||||
@@ -126,17 +125,17 @@ pub fn lseek(fd: i32, off: i64, whence: u32) i64 {
|
||||
}
|
||||
|
||||
/// Stat `path`. Returns 0 or -1.
|
||||
pub fn stat(path: []const u8, out: *protocol.Stat) i32 {
|
||||
pub fn stat(path: []const u8, out: *protocol.FileStatus) i32 {
|
||||
// Open, stat by node, close — simple and enough for now.
|
||||
const fd = open(path, 0);
|
||||
if (fd < 0) return -1;
|
||||
defer close(fd);
|
||||
const f = fdPtr(fd).?;
|
||||
const request = protocol.Request{ .operation = .stat, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
var sbuf: [@sizeOf(protocol.Stat)]u8 = undefined;
|
||||
const request = protocol.Request{ .operation = .status, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
var sbuf: [@sizeOf(protocol.FileStatus)]u8 = undefined;
|
||||
const r = transact(request, &.{}, &sbuf) orelse return -1;
|
||||
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.Stat)) return -1;
|
||||
out.* = std.mem.bytesToValue(protocol.Stat, sbuf[0..@sizeOf(protocol.Stat)]);
|
||||
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.FileStatus)) return -1;
|
||||
out.* = std.mem.bytesToValue(protocol.FileStatus, sbuf[0..@sizeOf(protocol.FileStatus)]);
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -3,13 +3,13 @@
|
||||
//! ownership of its hardware; the claim is the capability the kernel checks before
|
||||
//! mapping registers or routing an IRQ.
|
||||
|
||||
const danos = @import("danos");
|
||||
const device_abi = @import("device-abi");
|
||||
const sc = @import("system-call.zig");
|
||||
|
||||
pub const DeviceDescriptor = danos.DeviceDescriptor;
|
||||
pub const ResourceDescriptor = danos.ResourceDescriptor;
|
||||
pub const DeviceClass = danos.DeviceClass;
|
||||
pub const ResourceKind = danos.ResourceKind;
|
||||
pub const DeviceDescriptor = device_abi.DeviceDescriptor;
|
||||
pub const ResourceDescriptor = device_abi.ResourceDescriptor;
|
||||
pub const DeviceClass = device_abi.DeviceClass;
|
||||
pub const ResourceKind = device_abi.ResourceKind;
|
||||
|
||||
inline fn failed(r: usize) bool {
|
||||
return r > ~@as(usize, 0) - 4095;
|
||||
@@ -33,7 +33,7 @@ pub fn mmioMap(device_id: u64, resource_index: u64) ?usize {
|
||||
}
|
||||
|
||||
/// `DeviceDescriptor.parent` for a device with no parent.
|
||||
pub const no_parent = danos.no_parent;
|
||||
pub const no_parent = device_abi.no_parent;
|
||||
|
||||
/// Publish `descriptor` as a child of `parent_id`, which this process must have claimed.
|
||||
/// Returns the new device id. The child is left unclaimed, so whichever driver owns
|
||||
|
||||
@@ -13,10 +13,10 @@
|
||||
//! lock and larger alignments come when user programs gain threads.
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const system = @import("system.zig");
|
||||
const abi = @import("abi");
|
||||
const system_calls = @import("system.zig");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
const page_size = abi.page_size;
|
||||
|
||||
/// A block header, at the start of every block; while free it also links the
|
||||
/// free list via `next`.
|
||||
@@ -46,8 +46,8 @@ fn payloadOf(block: *Block) [*]u8 {
|
||||
/// grants usually are adjacent). Returns false if the kernel is out of memory.
|
||||
fn grow(minimum_bytes: usize) bool {
|
||||
const bytes = alignUp(@max(minimum_bytes, chunk), page_size);
|
||||
const ret = system.mmap(bytes, system.PROT_READ | system.PROT_WRITE);
|
||||
if (system.mmapFailed(ret)) return false;
|
||||
const ret = system_calls.mmap(bytes, system_calls.PROT_READ | system_calls.PROT_WRITE);
|
||||
if (system_calls.mmapFailed(ret)) return false;
|
||||
|
||||
const block: *Block = @ptrFromInt(ret);
|
||||
block.size = bytes;
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
//! reached this way. The server side (`replyWait`, which returns two values) is
|
||||
//! added with the first server binary.
|
||||
|
||||
const danos = @import("danos");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
|
||||
/// A small-int handle into the calling process's handle table.
|
||||
@@ -30,13 +30,13 @@ pub fn createEndpoint() ?Handle {
|
||||
}
|
||||
|
||||
/// Publish endpoint `h` under a well-known service id so other processes find it.
|
||||
pub fn register(id: danos.ServiceId, h: Handle) bool {
|
||||
pub fn register(id: abi.ServiceId, h: Handle) bool {
|
||||
return !failed(sc.systemCall2(.ipc_register, @intFromEnum(id), h));
|
||||
}
|
||||
|
||||
/// Find the endpoint published under `id`, installing a handle to it in this
|
||||
/// process.
|
||||
pub fn lookup(id: danos.ServiceId) ?Handle {
|
||||
pub fn lookup(id: abi.ServiceId) ?Handle {
|
||||
const r = sc.systemCall1(.ipc_lookup, @intFromEnum(id));
|
||||
return if (failed(r)) null else r;
|
||||
}
|
||||
@@ -53,7 +53,7 @@ pub fn call(h: Handle, message: []const u8, reply: []u8) CallError!usize {
|
||||
/// Set in `Received.badge` when what arrived is an asynchronous notification — a
|
||||
/// bound device interrupt — rather than a client's message. The low bits carry the
|
||||
/// GSI. See `isNotification`.
|
||||
pub const notify_badge_bit: u64 = danos.notify_badge_bit;
|
||||
pub const notify_badge_bit: u64 = abi.notify_badge_bit;
|
||||
|
||||
/// The result of a `replyWait`: the request length and the sender's badge (a
|
||||
/// task id, or an IRQ notification if the high bit is set).
|
||||
@@ -84,7 +84,7 @@ pub fn replyWait(h: Handle, reply: []const u8, receive: []u8) Received {
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "+{rdx}" (rdx),
|
||||
: [n] "{rax}" (@intFromEnum(danos.SystemCall.ipc_reply_wait)),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.ipc_reply_wait)),
|
||||
[a0] "{rdi}" (h),
|
||||
[a1] "{rsi}" (@intFromPtr(reply.ptr)),
|
||||
[a3] "{r10}" (@intFromPtr(receive.ptr)),
|
||||
|
||||
@@ -17,9 +17,7 @@ pub const start = @import("start.zig");
|
||||
/// The VFS wire protocol (shared with the VFS server).
|
||||
pub const vfs_protocol = @import("vfs-protocol");
|
||||
/// POSIX-style file API: open/read/write/lseek/stat/close.
|
||||
pub const unistd = @import("unistd.zig");
|
||||
/// C stdio: fopen/fread/fwrite/fseek/ftell/fclose over unistd.
|
||||
pub const stdio = @import("stdio.zig");
|
||||
/// Device access for drivers: enumerate/claim/mmioMap.
|
||||
pub const device = @import("device.zig");
|
||||
|
||||
|
||||
@@ -6,8 +6,8 @@
|
||||
//! Note argument #3 goes in **r10, not rcx** — rcx is unavailable across the
|
||||
//! instruction, so the kernel reads the 4th argument from r10.
|
||||
|
||||
const danos = @import("danos");
|
||||
const SystemCall = danos.SystemCall;
|
||||
const abi = @import("abi");
|
||||
const SystemCall = abi.SystemCall;
|
||||
|
||||
pub inline fn systemCall0(n: SystemCall) usize {
|
||||
return asm volatile ("syscall"
|
||||
|
||||
@@ -1,15 +1,15 @@
|
||||
//! Typed system_call surface for user space — thin wrappers over the raw `system_call`
|
||||
//! stubs, one per kernel call. Numbers come from `danos.SystemCall`, the single
|
||||
//! stubs, one per kernel call. Numbers come from `abi.SystemCall`, the single
|
||||
//! source of truth shared with the kernel dispatcher.
|
||||
|
||||
const danos = @import("danos");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
|
||||
/// `mmap` protection flags (matching the usual C bit values). Grants are always
|
||||
/// readable+writable today; the kernel does not yet honour finer prot.
|
||||
pub const PROT_READ: usize = danos.prot_read;
|
||||
pub const PROT_WRITE: usize = danos.prot_write;
|
||||
pub const PROT_EXEC: usize = danos.prot_exec;
|
||||
pub const PROT_READ: usize = abi.prot_read;
|
||||
pub const PROT_WRITE: usize = abi.prot_write;
|
||||
pub const PROT_EXEC: usize = abi.prot_exec;
|
||||
|
||||
/// Give up the rest of this quantum.
|
||||
pub fn yield() void {
|
||||
@@ -33,6 +33,14 @@ pub fn exit(code: usize) noreturn {
|
||||
unreachable; // the kernel never returns from exit
|
||||
}
|
||||
|
||||
/// Start the binary bundled in the initial-ramdisk under `name` as a new ring-3
|
||||
/// process, returning true on success. This is how a supervisor (the device manager)
|
||||
/// launches a driver it matched — danos-native, not POSIX (a spawn/exec family comes
|
||||
/// with the process work later).
|
||||
pub fn spawn(name: []const u8) bool {
|
||||
return sc.systemCall2(.system_spawn, @intFromPtr(name.ptr), name.len) == 0;
|
||||
}
|
||||
|
||||
/// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable
|
||||
/// memory and return the base virtual address. On failure returns a value in the
|
||||
/// top page (see `mmapFailed`). The user heap grows through this call.
|
||||
|
||||
@@ -0,0 +1,61 @@
|
||||
//! The **kernel ↔ user** ABI: the core contract every user program speaks to the
|
||||
//! kernel — the system_call numbers, `mmap` protection flags, the page size those
|
||||
//! calls work in, and the IPC name-registry ids and notification bit. Shared by the
|
||||
//! kernel dispatcher (system/kernel/process.zig) and the user runtime library
|
||||
//! (library/runtime/), so the two can never drift.
|
||||
//!
|
||||
//! This is the *core* ABI; the device half — `DeviceDescriptor` and friends, which
|
||||
//! also cross this boundary — lives with the device sub-project as [[device-abi]]
|
||||
//! (system/devices/device-abi.zig). The loader↔kernel handoff is [[boot-handoff]].
|
||||
|
||||
/// Page size every `mmap`/`munmap` grant and the boot memory map are measured in.
|
||||
/// 4 KiB on every architecture danos targets so far. Part of the ABI because user
|
||||
/// code aligns to it (grants are page-granular) and the kernel guarantees it.
|
||||
pub const page_size = 4096;
|
||||
|
||||
/// The kernel system_call numbers — the single source of truth shared by the kernel
|
||||
/// dispatcher (system/kernel/process.zig) and the user runtime library, so the two
|
||||
/// can never drift. The set is deliberately microkernel-minimal: file/device I/O
|
||||
/// is not here — it lives in user-space servers reached through the IPC calls.
|
||||
/// The table grows one milestone at a time; see docs/syscall.md.
|
||||
pub const SystemCall = enum(u64) {
|
||||
exit = 0, // exit(code): end the calling process
|
||||
yield = 1, // yield(): give up the rest of this quantum
|
||||
debug_write = 2, // debug_write(ptr, len): raw bytes to the kernel log (bring-up only)
|
||||
sleep = 3, // sleep(ms): block the caller for ms milliseconds
|
||||
mmap = 4, // mmap(len, prot) -> base: grant zeroed, page-aligned user pages
|
||||
munmap = 5, // munmap(base, len): release pages from a prior mmap
|
||||
create_endpoint = 6, // create_endpoint() -> handle: a new IPC endpoint
|
||||
ipc_register = 7, // ipc_register(service_id, handle): publish an endpoint by well-known id
|
||||
ipc_lookup = 8, // ipc_lookup(service_id) -> handle: find a published endpoint
|
||||
ipc_call = 9, // ipc_call(h, message, len, reply, cap) -> reply_len: send + block for reply
|
||||
ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, receive, cap) -> receive_len (+badge in rdx)
|
||||
device_enumerate = 11, // device_enumerate(buffer, maximum) -> count: snapshot the device table
|
||||
device_claim = 12, // device_claim(id) -> ok: take exclusive ownership of a device
|
||||
mmio_map = 13, // mmio_map(id, resource_index) -> vaddr: map a claimed device's MMIO into this AS
|
||||
irq_bind = 14, // irq_bind(id, resource_index, endpoint): deliver a device IRQ as an IPC notification
|
||||
irq_ack = 15, // irq_ack(id, resource_index): re-arm a bound IRQ after servicing it
|
||||
device_register = 16, // device_register(parent_id, descriptor) -> id: publish a child of a device you claimed
|
||||
system_spawn = 17, // system_spawn(name_ptr, name_len) -> 0: start a named initial-ramdisk binary as a new ring-3 process
|
||||
_,
|
||||
};
|
||||
|
||||
/// Set in the badge returned by `ipc_reply_wait` when what arrived is an
|
||||
/// **asynchronous notification** (today: a device interrupt bound with `irq_bind`)
|
||||
/// rather than a message from a client. There is no payload and no reply owed; the
|
||||
/// low bits carry the source, a GSI. Shared so the kernel's ISR and the driver's
|
||||
/// event loop can't disagree about which bit means "the hardware spoke".
|
||||
pub const notify_badge_bit: u64 = 1 << 63;
|
||||
|
||||
/// Well-known IPC service ids for the bootstrap name registry (create_endpoint +
|
||||
/// ipc_register/ipc_lookup). Small integers, so no string interning is needed
|
||||
/// during bring-up. The VFS server registers under `vfs`; clients look it up.
|
||||
pub const ServiceId = enum(u32) {
|
||||
vfs = 1,
|
||||
_,
|
||||
};
|
||||
|
||||
/// Protection flags for `mmap` (matching the usual C bit values).
|
||||
pub const prot_read: u64 = 1;
|
||||
pub const prot_write: u64 = 2;
|
||||
pub const prot_exec: u64 = 4;
|
||||
@@ -1,8 +1,11 @@
|
||||
//! Shared definitions that form the contract between a bootloader
|
||||
//! (boot/, e.g. efi.zig built as BOOTX64.efi) and the kernel (system/kernel/main.zig).
|
||||
//! The **loader ↔ kernel** contract: everything a bootloader (boot/, e.g. efi.zig
|
||||
//! built as BOOTX64.efi) and the kernel (system/kernel/kernel.zig) must agree on to
|
||||
//! hand control over — the handoff structures the loader fills in, plus the kernel's
|
||||
//! virtual-memory layout and the physical↔virtual addressing both sides use.
|
||||
//!
|
||||
//! Both binaries import this as the "danos" module, so the handoff layout is
|
||||
//! defined in exactly one place.
|
||||
//! Both binaries import this as the `boot-handoff` module, so the layout is defined
|
||||
//! in exactly one place. **User space never sees this** — the kernel↔user contract is
|
||||
//! [[abi]] (system/abi.zig); device types are [[device-abi]] (system/devices/device-abi.zig).
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
@@ -41,10 +44,6 @@ pub const Framebuffer = extern struct {
|
||||
}
|
||||
};
|
||||
|
||||
/// Page size the memory map is measured in. 4 KiB on every architecture danos
|
||||
/// targets so far.
|
||||
pub const page_size = 4096;
|
||||
|
||||
/// The kernel's virtual-memory layout (higher-half). The kernel is linked at
|
||||
/// `kernel_virt_base` but loaded at a low physical address; all of RAM (and the
|
||||
/// device MMIO windows) is also mapped at `physmap_base + physical`, so the kernel
|
||||
@@ -58,105 +57,6 @@ pub const page_size = 4096;
|
||||
pub const physmap_base: u64 = 0xFFFF_8800_0000_0000;
|
||||
pub const kernel_virt_base: u64 = 0xFFFF_FFFF_8000_0000;
|
||||
|
||||
/// The kernel system_call numbers — the single source of truth shared by the kernel
|
||||
/// dispatcher (system/kernel/process.zig) and the user runtime library, so the two
|
||||
/// can never drift. The set is deliberately microkernel-minimal: file/device I/O
|
||||
/// is not here — it lives in user-space servers reached through the IPC calls.
|
||||
/// The table grows one milestone at a time; see docs/syscall.md.
|
||||
pub const SystemCall = enum(u64) {
|
||||
exit = 0, // exit(code): end the calling process
|
||||
yield = 1, // yield(): give up the rest of this quantum
|
||||
debug_write = 2, // debug_write(ptr, len): raw bytes to the kernel log (bring-up only)
|
||||
sleep = 3, // sleep(ms): block the caller for ms milliseconds
|
||||
mmap = 4, // mmap(len, prot) -> base: grant zeroed, page-aligned user pages
|
||||
munmap = 5, // munmap(base, len): release pages from a prior mmap
|
||||
create_endpoint = 6, // create_endpoint() -> handle: a new IPC endpoint
|
||||
ipc_register = 7, // ipc_register(service_id, handle): publish an endpoint by well-known id
|
||||
ipc_lookup = 8, // ipc_lookup(service_id) -> handle: find a published endpoint
|
||||
ipc_call = 9, // ipc_call(h, message, len, reply, cap) -> reply_len: send + block for reply
|
||||
ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, receive, cap) -> receive_len (+badge in rdx)
|
||||
device_enumerate = 11, // device_enumerate(buffer, maximum) -> count: snapshot the device table
|
||||
device_claim = 12, // device_claim(id) -> ok: take exclusive ownership of a device
|
||||
mmio_map = 13, // mmio_map(id, resource_index) -> vaddr: map a claimed device's MMIO into this AS
|
||||
irq_bind = 14, // irq_bind(id, resource_index, endpoint): deliver a device IRQ as an IPC notification
|
||||
irq_ack = 15, // irq_ack(id, resource_index): re-arm a bound IRQ after servicing it
|
||||
device_register = 16, // device_register(parent_id, descriptor) -> id: publish a child of a device you claimed
|
||||
_,
|
||||
};
|
||||
|
||||
/// Set in the badge returned by `ipc_reply_wait` when what arrived is an
|
||||
/// **asynchronous notification** (today: a device interrupt bound with `irq_bind`)
|
||||
/// rather than a message from a client. There is no payload and no reply owed; the
|
||||
/// low bits carry the source, a GSI. Shared so the kernel's ISR and the driver's
|
||||
/// event loop can't disagree about which bit means "the hardware spoke".
|
||||
pub const notify_badge_bit: u64 = 1 << 63;
|
||||
|
||||
/// A device class, mirroring system/devices/device-model.zig's `DeviceClass` **in order**
|
||||
/// (its `@intFromEnum` values cross the system_call boundary in `DeviceDescriptor.class`).
|
||||
/// Keep the two in sync.
|
||||
pub const DeviceClass = enum(u32) {
|
||||
root,
|
||||
processor,
|
||||
interrupt_controller,
|
||||
timer,
|
||||
pci_host_bridge,
|
||||
pci_device,
|
||||
acpi_device,
|
||||
unknown,
|
||||
};
|
||||
|
||||
/// A resource kind, mirroring system/devices/device-model.zig's `ResourceKind` in order.
|
||||
pub const ResourceKind = enum(u32) {
|
||||
memory,
|
||||
io_port,
|
||||
irq,
|
||||
bus_range,
|
||||
};
|
||||
|
||||
/// One device resource, as handed to a user-space driver (flat, extern).
|
||||
pub const ResourceDescriptor = extern struct {
|
||||
kind: u64, // a ResourceKind value
|
||||
start: u64,
|
||||
len: u64,
|
||||
};
|
||||
|
||||
pub const maximum_device_resources = 8;
|
||||
|
||||
/// `DeviceDescriptor.parent` for a device with no parent — a root of the device tree.
|
||||
pub const no_parent: u64 = ~@as(u64, 0);
|
||||
|
||||
/// A device, as snapshotted for user space by `device_enumerate`. A driver scans
|
||||
/// these to find the hardware it owns, claims it, and maps its MMIO.
|
||||
///
|
||||
/// `parent` makes the table a tree rather than a list, which is what a **bus driver**
|
||||
/// needs: it claims the bus, finds the devices below it, and publishes any it
|
||||
/// discovers itself with `device_register`. A registered child's resources must lie
|
||||
/// within its parent's (the kernel enforces this) — that containment is what makes
|
||||
/// delegation safe, since a device descriptor is otherwise a licence to map physical
|
||||
/// memory.
|
||||
pub const DeviceDescriptor = extern struct {
|
||||
id: u64,
|
||||
parent: u64, // a device id, or `no_parent`
|
||||
class: u64, // a DeviceClass value
|
||||
hid_len: u64,
|
||||
resource_count: u64,
|
||||
hid: [8]u8,
|
||||
resources: [maximum_device_resources]ResourceDescriptor,
|
||||
};
|
||||
|
||||
/// Well-known IPC service ids for the bootstrap name registry (create_endpoint +
|
||||
/// ipc_register/ipc_lookup). Small integers, so no string interning is needed
|
||||
/// during bring-up. The VFS server registers under `vfs`; clients look it up.
|
||||
pub const ServiceId = enum(u32) {
|
||||
vfs = 1,
|
||||
_,
|
||||
};
|
||||
|
||||
/// Protection flags for `mmap` (matching the usual C bit values).
|
||||
pub const prot_read: u64 = 1;
|
||||
pub const prot_write: u64 = 2;
|
||||
pub const prot_exec: u64 = 4;
|
||||
|
||||
/// Physical address -> its virtual address in the physmap. The single way the
|
||||
/// kernel dereferences a physical address once paging is up.
|
||||
///
|
||||
@@ -204,7 +104,7 @@ pub const MemoryKind = enum(u32) {
|
||||
/// firmware's variable descriptor-stride to worry about.
|
||||
pub const MemoryRegion = extern struct {
|
||||
base: u64, // physical start address
|
||||
pages: u64, // length in `page_size` units
|
||||
pages: u64, // length in 4 KiB pages (the [[abi]] `page_size` unit)
|
||||
kind: MemoryKind,
|
||||
_pad: u32 = 0,
|
||||
};
|
||||
@@ -242,15 +142,15 @@ pub const BootInformation = extern struct {
|
||||
/// A device-tree boot path leaves this 0 and (later) fills a `device_tree_blob`
|
||||
/// field instead, so the kernel discovers devices without knowing what booted it.
|
||||
acpi_rsdp: u64 = 0,
|
||||
/// The raw `/sbin/init` ELF image, read off the boot volume by the loader
|
||||
/// The raw `/system/services/init` ELF image, read off the boot volume by the loader
|
||||
/// into memory that survives the handoff (classified reserved, so the kernel
|
||||
/// identity-maps it and never allocates over it). 0/0 = no init found — the
|
||||
/// kernel boots without user space. Grows into a full initrd handoff later.
|
||||
/// kernel boots without user space. Grows into a full initial_ramdisk handoff later.
|
||||
init_base: u64 = 0,
|
||||
init_len: u64 = 0,
|
||||
/// The initrd image (a bundle of extra user binaries — the VFS server and
|
||||
/// The initial_ramdisk image (a bundle of extra user binaries — the VFS server and
|
||||
/// device drivers), read off the boot volume into memory that survives the
|
||||
/// handoff, same as `init` above. 0/0 = no initrd. See system/initrd.zig.
|
||||
initrd_base: u64 = 0,
|
||||
initrd_len: u64 = 0,
|
||||
/// handoff, same as `init` above. 0/0 = no initial_ramdisk. See system/initial-ramdisk.zig.
|
||||
initial_ramdisk_base: u64 = 0,
|
||||
initial_ramdisk_len: u64 = 0,
|
||||
};
|
||||
+13
-12
@@ -15,7 +15,8 @@
|
||||
//! the `Hal.mapMmio` callback the caller supplies (the architecture VMM's map primitive).
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const abi = @import("abi");
|
||||
const parameters = @import("parameters");
|
||||
const device_model = @import("device-model.zig");
|
||||
const aml = @import("aml/aml.zig");
|
||||
@@ -147,7 +148,7 @@ var aml_block_count: usize = 0;
|
||||
|
||||
fn addAmlBlock(sdt_physical: u64) void {
|
||||
if (aml_block_count >= aml_block_physical.len or sdt_physical == 0) return;
|
||||
const h: *const SystemDescriptorTableHeader = @ptrFromInt(danos.physicalToVirtual(sdt_physical));
|
||||
const h: *const SystemDescriptorTableHeader = @ptrFromInt(boot_handoff.physicalToVirtual(sdt_physical));
|
||||
if (h.length <= @sizeOf(SystemDescriptorTableHeader)) return;
|
||||
aml_block_physical[aml_block_count] = sdt_physical + @sizeOf(SystemDescriptorTableHeader);
|
||||
aml_block_len[aml_block_count] = h.length - @sizeOf(SystemDescriptorTableHeader);
|
||||
@@ -376,14 +377,14 @@ pub fn discover(rsdp_physical: u64, device_tree: *DeviceTree, hal: Hal) !void {
|
||||
dsdt_physical = 0;
|
||||
aml_block_count = 0;
|
||||
|
||||
const rsdp: *const RootSystemDescriptionPointer = @ptrFromInt(danos.physicalToVirtual(rsdp_physical));
|
||||
const rsdp: *const RootSystemDescriptionPointer = @ptrFromInt(boot_handoff.physicalToVirtual(rsdp_physical));
|
||||
if (!std.mem.eql(u8, &rsdp.signature, "RSD PTR ")) return error.BadRsdpSignature;
|
||||
// Revision 0 checksums only the first 20 bytes (the v1.0 RSDP).
|
||||
if (!checksumOk(@ptrFromInt(danos.physicalToVirtual(rsdp_physical)), 20)) return error.BadRsdpChecksum;
|
||||
if (!checksumOk(@ptrFromInt(boot_handoff.physicalToVirtual(rsdp_physical)), 20)) return error.BadRsdpChecksum;
|
||||
|
||||
if (rsdp.revision >= 2) {
|
||||
const xsdp: *const ExtendedSystemDescriptorPointer = @ptrFromInt(danos.physicalToVirtual(rsdp_physical));
|
||||
if (!checksumOk(@ptrFromInt(danos.physicalToVirtual(rsdp_physical)), xsdp.length)) return error.BadXsdpChecksum;
|
||||
const xsdp: *const ExtendedSystemDescriptorPointer = @ptrFromInt(boot_handoff.physicalToVirtual(rsdp_physical));
|
||||
if (!checksumOk(@ptrFromInt(boot_handoff.physicalToVirtual(rsdp_physical)), xsdp.length)) return error.BadXsdpChecksum;
|
||||
try walkRoot(u64, xsdp.extended_system_descriptor_table_address, device_tree, hal);
|
||||
} else {
|
||||
try walkRoot(u32, rsdp.root_system_description_table_address, device_tree, hal);
|
||||
@@ -393,7 +394,7 @@ pub fn discover(rsdp_physical: u64, device_tree: *DeviceTree, hal: Hal) !void {
|
||||
// read the sleep types from it.
|
||||
var blocks: [aml_block_physical.len][]const u8 = undefined;
|
||||
for (0..aml_block_count) |i| {
|
||||
blocks[i] = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(aml_block_physical[i])))[0..aml_block_len[i]];
|
||||
blocks[i] = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(aml_block_physical[i])))[0..aml_block_len[i]];
|
||||
}
|
||||
const active = blocks[0..aml_block_count];
|
||||
if (aml.parse(device_tree.allocator, active)) |pr| {
|
||||
@@ -412,11 +413,11 @@ pub fn discover(rsdp_physical: u64, device_tree: *DeviceTree, hal: Hal) !void {
|
||||
/// Walk the RSDT (Entry = u32) or XSDT (Entry = u64): validate it, then dispatch
|
||||
/// each SDT it points at. A bad individual table is skipped, not fatal.
|
||||
fn walkRoot(comptime Entry: type, root_physical: u64, device_tree: *DeviceTree, hal: Hal) !void {
|
||||
const header: *const SystemDescriptorTableHeader = @ptrFromInt(danos.physicalToVirtual(root_physical));
|
||||
if (!checksumOk(@ptrFromInt(danos.physicalToVirtual(root_physical)), header.length)) return error.BadRootChecksum;
|
||||
const header: *const SystemDescriptorTableHeader = @ptrFromInt(boot_handoff.physicalToVirtual(root_physical));
|
||||
if (!checksumOk(@ptrFromInt(boot_handoff.physicalToVirtual(root_physical)), header.length)) return error.BadRootChecksum;
|
||||
|
||||
const count = (header.length - @sizeOf(SystemDescriptorTableHeader)) / @sizeOf(Entry);
|
||||
const base: [*]const u8 = @ptrFromInt(danos.physicalToVirtual(root_physical));
|
||||
const base: [*]const u8 = @ptrFromInt(boot_handoff.physicalToVirtual(root_physical));
|
||||
const entries: [*]align(1) const Entry = @ptrCast(base + @sizeOf(SystemDescriptorTableHeader));
|
||||
|
||||
for (entries[0..count]) |ent| {
|
||||
@@ -427,7 +428,7 @@ fn walkRoot(comptime Entry: type, root_physical: u64, device_tree: *DeviceTree,
|
||||
|
||||
/// Dispatch a single SDT on its signature.
|
||||
fn handleTable(device_tree: *DeviceTree, hal: Hal, sdt_physical: u64) !void {
|
||||
const header: *const SystemDescriptorTableHeader = @ptrFromInt(danos.physicalToVirtual(sdt_physical));
|
||||
const header: *const SystemDescriptorTableHeader = @ptrFromInt(boot_handoff.physicalToVirtual(sdt_physical));
|
||||
const sig = header.signature;
|
||||
if (std.mem.eql(u8, &sig, &APIC)) {
|
||||
try parseMadt(device_tree, header);
|
||||
@@ -1120,7 +1121,7 @@ fn pciConfigurationPtr(alloc: McfgAllocation, hal: Hal, bus: u8, device: u8, fun
|
||||
(@as(u64, function) << 12);
|
||||
// Map the configuration page (writable, for BAR sizing) and use the virtual
|
||||
// address the HAL hands back.
|
||||
return @ptrFromInt(hal.mapMmio(physical, danos.page_size, true));
|
||||
return @ptrFromInt(hal.mapMmio(physical, abi.page_size, true));
|
||||
}
|
||||
|
||||
/// Read a little-endian integer at `off` from a (possibly unaligned) byte pointer.
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
//!
|
||||
//! This module has two stages. `parser.zig` walks the entire byte stream and
|
||||
//! records every named object into a namespace tree (`namespace.zig`), capturing
|
||||
//! method bodies and field/region layout. `interp.zig` then *evaluates* control
|
||||
//! method bodies and field/region layout. `interpreter.zig` then *evaluates* control
|
||||
//! methods on demand — running operators, control flow, and OperationRegion field
|
||||
//! access — so callers can resolve device status (`_STA`), current resource
|
||||
//! settings (`_CRS`), sleep states (`_Sx`), and the like against the live namespace.
|
||||
@@ -17,10 +17,10 @@ pub const Node = @import("namespace.zig").Node;
|
||||
pub const NodeKind = @import("namespace.zig").NodeKind;
|
||||
|
||||
/// The AML evaluator: interprets control methods (and reads Names/Fields) far
|
||||
/// enough for device discovery. See `interp.zig`.
|
||||
pub const Interpreter = @import("interp.zig").Interpreter;
|
||||
pub const Object = @import("interp.zig").Object;
|
||||
pub const EvaluateHal = @import("interp.zig").Hal;
|
||||
/// enough for device discovery. See `interpreter.zig`.
|
||||
pub const Interpreter = @import("interpreter.zig").Interpreter;
|
||||
pub const Object = @import("interpreter.zig").Object;
|
||||
pub const EvaluateHal = @import("interpreter.zig").Hal;
|
||||
|
||||
/// The SLP_TYP values written to PM1a/PM1b control to enter a sleep state.
|
||||
pub const SleepType = struct {
|
||||
|
||||
@@ -0,0 +1,76 @@
|
||||
//! The **device ABI**: the flat, `extern` device types that cross the system_call
|
||||
//! boundary — what `device_enumerate` hands a user-space driver, what
|
||||
//! `device_register` takes back. This is the devices sub-project's *public
|
||||
//! interface*, exposed as its own `device-abi` module the same way the VFS server
|
||||
//! exposes `vfs-protocol` — so both the kernel and user space depend on the contract
|
||||
//! by name, and neither reaches into the other's files.
|
||||
//!
|
||||
//! It is also the **single source of truth** for `DeviceClass` and `ResourceKind`:
|
||||
//! the kernel's rich, pointer-based device tree (system/devices/device-model.zig,
|
||||
//! which user space must never import) re-exports these, so the enum that a driver
|
||||
//! matches on and the enum the kernel classifies with are the *same* type — no
|
||||
//! hand-kept "mirror in order" to drift. The core kernel↔user ABI is [[abi]]; the
|
||||
//! loader↔kernel handoff is [[boot-handoff]].
|
||||
|
||||
/// A coarse classification of a device, independent of the describing firmware.
|
||||
/// Kept small on purpose; refine as real drivers arrive. `enum(u32)` because the
|
||||
/// `@intFromEnum` value crosses the system_call boundary in `DeviceDescriptor.class`.
|
||||
pub const DeviceClass = enum(u32) {
|
||||
/// The synthetic root every discovered device hangs beneath.
|
||||
root,
|
||||
processor,
|
||||
interrupt_controller,
|
||||
timer,
|
||||
/// A PCI(e) host bridge — the root of a PCI segment (owns an ECAM window).
|
||||
pci_host_bridge,
|
||||
/// A single PCI function.
|
||||
pci_device,
|
||||
/// A device named in the ACPI namespace (from the DSDT/SSDT), carrying a
|
||||
/// hardware ID (`_HID`) and, where static, current resource settings (`_CRS`).
|
||||
acpi_device,
|
||||
unknown,
|
||||
};
|
||||
|
||||
/// The kind of hardware resource a device occupies. `enum(u32)` for the same
|
||||
/// boundary-crossing reason as `DeviceClass` (see `ResourceDescriptor.kind`).
|
||||
pub const ResourceKind = enum(u32) {
|
||||
/// A memory-mapped I/O window: `start` is the physical base, `len` its size.
|
||||
memory,
|
||||
/// A legacy I/O-port range: `start` is the first port, `len` the count.
|
||||
io_port,
|
||||
/// An interrupt: `start` is the global system interrupt (GSI), `len` is 1.
|
||||
irq,
|
||||
/// A range of bus numbers owned by a bridge: `start`..`start+len`.
|
||||
bus_range,
|
||||
};
|
||||
|
||||
/// One device resource, as handed to a user-space driver (flat, extern).
|
||||
pub const ResourceDescriptor = extern struct {
|
||||
kind: u64, // a ResourceKind value
|
||||
start: u64,
|
||||
len: u64,
|
||||
};
|
||||
|
||||
pub const maximum_device_resources = 8;
|
||||
|
||||
/// `DeviceDescriptor.parent` for a device with no parent — a root of the device tree.
|
||||
pub const no_parent: u64 = ~@as(u64, 0);
|
||||
|
||||
/// A device, as snapshotted for user space by `device_enumerate`. A driver scans
|
||||
/// these to find the hardware it owns, claims it, and maps its MMIO.
|
||||
///
|
||||
/// `parent` makes the table a tree rather than a list, which is what a **bus driver**
|
||||
/// needs: it claims the bus, finds the devices below it, and publishes any it
|
||||
/// discovers itself with `device_register`. A registered child's resources must lie
|
||||
/// within its parent's (the kernel enforces this) — that containment is what makes
|
||||
/// delegation safe, since a device descriptor is otherwise a licence to map physical
|
||||
/// memory.
|
||||
pub const DeviceDescriptor = extern struct {
|
||||
id: u64,
|
||||
parent: u64, // a device id, or `no_parent`
|
||||
class: u64, // a DeviceClass value
|
||||
hid_len: u64,
|
||||
resource_count: u64,
|
||||
hid: [8]u8,
|
||||
resources: [maximum_device_resources]ResourceDescriptor,
|
||||
};
|
||||
@@ -12,6 +12,7 @@
|
||||
//! on top of this — nothing here presumes them.
|
||||
|
||||
const std = @import("std");
|
||||
const device_abi = @import("device-abi");
|
||||
|
||||
/// The hardware primitives a discovery backend needs but can't express portably.
|
||||
/// The kernel injects an implementation (the architecture VMM + port I/O), so the device
|
||||
@@ -26,17 +27,10 @@ pub const Hal = struct {
|
||||
pioWrite: *const fn (width: u8, port: u16, value: u32) void,
|
||||
};
|
||||
|
||||
/// The kind of hardware resource a device occupies.
|
||||
pub const ResourceKind = enum {
|
||||
/// A memory-mapped I/O window: `start` is the physical base, `len` its size.
|
||||
memory,
|
||||
/// A legacy I/O-port range: `start` is the first port, `len` the count.
|
||||
io_port,
|
||||
/// An interrupt: `start` is the global system interrupt (GSI), `len` is 1.
|
||||
irq,
|
||||
/// A range of bus numbers owned by a bridge: `start`..`start+len`.
|
||||
bus_range,
|
||||
};
|
||||
/// The kind of hardware resource a device occupies. Canonically defined by the
|
||||
/// device ABI (system/devices/device-abi.zig) and re-exported here, so the kernel's
|
||||
/// internal tree and the descriptors it hands user space share one enum.
|
||||
pub const ResourceKind = device_abi.ResourceKind;
|
||||
|
||||
/// One hardware resource claimed by a device.
|
||||
pub const Resource = struct {
|
||||
@@ -46,22 +40,10 @@ pub const Resource = struct {
|
||||
};
|
||||
|
||||
/// A coarse classification of a device, independent of the describing firmware.
|
||||
/// Kept small on purpose; refine as real drivers arrive.
|
||||
pub const DeviceClass = enum {
|
||||
/// The synthetic root every discovered device hangs beneath.
|
||||
root,
|
||||
processor,
|
||||
interrupt_controller,
|
||||
timer,
|
||||
/// A PCI(e) host bridge — the root of a PCI segment (owns an ECAM window).
|
||||
pci_host_bridge,
|
||||
/// A single PCI function.
|
||||
pci_device,
|
||||
/// A device named in the ACPI namespace (from the DSDT/SSDT), carrying a
|
||||
/// hardware ID (`_HID`) and, where static, current resource settings (`_CRS`).
|
||||
acpi_device,
|
||||
unknown,
|
||||
};
|
||||
/// Canonically defined by the device ABI (system/devices/device-abi.zig) and
|
||||
/// re-exported here — the enum a driver matches on and the one the kernel classifies
|
||||
/// with are the same type. Kept small on purpose; refine as real drivers arrive.
|
||||
pub const DeviceClass = device_abi.DeviceClass;
|
||||
|
||||
/// Firmware-independent identity. Each backend fills only the fields it knows;
|
||||
/// the rest stay null. The generic layer never branches on *how* an id was
|
||||
|
||||
@@ -9,7 +9,7 @@
|
||||
//! compile-time choice.
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const device_model = @import("device-model.zig");
|
||||
const acpi = @import("acpi.zig");
|
||||
const power = @import("power.zig");
|
||||
@@ -65,7 +65,7 @@ pub fn cpusDropped() usize {
|
||||
/// ACPI registers); pass the architecture implementation. Errors leave nothing to clean up
|
||||
/// beyond the tree's own allocations.
|
||||
pub fn discover(
|
||||
boot_information: *const danos.BootInformation,
|
||||
boot_information: *const boot_handoff.BootInformation,
|
||||
allocator: std.mem.Allocator,
|
||||
hal: Hal,
|
||||
) !DeviceTree {
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
//! /sbin/busd — a user-space **bus driver**, and the smallest honest example of one.
|
||||
//! /system/drivers/bus — a user-space **bus driver**, and the smallest honest example of one.
|
||||
//!
|
||||
//! A bus driver owns a device that *contains other devices*, enumerates them by some
|
||||
//! bus-specific protocol, and publishes each one into the kernel's device table so a
|
||||
@@ -6,10 +6,10 @@
|
||||
//! the "bus" is the HPET's register block and the "devices" are its comparators, each
|
||||
//! a 0x20-byte window at 0x100 + 0x20*n that can be driven independently.
|
||||
//!
|
||||
//! It's a toy bus, but nothing about the mechanism is: `busd` reads how many children
|
||||
//! It's a toy bus, but nothing about the mechanism is: `bus` reads how many children
|
||||
//! exist from the hardware (GENERAL_CAP bits [12:8]), publishes one `DeviceDescriptor` per
|
||||
//! child with a sub-window of its own MMIO plus the shared IRQ, and the kernel checks
|
||||
//! every one of those resources is contained in what `busd` was granted. A comparator
|
||||
//! every one of those resources is contained in what `bus` was granted. A comparator
|
||||
//! driver then claims a child and maps only *its* registers — not the whole block.
|
||||
//!
|
||||
//! It also proves the negative: registering a child whose window escapes the parent's
|
||||
@@ -65,30 +65,30 @@ fn firstChildOf(buffer: []device.DeviceDescriptor, total: usize, parent_id: u64)
|
||||
|
||||
pub fn main() void {
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("busd: out of memory\n");
|
||||
_ = runtime.system.write("bus: out of memory\n");
|
||||
return;
|
||||
};
|
||||
|
||||
const parent = findHpet(buffer) orelse {
|
||||
_ = runtime.system.write("busd: no HPET\n");
|
||||
_ = runtime.system.write("bus: no HPET\n");
|
||||
return;
|
||||
};
|
||||
const resource = resourcesOf(parent);
|
||||
|
||||
// Claim the bus. Everything below is subdivision of what this claim granted.
|
||||
//
|
||||
// Claims are exclusive, and at a normal boot the kernel spawns every initrd
|
||||
// binary — so hpetd may own the HPET already. That's not an error, it's the
|
||||
// Claims are exclusive, and at a normal boot the kernel spawns every initial_ramdisk
|
||||
// binary — so hpet may own the HPET already. That's not an error, it's the
|
||||
// capability model working: exit quietly and leave the device to its owner. The
|
||||
// `bus` test spawns busd alone, so there it wins the claim.
|
||||
// `bus` test spawns bus alone, so there it wins the claim.
|
||||
if (!device.claim(parent.id)) {
|
||||
_ = runtime.system.write("busd: HPET already claimed by another driver, nothing to do\n");
|
||||
_ = runtime.system.write("bus: HPET already claimed by another driver, nothing to do\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Enumerate the bus: ask the hardware how many children it has.
|
||||
const base = device.mmioMap(parent.id, 0) orelse {
|
||||
_ = runtime.system.write("busd: mmio_map failed\n");
|
||||
_ = runtime.system.write("bus: mmio_map failed\n");
|
||||
return;
|
||||
};
|
||||
const cap: *volatile u64 = @ptrFromInt(base + register_general_cap);
|
||||
@@ -112,7 +112,7 @@ pub fn main() void {
|
||||
}
|
||||
|
||||
if (device.register(parent.id, &child) == null) {
|
||||
_ = runtime.system.write("busd: register failed\n");
|
||||
_ = runtime.system.write("bus: register failed\n");
|
||||
return;
|
||||
}
|
||||
published += 1;
|
||||
@@ -133,11 +133,11 @@ pub fn main() void {
|
||||
.len = 0x1000,
|
||||
};
|
||||
if (device.register(parent.id, &rogue) != null) {
|
||||
_ = runtime.system.write("busd: FAIL out-of-window child was accepted\n");
|
||||
_ = runtime.system.write("bus: FAIL out-of-window child was accepted\n");
|
||||
return;
|
||||
}
|
||||
if (device.enumerate(buffer) != before) {
|
||||
_ = runtime.system.write("busd: FAIL rogue child leaked into the table\n");
|
||||
_ = runtime.system.write("bus: FAIL rogue child leaked into the table\n");
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -149,18 +149,18 @@ pub fn main() void {
|
||||
if (d.parent != parent.id) continue;
|
||||
const w = d.resources[0];
|
||||
if (w.start < resource.mmio.start or w.len >= resource.mmio.len) {
|
||||
_ = runtime.system.write("busd: FAIL child window is not inside the bus\n");
|
||||
_ = runtime.system.write("bus: FAIL child window is not inside the bus\n");
|
||||
return;
|
||||
}
|
||||
seen += 1;
|
||||
}
|
||||
if (seen != published) {
|
||||
_ = runtime.system.write("busd: FAIL child count mismatch\n");
|
||||
_ = runtime.system.write("bus: FAIL child count mismatch\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Delegation, end to end: claim a child and map *it*. A real class driver would be
|
||||
// a different process; here busd plays both parts, which exercises the same path.
|
||||
// a different process; here bus plays both parts, which exercises the same path.
|
||||
// The child's window is 0x20 bytes at parent+0x100, so the register it sees at
|
||||
// offset 0 must be the same timer-0 configuration register the bus sees at 0x100.
|
||||
//
|
||||
@@ -168,21 +168,21 @@ pub fn main() void {
|
||||
// 4 KiB the HPET lives in — the granularity limit documented in docs/drivers.md.
|
||||
// The *resource* is narrow even though the page isn't.)
|
||||
const child_id = firstChildOf(buffer, device.enumerate(buffer), parent.id) orelse {
|
||||
_ = runtime.system.write("busd: FAIL no child to claim\n");
|
||||
_ = runtime.system.write("bus: FAIL no child to claim\n");
|
||||
return;
|
||||
};
|
||||
if (!device.claim(child_id)) {
|
||||
_ = runtime.system.write("busd: FAIL could not claim own child\n");
|
||||
_ = runtime.system.write("bus: FAIL could not claim own child\n");
|
||||
return;
|
||||
}
|
||||
const child_base = device.mmioMap(child_id, 0) orelse {
|
||||
_ = runtime.system.write("busd: FAIL child mmio_map refused\n");
|
||||
_ = runtime.system.write("bus: FAIL child mmio_map refused\n");
|
||||
return;
|
||||
};
|
||||
const via_child: *volatile u64 = @ptrFromInt(child_base);
|
||||
const via_bus: *volatile u64 = @ptrFromInt(base + 0x100);
|
||||
if (via_child.* != via_bus.*) {
|
||||
_ = runtime.system.write("busd: FAIL child window does not alias the bus register\n");
|
||||
_ = runtime.system.write("bus: FAIL child window does not alias the bus register\n");
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -195,12 +195,12 @@ pub fn main() void {
|
||||
_ = runtime.system.munmap(scratch, 0x1000);
|
||||
const descriptor: *const device.DeviceDescriptor = @ptrFromInt(scratch);
|
||||
if (device.register(parent.id, descriptor) != null) {
|
||||
_ = runtime.system.write("busd: FAIL register accepted an unmapped descriptor\n");
|
||||
_ = runtime.system.write("bus: FAIL register accepted an unmapped descriptor\n");
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
_ = runtime.system.write("busd: ok\n");
|
||||
_ = runtime.system.write("bus: ok\n");
|
||||
while (true) runtime.system.sleep(1000);
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
//! /sbin/hpetd — a user-space HPET driver. It proves the whole driver model end to
|
||||
//! /system/drivers/hpet — a user-space HPET driver. It proves the whole driver model end to
|
||||
//! end: enumerate the device table, find the HPET, claim it, map its registers into
|
||||
//! this ring-3 address space (strong-uncacheable), **bind its interrupt to an IPC
|
||||
//! endpoint**, then sit blocked in `replyWait` until the hardware wakes it.
|
||||
@@ -13,8 +13,8 @@
|
||||
//! the full cycle to be correct:
|
||||
//!
|
||||
//! kernel ISR mask the GSI -> EOI -> notify this endpoint
|
||||
//! hpetd wake, clear GENERAL_INT_STATUS (deasserts the line), re-arm
|
||||
//! hpetd irq_ack -> kernel unmasks the GSI
|
||||
//! hpet wake, clear GENERAL_INT_STATUS (deasserts the line), re-arm
|
||||
//! hpet irq_ack -> kernel unmasks the GSI
|
||||
//!
|
||||
//! Clear the status bit *before* acking, or the line is still asserted when the
|
||||
//! kernel unmasks and the I/O APIC redelivers forever.
|
||||
@@ -64,7 +64,7 @@ fn findHpet(buffer: []device.DeviceDescriptor) ?Found {
|
||||
for (buffer[0..n]) |d| {
|
||||
if (d.class != @intFromEnum(device.DeviceClass.timer)) continue;
|
||||
// Skip comparator children a bus driver may have published below the block
|
||||
// (see system/drivers/busd/busd.zig) — we want the register block itself.
|
||||
// (see system/drivers/bus/bus.zig) — we want the register block itself.
|
||||
if (d.parent != device.no_parent) continue;
|
||||
var mmio: ?u64 = null;
|
||||
var irq: ?u64 = null;
|
||||
@@ -85,21 +85,21 @@ fn findHpet(buffer: []device.DeviceDescriptor) ?Found {
|
||||
pub fn main() void {
|
||||
// Enumerate into a heap buffer (too big for the one-page user stack).
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 32) catch {
|
||||
_ = runtime.system.write("hpetd: out of memory\n");
|
||||
_ = runtime.system.write("hpet: out of memory\n");
|
||||
return;
|
||||
};
|
||||
|
||||
const hpet = findHpet(buffer) orelse {
|
||||
_ = runtime.system.write("hpetd: no HPET with an IRQ\n");
|
||||
_ = runtime.system.write("hpet: no HPET with an IRQ\n");
|
||||
return;
|
||||
};
|
||||
|
||||
if (!device.claim(hpet.device_id)) {
|
||||
_ = runtime.system.write("hpetd: claim failed\n");
|
||||
_ = runtime.system.write("hpet: claim failed\n");
|
||||
return;
|
||||
}
|
||||
const base = device.mmioMap(hpet.device_id, hpet.mmio) orelse {
|
||||
_ = runtime.system.write("hpetd: mmio_map failed\n");
|
||||
_ = runtime.system.write("hpet: mmio_map failed\n");
|
||||
return;
|
||||
};
|
||||
|
||||
@@ -108,7 +108,7 @@ pub fn main() void {
|
||||
const gsi = hpet.gsi;
|
||||
|
||||
const endpoint = ipc.createEndpoint() orelse {
|
||||
_ = runtime.system.write("hpetd: create_endpoint failed\n");
|
||||
_ = runtime.system.write("hpet: create_endpoint failed\n");
|
||||
return;
|
||||
};
|
||||
|
||||
@@ -116,7 +116,7 @@ pub fn main() void {
|
||||
// Counter period, so we can arm the comparator a fixed wall-clock distance out.
|
||||
const femtos_per_tick = register(base, register_general_cap).* >> 32;
|
||||
if (femtos_per_tick == 0) {
|
||||
_ = runtime.system.write("hpetd: bad HPET period\n");
|
||||
_ = runtime.system.write("hpet: bad HPET period\n");
|
||||
return;
|
||||
}
|
||||
const ticks_per_ms = 1_000_000_000_000 / femtos_per_tick;
|
||||
@@ -139,10 +139,10 @@ pub fn main() void {
|
||||
register(base, register_general_configuration).* |= configuration_enable;
|
||||
|
||||
if (!device.irqBind(hpet.device_id, hpet.irq, endpoint)) {
|
||||
_ = runtime.system.write("hpetd: irq_bind failed\n");
|
||||
_ = runtime.system.write("hpet: irq_bind failed\n");
|
||||
return;
|
||||
}
|
||||
_ = runtime.system.write("hpetd: bound, sleeping until the hardware speaks\n");
|
||||
_ = runtime.system.write("hpet: bound, sleeping until the hardware speaks\n");
|
||||
|
||||
// --- the driver loop -----------------------------------------------------
|
||||
// Blocked in replyWait. No polling, no spinning: the next line of this function
|
||||
@@ -170,14 +170,14 @@ pub fn main() void {
|
||||
register(base, register_timer0_configuration).* &= ~tn_int_enb;
|
||||
}
|
||||
|
||||
_ = runtime.system.write("hpetd: irq\n");
|
||||
_ = runtime.system.write("hpet: irq\n");
|
||||
if (!device.irqAck(hpet.device_id, hpet.irq)) {
|
||||
_ = runtime.system.write("hpetd: irq_ack failed\n");
|
||||
_ = runtime.system.write("hpet: irq_ack failed\n");
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
_ = runtime.system.write("hpetd: ok\n");
|
||||
_ = runtime.system.write("hpet: ok\n");
|
||||
while (true) runtime.system.sleep(1000);
|
||||
}
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
//! The initrd (initial ramdisk) container format — shared by the build-time
|
||||
//! packer (tools/mkinitrd.zig) and the kernel that unpacks it. Deliberately
|
||||
//! The initial_ramdisk (initial ramdisk) container format — shared by the build-time
|
||||
//! packer (tools/make-initial-ramdisk.py) and the kernel that unpacks it. Deliberately
|
||||
//! trivial: a header, a table of fixed-size entries, then the concatenated file
|
||||
//! blobs. We own both producer and consumer, so it need be no fancier.
|
||||
//!
|
||||
@@ -10,7 +10,7 @@
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// "DNRD" — identifies a danos initrd image.
|
||||
/// "DNRD" — identifies a danos initial_ramdisk image.
|
||||
pub const magic: u32 = 0x444E5244;
|
||||
|
||||
pub const Header = extern struct {
|
||||
@@ -24,7 +24,7 @@ pub const Entry = extern struct {
|
||||
len: u64, // blob length in bytes
|
||||
};
|
||||
|
||||
/// A validated view over an initrd image. `init` checks the magic and that the
|
||||
/// A validated view over an initial_ramdisk image. `init` checks the magic and that the
|
||||
/// entry table fits; `entry` bounds-checks each blob against the image.
|
||||
pub const Reader = struct {
|
||||
image: []const u8,
|
||||
@@ -9,7 +9,7 @@
|
||||
//! map). Every interrupt must be acknowledged with an end-of-interrupt write, or
|
||||
//! the LAPIC won't deliver the next one.
|
||||
|
||||
const danos = @import("danos");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const io = @import("io.zig");
|
||||
const paging = @import("paging.zig");
|
||||
|
||||
@@ -121,7 +121,7 @@ pub fn init() void {
|
||||
|
||||
const msr = io.rdmsr(ia32_apic_base_msr);
|
||||
// Reach the LAPIC through the physmap (paging.init maps its page there).
|
||||
base = @intCast(danos.physicalToVirtual(msr & 0xFFFFF000)); // physical base is bits 12+
|
||||
base = @intCast(boot_handoff.physicalToVirtual(msr & 0xFFFFF000)); // physical base is bits 12+
|
||||
io.wrmsr(ia32_apic_base_msr, msr | (1 << 11)); // global enable
|
||||
|
||||
write(register_spurious, 0x100 | spurious_vector); // bit 8 = software enable
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
//! build.zig — no change to the generic code. Keep everything CPU-specific here
|
||||
//! (halt, the descriptor tables, later paging), and nothing generic.
|
||||
|
||||
const danos = @import("danos");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const parameters = @import("parameters");
|
||||
const gdt = @import("gdt.zig");
|
||||
const tss = @import("tss.zig");
|
||||
@@ -132,7 +132,7 @@ pub fn init() void {
|
||||
/// Build the kernel's own page tables (with real permissions) and switch onto
|
||||
/// them. Needs the frame allocator and the boot info (for the memory map and the
|
||||
/// kernel's segment layout). Call once the frame allocator is up.
|
||||
pub fn enablePaging(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_information: *const danos.BootInformation) void {
|
||||
pub fn enablePaging(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_information: *const boot_handoff.BootInformation) void {
|
||||
paging.init(allocFrame, freeFrame, boot_information);
|
||||
}
|
||||
|
||||
|
||||
@@ -10,10 +10,11 @@
|
||||
//! Everything is 4 KiB pages — precise and simple; the extra table memory is
|
||||
//! negligible against available RAM.
|
||||
|
||||
const danos = @import("danos");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const abi = @import("abi");
|
||||
const io = @import("io.zig");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
const page_size = abi.page_size;
|
||||
|
||||
// Page-table entry bits.
|
||||
const present: u64 = 1 << 0;
|
||||
@@ -58,7 +59,7 @@ const bootstrap_physmap_limit: u64 = 4 << 30;
|
||||
/// both the loader's bootstrap tables and the kernel's own, which share the
|
||||
/// physmap base.
|
||||
fn tableAt(physical: u64) *[512]u64 {
|
||||
return @ptrFromInt(danos.physicalToVirtual(physical));
|
||||
return @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
||||
}
|
||||
|
||||
fn allocTable() u64 {
|
||||
@@ -102,12 +103,12 @@ fn mapRangePhysmap(pml4: u64, physical_base: u64, len: u64, flags: u64) void {
|
||||
var address = physical_base & ~@as(u64, page_size - 1);
|
||||
const end = physical_base + len;
|
||||
while (address < end) : (address += page_size) {
|
||||
mapPage(pml4, danos.physicalToVirtual(address), address, flags);
|
||||
mapPage(pml4, boot_handoff.physicalToVirtual(address), address, flags);
|
||||
}
|
||||
}
|
||||
|
||||
fn regions(mm: danos.MemoryMap) []const danos.MemoryRegion {
|
||||
return @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physicalToVirtual(mm.regions)))[0..mm.len];
|
||||
fn regions(mm: boot_handoff.MemoryMap) []const boot_handoff.MemoryRegion {
|
||||
return @as([*]const boot_handoff.MemoryRegion, @ptrFromInt(boot_handoff.physicalToVirtual(mm.regions)))[0..mm.len];
|
||||
}
|
||||
|
||||
/// Enable the NX bit in the page-table format (EFER.NXE). Must happen before we
|
||||
@@ -118,7 +119,7 @@ fn enableNx() void {
|
||||
}
|
||||
|
||||
/// Build the address space and switch onto it.
|
||||
pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_information: *const danos.BootInformation) void {
|
||||
pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot_information: *const boot_handoff.BootInformation) void {
|
||||
alloc_frame = allocFrame;
|
||||
free_frame = freeFrame;
|
||||
enableNx();
|
||||
@@ -136,7 +137,7 @@ pub fn init(allocFrame: *const fn () ?u64, freeFrame: *const fn (u64) void, boot
|
||||
// the kernel touches directly), RW + NX.
|
||||
const fb = boot_information.framebuffer;
|
||||
mapRangePhysmap(pml4, fb.base, @as(u64, fb.height) * fb.pitch, present | writable | no_execute);
|
||||
mapPage(pml4, danos.physicalToVirtual(0xFEE00000), 0xFEE00000, present | writable | no_execute);
|
||||
mapPage(pml4, boot_handoff.physicalToVirtual(0xFEE00000), 0xFEE00000, present | writable | no_execute);
|
||||
|
||||
// 3. The kernel's own segments at their higher-half link addresses, mapped
|
||||
// to their low physical load addresses with real ELF permissions: code
|
||||
@@ -206,11 +207,11 @@ pub fn mapMmio(physical: u64, len: u64, writable_page: bool) u64 {
|
||||
const last = physical + (if (len == 0) 1 else len) - 1;
|
||||
var address = first;
|
||||
while (address <= (last & ~@as(u64, page_size - 1))) : (address += page_size) {
|
||||
const virtual = danos.physicalToVirtual(address);
|
||||
const virtual = boot_handoff.physicalToVirtual(address);
|
||||
mapPage(kernel_pml4, virtual, address, flags);
|
||||
invalidate(virtual);
|
||||
}
|
||||
return danos.physicalToVirtual(physical);
|
||||
return boot_handoff.physicalToVirtual(physical);
|
||||
}
|
||||
|
||||
/// Like `descend`, but also sets the U/S bit on the intermediate entry (new or
|
||||
|
||||
@@ -13,7 +13,7 @@
|
||||
//! Once a core has its own descriptor tables, LAPIC, and timer, it calls the generic
|
||||
//! scheduler entry and joins the run loop — mechanism here, policy there.
|
||||
|
||||
const danos = @import("danos");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const io = @import("io.zig");
|
||||
const gdt = @import("gdt.zig");
|
||||
const tss = @import("tss.zig");
|
||||
@@ -85,7 +85,7 @@ fn arm() void {
|
||||
const start = @extern([*]const u8, .{ .name = "ap_trampoline_start" });
|
||||
const end = @extern([*]const u8, .{ .name = "ap_trampoline_end" });
|
||||
const len = @intFromPtr(end) - @intFromPtr(start);
|
||||
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(tramp_physical));
|
||||
const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(tramp_physical));
|
||||
@memcpy(destination[0..len], start[0..len]);
|
||||
}
|
||||
|
||||
@@ -95,7 +95,7 @@ fn arm() void {
|
||||
/// reported in — it's long past the trampoline by then, in the kernel image; a
|
||||
/// core that never answered is dead and can't be mid-climb.
|
||||
fn disarm() void {
|
||||
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(tramp_physical));
|
||||
const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(tramp_physical));
|
||||
@memset(destination[0..page_size], 0);
|
||||
paging.unmap(tramp_physical); // drop the transient low identity mapping
|
||||
}
|
||||
@@ -107,7 +107,7 @@ fn disarm() void {
|
||||
fn param(comptime name: []const u8) *align(1) volatile u64 {
|
||||
const start = @intFromPtr(@extern([*]const u8, .{ .name = "ap_trampoline_start" }));
|
||||
const sym = @intFromPtr(@extern([*]const u8, .{ .name = name }));
|
||||
return @ptrFromInt(danos.physicalToVirtual(tramp_physical + (sym - start)));
|
||||
return @ptrFromInt(boot_handoff.physicalToVirtual(tramp_physical + (sym - start)));
|
||||
}
|
||||
|
||||
/// Wake the core with Local APIC id `apic_id` as dense CPU `index`, hand it
|
||||
|
||||
@@ -14,7 +14,7 @@
|
||||
//! never assumes a display exists.
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
|
||||
/// The one framebuffer console, valid only when `con_present`.
|
||||
var con: Console = undefined;
|
||||
@@ -22,7 +22,7 @@ var con_present: bool = false;
|
||||
|
||||
/// Set up the console over `fb`, or mark it absent if there's no usable
|
||||
/// framebuffer. Clears the screen when present.
|
||||
pub fn init(fb: danos.Framebuffer) void {
|
||||
pub fn init(fb: boot_handoff.Framebuffer) void {
|
||||
if (!fb.present()) {
|
||||
con_present = false;
|
||||
return;
|
||||
@@ -58,7 +58,7 @@ const glyph_bytes = glyph_h; // 8 pixels wide => 1 byte per row
|
||||
const glyph_data = 32; // PSF2 header size
|
||||
|
||||
pub const Console = struct {
|
||||
fb: danos.Framebuffer,
|
||||
fb: boot_handoff.Framebuffer,
|
||||
cols: u32,
|
||||
rows: u32,
|
||||
col: u32 = 0,
|
||||
@@ -66,12 +66,12 @@ pub const Console = struct {
|
||||
fg: u32 = 0x00c8_c8c8, // light grey
|
||||
bg: u32 = 0x0000_0000, // black
|
||||
|
||||
pub fn init(fb: danos.Framebuffer) Console {
|
||||
pub fn init(fb: boot_handoff.Framebuffer) Console {
|
||||
// Reach the framebuffer through the physmap, so the pointer stays valid
|
||||
// once the low identity map is gone. The base is mapped by both the
|
||||
// loader's bootstrap tables and paging.init.
|
||||
var mapped = fb;
|
||||
if (fb.base != 0) mapped.base = danos.physicalToVirtual(fb.base);
|
||||
if (fb.base != 0) mapped.base = boot_handoff.physicalToVirtual(fb.base);
|
||||
return .{
|
||||
.fb = mapped,
|
||||
.cols = fb.width / glyph_w,
|
||||
|
||||
@@ -20,7 +20,7 @@
|
||||
|
||||
const std = @import("std");
|
||||
const platform = @import("platform");
|
||||
const danos = @import("danos");
|
||||
const device_abi = @import("device-abi");
|
||||
|
||||
const maximum_devices = 64;
|
||||
|
||||
@@ -32,7 +32,7 @@ const maximum_devices = 64;
|
||||
/// `device_release` to reclaim on exit) is future work — see docs/driver-model.md.
|
||||
const maximum_children_per_parent = 16;
|
||||
|
||||
var devices: [maximum_devices]danos.DeviceDescriptor = undefined;
|
||||
var devices: [maximum_devices]device_abi.DeviceDescriptor = undefined;
|
||||
var claimed: [maximum_devices]?u32 = .{null} ** maximum_devices; // owner task id, or null
|
||||
var count: usize = 0;
|
||||
|
||||
@@ -46,13 +46,13 @@ pub fn init(device_tree: *const platform.DeviceTree) void {
|
||||
count = 0;
|
||||
dropped = 0;
|
||||
for (&claimed) |*c| c.* = null;
|
||||
walk(device_tree.root, danos.no_parent);
|
||||
walk(device_tree.root, device_abi.no_parent);
|
||||
}
|
||||
|
||||
/// Record `node` (unless it's the synthetic root) and recurse, threading the id we
|
||||
/// assigned it down to its children as their parent.
|
||||
fn walk(node: *platform.Device, parent_id: u64) void {
|
||||
const id = if (node.class == .root) danos.no_parent else record(node, parent_id);
|
||||
const id = if (node.class == .root) device_abi.no_parent else record(node, parent_id);
|
||||
var child = node.first_child;
|
||||
while (child) |c| : (child = c.next_sibling) walk(c, id);
|
||||
}
|
||||
@@ -60,16 +60,16 @@ fn walk(node: *platform.Device, parent_id: u64) void {
|
||||
fn record(node: *platform.Device, parent_id: u64) u64 {
|
||||
if (count >= maximum_devices) {
|
||||
dropped += 1;
|
||||
return danos.no_parent; // children of a dropped node become roots, not orphans
|
||||
return device_abi.no_parent; // children of a dropped node become roots, not orphans
|
||||
}
|
||||
var d = std.mem.zeroes(danos.DeviceDescriptor);
|
||||
var d = std.mem.zeroes(device_abi.DeviceDescriptor);
|
||||
d.id = count;
|
||||
d.parent = parent_id;
|
||||
d.class = @intFromEnum(node.class);
|
||||
const h = node.hid();
|
||||
d.hid_len = @min(h.len, d.hid.len);
|
||||
@memcpy(d.hid[0..d.hid_len], h[0..d.hid_len]);
|
||||
const rc = @min(node.resource_count, danos.maximum_device_resources);
|
||||
const rc = @min(node.resource_count, device_abi.maximum_device_resources);
|
||||
d.resource_count = rc;
|
||||
for (0..rc) |i| {
|
||||
const r = node.resources[i];
|
||||
@@ -82,7 +82,7 @@ fn record(node: *platform.Device, parent_id: u64) u64 {
|
||||
|
||||
/// Copy up to `out.len` device descriptors into `out`; returns the total count
|
||||
/// available (which may exceed `out.len`).
|
||||
pub fn enumerate(out: []danos.DeviceDescriptor) usize {
|
||||
pub fn enumerate(out: []device_abi.DeviceDescriptor) usize {
|
||||
const n = @min(count, out.len);
|
||||
@memcpy(out[0..n], devices[0..n]);
|
||||
return count;
|
||||
@@ -104,7 +104,7 @@ pub fn ownerOf(id: u64) ?u32 {
|
||||
}
|
||||
|
||||
/// Resource `index` of device `id`, or null if out of range.
|
||||
pub fn resourceOf(id: u64, index: u64) ?danos.ResourceDescriptor {
|
||||
pub fn resourceOf(id: u64, index: u64) ?device_abi.ResourceDescriptor {
|
||||
if (id >= count) return null;
|
||||
const d = &devices[@intCast(id)];
|
||||
if (index >= d.resource_count) return null;
|
||||
@@ -115,9 +115,9 @@ pub fn resourceOf(id: u64, index: u64) ?danos.ResourceDescriptor {
|
||||
/// interval containment; for an irq it's equality, since an interrupt line is not
|
||||
/// divisible. Zero-length child ranges are refused — an empty window is meaningless
|
||||
/// and would otherwise vacuously "fit" anywhere.
|
||||
fn contains(parent: danos.ResourceDescriptor, child: danos.ResourceDescriptor) bool {
|
||||
fn contains(parent: device_abi.ResourceDescriptor, child: device_abi.ResourceDescriptor) bool {
|
||||
if (parent.kind != child.kind) return false;
|
||||
if (child.kind == @intFromEnum(danos.ResourceKind.irq)) return parent.start == child.start;
|
||||
if (child.kind == @intFromEnum(device_abi.ResourceKind.irq)) return parent.start == child.start;
|
||||
if (child.len == 0 or parent.len == 0) return false;
|
||||
// No overflow: a resource that wraps the address space is not containable.
|
||||
const child_end = std.math.add(u64, child.start, child.len) catch return false;
|
||||
@@ -149,10 +149,10 @@ fn childCount(parent_id: u64) usize {
|
||||
/// `owner` must have claimed `parent_id`, and every resource in `descriptor` must be
|
||||
/// contained in a parent resource of the same kind. A device with no resources is
|
||||
/// fine and common: a USB device is addressed through its controller, not by MMIO.
|
||||
pub fn register(parent_id: u64, owner: u32, descriptor: *const danos.DeviceDescriptor) RegisterError!u64 {
|
||||
pub fn register(parent_id: u64, owner: u32, descriptor: *const device_abi.DeviceDescriptor) RegisterError!u64 {
|
||||
const parent_owner = ownerOf(parent_id) orelse return error.BadParent;
|
||||
if (parent_owner != owner) return error.BadParent;
|
||||
if (descriptor.resource_count > danos.maximum_device_resources) return error.TooManyResources;
|
||||
if (descriptor.resource_count > device_abi.maximum_device_resources) return error.TooManyResources;
|
||||
if (childCount(parent_id) >= maximum_children_per_parent) return error.TooManyChildren;
|
||||
if (count >= maximum_devices) return error.NoSpace;
|
||||
|
||||
@@ -166,7 +166,7 @@ pub fn register(parent_id: u64, owner: u32, descriptor: *const danos.DeviceDescr
|
||||
if (!ok) return error.NotContained;
|
||||
}
|
||||
|
||||
var d = std.mem.zeroes(danos.DeviceDescriptor);
|
||||
var d = std.mem.zeroes(device_abi.DeviceDescriptor);
|
||||
d.id = count;
|
||||
d.parent = parent_id;
|
||||
d.class = descriptor.class;
|
||||
@@ -13,11 +13,11 @@
|
||||
//! interrupt handlers (ours don't). A lock comes with threads/SMP.
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const abi = @import("abi");
|
||||
const architecture = @import("architecture");
|
||||
const pmm = @import("pmm.zig");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
const page_size = abi.page_size;
|
||||
|
||||
/// Virtual base of the heap: the start of the higher half, which is unmapped and
|
||||
/// well clear of the identity-mapped low half. (Canonical on x86_64; an architecture that
|
||||
|
||||
@@ -22,13 +22,14 @@
|
||||
//! copy is a later security-track item, matching the existing debug_write gap.
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const abi = @import("abi");
|
||||
const architecture = @import("architecture");
|
||||
const scheduler = @import("scheduler.zig");
|
||||
const sync = @import("sync.zig");
|
||||
const heap = @import("heap.zig");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
const page_size = abi.page_size;
|
||||
const Task = scheduler.Task;
|
||||
|
||||
/// Largest message a single call/reply may carry. Bumping it is trivial; kept
|
||||
@@ -50,8 +51,8 @@ pub const ENOMEM: i64 = 6; // out of memory
|
||||
/// message from a client — there is no reply owed. The low bits carry the source
|
||||
/// (a GSI for IRQs). Posted by `notifyFromIsr`, from the ISR in system/kernel/irq.zig;
|
||||
/// the message path uses a plain task-id badge with this bit clear. Defined in the
|
||||
/// shared contract (system/danos.zig), because ring 3 has to test the same bit.
|
||||
pub const notify_badge_bit: u64 = danos.notify_badge_bit;
|
||||
/// shared kernel↔user ABI (system/abi.zig), because ring 3 has to test the same bit.
|
||||
pub const notify_badge_bit: u64 = abi.notify_badge_bit;
|
||||
|
||||
/// End of the user (low) canonical half — user buffers must lie below it.
|
||||
const user_half_end: u64 = 0x0000_8000_0000_0000;
|
||||
@@ -124,8 +125,8 @@ fn copyAcross(source_as: u64, source_va: u64, destination_as: u64, destination_v
|
||||
const s_left = page_size - ((source_va + off) & (page_size - 1));
|
||||
const d_left = page_size - ((destination_va + off) & (page_size - 1));
|
||||
const n = @min(@min(s_left, d_left), len - off);
|
||||
const source: [*]const u8 = @ptrFromInt(danos.physicalToVirtual(s));
|
||||
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(d));
|
||||
const source: [*]const u8 = @ptrFromInt(boot_handoff.physicalToVirtual(s));
|
||||
const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(d));
|
||||
@memcpy(destination[0..n], source[0..n]);
|
||||
off += n;
|
||||
}
|
||||
@@ -147,7 +148,7 @@ pub fn copyFromUser(user_as: u64, user_va: u64, destination: []u8) bool {
|
||||
const s = architecture.translate(user_as, user_va + off) orelse return false;
|
||||
const s_left = page_size - ((user_va + off) & (page_size - 1));
|
||||
const n = @min(s_left, destination.len - off);
|
||||
const source: [*]const u8 = @ptrFromInt(danos.physicalToVirtual(s));
|
||||
const source: [*]const u8 = @ptrFromInt(boot_handoff.physicalToVirtual(s));
|
||||
@memcpy(destination[off..][0..n], source[0..n]);
|
||||
off += n;
|
||||
}
|
||||
|
||||
@@ -22,7 +22,7 @@
|
||||
//!
|
||||
//! Binding is capability-gated exactly like `mmio_map`: the caller must have
|
||||
//! `device_claim`ed the device, and the GSI must come from one of that device's `irq`
|
||||
//! resources in the discovered device table (system/kernel/device-service.zig). A driver can
|
||||
//! resources in the discovered device table (system/kernel/devices-broker.zig). A driver can
|
||||
//! therefore never bind an interrupt it doesn't own — a raw-GSI system_call would let
|
||||
//! any process steal the keyboard's line.
|
||||
//!
|
||||
@@ -152,7 +152,7 @@ pub fn bind(gsi: u32, endpoint: *ipc_sync.Endpoint, owner: u32) BindError!void {
|
||||
bound_owner[gsi] = owner;
|
||||
|
||||
// Level-triggered, active-high. Level is the general case a driver must survive
|
||||
// (and what hpetd configures its comparator for); an edge source simply never
|
||||
// (and what hpet configures its comparator for); an edge source simply never
|
||||
// leaves the line asserted, so the mask/ack cycle is harmless there.
|
||||
//
|
||||
// Hardcoded for now: a device whose MADT interrupt-source override declares the
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const abi = @import("abi");
|
||||
const parameters = @import("parameters");
|
||||
const architecture = @import("architecture");
|
||||
const console = @import("console.zig");
|
||||
@@ -8,20 +9,19 @@ const pmm = @import("pmm.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const scheduler = @import("scheduler.zig");
|
||||
const process = @import("process.zig");
|
||||
const device_service = @import("device-service.zig");
|
||||
const devices_broker = @import("devices-broker.zig");
|
||||
const irq = @import("irq.zig");
|
||||
const initrd = @import("initrd");
|
||||
const platform = @import("platform");
|
||||
const tests = @import("tests.zig");
|
||||
const build_options = @import("build_options");
|
||||
const BootInformation = danos.BootInformation;
|
||||
const BootInformation = boot_handoff.BootInformation;
|
||||
|
||||
/// The calling convention used to enter the kernel. Pinned to SystemV explicitly:
|
||||
/// the bootloader is built for the UEFI target, whose C convention is Microsoft
|
||||
/// x64 (first argument in RCX), while the kernel is SystemV (first argument in
|
||||
/// RDI). Both sides reference this so the `boot_information` pointer lands in the
|
||||
/// register the other expects. `danos.kernel_abi` re-exports it to the loader.
|
||||
pub const kernel_abi = danos.kernel_abi;
|
||||
/// register the other expects. `boot_handoff.kernel_abi` re-exports it to the loader.
|
||||
pub const kernel_abi = boot_handoff.kernel_abi;
|
||||
|
||||
// POST/checkpoint codes emitted to I/O port 0x80 at boot milestones — the
|
||||
// last-resort progress signal on a machine with no text output at all.
|
||||
@@ -50,7 +50,7 @@ var ap_trampoline_page: u64 = 0;
|
||||
/// half. `boot_information` (also low) is reached through the physmap — its base is the
|
||||
/// same under the loader's bootstrap tables and the kernel's own.
|
||||
export fn kmainEntry(boot_information: *const BootInformation) callconv(kernel_abi) noreturn {
|
||||
kmain(@ptrFromInt(danos.physicalToVirtual(@intFromPtr(boot_information))));
|
||||
kmain(@ptrFromInt(boot_handoff.physicalToVirtual(@intFromPtr(boot_information))));
|
||||
}
|
||||
|
||||
fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
@@ -91,7 +91,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
|
||||
// Summarise the physical memory the loader handed us. The array is danos's
|
||||
// own MemoryRegion, so this is a plain slice — no firmware layout in sight.
|
||||
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physicalToVirtual(boot_information.memory_map.regions)))[0..boot_information.memory_map.len];
|
||||
const regions = @as([*]const boot_handoff.MemoryRegion, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.memory_map.regions)))[0..boot_information.memory_map.len];
|
||||
var usable_pages: u64 = 0;
|
||||
var reserved_pages: u64 = 0; // reserved RAM only — MMIO is device space, not RAM
|
||||
for (regions) |r| {
|
||||
@@ -102,7 +102,7 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
}
|
||||
}
|
||||
const total_pages = usable_pages + reserved_pages;
|
||||
const total_bytes = total_pages * danos.page_size;
|
||||
const total_bytes = total_pages * abi.page_size;
|
||||
const gib = 1 << 30;
|
||||
|
||||
log.write("\ndanos: physical memory\n");
|
||||
@@ -161,10 +161,10 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
|
||||
// Snapshot the device tree for user-space drivers (device_enumerate/claim/
|
||||
// mmio_map operate on this flat, id-indexed table + claim map).
|
||||
device_service.init(&device_tree);
|
||||
if (device_service.dropped > 0) {
|
||||
devices_broker.init(&device_tree);
|
||||
if (devices_broker.dropped > 0) {
|
||||
// Otherwise entirely silent: drivers would just never see that hardware.
|
||||
log.print("danos: WARNING {d} device(s) dropped — table full\n", .{device_service.dropped});
|
||||
log.print("danos: WARNING {d} device(s) dropped — table full\n", .{devices_broker.dropped});
|
||||
}
|
||||
|
||||
// Install the device-IRQ trampolines, so a driver's irq_bind has vectors to
|
||||
@@ -271,50 +271,42 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
log.checkpoint(cp_running);
|
||||
status("kernel initialised.\n");
|
||||
|
||||
// Hand over to user space: load /sbin/init (read off the boot volume by the
|
||||
// loader) and spawn it as a real ring-3 process, PID 1. It runs on its own
|
||||
// address space, preemptively, alongside the kernel — no cooperative
|
||||
// borrowing. This boot context then becomes the BSP's idle loop.
|
||||
// Publish the initial-ramdisk so user space can `system_spawn` its bundled
|
||||
// binaries by name. The kernel no longer launches them itself: init is the
|
||||
// service supervisor and the device manager spawns the drivers it discovers.
|
||||
publishInitialRamdisk(boot_information);
|
||||
|
||||
// Hand over to user space: load /system/services/init (read off the boot volume by
|
||||
// the loader) and spawn it as a real ring-3 process, PID 1. As the supervisor it
|
||||
// brings up the system services (the VFS server, the device manager); the device
|
||||
// manager then discovers the hardware and spawns each driver. init runs on its own
|
||||
// address space, preemptively — this boot context becomes the BSP's idle loop.
|
||||
if (boot_information.init_len != 0) {
|
||||
status("starting /sbin/init...\n");
|
||||
const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
||||
status("starting /system/services/init...\n");
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
||||
process.spawnProcess(image, 4) catch |err| {
|
||||
statusPrint("/sbin/init failed to load: {s}\n", .{@errorName(err)});
|
||||
statusPrint("/system/services/init failed to load: {s}\n", .{@errorName(err)});
|
||||
};
|
||||
} else {
|
||||
status("no /sbin/init on the boot volume.\n");
|
||||
status("no /system/services/init on the boot volume.\n");
|
||||
}
|
||||
|
||||
// Spawn the extra user binaries the loader ferried in the initrd (the VFS
|
||||
// server, and later device drivers). For now the kernel launches them all;
|
||||
// once init is a real service supervisor it will spawn them itself (system_spawn).
|
||||
startInitrdBinaries(boot_information);
|
||||
|
||||
// Become the idle task: drop below every real task and halt until an
|
||||
// interrupt. The timer keeps preempting into init and any other work.
|
||||
scheduler.setPriority(0);
|
||||
status("\nkernel idle; /sbin/init is running.\n");
|
||||
status("\nkernel idle; user space is running.\n");
|
||||
architecture.halt();
|
||||
}
|
||||
|
||||
/// Spawn every program bundled in the initrd as its own ring-3 process. A bad
|
||||
/// image or a program that fails to load is logged and skipped — the rest of the
|
||||
/// system still runs.
|
||||
fn startInitrdBinaries(boot_information: *const danos.BootInformation) void {
|
||||
if (boot_information.initrd_len == 0) return;
|
||||
const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.initrd_base)))[0..boot_information.initrd_len];
|
||||
const rd = initrd.Reader.init(image) orelse {
|
||||
status("initrd: bad image, skipping\n");
|
||||
return;
|
||||
};
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
statusPrint("starting /sbin/{s} (from initrd)...\n", .{item.name});
|
||||
process.spawnProcess(item.blob, 4) catch |err| {
|
||||
statusPrint("initrd: {s} failed to load: {s}\n", .{ item.name, @errorName(err) });
|
||||
};
|
||||
}
|
||||
/// Publish the initial-ramdisk image to the process layer so user space can
|
||||
/// `system_spawn` its bundled binaries by name. The kernel used to spawn every
|
||||
/// bundled program here; now init (the service supervisor) and the device manager
|
||||
/// (drivers) own that, so this only hands the image over — nothing is launched from
|
||||
/// the kernel.
|
||||
fn publishInitialRamdisk(boot_information: *const boot_handoff.BootInformation) void {
|
||||
if (boot_information.initial_ramdisk_len == 0) return;
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
process.setInitialRamdisk(image);
|
||||
}
|
||||
|
||||
/// Wake the application processors the firmware left parked. Allocates the low
|
||||
@@ -386,11 +378,11 @@ fn statusPrint(comptime fmt: []const u8, args: anytype) void {
|
||||
|
||||
/// Frames (4 KiB pages) to whole MiB.
|
||||
fn mib(pages: u64) u64 {
|
||||
return pages * danos.page_size / (1024 * 1024);
|
||||
return pages * abi.page_size / (1024 * 1024);
|
||||
}
|
||||
|
||||
fn kib(frames: u64) u64 {
|
||||
return frames * danos.page_size / (1024);
|
||||
return frames * abi.page_size / (1024);
|
||||
}
|
||||
|
||||
/// Report a CPU exception and halt **this core**. There's no fault recovery yet, so
|
||||
@@ -2,14 +2,15 @@
|
||||
//! 4 KiB physical frames — the primitive every later memory feature (page
|
||||
//! tables, the heap) is built on top of.
|
||||
//!
|
||||
//! This is generic kernel code: it works on the neutral `danos.MemoryRegion`
|
||||
//! This is generic kernel code: it works on the neutral `boot_handoff.MemoryRegion`
|
||||
//! array the loader hands over (see docs/memory-map.md), so it carries no UEFI
|
||||
//! and nothing architecture-specific beyond the 4 KiB page.
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const abi = @import("abi");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
const page_size = abi.page_size;
|
||||
|
||||
/// One bit per frame, covering physical RAM from 0 up to the highest usable
|
||||
/// address: 1 = used/unavailable, 0 = free. The bitmap itself lives in a frame
|
||||
@@ -48,8 +49,8 @@ inline fn setFree(frame: usize) void {
|
||||
bitmap[frame >> 3] &= ~(@as(u8, 1) << bit(frame));
|
||||
}
|
||||
|
||||
fn regions(map: danos.MemoryMap) []const danos.MemoryRegion {
|
||||
return @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physicalToVirtual(map.regions)))[0..map.len];
|
||||
fn regions(map: boot_handoff.MemoryMap) []const boot_handoff.MemoryRegion {
|
||||
return @as([*]const boot_handoff.MemoryRegion, @ptrFromInt(boot_handoff.physicalToVirtual(map.regions)))[0..map.len];
|
||||
}
|
||||
|
||||
/// Build the allocator from the loader's memory map. Reaches physical memory
|
||||
@@ -59,7 +60,7 @@ fn regions(map: danos.MemoryMap) []const danos.MemoryRegion {
|
||||
/// region (lowest address), which must sit under the bootstrap physmap's reach
|
||||
/// (4 GiB); it always does, as both this and the page-table allocator scan from
|
||||
/// low addresses up.
|
||||
pub fn init(map: danos.MemoryMap) void {
|
||||
pub fn init(map: boot_handoff.MemoryMap) void {
|
||||
const regs = regions(map);
|
||||
|
||||
// 1. Size the bitmap to cover every frame up to the highest RAM address —
|
||||
@@ -91,7 +92,7 @@ pub fn init(map: danos.MemoryMap) void {
|
||||
}
|
||||
}
|
||||
const bitmap_base = storage orelse @panic("pmm: no region large enough for the frame bitmap");
|
||||
bitmap = @as([*]u8, @ptrFromInt(danos.physicalToVirtual(bitmap_base)))[0..bitmap_bytes];
|
||||
bitmap = @as([*]u8, @ptrFromInt(boot_handoff.physicalToVirtual(bitmap_base)))[0..bitmap_bytes];
|
||||
|
||||
// 3. Start with everything marked used, then free the usable regions. Doing
|
||||
// it this way means every gap, reserved span and MMIO hole is unallocatable
|
||||
|
||||
+65
-22
@@ -3,7 +3,7 @@
|
||||
//! loader; in-kernel code is linked into the kernel image, not loaded here.
|
||||
//!
|
||||
//! Two entry points:
|
||||
//! - `spawnProcess` loads a user ELF (`/sbin/init`, and later servers/drivers)
|
||||
//! - `spawnProcess` loads a user ELF (`/system/services/init`, and later servers/drivers)
|
||||
//! into a fresh address space and schedules it as a real preemptive ring-3
|
||||
//! process on its own page tables. This is the production path.
|
||||
//! - `run` executes a raw code blob (the user-pf isolation test program) on the
|
||||
@@ -21,18 +21,21 @@
|
||||
|
||||
const std = @import("std");
|
||||
const elf = std.elf;
|
||||
const danos = @import("danos");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const abi = @import("abi");
|
||||
const device_abi = @import("device-abi");
|
||||
const architecture = @import("architecture");
|
||||
const pmm = @import("pmm.zig");
|
||||
const scheduler = @import("scheduler.zig");
|
||||
const sync = @import("sync.zig");
|
||||
const ipc = @import("ipc-synchronous.zig");
|
||||
const device_service = @import("device-service.zig");
|
||||
const devices_broker = @import("devices-broker.zig");
|
||||
const irq = @import("irq.zig");
|
||||
const initial_ramdisk = @import("initial-ramdisk");
|
||||
const log = @import("log.zig");
|
||||
|
||||
const page_size = danos.page_size;
|
||||
const SystemCall = danos.SystemCall;
|
||||
const page_size = abi.page_size;
|
||||
const SystemCall = abi.SystemCall;
|
||||
|
||||
/// User virtual addresses. PML4 index 224 — a user-exclusive region, far from
|
||||
/// the identity map (low indices) and the vmm test address (index 128), so
|
||||
@@ -79,7 +82,18 @@ pub var write_from_user: bool = false;
|
||||
pub var write_count: u64 = 0; // total write syscalls served (for the heartbeat tests)
|
||||
pub var exit_code: u64 = 0;
|
||||
|
||||
/// The system_call surface, dispatched on the saved system_call number (`danos.SystemCall`).
|
||||
/// The initial-ramdisk image, recorded at boot so `system_spawn` can find bundled
|
||||
/// binaries by name. Null until `setInitialRamdisk` runs; `system_spawn` then fails
|
||||
/// cleanly rather than reaching into unset memory.
|
||||
var ramdisk_image: ?[]const u8 = null;
|
||||
|
||||
/// Record the initial-ramdisk image (the kernel already holds it from the boot
|
||||
/// handoff) so a user-space supervisor can `system_spawn` binaries out of it.
|
||||
pub fn setInitialRamdisk(image: []const u8) void {
|
||||
ramdisk_image = image;
|
||||
}
|
||||
|
||||
/// The system_call surface, dispatched on the saved system_call number (`abi.SystemCall`).
|
||||
/// This is the microkernel-minimal set — memory + scheduling only; file/device
|
||||
/// I/O will arrive as IPC to user-space servers (docs/syscall.md). The result is
|
||||
/// written back into the trap frame, since the entry paths restore user registers
|
||||
@@ -135,6 +149,7 @@ fn system_call(state: *architecture.CpuState) void {
|
||||
.irq_bind => systemIrqBind(state),
|
||||
.irq_ack => systemIrqAck(state),
|
||||
.device_register => systemDeviceRegister(state),
|
||||
.system_spawn => systemSpawn(state),
|
||||
_ => fail(state),
|
||||
}
|
||||
}
|
||||
@@ -203,15 +218,15 @@ fn systemDeviceEnumerate(state: *architecture.CpuState) void {
|
||||
const maximum = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0 or buffer_ptr >= user_half_end) return fail(state);
|
||||
const sz = @sizeOf(danos.DeviceDescriptor);
|
||||
const sz = @sizeOf(device_abi.DeviceDescriptor);
|
||||
const cap = @min(maximum, (user_half_end - buffer_ptr) / sz); // clamp to the user half
|
||||
const out: [*]danos.DeviceDescriptor = @ptrFromInt(buffer_ptr);
|
||||
architecture.setSystemCallResult(state, device_service.enumerate(out[0..@intCast(cap)]));
|
||||
const out: [*]device_abi.DeviceDescriptor = @ptrFromInt(buffer_ptr);
|
||||
architecture.setSystemCallResult(state, devices_broker.enumerate(out[0..@intCast(cap)]));
|
||||
}
|
||||
|
||||
/// device_claim(id) -> 0/-1: take exclusive ownership of a device for this process.
|
||||
fn systemDeviceClaim(state: *architecture.CpuState) void {
|
||||
if (device_service.claim(architecture.systemCallArg(state, 0), scheduler.current().id))
|
||||
if (devices_broker.claim(architecture.systemCallArg(state, 0), scheduler.current().id))
|
||||
architecture.setSystemCallResult(state, 0)
|
||||
else
|
||||
fail(state);
|
||||
@@ -225,10 +240,10 @@ fn systemMmioMap(state: *architecture.CpuState) void {
|
||||
const resource_index = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
const owner = device_service.ownerOf(device_id) orelse return fail(state);
|
||||
const owner = devices_broker.ownerOf(device_id) orelse return fail(state);
|
||||
if (owner != t.id) return fail(state); // not claimed by this process
|
||||
const r = device_service.resourceOf(device_id, resource_index) orelse return fail(state);
|
||||
if (r.kind != @intFromEnum(danos.ResourceKind.memory)) return fail(state);
|
||||
const r = devices_broker.resourceOf(device_id, resource_index) orelse return fail(state);
|
||||
if (r.kind != @intFromEnum(device_abi.ResourceKind.memory)) return fail(state);
|
||||
|
||||
if (t.device_map_next == 0) t.device_map_next = device_arena_base;
|
||||
const first = r.start & ~@as(u64, page_size - 1);
|
||||
@@ -260,13 +275,41 @@ fn systemDeviceRegister(state: *architecture.CpuState) void {
|
||||
const t = scheduler.current();
|
||||
if (t.aspace == 0) return fail(state);
|
||||
|
||||
var descriptor: danos.DeviceDescriptor = undefined;
|
||||
var descriptor: device_abi.DeviceDescriptor = undefined;
|
||||
if (!ipc.copyFromUser(t.aspace, descriptor_ptr, std.mem.asBytes(&descriptor))) return fail(state);
|
||||
|
||||
const id = device_service.register(parent_id, t.id, &descriptor) catch return fail(state);
|
||||
const id = devices_broker.register(parent_id, t.id, &descriptor) catch return fail(state);
|
||||
architecture.setSystemCallResult(state, id);
|
||||
}
|
||||
|
||||
/// system_spawn(name_ptr, name_len) -> 0 on success, -1 on failure. Load the binary
|
||||
/// bundled in the initial-ramdisk under `name` as a fresh ring-3 process. This is the
|
||||
/// mechanism a user-space supervisor (the device manager) uses to start a driver it
|
||||
/// matched: discovery and policy stay in user space, the kernel only spawns.
|
||||
///
|
||||
/// Ungated for now — any process may spawn any bundled binary. A capability (only a
|
||||
/// supervisor holds the right to spawn) belongs here once the model grows one; see
|
||||
/// docs/driver-model.md. The name is bounds-checked into the user half exactly like
|
||||
/// `debug_write`, and an unknown name or a load failure returns -1.
|
||||
fn systemSpawn(state: *architecture.CpuState) void {
|
||||
const ptr = architecture.systemCallArg(state, 0);
|
||||
const len = architecture.systemCallArg(state, 1);
|
||||
if (len == 0 or len > 64 or ptr >= user_half_end or ptr + len > user_half_end) return fail(state);
|
||||
const image = ramdisk_image orelse return fail(state);
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse return fail(state);
|
||||
|
||||
const name = @as([*]const u8, @ptrFromInt(ptr))[0..len];
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!std.mem.eql(u8, item.name, name)) continue;
|
||||
spawnProcess(item.blob, 4) catch return fail(state);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
return;
|
||||
}
|
||||
fail(state); // no bundled binary by that name
|
||||
}
|
||||
|
||||
/// Drop every IRQ binding `t` made. Called on exit, before the handle table is closed
|
||||
/// (which is what frees the endpoints an ISR would otherwise notify into).
|
||||
fn releaseIrqs(t: *scheduler.Task) void {
|
||||
@@ -281,10 +324,10 @@ fn releaseIrqs(t: *scheduler.Task) void {
|
||||
/// by discovery. Neither a raw GSI nor an unclaimed device can get through — which
|
||||
/// is why irq_bind takes a resource index and not an interrupt number.
|
||||
fn ownedGsi(t: *scheduler.Task, device_id: u64, resource_index: u64) ?u32 {
|
||||
const owner = device_service.ownerOf(device_id) orelse return null;
|
||||
const owner = devices_broker.ownerOf(device_id) orelse return null;
|
||||
if (owner != t.id) return null;
|
||||
const r = device_service.resourceOf(device_id, resource_index) orelse return null;
|
||||
if (r.kind != @intFromEnum(danos.ResourceKind.irq)) return null;
|
||||
const r = devices_broker.resourceOf(device_id, resource_index) orelse return null;
|
||||
if (r.kind != @intFromEnum(device_abi.ResourceKind.irq)) return null;
|
||||
if (r.start >= irq.maximum_gsi) return null;
|
||||
return @intCast(r.start);
|
||||
}
|
||||
@@ -373,7 +416,7 @@ fn systemMmap(state: *architecture.CpuState) void {
|
||||
}
|
||||
|
||||
for (frames[0..pages], 0..) |frame, i| {
|
||||
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(frame));
|
||||
const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(frame));
|
||||
@memset(destination[0..page_size], 0); // hand out zeroed memory
|
||||
architecture.mapUserPageInto(t.aspace, base + i * page_size, frame, true, false); // RW + NX
|
||||
}
|
||||
@@ -428,7 +471,7 @@ pub fn run(blob: []const u8) RunError!void {
|
||||
// Fill the code frame through the physmap (supervisor RW): the user-facing
|
||||
// mapping is read-only, and this also sidesteps CR0.WP/SMAP. The tail is
|
||||
// padded with int3 so a stray jump traps instead of sliding.
|
||||
const code: [*]u8 = @ptrFromInt(danos.physicalToVirtual(code_frame));
|
||||
const code: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(code_frame));
|
||||
@memcpy(code[0..blob.len], blob);
|
||||
@memset(code[blob.len..page_size], 0xCC);
|
||||
|
||||
@@ -446,7 +489,7 @@ pub fn run(blob: []const u8) RunError!void {
|
||||
pmm.free(stack_frame);
|
||||
}
|
||||
|
||||
// --- user ELF loading (/sbin/init) ------------------------------------------
|
||||
// --- user ELF loading (/system/services/init) ------------------------------------------
|
||||
|
||||
pub const InitError = error{
|
||||
BadElf, // malformed/inapplicable image (magic, class, machine, type, bounds)
|
||||
@@ -542,7 +585,7 @@ fn parseSegments(image: []const u8, segs: *[maximum_segments]Segment) InitError!
|
||||
/// frame mapped into it — so no per-page rollback list is needed here.
|
||||
fn loadPageInto(aspace: u64, image: []const u8, seg: Segment, page_index: u64) InitError!void {
|
||||
const frame = pmm.alloc() orelse return error.OutOfMemory;
|
||||
const destination: [*]u8 = @ptrFromInt(danos.physicalToVirtual(frame));
|
||||
const destination: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(frame));
|
||||
@memset(destination[0..page_size], 0);
|
||||
const page_off = page_index * page_size;
|
||||
if (page_off < seg.filesz) {
|
||||
|
||||
+132
-80
@@ -10,9 +10,11 @@
|
||||
//! exception report the handler prints (which also reaches serial).
|
||||
|
||||
const std = @import("std");
|
||||
const danos = @import("danos");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const abi = @import("abi");
|
||||
const device_abi = @import("device-abi");
|
||||
const architecture = @import("architecture");
|
||||
const device_service = @import("device-service.zig");
|
||||
const devices_broker = @import("devices-broker.zig");
|
||||
const platform = @import("platform");
|
||||
const pmm = @import("pmm.zig");
|
||||
const heap = @import("heap.zig");
|
||||
@@ -22,7 +24,7 @@ const ipcsync = @import("ipc-synchronous.zig");
|
||||
const irq = @import("irq.zig");
|
||||
const sync = @import("sync.zig");
|
||||
const process = @import("process.zig");
|
||||
const initrd = @import("initrd");
|
||||
const initial_ramdisk = @import("initial-ramdisk");
|
||||
|
||||
/// Formatted write straight to serial, independent of the framebuffer console.
|
||||
fn log(comptime fmt: []const u8, args: anytype) void {
|
||||
@@ -108,8 +110,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
initTest(boot_information);
|
||||
} else if (eql(case, "process")) {
|
||||
processTest(boot_information);
|
||||
} else if (eql(case, "initrd")) {
|
||||
initrdTest(boot_information);
|
||||
} else if (eql(case, "initial-ramdisk")) {
|
||||
initialRamdiskTest(boot_information);
|
||||
} else if (eql(case, "vfs")) {
|
||||
vfsTest(boot_information);
|
||||
} else if (eql(case, "hpet")) {
|
||||
@@ -120,6 +122,8 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
irqFreeTest();
|
||||
} else if (eql(case, "bus")) {
|
||||
busTest(boot_information);
|
||||
} else if (eql(case, "device-manager")) {
|
||||
deviceManagerTest(boot_information);
|
||||
} else if (eql(case, "poweroff")) {
|
||||
powerTest(.off);
|
||||
} else if (eql(case, "reboot")) {
|
||||
@@ -153,7 +157,7 @@ fn powerTest(comptime action: enum { off, reboot }) void {
|
||||
result();
|
||||
}
|
||||
|
||||
const BootInformation = danos.BootInformation;
|
||||
const BootInformation = boot_handoff.BootInformation;
|
||||
|
||||
fn eql(a: []const u8, b: []const u8) bool {
|
||||
return std.mem.eql(u8, a, b);
|
||||
@@ -165,7 +169,7 @@ fn smoke(boot_information: *const BootInformation) void {
|
||||
|
||||
// The memory map has some usable RAM.
|
||||
const mm = boot_information.memory_map;
|
||||
const regions = @as([*]const danos.MemoryRegion, @ptrFromInt(danos.physicalToVirtual(mm.regions)))[0..mm.len];
|
||||
const regions = @as([*]const boot_handoff.MemoryRegion, @ptrFromInt(boot_handoff.physicalToVirtual(mm.regions)))[0..mm.len];
|
||||
var usable: u64 = 0;
|
||||
for (regions) |r| {
|
||||
if (r.kind == .usable) usable += r.pages;
|
||||
@@ -177,7 +181,7 @@ fn smoke(boot_information: *const BootInformation) void {
|
||||
const b = pmm.alloc();
|
||||
check("alloc returns a frame", a != null);
|
||||
check("alloc returns distinct frames", a != null and b != null and a.? != b.?);
|
||||
check("frames are page-aligned", (a orelse 1) % danos.page_size == 0);
|
||||
check("frames are page-aligned", (a orelse 1) % abi.page_size == 0);
|
||||
|
||||
// Freeing restores the count.
|
||||
const before = pmm.stats().free_frames;
|
||||
@@ -187,7 +191,7 @@ fn smoke(boot_information: *const BootInformation) void {
|
||||
|
||||
// Paging is active on our own tables (the root is non-zero and page-aligned).
|
||||
const root = architecture.activePageTable();
|
||||
check("paging active (page-table root set)", root != 0 and root % danos.page_size == 0);
|
||||
check("paging active (page-table root set)", root != 0 and root % abi.page_size == 0);
|
||||
|
||||
result();
|
||||
}
|
||||
@@ -573,7 +577,7 @@ fn smpTest() void {
|
||||
const tramp = architecture.trampolinePage();
|
||||
check("trampoline frame reserved", tramp != 0);
|
||||
if (tramp != 0) {
|
||||
const bytes: [*]const u8 = @ptrFromInt(danos.physicalToVirtual(tramp));
|
||||
const bytes: [*]const u8 = @ptrFromInt(boot_handoff.physicalToVirtual(tramp));
|
||||
var zeroed = true;
|
||||
for (0..4096) |b| {
|
||||
if (bytes[b] != 0) zeroed = false;
|
||||
@@ -757,7 +761,7 @@ fn userMemTest() void {
|
||||
var mapped: usize = 0;
|
||||
while (mapped < npages) : (mapped += 1) {
|
||||
frames[mapped] = pmm.alloc() orelse break;
|
||||
architecture.mapUserPageInto(aspace, arena + mapped * danos.page_size, frames[mapped], true, false);
|
||||
architecture.mapUserPageInto(aspace, arena + mapped * abi.page_size, frames[mapped], true, false);
|
||||
}
|
||||
check("granted three user pages", mapped == npages);
|
||||
|
||||
@@ -765,13 +769,13 @@ fn userMemTest() void {
|
||||
var translate_ok = true;
|
||||
var rw_ok = true;
|
||||
for (0..npages) |i| {
|
||||
const va = arena + i * danos.page_size;
|
||||
const va = arena + i * abi.page_size;
|
||||
const physical = architecture.translate(aspace, va) orelse {
|
||||
translate_ok = false;
|
||||
continue;
|
||||
};
|
||||
if (physical != frames[i]) translate_ok = false;
|
||||
const p: [*]u8 = @ptrFromInt(danos.physicalToVirtual(physical));
|
||||
const p: [*]u8 = @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
||||
p[0] = 0xA5;
|
||||
if (p[0] != 0xA5) rw_ok = false;
|
||||
}
|
||||
@@ -780,7 +784,7 @@ fn userMemTest() void {
|
||||
|
||||
// Release them the way munmap does, then tear down the address space.
|
||||
for (0..npages) |i| {
|
||||
const va = arena + i * danos.page_size;
|
||||
const va = arena + i * abi.page_size;
|
||||
if (architecture.translate(aspace, va)) |physical| {
|
||||
architecture.unmapUserPageInto(aspace, va);
|
||||
pmm.free(physical);
|
||||
@@ -864,7 +868,7 @@ fn procWorker() void {
|
||||
scheduler.exit();
|
||||
}
|
||||
|
||||
/// Real processes: load /sbin/init as TWO scheduled ring-3 processes, each with
|
||||
/// Real processes: load /system/services/init as TWO scheduled ring-3 processes, each with
|
||||
/// its own address space at the same virtual addresses, running concurrently
|
||||
/// with a kernel task. Both must make heartbeat syscalls from CPL 3 — which can
|
||||
/// only happen if each runs on its own page tables (CR3 switched correctly per
|
||||
@@ -872,12 +876,12 @@ fn procWorker() void {
|
||||
/// strongest cheap proof of address-space isolation.
|
||||
fn processTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: process\n", .{});
|
||||
check("bootloader handed over sbin/init", boot_information.init_len != 0);
|
||||
check("bootloader handed over /system/services/init", boot_information.init_len != 0);
|
||||
if (boot_information.init_len == 0) {
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
||||
|
||||
process.write_count = 0;
|
||||
process.write_from_user = false;
|
||||
@@ -916,19 +920,19 @@ fn userPfTest() void {
|
||||
log("DANOS-TEST-RESULT: FAIL (user read of kernel memory did not fault)\n", .{});
|
||||
}
|
||||
|
||||
/// The full PID-1 path: the bootloader read sbin/init off the boot volume and
|
||||
/// The full PID-1 path: the bootloader read /system/services/init off the boot volume and
|
||||
/// handed it over; load it as a user ELF and spawn it as a real ring-3 process
|
||||
/// — the same call the normal boot path makes — then confirm it beats. init
|
||||
/// heartbeats forever, so this proves it reaches ring 3, makes repeated syscalls
|
||||
/// (write + sleep), and stays alive rather than exiting.
|
||||
fn initTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: init\n", .{});
|
||||
check("bootloader handed over sbin/init", boot_information.init_len != 0);
|
||||
check("bootloader handed over /system/services/init", boot_information.init_len != 0);
|
||||
if (boot_information.init_len == 0) {
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.init_base)))[0..boot_information.init_len];
|
||||
process.write_count = 0;
|
||||
const spawned = if (process.spawnProcess(image, 4)) true else |err| blk: {
|
||||
log("DANOS-INIT-ERR: {s}\n", .{@errorName(err)});
|
||||
@@ -952,24 +956,24 @@ fn initTest(boot_information: *const BootInformation) void {
|
||||
result();
|
||||
}
|
||||
|
||||
/// The initrd path: the bootloader handed over an image bundling extra user
|
||||
/// The initial_ramdisk path: the bootloader handed over an image bundling extra user
|
||||
/// binaries; parse it, spawn every program, and confirm one (the vfs stub)
|
||||
/// reaches ring 3 and heartbeats — proving the whole ferry-parse-spawn pipeline.
|
||||
fn initrdTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: initrd\n", .{});
|
||||
check("bootloader handed over an initrd", boot_information.initrd_len != 0);
|
||||
if (boot_information.initrd_len == 0) {
|
||||
fn initialRamdiskTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: initial_ramdisk\n", .{});
|
||||
check("bootloader handed over an initial_ramdisk", boot_information.initial_ramdisk_len != 0);
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.initrd_base)))[0..boot_information.initrd_len];
|
||||
const rd = initrd.Reader.init(image) orelse {
|
||||
check("initrd image is valid", false);
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
check("initrd image is valid", true);
|
||||
check("initrd contains at least one binary", rd.count >= 1);
|
||||
check("initial_ramdisk image is valid", true);
|
||||
check("initial_ramdisk contains at least one binary", rd.count >= 1);
|
||||
|
||||
process.write_count = 0;
|
||||
process.write_from_user = false;
|
||||
@@ -981,7 +985,7 @@ fn initrdTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-INITRD-ERR: {s}: {s}\n", .{ item.name, @errorName(err) });
|
||||
}
|
||||
}
|
||||
check("every initrd binary spawned", spawned == rd.count);
|
||||
check("every initial_ramdisk binary spawned", spawned == rd.count);
|
||||
|
||||
// Wait for the spawned programs to run and make syscalls (they write + sleep).
|
||||
scheduler.setPriority(1);
|
||||
@@ -989,33 +993,33 @@ fn initrdTest(boot_information: *const BootInformation) void {
|
||||
while (process.write_count < 2 and architecture.millis() < deadline) scheduler.yield();
|
||||
scheduler.setPriority(4);
|
||||
|
||||
check("initrd processes ran and made syscalls (>=2)", process.write_count >= 2);
|
||||
check("initial_ramdisk processes ran and made syscalls (>=2)", process.write_count >= 2);
|
||||
check("syscalls came from user mode (CPL 3)", process.write_from_user);
|
||||
result();
|
||||
}
|
||||
|
||||
/// The full VFS path: spawn the user-space VFS server and a client from the
|
||||
/// initrd. The client opens a file through the runtime file API, writes, seeks, reads
|
||||
/// initial_ramdisk. The client opens a file through the runtime file API, writes, seeks, reads
|
||||
/// it back, and — only if the round trip matched — heartbeats "vfstest: ok". So
|
||||
/// seeing that marker proves client open/write/read reached the server over IPC
|
||||
/// and came back correct. (The client retries until the server registers.)
|
||||
fn vfsTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: vfs\n", .{});
|
||||
if (boot_information.initrd_len == 0) {
|
||||
check("bootloader handed over an initrd", false);
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.initrd_base)))[0..boot_information.initrd_len];
|
||||
const rd = initrd.Reader.init(image) orelse {
|
||||
check("initrd image is valid", false);
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
process.write_count = 0;
|
||||
process.write_from_user = false;
|
||||
// Spawn just the server and its client (other initrd binaries would write to
|
||||
// Spawn just the server and its client (other initial_ramdisk binaries would write to
|
||||
// the shared evidence buffer and confuse the marker check).
|
||||
_ = spawnNamed(rd, "vfs");
|
||||
_ = spawnNamed(rd, "vfs-test");
|
||||
@@ -1037,9 +1041,9 @@ fn vfsTest(boot_information: *const BootInformation) void {
|
||||
result();
|
||||
}
|
||||
|
||||
/// Spawn the initrd binary named `name` as a ring-3 process. Returns false if it
|
||||
/// Spawn the initial_ramdisk binary named `name` as a ring-3 process. Returns false if it
|
||||
/// isn't in the image or fails to load.
|
||||
fn spawnNamed(rd: initrd.Reader, name: []const u8) bool {
|
||||
fn spawnNamed(rd: initial_ramdisk.Reader, name: []const u8) bool {
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
@@ -1051,36 +1055,36 @@ fn spawnNamed(rd: initrd.Reader, name: []const u8) bool {
|
||||
}
|
||||
|
||||
/// IO passthrough + IRQ-as-IPC: a user-space driver drives real hardware and is
|
||||
/// *woken by it*. Spawn hpetd, which claims the HPET, maps its registers into its
|
||||
/// *woken by it*. Spawn hpet, which claims the HPET, maps its registers into its
|
||||
/// own ring-3 address space, arms a level-triggered comparator, binds the interrupt
|
||||
/// to an IPC endpoint, and then blocks. It prints "hpetd: ok" only after being woken
|
||||
/// to an IPC endpoint, and then blocks. It prints "hpet: ok" only after being woken
|
||||
/// `target_ticks` times — it cannot reach that line by polling, because the loop's
|
||||
/// only exit is through `replyWait` returning a notification badge.
|
||||
///
|
||||
/// The interesting assertion is the last one, which doesn't trust hpetd at all: it
|
||||
/// The interesting assertion is the last one, which doesn't trust hpet at all: it
|
||||
/// reads the I/O APIC's redirection entry back and checks the kernel really routed
|
||||
/// the line (our vector, level-triggered) and really left it unmasked after the
|
||||
/// driver's final `irq_ack`. hpetd disables its comparator on the last interrupt, so
|
||||
/// driver's final `irq_ack`. hpet disables its comparator on the last interrupt, so
|
||||
/// that state is quiescent and not a race.
|
||||
fn hpetTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: hpet\n", .{});
|
||||
if (boot_information.initrd_len == 0) {
|
||||
check("bootloader handed over an initrd", false);
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.initrd_base)))[0..boot_information.initrd_len];
|
||||
const rd = initrd.Reader.init(image) orelse {
|
||||
check("initrd image is valid", false);
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
process.write_count = 0;
|
||||
process.write_from_user = false;
|
||||
check("hpetd spawned from the initrd", spawnNamed(rd, "hpetd"));
|
||||
check("hpet spawned from the initial_ramdisk", spawnNamed(rd, "hpet"));
|
||||
|
||||
const prefix = "hpetd: ok";
|
||||
const prefix = "hpet: ok";
|
||||
scheduler.setPriority(1);
|
||||
const deadline = architecture.millis() + 10000;
|
||||
while (architecture.millis() < deadline) {
|
||||
@@ -1113,14 +1117,14 @@ fn hpetRouteOk() bool {
|
||||
|
||||
/// The GSI discovery recorded for the HPET, from the same device table the driver saw.
|
||||
fn hpetGsi() ?u32 {
|
||||
var buffer: [16]danos.DeviceDescriptor = undefined;
|
||||
const n = @min(device_service.enumerate(&buffer), buffer.len);
|
||||
var buffer: [16]device_abi.DeviceDescriptor = undefined;
|
||||
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
||||
for (buffer[0..n]) |d| {
|
||||
if (d.class != @intFromEnum(danos.DeviceClass.timer)) continue;
|
||||
if (d.parent != danos.no_parent) continue; // the block, not a comparator child
|
||||
if (d.class != @intFromEnum(device_abi.DeviceClass.timer)) continue;
|
||||
if (d.parent != device_abi.no_parent) continue; // the block, not a comparator child
|
||||
for (0..d.resource_count) |j| {
|
||||
const r = d.resources[j];
|
||||
if (r.kind == @intFromEnum(danos.ResourceKind.irq)) return @intCast(r.start);
|
||||
if (r.kind == @intFromEnum(device_abi.ResourceKind.irq)) return @intCast(r.start);
|
||||
}
|
||||
}
|
||||
return null;
|
||||
@@ -1130,34 +1134,34 @@ fn hpetGsi() ?u32 {
|
||||
/// them from the hardware, and publishes each as a child via `device_register` — the
|
||||
/// primitive a PCI bridge or USB hub driver is built from.
|
||||
///
|
||||
/// `busd` treats the HPET's register block as a bus and its comparators as children,
|
||||
/// `bus` treats the HPET's register block as a bus and its comparators as children,
|
||||
/// giving each a 0x20 sub-window. It checks its own work (children come back from the
|
||||
/// table with the right parent and a strictly narrower window) and, importantly, that
|
||||
/// the kernel **refuses** a child whose window escapes the parent's — without that,
|
||||
/// `device_register` would be a system_call for mapping arbitrary physical memory. It prints
|
||||
/// "busd: ok" only if all of that holds.
|
||||
/// "bus: ok" only if all of that holds.
|
||||
///
|
||||
/// The kernel-side check here is the one busd can't make: that the children really did
|
||||
/// The kernel-side check here is the one bus can't make: that the children really did
|
||||
/// land in the device table with the containment invariant intact.
|
||||
fn busTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: bus\n", .{});
|
||||
if (boot_information.initrd_len == 0) {
|
||||
check("bootloader handed over an initrd", false);
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(danos.physicalToVirtual(boot_information.initrd_base)))[0..boot_information.initrd_len];
|
||||
const rd = initrd.Reader.init(image) orelse {
|
||||
check("initrd image is valid", false);
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
process.write_count = 0;
|
||||
process.write_from_user = false;
|
||||
check("busd spawned from the initrd", spawnNamed(rd, "busd"));
|
||||
check("bus spawned from the initial_ramdisk", spawnNamed(rd, "bus"));
|
||||
|
||||
const prefix = "busd: ok";
|
||||
const prefix = "bus: ok";
|
||||
scheduler.setPriority(1);
|
||||
const deadline = architecture.millis() + 10000;
|
||||
while (architecture.millis() < deadline) {
|
||||
@@ -1173,7 +1177,55 @@ fn busTest(boot_information: *const BootInformation) void {
|
||||
result();
|
||||
}
|
||||
|
||||
/// Every child `busd` registered must have each of its resources inside a parent
|
||||
/// The device manager (a ring-3 service) enumerates /system/devices, matches each
|
||||
/// device to a driver, and — eventually — spawns it. This increment only checks the
|
||||
/// discovery+matching half: it must find the HPET (a timer) and decide `hpet` serves
|
||||
/// it, printing "device-manager: ok". It uses no special privilege — the same
|
||||
/// `device_enumerate` any process could call. (Spawning is the next increment.)
|
||||
fn deviceManagerTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: device-manager\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
// Let `system_spawn` find bundled binaries by name (the normal boot path does
|
||||
// this too). Only the device-manager is spawned here — so if `hpet` runs at all,
|
||||
// it's because the manager discovered the timer, matched, and spawned it.
|
||||
process.setInitialRamdisk(image);
|
||||
|
||||
process.write_count = 0;
|
||||
process.write_from_user = false;
|
||||
check("device-manager spawned from the initial_ramdisk", spawnNamed(rd, "device-manager"));
|
||||
|
||||
// End-to-end proof: the driver the manager spawned reaches its own live marker.
|
||||
// `hpet: ok` is hpet's final, stable message (it claims the timer, maps its MMIO,
|
||||
// binds its IRQ, services one, then sleeps) — nothing overwrites the buffer after,
|
||||
// so it's race-free to poll for. Its arrival means the whole
|
||||
// discover -> match -> system_spawn -> driver-up chain worked.
|
||||
const prefix = "hpet: ok";
|
||||
scheduler.setPriority(1);
|
||||
const deadline = architecture.millis() + 10000;
|
||||
while (architecture.millis() < deadline) {
|
||||
if (process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix)) break;
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
|
||||
const ok = process.write_len >= prefix.len and eql(process.write_buffer[0..prefix.len], prefix);
|
||||
check("device manager matched the timer and system_spawn'd hpet, which came up", ok);
|
||||
check("its syscalls came from user mode (CPL 3)", process.write_from_user);
|
||||
result();
|
||||
}
|
||||
|
||||
/// Every child `bus` registered must have each of its resources inside a parent
|
||||
/// resource of the same kind — the invariant `device_register` exists to maintain,
|
||||
/// checked from the kernel's own table rather than the driver's word for it.
|
||||
///
|
||||
@@ -1181,8 +1233,8 @@ fn busTest(boot_information: *const BootInformation) void {
|
||||
/// trusted and doesn't obey containment: a PCI function's BAR is not inside its host
|
||||
/// bridge's `bus_range`, because a bus-number range isn't an address window.
|
||||
fn childrenContained() bool {
|
||||
var buffer: [64]danos.DeviceDescriptor = undefined;
|
||||
const n = @min(device_service.enumerate(&buffer), buffer.len);
|
||||
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
||||
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
||||
|
||||
const bus_id = hpetDeviceId() orelse return false;
|
||||
const p = buffer[@intCast(bus_id)];
|
||||
@@ -1197,7 +1249,7 @@ fn childrenContained() bool {
|
||||
for (0..p.resource_count) |j| {
|
||||
const pr = p.resources[j];
|
||||
if (pr.kind != r.kind) continue;
|
||||
if (r.kind == @intFromEnum(danos.ResourceKind.irq)) {
|
||||
if (r.kind == @intFromEnum(device_abi.ResourceKind.irq)) {
|
||||
if (pr.start == r.start) ok = true;
|
||||
} else if (r.len != 0 and r.start >= pr.start and
|
||||
r.start + r.len <= pr.start + pr.len) ok = true;
|
||||
@@ -1205,18 +1257,18 @@ fn childrenContained() bool {
|
||||
if (!ok) return false;
|
||||
}
|
||||
}
|
||||
return children > 0; // busd must have published at least one
|
||||
return children > 0; // bus must have published at least one
|
||||
}
|
||||
|
||||
/// Device id of the HPET (the bus busd claims), from the same table drivers see.
|
||||
/// Device id of the HPET (the bus bus claims), from the same table drivers see.
|
||||
fn hpetDeviceId() ?u64 {
|
||||
var buffer: [64]danos.DeviceDescriptor = undefined;
|
||||
const n = @min(device_service.enumerate(&buffer), buffer.len);
|
||||
var buffer: [64]device_abi.DeviceDescriptor = undefined;
|
||||
const n = @min(devices_broker.enumerate(&buffer), buffer.len);
|
||||
for (buffer[0..n]) |d| {
|
||||
if (d.class != @intFromEnum(danos.DeviceClass.timer)) continue;
|
||||
if (d.parent != danos.no_parent) continue; // a comparator child, not the block
|
||||
if (d.class != @intFromEnum(device_abi.DeviceClass.timer)) continue;
|
||||
if (d.parent != device_abi.no_parent) continue; // a comparator child, not the block
|
||||
for (0..d.resource_count) |j| {
|
||||
if (d.resources[j].kind == @intFromEnum(danos.ResourceKind.memory)) return d.id;
|
||||
if (d.resources[j].kind == @intFromEnum(device_abi.ResourceKind.memory)) return d.id;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
@@ -1226,7 +1278,7 @@ fn hpetDeviceId() ?u64 {
|
||||
/// (so a dead driver's device goes quiet instead of storming) and the slot cleared
|
||||
/// (so an ISR never posts a notification into the endpoint that is about to be freed).
|
||||
///
|
||||
/// This is the path `hpetd` never takes — it runs forever — so it gets its own test.
|
||||
/// This is the path `hpet` never takes — it runs forever — so it gets its own test.
|
||||
/// Two properties, both read back from the hardware rather than from our own state:
|
||||
///
|
||||
/// 1. A bound GSI is routed and unmasked.
|
||||
@@ -1310,7 +1362,7 @@ fn ioPassTest() void {
|
||||
return;
|
||||
};
|
||||
// Map it the way mmio_map does (device grant), then tear the space down.
|
||||
architecture.mapUserDeviceInto(aspace, process.device_arena_base, frame, danos.page_size);
|
||||
architecture.mapUserDeviceInto(aspace, process.device_arena_base, frame, abi.page_size);
|
||||
architecture.destroyAddressSpace(aspace);
|
||||
|
||||
// The page tables were reclaimed; the device-granted frame must not have been.
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
//! /system/services/device-manager — the ring-3 process that turns the device
|
||||
//! tree into a running system. The kernel enumerates the hardware and enforces the
|
||||
//! claim capability (mechanism); this decides *which driver serves which device*
|
||||
//! and, eventually, spawns it (policy). Keeping that split in user space is the
|
||||
//! whole point of the microkernel: the manager is an ordinary, restartable process
|
||||
//! with no special privilege — it uses the same `device_*` system calls any process
|
||||
//! could ([drivers.md](../../../docs/drivers.md), [driver-model.md]).
|
||||
//!
|
||||
//! Increment 2 (this file): enumerate /system/devices, *match* each device to a
|
||||
//! driver, and *spawn* it with `system_spawn` — the kernel loads the named binary
|
||||
//! from the initial-ramdisk as a fresh ring-3 process. On QEMU this discovers the
|
||||
//! HPET, decides `hpet` serves it, and brings that driver all the way up. (The
|
||||
//! kernel still auto-spawns the whole initial-ramdisk at boot; increment 3 removes
|
||||
//! that redundancy so the manager is the sole owner of driver spawning.)
|
||||
|
||||
const runtime = @import("runtime");
|
||||
const device = runtime.device;
|
||||
|
||||
/// The driver that serves each device class — the policy table. In a fuller system
|
||||
/// this comes from the drivers describing what they bind (or a manifest under
|
||||
/// /system/drivers); for now it is a small static map, which is enough to prove the
|
||||
/// manager reads the tree and decides. `null` = no driver for this class yet.
|
||||
fn driverFor(class: u64) ?[]const u8 {
|
||||
if (class == @intFromEnum(device.DeviceClass.timer)) return "hpet"; // the HPET
|
||||
return null;
|
||||
}
|
||||
|
||||
pub fn main() void {
|
||||
// Enumerate into a heap buffer (too big for the one-page user stack).
|
||||
const buffer = runtime.allocator().alloc(device.DeviceDescriptor, 64) catch {
|
||||
_ = runtime.system.write("device-manager: out of memory\n");
|
||||
return;
|
||||
};
|
||||
const total = device.enumerate(buffer);
|
||||
const count = @min(total, buffer.len);
|
||||
|
||||
var matched: usize = 0;
|
||||
for (buffer[0..count]) |descriptor| {
|
||||
const driver_name = driverFor(descriptor.class) orelse continue;
|
||||
matched += 1;
|
||||
if (runtime.system.spawn(driver_name)) {
|
||||
_ = runtime.system.write("device-manager: spawned ");
|
||||
_ = runtime.system.write(driver_name);
|
||||
_ = runtime.system.write("\n");
|
||||
} else {
|
||||
_ = runtime.system.write("device-manager: failed to spawn ");
|
||||
_ = runtime.system.write(driver_name);
|
||||
_ = runtime.system.write("\n");
|
||||
}
|
||||
}
|
||||
|
||||
if (matched == 0) {
|
||||
_ = runtime.system.write("device-manager: no matchable devices\n");
|
||||
return;
|
||||
}
|
||||
_ = runtime.system.write("device-manager: ok\n");
|
||||
while (true) runtime.system.sleep(1000);
|
||||
}
|
||||
|
||||
pub const panic = runtime.panic;
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
@@ -1,17 +1,24 @@
|
||||
//! /sbin/init — the first user-space program, PID 1. Built as its own
|
||||
//! freestanding binary (see build.zig), shipped on the boot volume at sbin/init,
|
||||
//! /system/services/init — the first user-space program, PID 1. Built as its own
|
||||
//! freestanding binary (see build.zig), shipped on the boot volume at /system/services/init,
|
||||
//! loaded by the bootloader, and started in ring 3 as a scheduled process by the
|
||||
//! kernel (system/kernel/process.zig). It links against the shared user runtime
|
||||
//! library `runtime` and talks to the kernel only through `runtime`'s system_call wrappers.
|
||||
//!
|
||||
//! Today it proves the C-convention heap works, then settles into a heartbeat:
|
||||
//! it prints a line and sleeps, forever — enough to show the system reaches user
|
||||
//! space and stays alive with a real process scheduled alongside the kernel's
|
||||
//! idle loop. It grows into the real init (service supervision) once there are
|
||||
//! other user programs to supervise.
|
||||
//! It proves the C-convention heap works, then — as PID 1 — acts as the system's
|
||||
//! **service supervisor**: it spawns the user-space services danos brings up at boot
|
||||
//! (the VFS server, the device manager), and settles into a heartbeat so it stays
|
||||
//! alive as the root of user space. Drivers are *not* its job: the device manager
|
||||
//! discovers the hardware and spawns those. This is the service half of the
|
||||
//! service/driver spawn split (docs/driver-model.md).
|
||||
|
||||
const runtime = @import("runtime");
|
||||
|
||||
/// The system services init brings up at boot, in order. This is init's policy — the
|
||||
/// microkernel keeps such choices in user space, not the kernel. Drivers are absent
|
||||
/// on purpose: the device manager owns those. (A future init reads this from a
|
||||
/// manifest under /system/services instead of a hardcoded list.)
|
||||
const boot_services = [_][]const u8{ "vfs", "device-manager" };
|
||||
|
||||
pub fn main() void {
|
||||
// Prove the heap end to end: allocate through the runtime allocator (which
|
||||
// mmaps pages from the kernel and carves them with the free list), write into
|
||||
@@ -27,6 +34,13 @@ pub fn main() void {
|
||||
gpa.free(buffer);
|
||||
} else |_| {}
|
||||
|
||||
// Bring up the boot services. Best-effort and silent: each service announces its
|
||||
// own readiness (`vfs: ready`, ...), and in an isolation test that runs init with
|
||||
// no initial-ramdisk the spawns simply no-op rather than deranging the heartbeat.
|
||||
for (boot_services) |service| {
|
||||
_ = runtime.system.spawn(service);
|
||||
}
|
||||
|
||||
while (true) {
|
||||
_ = runtime.system.write("init: heartbeat\n");
|
||||
runtime.system.sleep(1000);
|
||||
|
||||
@@ -1,18 +1,22 @@
|
||||
//! The VFS wire protocol — the message format spoken between a client (via the
|
||||
//! `runtime` file API) and the user-space VFS server over IPC. A request is a fixed
|
||||
//! `Request` header followed by an inline payload (a path, or write bytes); a
|
||||
//! reply is a fixed `Reply` header followed by an inline payload (read bytes, or
|
||||
//! a Stat). Everything fits in one IPC message (<= ipc MESSAGE_MAXIMUM = 256 bytes).
|
||||
//! The VFS wire protocol — the message format spoken between a client (via the file
|
||||
//! API) and the user-space VFS server over IPC. A request is a fixed `Request` header
|
||||
//! followed by an inline payload (a path, or write bytes); a reply is a fixed `Reply`
|
||||
//! header followed by an inline payload (read bytes, or a FileStatus). Everything fits
|
||||
//! in one IPC message (<= ipc MESSAGE_MAXIMUM = 256 bytes).
|
||||
//!
|
||||
//! This is a danos-native contract, so it uses danos names throughout — the POSIX
|
||||
//! spellings (`stat`, `O_CREAT`, ...) live only in the POSIX layer
|
||||
//! (library/posix/unistd.zig), which translates to these.
|
||||
//!
|
||||
//! This is user-space only — the kernel knows nothing of files or paths; it only
|
||||
//! moves the bytes. Shared by library/runtime/unistd.zig (client) and system/services/vfs/vfs.zig (server).
|
||||
//! moves the bytes. Shared by library/posix/unistd.zig (client) and system/services/vfs/vfs.zig (server).
|
||||
|
||||
pub const Operation = enum(u32) {
|
||||
open, // open(path) -> node id
|
||||
close, // close(node)
|
||||
read, // read(node, offset, len) -> bytes
|
||||
write, // write(node, offset, bytes) -> count
|
||||
stat, // stat(node) -> Stat
|
||||
status, // status(node) -> FileStatus
|
||||
};
|
||||
|
||||
/// Request header. `node` is the server-side open-file id (from a prior open);
|
||||
@@ -28,7 +32,7 @@ pub const Request = extern struct {
|
||||
|
||||
/// Reply header. `status` is 0 on success or a negative errno; `node` is the new
|
||||
/// open-file id (for `open`); `len` is the payload length (bytes read, or the
|
||||
/// Stat size).
|
||||
/// FileStatus size).
|
||||
pub const Reply = extern struct {
|
||||
status: i32,
|
||||
_padding: u32 = 0,
|
||||
@@ -37,7 +41,9 @@ pub const Reply = extern struct {
|
||||
_padding2: u32 = 0,
|
||||
};
|
||||
|
||||
pub const Stat = extern struct {
|
||||
/// A file's metadata (the danos-native answer to a `status` request). The POSIX
|
||||
/// layer maps this onto `struct stat`.
|
||||
pub const FileStatus = extern struct {
|
||||
size: u64,
|
||||
kind: u32,
|
||||
_padding: u32 = 0,
|
||||
@@ -49,5 +55,5 @@ pub const reply_size: usize = @sizeOf(Reply);
|
||||
/// Largest inline payload that still fits one IPC message alongside a header.
|
||||
pub const maximum_payload: usize = message_maximum - request_size;
|
||||
|
||||
/// Open flags.
|
||||
pub const O_CREAT: u32 = 1;
|
||||
/// Open flags (danos-native; the POSIX layer maps `O_CREAT` onto `create`).
|
||||
pub const create: u32 = 1;
|
||||
|
||||
@@ -1,13 +1,13 @@
|
||||
//! /sbin/vfstest — a client that proves the VFS round trip end to end: open a
|
||||
//! /system/services/vfs/vfs-test — a client that proves the VFS round trip end to end: open a
|
||||
//! file through the `runtime` file API, write to it, seek back, read it, and compare.
|
||||
//! On success it heartbeats "vfstest: ok" so the kernel test can observe it;
|
||||
//! on failure it reports what went wrong. Shipped in the initrd alongside vfs.
|
||||
//! on failure it reports what went wrong. Shipped in the initial_ramdisk alongside vfs.
|
||||
|
||||
const std = @import("std");
|
||||
const runtime = @import("runtime");
|
||||
|
||||
pub fn main() void {
|
||||
const u = runtime.unistd;
|
||||
const u = @import("posix").unistd;
|
||||
const payload = "hello-vfs";
|
||||
|
||||
// The VFS server may not have registered yet — retry open until it's up.
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
//! /sbin/vfs — the user-space VFS server. Shipped in the initrd, spawned as a
|
||||
//! system/services/vfs — the user-space VFS server. Shipped in the initial_ramdisk, spawned as a
|
||||
//! ring-3 process, and reached by every other process through IPC (the `runtime`
|
||||
//! file API marshals open/read/write/stat/close into calls to this server's
|
||||
//! endpoint, published under the well-known `vfs` service id).
|
||||
@@ -101,10 +101,10 @@ fn handle(message: []const u8, out: []u8) usize {
|
||||
if (off + n > nd.size) nd.size = off + n;
|
||||
return writeReply(out, .{ .status = 0, .len = @intCast(n) }, &.{});
|
||||
},
|
||||
.stat => {
|
||||
.status => {
|
||||
const of = openAt(request.node) orelse return fail(out);
|
||||
const st = protocol.Stat{ .size = nodes[of.node].size, .kind = 0 };
|
||||
return writeReply(out, .{ .status = 0, .len = @sizeOf(protocol.Stat) }, std.mem.asBytes(&st));
|
||||
const st = protocol.FileStatus{ .size = nodes[of.node].size, .kind = 0 };
|
||||
return writeReply(out, .{ .status = 0, .len = @sizeOf(protocol.FileStatus) }, std.mem.asBytes(&st));
|
||||
},
|
||||
.close => {
|
||||
if (request.node < opens.len) opens[@intCast(request.node)].used = false;
|
||||
|
||||
+27
-16
@@ -55,11 +55,14 @@ ARCHES = {
|
||||
"/opt/homebrew/share/qemu/edk2-i386-vars.fd", # macOS Homebrew (Apple Silicon)
|
||||
"/usr/local/share/qemu/edk2-i386-vars.fd", # macOS Homebrew (Intel)
|
||||
],
|
||||
"efi_app": ("EFI/BOOT/BOOTX64.efi", "BOOTX64.efi"), # (dest in ESP, name in zig-out/bin)
|
||||
"kernel": ("kernel", "kernel"),
|
||||
# Further files shipped on the ESP: the init user program and the initrd
|
||||
# (VFS server + drivers), both copied from zig-out/bin.
|
||||
"extra": [("sbin/init", "init"), ("initrd.img", "initrd.img")],
|
||||
# zig-out is a FHS-shaped image and the boot volume; the harness copies the
|
||||
# boot-critical files from their FHS paths into a fresh ESP with the same
|
||||
# layout. (dest in ESP, source path under zig-out) — identical here.
|
||||
"efi_app": ("EFI/BOOT/BOOTX64.efi", "EFI/BOOT/BOOTX64.efi"),
|
||||
"kernel": ("system/kernel", "system/kernel"),
|
||||
# The init user program and the initial-ramdisk (VFS server + drivers).
|
||||
"extra": [("system/services/init", "system/services/init"),
|
||||
("boot/initial-ramdisk.img", "boot/initial-ramdisk.img")],
|
||||
# Built as a function so we can splice in per-run paths.
|
||||
"qemu_args": lambda a, esp, vars_fd, serial: [
|
||||
"-machine", "q35", "-m", "128M",
|
||||
@@ -166,21 +169,21 @@ CASES = [
|
||||
{"name": "user-pf",
|
||||
"expect": r"page fault \(vector 14\)[\s\S]*error code : 0x5[\s\S]*IP\s*: 0x00007000000000",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# The real user binary: the bootloader ships sbin/init off the ESP, the
|
||||
# The real user binary: the bootloader ships /system/services/init off the ESP, the
|
||||
# kernel loads the ELF and runs it in ring 3, and it writes + exits cleanly.
|
||||
{"name": "init",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Real processes: /sbin/init loaded as a scheduled ring-3 process with its
|
||||
# Real processes: /system/services/init loaded as a scheduled ring-3 process with its
|
||||
# own address space, run twice (create/exit/teardown/recreate), coexisting
|
||||
# with a kernel task under preemption.
|
||||
{"name": "process",
|
||||
"smp": 4,
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# The initrd: the loader ferries a bundle of user binaries; the kernel parses
|
||||
# The initial_ramdisk: the loader ferries a bundle of user binaries; the kernel parses
|
||||
# it and spawns each as a ring-3 process (here the VFS-server stub heartbeats).
|
||||
{"name": "initrd",
|
||||
{"name": "initial-ramdisk",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# The user-space VFS: a client opens/writes/reads a file through the rt file
|
||||
@@ -194,6 +197,12 @@ CASES = [
|
||||
{"name": "hpet",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Device manager: a ring-3 service enumerates /system/devices and matches each
|
||||
# device to a driver (discovery + policy in user space). This increment logs the
|
||||
# decision; spawning follows.
|
||||
{"name": "device-manager",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# Bus driver: a user process claims a device, enumerates its children from the
|
||||
# hardware, and publishes each with dev_register — and the kernel refuses a child
|
||||
# whose window escapes the parent's (else dev_register maps arbitrary memory).
|
||||
@@ -202,7 +211,7 @@ CASES = [
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# IRQ teardown: an exiting driver's line is masked and its slot cleared (so no
|
||||
# ISR notifies a freed endpoint), and a sibling owner sharing that endpoint
|
||||
# keeps its own binding. The path hpetd never takes, since it runs forever.
|
||||
# keeps its own binding. The path hpet never takes, since it runs forever.
|
||||
{"name": "irqfree",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
@@ -237,14 +246,16 @@ def make_esp(arch):
|
||||
esp = os.path.join(WORK, "esp")
|
||||
if os.path.exists(esp):
|
||||
shutil.rmtree(esp)
|
||||
efi_dest, efi_name = arch["efi_app"]
|
||||
kern_dest, kern_name = arch["kernel"]
|
||||
efi_dest, efi_src = arch["efi_app"]
|
||||
kern_dest, kern_src = arch["kernel"]
|
||||
fhs = os.path.join(REPO, "zig-out") # zig-out is the FHS image
|
||||
os.makedirs(os.path.join(esp, os.path.dirname(efi_dest)), exist_ok=True)
|
||||
shutil.copy(os.path.join(REPO, "zig-out", "bin", efi_name), os.path.join(esp, efi_dest))
|
||||
shutil.copy(os.path.join(REPO, "zig-out", "bin", kern_name), os.path.join(esp, kern_dest))
|
||||
for dest, name in arch.get("extra", []):
|
||||
os.makedirs(os.path.join(esp, os.path.dirname(kern_dest)), exist_ok=True)
|
||||
shutil.copy(os.path.join(fhs, efi_src), os.path.join(esp, efi_dest))
|
||||
shutil.copy(os.path.join(fhs, kern_src), os.path.join(esp, kern_dest))
|
||||
for dest, src in arch.get("extra", []):
|
||||
os.makedirs(os.path.join(esp, os.path.dirname(dest)), exist_ok=True)
|
||||
shutil.copy(os.path.join(REPO, "zig-out", "bin", name), os.path.join(esp, dest))
|
||||
shutil.copy(os.path.join(fhs, src), os.path.join(esp, dest))
|
||||
return esp
|
||||
|
||||
|
||||
|
||||
@@ -1,10 +1,10 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Build-time initrd packer. Concatenates user binaries into one image the
|
||||
"""Build-time initial_ramdisk packer. Concatenates user binaries into one image the
|
||||
bootloader ferries to the kernel.
|
||||
|
||||
Usage: mkinitrd.py <out.img> [<name> <file>]...
|
||||
Usage: make-initial-ramdisk.py <out.img> [<name> <file>]...
|
||||
|
||||
Image layout (little-endian), mirroring src/user/proto/initrd.zig:
|
||||
Image layout (little-endian), mirroring src/user/proto/initial-ramdisk.zig:
|
||||
Header : magic u32 ("DNRD"=0x444E5244), count u32
|
||||
Entry*N : name [32]u8 (NUL-padded), offset u64, len u64
|
||||
blobs : each entry's file bytes at its offset
|
||||
@@ -21,7 +21,7 @@ def main() -> int:
|
||||
out_path = sys.argv[1]
|
||||
rest = sys.argv[2:]
|
||||
if len(rest) % 2 != 0:
|
||||
sys.stderr.write("usage: mkinitrd.py <out.img> [<name> <file>]...\n")
|
||||
sys.stderr.write("usage: make-initial-ramdisk.py <out.img> [<name> <file>]...\n")
|
||||
return 2
|
||||
items = [(rest[i], rest[i + 1]) for i in range(0, len(rest), 2)]
|
||||
|
||||
Reference in New Issue
Block a user