Re-organize the source tree as a monorepo mirroring the FHS

The source layout now mirrors the runtime filesystem hierarchy
(docs/danos-file-system-hierarchy-FSH.md): what lives under system/ in the
source is what a running danos represents under /system. Each service and
driver is a sub-project directory that is its own Zig module — cross-project
references go by module name, never by a path into another project's files.

Moves (all git mv, history preserved):
- src/            -> system/            (danos internals; the self-representation)
    root.zig      -> danos.zig          (the kernel<->user contract module)
    kernel/arch/  -> kernel/architecture/   (arch -> architecture)
    device/       -> devices/           (what /system/devices reflects)
    boot/         -> /boot              (the loaders, top level)
- sbin/           -> split by role:
    init, vfs     -> system/services/<name>/<name>.zig
    hpetd, busd   -> system/drivers/<name>/<name>.zig
    vfs-test      -> system/services/vfs/vfs-test.zig  (inside the vfs project)
- lib/            -> library/runtime/   (room for other libraries beside runtime)

The VFS wire protocol becomes its own module, system/services/vfs/protocol.zig
("vfs-protocol"): the vfs sub-project exposes its interface, and the runtime's
file layer imports it by name. First instance of the "protocol module" pattern
(docs/driver-model.md); usb/block will expose theirs the same way.

Also: fix a naming-standard violation in the protocol — Op -> Operation (and
req -> request, _pad -> _padding). Docs updated: /system/services added to the
FHS doc, a repository-layout section added to the docs index, and stale source
paths swept across comments and docs.

Runtime boot paths are unchanged (the bootloader still loads /sbin/init);
aligning the runtime filesystem to the FHS is a separate follow-up. Suite 35/35
plus host tests green.
This commit is contained in:
Daniel Samson
2026-07-10 12:55:56 +01:00
parent 15b70856c9
commit 8754d4e46a
83 changed files with 334 additions and 177 deletions
+2 -2
View File
@@ -2,8 +2,8 @@
Codename: Shodan Codename: Shodan
Version: 1 Version: 1
A small operating system, written from scratch in Zig — a bootloader (`src/boot/`) A small operating system, written from scratch in Zig — a bootloader (`boot/`)
and a microkernel (`src/kernel/`), sharing a neutral handoff contract (`src/root.zig`). and a microkernel (`system/kernel/`), sharing a neutral handoff contract (`system/danos.zig`).
It boots x86-64 via UEFI, and so far has a framebuffer console, a physical frame It boots x86-64 via UEFI, and so far has a framebuffer console, a physical frame
allocator, its own paging with W^X permissions, interrupt/exception handling, a allocator, its own paging with W^X permissions, interrupt/exception handling, a
LAPIC timer, a kernel heap, a fixed-priority preemptive scheduler, and in-kernel IPC LAPIC timer, a kernel heap, a fixed-priority preemptive scheduler, and in-kernel IPC
View File
+34 -24
View File
@@ -77,7 +77,7 @@ fn addUserBinary(
}, },
}), }),
}); });
exe.setLinkerScript(b.path("lib/user.ld")); exe.setLinkerScript(b.path("library/runtime/user.ld"));
exe.entry = .{ .symbol_name = "_start" }; exe.entry = .{ .symbol_name = "_start" };
exe.image_base = 0x7000_0000_0000; exe.image_base = 0x7000_0000_0000;
exe.use_llvm = true; exe.use_llvm = true;
@@ -95,21 +95,21 @@ pub fn build(b: *std.Build) void {
// so the module inherits the target of whichever binary imports it — the // so the module inherits the target of whichever binary imports it — the
// freestanding kernel or the UEFI bootloader. // freestanding kernel or the UEFI bootloader.
const danos_module = b.addModule("danos", .{ const danos_module = b.addModule("danos", .{
.root_source_file = b.path("src/root.zig"), .root_source_file = b.path("system/danos.zig"),
}); });
// Kernel tunables (maximum_cpus, stack sizes, tick rate). A dependency-free module of // Kernel tunables (maximum_cpus, stack sizes, tick rate). A dependency-free module of
// compile-time constants, imported wherever a knob is read; keeps the trade-offs // compile-time constants, imported wherever a knob is read; keeps the trade-offs
// in one place instead of scattered across the tree. See src/configuration.zig. // in one place instead of scattered across the tree. See system/parameters.zig.
const parameters_module = b.addModule("parameters", .{ const parameters_module = b.addModule("parameters", .{
.root_source_file = b.path("src/parameters.zig"), .root_source_file = b.path("system/parameters.zig"),
}); });
// Architecture-specific kernel code (CPU ops, entry, later GDT/IDT/paging). // Architecture-specific kernel code (CPU ops, entry, later GDT/IDT/paging).
// The generic kernel imports this as "architecture" and never names x86_64, so a new // The generic kernel imports this as "architecture" and never names x86_64, so a new
// architecture is a matter of pointing this module at a different directory. // architecture is a matter of pointing this module at a different directory.
const architecture_module = b.addModule("architecture", .{ const architecture_module = b.addModule("architecture", .{
.root_source_file = b.path("src/kernel/arch/x86_64/cpu.zig"), .root_source_file = b.path("system/kernel/architecture/x86_64/cpu.zig"),
.imports = &.{ .imports = &.{
.{ .name = "danos", .module = danos_module }, // paging uses the shared BootInformation/memory-map types .{ .name = "danos", .module = danos_module }, // paging uses the shared BootInformation/memory-map types
.{ .name = "parameters", .module = parameters_module }, // maximum_cpus, ist_stack_size, timer_hz .{ .name = "parameters", .module = parameters_module }, // maximum_cpus, ist_stack_size, timer_hz
@@ -117,45 +117,55 @@ pub fn build(b: *std.Build) void {
}); });
// CPU-exception stubs — real assembly, since they need cross-symbol // CPU-exception stubs — real assembly, since they need cross-symbol
// jumps/calls that Zig inline asm can't express (see the file's header). // jumps/calls that Zig inline asm can't express (see the file's header).
architecture_module.addAssemblyFile(b.path("src/kernel/arch/x86_64/isr.s")); architecture_module.addAssemblyFile(b.path("system/kernel/architecture/x86_64/isr.s"));
// The AP bring-up trampoline: 16-/32-/64-bit mode-switch code that can't be // The AP bring-up trampoline: 16-/32-/64-bit mode-switch code that can't be
// inline asm (it runs relocated to a low page, not at its link address). // inline asm (it runs relocated to a low page, not at its link address).
architecture_module.addAssemblyFile(b.path("src/kernel/arch/x86_64/trampoline.s")); architecture_module.addAssemblyFile(b.path("system/kernel/architecture/x86_64/trampoline.s"));
// Firmware-agnostic device discovery. The generic kernel imports this as // Firmware-agnostic device discovery. The generic kernel imports this as
// "platform" and asks it to enumerate hardware into a backend-neutral device // "platform" and asks it to enumerate hardware into a backend-neutral device
// tree, never naming ACPI (or, later, device-tree) — the same discipline the // tree, never naming ACPI (or, later, device-tree) — the same discipline the
// architecture module applies to CPU code. The backend is selected at runtime from // architecture module applies to CPU code. The backend is selected at runtime from
// the boot handoff (see src/device/platform.zig). // the boot handoff (see system/devices/platform.zig).
const platform_module = b.addModule("platform", .{ const platform_module = b.addModule("platform", .{
.root_source_file = b.path("src/device/platform.zig"), .root_source_file = b.path("system/devices/platform.zig"),
.imports = &.{ .imports = &.{
.{ .name = "danos", .module = danos_module }, // BootInformation (carries the ACPI RSDP) .{ .name = "danos", .module = danos_module }, // BootInformation (carries the ACPI RSDP)
.{ .name = "parameters", .module = parameters_module }, // maximum_cpus (the discovery pool) .{ .name = "parameters", .module = parameters_module }, // maximum_cpus (the discovery pool)
}, },
}); });
// The VFS wire protocol: the vfs sub-project's public interface, exposed as its
// own module. Both the vfs server and the runtime's file layer (unistd/stdio)
// depend on this contract by name — neither reaches into the other's files. This
// is the first "protocol module" (see docs/driver-model.md); usb/block will
// expose theirs the same way.
const vfs_protocol_module = b.addModule("vfs-protocol", .{
.root_source_file = b.path("system/services/vfs/protocol.zig"),
});
// The user-space runtime library (a nascent libc): system_call wrappers, the // The user-space runtime library (a nascent libc): system_call wrappers, the
// C-convention heap, IPC helpers, the process start shim. Compiled into every // C-convention heap, IPC helpers, the process start shim. Compiled into every
// user binary (see addUserBinary), so it inherits each exe's `.large` code // user binary (see addUserBinary), so it inherits each exe's `.large` code
// model — do NOT set a target/code_model here. It imports `danos` for the // model — do NOT set a target/code_model here. It imports `danos` for the
// shared SystemCall numbers. // shared SystemCall numbers and `vfs-protocol` for the file API.
const runtime_module = b.addModule("runtime", .{ const runtime_module = b.addModule("runtime", .{
.root_source_file = b.path("lib/runtime.zig"), .root_source_file = b.path("library/runtime/runtime.zig"),
.imports = &.{ .imports = &.{
.{ .name = "danos", .module = danos_module }, .{ .name = "danos", .module = danos_module },
.{ .name = "vfs-protocol", .module = vfs_protocol_module },
}, },
}); });
// The initrd container format, shared by the kernel (unpacks it) and the // The initrd container format, shared by the kernel (unpacks it) and the
// build-time packer tools/mkinitrd.zig (produces it). No dependencies. // build-time packer tools/mkinitrd.zig (produces it). No dependencies.
const initrd_module = b.addModule("initrd", .{ const initrd_module = b.addModule("initrd", .{
.root_source_file = b.path("src/user/protocol/initrd.zig"), .root_source_file = b.path("system/initrd.zig"),
}); });
// Compile-time configuration the kernel reads as `@import("build_options")`. The // Compile-time configuration the kernel reads as `@import("build_options")`. The
// QEMU test harness sets -Dtest-case=<name> to run one self-test at boot. // QEMU test harness sets -Dtest-case=<name> to run one self-test at boot.
const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see src/kernel/tests.zig)"); const test_case = b.option([]const u8, "test-case", "Kernel self-test case to run at boot (see system/kernel/tests.zig)");
const build_options = b.addOptions(); const build_options = b.addOptions();
build_options.addOption(?[]const u8, "test_case", test_case); build_options.addOption(?[]const u8, "test_case", test_case);
const build_options_module = build_options.createModule(); const build_options_module = build_options.createModule();
@@ -173,7 +183,7 @@ pub fn build(b: *std.Build) void {
const exe = b.addExecutable(.{ const exe = b.addExecutable(.{
.name = "kernel", .name = "kernel",
.root_module = b.createModule(.{ .root_module = b.createModule(.{
.root_source_file = b.path("src/kernel/main.zig"), .root_source_file = b.path("system/kernel/main.zig"),
.target = kernel_target, .target = kernel_target,
.optimize = optimize, .optimize = optimize,
.code_model = .kernel, // kernel runs in the top 2 GiB (higher half) .code_model = .kernel, // kernel runs in the top 2 GiB (higher half)
@@ -192,7 +202,7 @@ pub fn build(b: *std.Build) void {
}, },
}), }),
}); });
exe.setLinkerScript(b.path("src/kernel/arch/x86_64/linker.ld")); exe.setLinkerScript(b.path("system/kernel/architecture/x86_64/linker.ld"));
exe.entry = .{ .symbol_name = "_start" }; exe.entry = .{ .symbol_name = "_start" };
// The self-hosted linker ignores parts of the linker script (PHDRS, // The self-hosted linker ignores parts of the linker script (PHDRS,
// /DISCARD/, AT(), section order); the higher-half layout depends on the // /DISCARD/, AT(), section order); the higher-half layout depends on the
@@ -210,17 +220,17 @@ pub fn build(b: *std.Build) void {
// Built by the shared user-binary recipe (see addUserBinary): freestanding, // Built by the shared user-binary recipe (see addUserBinary): freestanding,
// linked into the kernel's user region against the `runtime` runtime library, and // linked into the kernel's user region against the `runtime` runtime library, and
// started in ring 3 by the kernel's user-ELF loader. // started in ring 3 by the kernel's user-ELF loader.
const init_exe = addUserBinary(b, kernel_target, runtime_module, "init", "sbin/init.zig"); const init_exe = addUserBinary(b, kernel_target, runtime_module, "init", "system/services/init/init.zig");
b.installArtifact(init_exe); b.installArtifact(init_exe);
// --- initrd: a bundle of extra user binaries (VFS server + drivers) --- // --- initrd: a bundle of extra user binaries (VFS server + drivers) ---
// Each is built by the same user-binary recipe, then packed into one image by // Each is built by the same user-binary recipe, then packed into one image by
// the host-side mkinitrd tool. The bootloader ferries the image to the kernel, // the host-side mkinitrd tool. The bootloader ferries the image to the kernel,
// which unpacks it and spawns each program (src/user/protocol/initrd.zig). // which unpacks it and spawns each program (system/initrd.zig).
const vfs_exe = addUserBinary(b, kernel_target, runtime_module, "vfs", "sbin/vfs.zig"); const vfs_exe = addUserBinary(b, kernel_target, runtime_module, "vfs", "system/services/vfs/vfs.zig");
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, "vfs-test", "sbin/vfs-test.zig"); const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, "vfs-test", "system/services/vfs/vfs-test.zig");
const hpetd_exe = addUserBinary(b, kernel_target, runtime_module, "hpetd", "sbin/hpetd.zig"); const hpetd_exe = addUserBinary(b, kernel_target, runtime_module, "hpetd", "system/drivers/hpetd/hpetd.zig");
const busd_exe = addUserBinary(b, kernel_target, runtime_module, "busd", "sbin/busd.zig"); const busd_exe = addUserBinary(b, kernel_target, runtime_module, "busd", "system/drivers/busd/busd.zig");
// Pack the user binaries into the initrd image with the host-side Python tool // Pack the user binaries into the initrd image with the host-side Python tool
// (the container format is trivial, and Python sidesteps std API churn). Args: // (the container format is trivial, and Python sidesteps std API churn). Args:
@@ -242,13 +252,13 @@ pub fn build(b: *std.Build) void {
const initrd_install = b.addInstallFile(initrd_img, "bin/initrd.img"); const initrd_install = b.addInstallFile(initrd_img, "bin/initrd.img");
b.getInstallStep().dependOn(&initrd_install.step); b.getInstallStep().dependOn(&initrd_install.step);
// Boot methods live in src/boot/, one per way of getting the kernel running. // Boot methods live in boot/, one per way of getting the kernel running.
// Each is its own binary/entry (a loader is built for its own target); today // Each is its own binary/entry (a loader is built for its own target); today
// that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis. // that's UEFI for x86-64, with room for e.g. a device-tree path for the Pis.
const efiexe = b.addExecutable(.{ const efiexe = b.addExecutable(.{
.name = "BOOTX64", .name = "BOOTX64",
.root_module = b.createModule(.{ .root_module = b.createModule(.{
.root_source_file = b.path("src/boot/efi.zig"), .root_source_file = b.path("boot/efi.zig"),
.target = b.resolveTargetQuery(.{ .target = b.resolveTargetQuery(.{
.cpu_arch = .x86_64, .cpu_arch = .x86_64,
.os_tag = .uefi, .os_tag = .uefi,
@@ -362,7 +372,7 @@ pub fn build(b: *std.Build) void {
// here (compiled for the host rather than inheriting a freestanding target). // here (compiled for the host rather than inheriting a freestanding target).
const mod_tests = b.addTest(.{ const mod_tests = b.addTest(.{
.root_module = b.createModule(.{ .root_module = b.createModule(.{
.root_source_file = b.path("src/root.zig"), .root_source_file = b.path("system/danos.zig"),
.target = target, .target = target,
.optimize = optimize, .optimize = optimize,
}), }),
+48 -18
View File
@@ -116,27 +116,57 @@ other over IPC **endpoints** ([ipc.md](ipc.md)), and a **[driver](drivers.md)**
a device, maps its registers, and sleeps until the hardware interrupts it — which is a device, maps its registers, and sleeps until the hardware interrupts it — which is
the whole reason for the arrangement ([vision.md](vision.md)). the whole reason for the arrangement ([vision.md](vision.md)).
## Repository layout
danos is a **monorepo of sub-projects**. Each service or driver is a directory that is
its own Zig module — it can hold as many files as it needs, and other sub-projects
reach it *by module name*, never by a path into its files. The source tree deliberately
**mirrors the runtime FHS** ([danos-file-system-hierarchy-FSH.md](danos-file-system-hierarchy-FSH.md)):
what you see under `system/` in the source is what a running danos represents under
`/system`.
```
system/ → /system danos's own internals (the self-representation)
danos.zig the kernel↔user ABI contract (the `danos` module)
parameters.zig initrd.zig shared contracts
kernel/ IPC, memory, scheduling, the private syscall dispatch
architecture/x86_64/ the `architecture` module (never named by generic code)
devices/ the device model /system/devices reflects (+ aml/)
drivers/ hpetd/ busd/ one sub-project per driver → /system/drivers
services/ init/ vfs/ system servers → /system/services (vfs/ holds
vfs.zig, vfs-test.zig, protocol.zig)
library/ → /lib the runtime library (the stable application ABI)
boot/ → /boot the loaders
tools/ test/ host-side build + QEMU test harness
```
A sub-project exposes its **public interface as a module**: `system/services/vfs/` owns
the VFS wire protocol (`protocol.zig`, the `vfs-protocol` module), which the runtime's
file layer imports by name. `usb`/`block` drivers will expose their protocols the same
way.
## Source map ## Source map
| Area | Code | | Area | Code |
|------|------| |------|------|
| Boot methods (one per way of booting the kernel) | `src/boot/` — `efi.zig` (UEFI) → `BOOTX64.efi` | | Boot methods (one per way of booting the kernel) | `boot/` — `efi.zig` (UEFI) → `BOOTX64.efi` |
| Kernel entry, panic, bring-up | `src/kernel/main.zig` | | Kernel entry, panic, bring-up | `system/kernel/main.zig` |
| Shared loader↔kernel contract (`BootInfo`, `Framebuffer`, `MemoryMap`, `Syscall`, ABI) | `src/root.zig` | | Shared loader↔kernel contract (`BootInfo`, `Framebuffer`, `MemoryMap`, `Syscall`, ABI) | `system/danos.zig` |
| Physical frame allocator | `src/kernel/pmm.zig` | | Physical frame allocator | `system/kernel/pmm.zig` |
| Kernel heap (`std.mem.Allocator`) | `src/kernel/heap.zig` | | Kernel heap (`std.mem.Allocator`) | `system/kernel/heap.zig` |
| Scheduler (fixed-priority preemptive; blocking, wait queues) | `src/kernel/scheduler.zig` | | Scheduler (fixed-priority preemptive; blocking, wait queues) | `system/kernel/scheduler.zig` |
| Big kernel lock + interrupt-safe critical sections | `src/kernel/sync.zig` | | Big kernel lock + interrupt-safe critical sections | `system/kernel/sync.zig` |
| IPC channels between kernel threads (message passing) | `src/kernel/ipc.zig` | | IPC channels between kernel threads (message passing) | `system/kernel/ipc.zig` |
| IPC endpoints: cross-address-space call/reply, handles, notifications | `src/kernel/ipc-synchronous.zig` | | IPC endpoints: cross-address-space call/reply, handles, notifications | `system/kernel/ipc-synchronous.zig` |
| User processes: ELF loading, address spaces, the syscall table | `src/kernel/process.zig` | | User processes: ELF loading, address spaces, the syscall table | `system/kernel/process.zig` |
| Device tree + claim capability + `device_register` containment | `src/kernel/device-service.zig` | | Device tree + claim capability + `device_register` containment | `system/kernel/device-service.zig` |
| IRQ-as-IPC: routing a device interrupt to a driver's endpoint | `src/kernel/irq.zig` | | IRQ-as-IPC: routing a device interrupt to a driver's endpoint | `system/kernel/irq.zig` |
| Hardware discovery (ACPI/device tree) behind one neutral device model | `src/device/` | | Hardware discovery (ACPI/device tree) behind one neutral device model | `system/devices/` |
| Framebuffer text console (mirrors to serial) | `src/kernel/console.zig` | | Framebuffer text console (mirrors to serial) | `system/kernel/console.zig` |
| In-kernel test cases | `src/kernel/tests.zig` | | In-kernel test cases | `system/kernel/tests.zig` |
| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/IO-APIC/timer, serial, linker script) | `src/kernel/arch/x86_64/` | | Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/IO-APIC/timer, serial, linker script) | `system/kernel/architecture/x86_64/` |
| User runtime library (`rt`): syscalls, heap, stdio, IPC, device access | `lib/` | | Runtime library (`runtime`): syscall wrappers, heap, stdio, IPC, device access — the stable application ABI | `library/runtime/` |
| User-space programs shipped in the initrd (`init`, `vfs`, `hpetd` leaf driver, `busd` bus driver) | `sbin/` | | System services (init, the VFS server + its `protocol` module) | `system/services/` |
| Device drivers, one sub-project each (`hpetd` leaf driver, `busd` bus driver) | `system/drivers/` |
| Build + `run-x86-64` (QEMU/OVMF) | `build.zig` | | Build + `run-x86-64` (QEMU/OVMF) | `build.zig` |
| QEMU integration test harness | `test/qemu_test.py` | | QEMU integration test harness | `test/qemu_test.py` |
+5 -5
View File
@@ -16,13 +16,13 @@ the RSDT's address is a field *inside* the RSDP. The platform follows that point
UEFI configuration table UEFI configuration table
│ the loader reads the RSDP's physical address │ the loader reads the RSDP's physical address
▼ ▼
BootInfo.acpi_rsdp (u64, in the shared `danos` module) src/root.zig BootInfo.acpi_rsdp (u64, in the shared `danos` module) system/danos.zig
│ the kernel forwards the whole BootInfo │ the kernel forwards the whole BootInfo
▼ ▼
platform.discover(boot_info, …) src/device/platform.zig platform.discover(boot_info, …) system/devices/platform.zig
│ reads boot_info.acpi_rsdp, hands it to the ACPI backend │ reads boot_info.acpi_rsdp, hands it to the ACPI backend
▼ ▼
acpi.discover(rsdp_phys, …) src/device/acpi.zig acpi.discover(rsdp_phys, …) system/devices/acpi.zig
│ dereferences the RSDP, reads the pointer it contains │ dereferences the RSDP, reads the pointer it contains
▼ ▼
RSDP ──(a field in the struct)──► RSDT / XSDT ──► SDTs (MADT, MCFG, FADT, HPET, DSDT…) RSDP ──(a field in the struct)──► RSDT / XSDT ──► SDTs (MADT, MCFG, FADT, HPET, DSDT…)
@@ -35,7 +35,7 @@ successor the **XSDT** (ACPI 2.0+) — which in turn lists every other SDT.
## Step 1 — the loader finds the RSDP ## Step 1 — the loader finds the RSDP
Only the firmware knows where ACPI lives, so the RSDP must be grabbed while UEFI is Only the firmware knows where ACPI lives, so the RSDP must be grabbed while UEFI is
still up. `acpiRootSystemDescriptorPointer()` in `src/boot/efi.zig` walks the UEFI still up. `acpiRootSystemDescriptorPointer()` in `boot/efi.zig` walks the UEFI
**configuration table** for the ACPI GUID and returns the vendor pointer — the same **configuration table** for the ACPI GUID and returns the vendor pointer — the same
"grab it before `ExitBootServices`" pattern as the [framebuffer](framebuffer.md) and "grab it before `ExitBootServices`" pattern as the [framebuffer](framebuffer.md) and
the [memory map](memory-map.md). the [memory map](memory-map.md).
@@ -48,7 +48,7 @@ module at all** (it imports only the shared `danos` module). So instead of a cal
deposits a value in the handoff struct: deposits a value in the handoff struct:
```zig ```zig
// src/boot/efi.zig — while boot services are still up // boot/efi.zig — while boot services are still up
.acpi_rsdp = if (acpiRootSystemDescriptorPointer()) |p| @intFromPtr(p) else 0, .acpi_rsdp = if (acpiRootSystemDescriptorPointer()) |p| @intFromPtr(p) else 0,
``` ```
+12 -12
View File
@@ -13,7 +13,7 @@ runtime dispatch. `build.zig` exposes one architecture's code as a module called
```zig ```zig
const arch_mod = b.addModule("arch", .{ const arch_mod = b.addModule("arch", .{
.root_source_file = b.path("src/kernel/arch/x86_64/cpu.zig"), .root_source_file = b.path("system/kernel/architecture/x86_64/cpu.zig"),
}); });
``` ```
@@ -26,7 +26,7 @@ arch.halt(); // never says "x86_64"
``` ```
Adding a second architecture is then a build-time choice: create Adding a second architecture is then a build-time choice: create
`src/kernel/arch/aarch64/`, and point the `arch` module at it when the target CPU is `system/kernel/arch/aarch64/`, and point the `arch` module at it when the target CPU is
AArch64. `main.zig` and `console.zig` don't change. **That compiler-checked module AArch64. `main.zig` and `console.zig` don't change. **That compiler-checked module
boundary _is_ the architecture interface** — when a new arch is missing a function boundary _is_ the architecture interface** — when a new arch is missing a function
the generic kernel calls, the build fails and names exactly what's missing. the generic kernel calls, the build fails and names exactly what's missing.
@@ -36,7 +36,7 @@ the generic kernel calls, the build fails and names exactly what's missing.
The split follows a simple test: does it name a CPU instruction, a hardware The split follows a simple test: does it name a CPU instruction, a hardware
register, or a memory-management structure? If so, it's arch-specific. register, or a memory-management structure? If so, it's arch-specific.
| Arch-specific — `src/kernel/arch/x86_64/` | Generic — kernel core | | Arch-specific — `system/kernel/architecture/x86_64/` | Generic — kernel core |
|---|---| |---|---|
| `cpu.zig`: `halt()` (`hlt`), later GDT/IDT/paging | `console.zig` — pure pixel math, works anywhere | | `cpu.zig`: `halt()` (`hlt`), later GDT/IDT/paging | `console.zig` — pure pixel math, works anywhere |
| `linker.ld` — link layout, load address | `main.zig` — `kmain` orchestration, panic handler | | `linker.ld` — link layout, load address | `main.zig` — `kmain` orchestration, panic handler |
@@ -51,9 +51,9 @@ should end up on the generic side; the arch module stays small.
There are really two independent questions, and it's worth not conflating them: There are really two independent questions, and it's worth not conflating them:
- **CPU architecture** (x86_64 vs AArch64): instructions, MMU, interrupts → - **CPU architecture** (x86_64 vs AArch64): instructions, MMU, interrupts →
`src/kernel/arch/<cpu>/`. `system/kernel/arch/<cpu>/`.
- **Boot protocol** (UEFI vs Raspberry Pi firmware + device tree): handled - **Boot protocol** (UEFI vs Raspberry Pi firmware + device tree): handled
*separately*, because loaders are their own binaries. `src/boot/efi.zig` builds *separately*, because loaders are their own binaries. `boot/efi.zig` builds
`BOOTX64.efi`, a distinct executable from the kernel ELF. On a Pi there is no `BOOTX64.efi`, a distinct executable from the kernel ELF. On a Pi there is no
separate loader at all — the firmware jumps straight into the kernel with a separate loader at all — the firmware jumps straight into the kernel with a
device-tree pointer, so that entry work would live in the AArch64 arch code. device-tree pointer, so that entry work would live in the AArch64 arch code.
@@ -61,24 +61,24 @@ There are really two independent questions, and it's worth not conflating them:
## Current x86_64 contents ## Current x86_64 contents
- **`src/kernel/arch/x86_64/cpu.zig`** — the `arch` module root. Exposes `halt()` (see - **`system/kernel/architecture/x86_64/cpu.zig`** — the `arch` module root. Exposes `halt()` (see
[halting.md](halting.md)), `init()` (bring up the descriptor tables), [halting.md](halting.md)), `init()` (bring up the descriptor tables),
`enablePaging()`, `setFaultHandler`, `readCr2`/`readCr3`, and the `CpuState` `enablePaging()`, `setFaultHandler`, `readCr2`/`readCr3`, and the `CpuState`
trap frame. trap frame.
- **`src/kernel/arch/x86_64/gdt.zig`** / **`idt.zig`** / **`tss.zig`** — the GDT, IDT and - **`system/kernel/architecture/x86_64/gdt.zig`** / **`idt.zig`** / **`tss.zig`** — the GDT, IDT and
TSS plus CPU-exception handling (see [interrupts.md](interrupts.md)). TSS plus CPU-exception handling (see [interrupts.md](interrupts.md)).
- **`src/kernel/arch/x86_64/paging.zig`** — the kernel's page tables (see - **`system/kernel/architecture/x86_64/paging.zig`** — the kernel's page tables (see
[paging.md](paging.md)). [paging.md](paging.md)).
- **`src/kernel/arch/x86_64/apic.zig`** — the Local APIC and its timer, the source of - **`system/kernel/architecture/x86_64/apic.zig`** — the Local APIC and its timer, the source of
device interrupts (see [device-interrupts.md](device-interrupts.md)). device interrupts (see [device-interrupts.md](device-interrupts.md)).
- **`src/kernel/arch/x86_64/serial.zig`** / **`io.zig`** — the COM1 UART (the kernel's - **`system/kernel/architecture/x86_64/serial.zig`** / **`io.zig`** — the COM1 UART (the kernel's
machine-readable log channel, see [testing.md](testing.md)) and the shared machine-readable log channel, see [testing.md](testing.md)) and the shared
port-I/O + MSR primitives. port-I/O + MSR primitives.
- **`src/kernel/arch/x86_64/isr.s`** — the exception stubs, the `lgdt`/`lidt`/`ltr` load - **`system/kernel/architecture/x86_64/isr.s`** — the exception stubs, the `lgdt`/`lidt`/`ltr` load
helpers, and the context switch (`switch_context` / `task_trampoline`, see helpers, and the context switch (`switch_context` / `task_trampoline`, see
[scheduling.md](scheduling.md)) — real assembly, since Zig inline asm can't [scheduling.md](scheduling.md)) — real assembly, since Zig inline asm can't
express them. express them.
- **`src/kernel/arch/x86_64/linker.ld`** — the kernel link layout (fixed low load - **`system/kernel/architecture/x86_64/linker.ld`** — the kernel link layout (fixed low load
address, one PT_LOAD per permission set). address, one PT_LOAD per permission set).
The kernel entry point `_start` currently still lives in the generic `main.zig` as The kernel entry point `_start` currently still lives in the generic `main.zig` as
+4 -4
View File
@@ -18,7 +18,7 @@ matters for understanding why. This page maps the landscape so the
new ISA. new ISA.
They are as different from each other as either is from x86-64: separate registers, They are as different from each other as either is from x86-64: separate registers,
page-table formats, and calling conventions. Each needs its own `src/kernel/arch/<name>/`. page-table formats, and calling conventions. Each needs its own `system/kernel/arch/<name>/`.
## The Raspberry Pi models ## The Raspberry Pi models
@@ -57,16 +57,16 @@ the DTB/ACPI tells you what devices exist.
## What danos needs, layer by layer ## What danos needs, layer by layer
- **One CPU arch module: `src/kernel/arch/aarch64/`** — covering the Zero 2 W and Pi 3-5, - **One CPU arch module: `system/kernel/arch/aarch64/`** — covering the Zero 2 W and Pi 3-5,
providing the same `arch` interface as x86_64: `halt`, context switch, providing the same `arch` interface as x86_64: `halt`, context switch,
interrupt/exception vectors, page tables, a UART, a timer. No `src/kernel/arch/arm/` is interrupt/exception vectors, page tables, a UART, a timer. No `system/kernel/arch/arm/` is
planned (see the decision above), so there's a single ARM backend to write. planned (see the decision above), so there's a single ARM backend to write.
- **A device-tree boot path.** Since stock Pis boot via DTB, danos needs an entry - **A device-tree boot path.** Since stock Pis boot via DTB, danos needs an entry
that parses the DTB's `/memory` and `/reserved-memory` into the neutral that parses the DTB's `/memory` and `/reserved-memory` into the neutral
[`MemoryMap`](memory-map.md) — the same neutral handoff `efi.zig` produces, just [`MemoryMap`](memory-map.md) — the same neutral handoff `efi.zig` produces, just
from a different source. This is where keeping boot-protocol knowledge on the from a different source. This is where keeping boot-protocol knowledge on the
loader side (as we did for the UEFI memory-map classification) pays off. loader side (as we did for the UEFI memory-map classification) pays off.
- **The UEFI loader mostly carries over.** `src/boot/efi.zig` is largely - **The UEFI loader mostly carries over.** `boot/efi.zig` is largely
boot-*protocol* code (`std.os.uefi` protocol calls), not x86 code. Its only truly boot-*protocol* code (`std.os.uefi` protocol calls), not x86 code. Its only truly
x86-specific bits are the ELF machine check (`.X86_64`) and the SysV calling x86-specific bits are the ELF machine check (`.X86_64`) and the SysV calling
convention for the kernel jump. So an `aarch64`-UEFI target (QEMU `virt` + AAVMF) convention for the kernel jump. So an `aarch64`-UEFI target (QEMU `virt` + AAVMF)
+117
View File
@@ -0,0 +1,117 @@
# DanOS Filesystem Hierarchy Standard (DFHS)
Most modern Unix and Unix-like operating systems follow the FHS. DanOS has its own FHS structure which extends the unix FHS. This is provided by virtual file system driver (VFS).
## Directory structure
| Path | Description |
|-----------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------|
| / | Primary hierarchy root and root directory of the entire file system hierarchy. |
| /bin | Essential command binaries that need to be available in single-user mode, including to bring up the system or repair it, for all users (e.g., cat, ls, cp). |
| /boot | Boot loader files (e.g., EFI, initrd.img ). |
| /dev | POSIX Device files (e.g., /dev/null, /dev/disk0, /dev/tty, /dev/random). |
| /etc | Host-specific system-wide configuration files. |
| /home | Users' home directories, containing saved files, personal settings, etc. |
| /lib | Libraries essential for the binaries in /bin and /sbin. eg realtime, system, ipc etc. |
| /sbin | Essential system binaries (e.g init) |
| /srv | Site-specific data served by this system, such as data and scripts for web servers, data offered by FTP servers, and repositories for version control systems |
| /system | DanOS operating system files (similar idea to C:\Windows). A true representation of danos — its layout mirrors the source tree, so `/system` is what danos *is*. |
| /system/devices | danos virtual device tree e.g. similar to /sys on linux but with danos device tree conventions (the structures in the devices module) |
| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/hpetd) |
| /system/services | system-service binaries — the VFS server, init, and other user-mode servers (e.g. /system/services/vfs, /system/services/init) |
| /system/kernel | the kernel image |
| /tmp | Directory for temporary files (see also /var/tmp). Often not preserved between system reboots and may be severely size-restricted. |
| /usr | Secondary hierarchy for read-only user data; contains the majority of (multi-)user utilities and applications. Should be shareable and read-only. |
| /var | Variable files: files whose content is expected to continually change during normal operation of the system, such as logs, spool files, and temporary e-mail files. |
## File types
POSIX specifies the long format of the ls command to represent the Unix file type as the first letter for an entry.
| type | symbol | Description |
|-------------------|--------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
| regular | - | An ordinary file holding an uninterpreted byte stream. Reads and writes are positional, and the file grows on demand (e.g., a binary in /bin, a config file in /etc). |
| directory | d | A container mapping names to other files. It may only be modified through directory operations, never written to directly. |
| symbolic link | l | A file whose contents are a path that is resolved in its place. The target need not exist, and may cross mount points. |
| FIFO special | p | A named pipe: an in-order byte stream between processes, where writers block until a reader opens the other end. |
| block special | b | A device node addressed in fixed-size blocks with the kernel free to buffer and reorder access (e.g., /dev/disk0). |
| character special | c | A device node addressed as an unbuffered byte stream, delivered to the driver in order (e.g., /dev/tty, /dev/null). |
| socket | s | A named endpoint for bidirectional message-passing between processes, bound to a path rather than an address. |
## /dev
`/dev` holds the names through which processes reach devices. It is deliberately not
the device tree: the tree — every node discovered by ACPI or PCI enumeration, with its
resources and its parent — lives under [/system/devices](#directory-structure) and is
addressed by device id. `/dev` is the much smaller set of devices that have a driver
willing to serve them, addressed by name.
A device node is not a file the VFS can read. The bytes live in a driver process
([drivers.md](drivers.md)), so opening a `/dev` name has to resolve to that driver's
IPC endpoint, and subsequent reads and writes are calls against it. This is what
`system/services/vfs/vfs.zig` reserves for M10 and what the `Stat.kind` field is for; **none of it is
implemented today.** The current VFS is a flat, in-memory ramfs of eight nodes, with no
directories at all and `kind` hardcoded to zero. The three sections below describe the
intended shape, and are honest about which parts the kernel can already support.
### Character devices
A character device is a byte stream with no addressable position: bytes are delivered
to the driver in the order written, and a read consumes what is there. Terminals,
serial lines, keyboards and mice are all of this shape. These are the natural first
device nodes in danos, because a character driver needs nothing the kernel doesn't
already provide — it claims its device, maps its registers with `mmio_map`, and blocks
on `replyWait` for either an interrupt or a client request. `system/drivers/hpetd/hpetd.zig` is already
that program, minus the client half.
The obstacle is not the file type, it is which hardware a ring-3 driver can actually
drive. Port I/O is unavailable to user space — the TSS I/O permission bitmap is absent
and IOPL is never raised — so `in`/`out` from a driver is a #GP. That excludes the
16550 UART at `0x3F8` and PS/2 at `0x60`/`0x64`, which is to say it excludes the
obvious implementations of `/dev/tty`, `/dev/ttyS0` and a keyboard node. Until either
port I/O grants or a memory-mapped UART exist, serial output stays a kernel service
reached through the `write` system call rather than a file. A memory-mapped device such
as the framebuffer has no such problem and is the more likely first real entry here.
### Block devices
A block device is addressed in fixed-size blocks and, unlike a character device, the
layer above is free to buffer, reorder, coalesce and retry requests against it. Disks
and other persistent storage are the whole population of this class.
**danos cannot host a block driver at all today,** and the reason is worth stating
plainly because it is not a matter of unwritten code. Every storage controller worth
naming is a bus master: it is programmed by handing it the physical address of a
descriptor ring and left to read and write memory on its own. A ring-3 driver cannot
build such a ring, because `mmap` returns writeback-cached, physically discontiguous
pages and never discloses their physical address. Nor should it be allowed to: a device
programmed with an arbitrary physical address writes to arbitrary physical memory, and
page tables do not sit between a device and RAM — an IOMMU does. Granting a DMA-capable
device to a driver process, with no IOMMU programmed, is equivalent to granting ring 0,
which would forfeit the isolation that motivates user-space drivers in the first place.
Block devices therefore wait on DMA-capable memory, memory barriers, and VT-d/DMAR —
the M14–M16 work in [driver-model.md](driver-model.md). A ramdisk over the initrd is
the one block-shaped thing implementable now, and it needs no driver process.
### Pseudo-devices
A pseudo-device has the interface of a device and no hardware behind it: `/dev/null`
discarding writes and reading as end-of-file, `/dev/zero` reading as an endless run of
zero bytes, `/dev/full` failing writes with `ENOSPC`, `/dev/random` and `/dev/urandom`
yielding unpredictable bytes.
These are the only `/dev` entries danos can implement immediately, and they are the
sensible place to start, because they are exactly the entries that need no driver
process, no `device_claim`, no MMIO grant and no interrupt. The VFS server answers them
out of its own address space — `null` and `zero` are a few lines each in
`system/services/vfs/vfs.zig`'s `read` and `write` handlers. Doing so forces the two pieces of
structure that every later device node depends on and that the flat ramfs currently
lacks: a directory, so that `/dev/null` is a path rather than a name; and a populated
`Stat.kind`, so that a caller can tell a character device from a regular file.
`/dev/random` is the one that is not free. It needs an entropy source, and the honest
options on this kernel are `RDRAND`/`RDSEED` where CPUID advertises them, and the HPET
counter's low bits as a poor fallback. Neither is a seeded CSPRNG, and a `/dev/random`
that is merely unpredictable-looking is worse than none — nothing should be keyed from
it until it is a real one.
+1 -1
View File
@@ -18,7 +18,7 @@ Interrupt delivery on modern x86 goes through the **APIC**, not the legacy 8259
PIC. There are two halves; we only need one so far: PIC. There are two halves; we only need one so far:
- The **Local APIC** (per-CPU, memory-mapped at physical `0xFEE00000`) handles the - The **Local APIC** (per-CPU, memory-mapped at physical `0xFEE00000`) handles the
CPU's own timer and receives interrupts routed to it. `src/kernel/arch/x86_64/apic.zig`. CPU's own timer and receives interrupts routed to it. `system/kernel/architecture/x86_64/apic.zig`.
- The **IO-APIC** routes *external* device lines (keyboard, etc.) to LAPIC vectors. - The **IO-APIC** routes *external* device lines (keyboard, etc.) to LAPIC vectors.
Not needed for the timer — it'll arrive with the keyboard. Not needed for the timer — it'll arrive with the keyboard.
+9 -9
View File
@@ -32,7 +32,7 @@ plain bus driver with no controller — a USB hub — is also a real thing.
## The device table is the spine ## The device table is the spine
danos already has the right central structure. `src/kernel/device-service.zig` holds a table of danos already has the right central structure. `system/kernel/device-service.zig` holds a table of
`DeviceDesc`, each with a parent, a class, and a set of resources. Firmware discovery `DeviceDesc`, each with a parent, a class, and a set of resources. Firmware discovery
seeds it ([discovery.md](discovery.md)); `device_register` grows it. seeds it ([discovery.md](discovery.md)); `device_register` grows it.
@@ -57,7 +57,7 @@ is not an address window. Discovery is trusted; user space is not.
### What a bus driver looks like ### What a bus driver looks like
`sbin/busd.zig` is the smallest honest one. Its "bus" is the HPET's register block and `system/drivers/busd/busd.zig` is the smallest honest one. Its "bus" is the HPET's register block and
its "devices" are the block's comparators: its "devices" are the block's comparators:
```zig ```zig
@@ -94,8 +94,8 @@ A "family" is two modules, not one:
- **A protocol module** — the IPC message types that let a class driver talk to - **A protocol module** — the IPC message types that let a class driver talk to
*whatever* published its device. This is the part that makes class drivers portable. *whatever* published its device. This is the part that makes class drivers portable.
danos already has one of each: `lib/device.zig` is a logic module, danos already has one of each: `library/runtime/device.zig` is a logic module,
[`lib/vfs-protocol.zig`](lib/vfs-protocol.zig) is a protocol module shared by `sbin/vfs.zig` [`system/services/vfs/protocol.zig`](system/services/vfs/protocol.zig) is a protocol module shared by `system/services/vfs/vfs.zig`
and its clients. The pattern generalises directly: and its clients. The pattern generalises directly:
``` ```
@@ -106,7 +106,7 @@ lib/
pci.zig module "pci" — ECAM, BAR decode, capability walk pci.zig module "pci" — ECAM, BAR decode, capability walk
usb.zig module "usb" — descriptors, control transfers, hubs usb.zig module "usb" — descriptors, control transfers, hubs
proto/ proto/
vfs.zig module "proto.vfs" (today: lib/vfs-protocol.zig) vfs.zig module "proto.vfs" (today: system/services/vfs/protocol.zig)
block.zig module "proto.block" block.zig module "proto.block"
hid.zig module "proto.hid" hid.zig module "proto.hid"
@@ -183,7 +183,7 @@ const dev_ep = ipc.callCap(h, // ... mint a per-device endpoint,
**The blocker.** An HCD is a DMA-engine programmer. It needs a descriptor ring the **The blocker.** An HCD is a DMA-engine programmer. It needs a descriptor ring the
device can read, which means memory that is (a) physically contiguous, (b) at a device can read, which means memory that is (a) physically contiguous, (b) at a
physical address the driver knows, (c) of the right cacheability, and (d) pinned. physical address the driver knows, (c) of the right cacheability, and (d) pinned.
[`sysMmap`](src/kernel/process.zig) gives you *none* of the four: it calls `pmm.alloc()` [`sysMmap`](system/kernel/process.zig) gives you *none* of the four: it calls `pmm.alloc()`
once per page, maps writeback-cached, and never reveals a physical address. once per page, maps writeback-cached, and never reveals a physical address.
**The fix.** **The fix.**
@@ -217,7 +217,7 @@ doorbell.* = i; // volatile store to UC MMIO
// nothing stops the compiler reordering these; the device reads a stale descriptor // nothing stops the compiler reordering these; the device reads a stale descriptor
``` ```
So the rules, which belong in `lib/mmio.zig` and behind `arch`: So the rules, which belong in `library/mmio.zig` and behind `arch`:
| Situation | Required | | Situation | Required |
|---|---| |---|---|
@@ -241,12 +241,12 @@ with a compiler barrier alone. ARM is not, and [vision.md](vision.md) makes ARM
condition. Build the abstraction while there is one caller to fix. condition. Build the abstraction while there is one caller to fix.
(Zig note: `@fence` was **removed in 0.16**. Use `@atomicRmw(..., .seq_cst)` for a full (Zig note: `@fence` was **removed in 0.16**. Use `@atomicRmw(..., .seq_cst)` for a full
barrier, or per-arch inline asm — which is what `lib/mmio.zig` should hide.) barrier, or per-arch inline asm — which is what `library/mmio.zig` should hide.)
## M15 — interrupts for PCI devices ## M15 — interrupts for PCI devices
**The blocker, and it's a hard one.** No PCI device can take an interrupt today. **The blocker, and it's a hard one.** No PCI device can take an interrupt today.
[`addBars`](src/device/acpi.zig) records `.memory` and `.io_port` BARs and never an [`addBars`](system/devices/acpi.zig) records `.memory` and `.io_port` BARs and never an
`.irq`; there is no `_PRT` parsing anywhere in the tree. `hpetd` only works because the `.irq`; there is no `_PRT` parsing anywhere in the tree. `hpetd` only works because the
HPET advertises its own routing options in its own registers — a privilege no ordinary HPET advertises its own routing options in its own registers — a privilege no ordinary
device has. device has.
+4 -4
View File
@@ -22,7 +22,7 @@ say.*
## The capability: claim before touch ## The capability: claim before touch
The five driver syscalls (`src/root.zig`, dispatched in `src/kernel/process.zig`): The five driver syscalls (`system/danos.zig`, dispatched in `system/kernel/process.zig`):
| # | Call | Meaning | | # | Call | Meaning |
|---|------|---------| |---|------|---------|
@@ -138,7 +138,7 @@ Two properties worth knowing:
## A whole driver ## A whole driver
`sbin/hpetd.zig` is ~150 lines and does all of it. The shape: `system/drivers/hpetd/hpetd.zig` is ~150 lines and does all of it. The shape:
```zig ```zig
const hpet = findHpet(buf) orelse return; // device_enumerate, look for const hpet = findHpet(buf) orelse return; // device_enumerate, look for
@@ -206,7 +206,7 @@ bus driver may only ever subdivide what it already owns.
A device with **no resources** is legal and common. A USB device is reached through its A device with **no resources** is legal and common. A USB device is reached through its
controller, not by MMIO, so it gets `resource_count = 0`. controller, not by MMIO, so it gets `resource_count = 0`.
See [`sbin/busd.zig`](../sbin/busd.zig) for a complete one, and See [`system/drivers/busd/busd.zig`](../system/drivers/busd/busd.zig) for a complete one, and
[driver-model.md](driver-model.md) for how bus drivers, class drivers and host [driver-model.md](driver-model.md) for how bus drivers, class drivers and host
controller drivers fit together. controller drivers fit together.
@@ -265,7 +265,7 @@ Worth knowing before you write the second driver:
level-triggered redirection entry with remote-IRR set doesn't clear it on some level-triggered redirection entry with remote-IRR set doesn't clear it on some
chipsets, and the line never fires again. QEMU clears it on EOI regardless, so the chipsets, and the line never fires again. QEMU clears it on EOI regardless, so the
tests can't see this. Linux flushes remote-IRR by toggling the entry to edge and tests can't see this. Linux flushes remote-IRR by toggling the entry to edge and
back. See the note at the top of `src/kernel/irq.zig`. back. See the note at the top of `system/kernel/irq.zig`.
## Verifying it ## Verifying it
+6 -6
View File
@@ -10,7 +10,7 @@ that hands us a working CPU, a memory map, and a screen, and then gets out of th
way. way.
The key thing to understand: **UEFI is not our OS, it's a stepping stone.** It The key thing to understand: **UEFI is not our OS, it's a stepping stone.** It
exists to load *us*. Our `src/boot/efi.zig` is a UEFI *application* — a normal program exists to load *us*. Our `boot/efi.zig` is a UEFI *application* — a normal program
that the firmware runs — and its entire purpose is to gather what the kernel needs that the firmware runs — and its entire purpose is to gather what the kernel needs
and then jump into the kernel. and then jump into the kernel.
@@ -23,7 +23,7 @@ Partition (ESP)** and running a file at a well-known fallback path:
esp/EFI/BOOT/BOOTX64.efi <- the "removable media" default for x86-64 esp/EFI/BOOT/BOOTX64.efi <- the "removable media" default for x86-64
``` ```
That's exactly the layout `build.zig` assembles. It builds `src/boot/efi.zig` for the That's exactly the layout `build.zig` assembles. It builds `boot/efi.zig` for the
`uefi` target, installs it to `esp/EFI/BOOT/BOOTX64.efi`, and drops the kernel ELF `uefi` target, installs it to `esp/EFI/BOOT/BOOTX64.efi`, and drops the kernel ELF
at `esp/kernel`. The `run-x86-64` step then points QEMU at OVMF (UEFI firmware for at `esp/kernel`. The `run-x86-64` step then points QEMU at OVMF (UEFI firmware for
virtual machines) and presents that `esp/` directory to the guest as a FAT drive. virtual machines) and presents that `esp/` directory to the guest as a FAT drive.
@@ -139,7 +139,7 @@ kernel is freestanding and uses the **SysV AMD64** convention (first argument in
read garbage. read garbage.
So both sides pin the convention explicitly to SysV via the shared So both sides pin the convention explicitly to SysV via the shared
`danos.kernel_abi` (defined in `src/root.zig`). The loader's function-pointer type `danos.kernel_abi` (defined in `system/danos.zig`). The loader's function-pointer type
and the kernel's `_start` both reference it, so the pointer lands in the register and the kernel's `_start` both reference it, so the pointer lands in the register
the kernel expects. This is the whole reason `kernel_abi` lives in the shared the kernel expects. This is the whole reason `kernel_abi` lives in the shared
`danos` module: it's a contract both binaries must agree on. See `danos` module: it's a contract both binaries must agree on. See
@@ -149,7 +149,7 @@ the kernel expects. This is the whole reason `kernel_abi` lives in the shared
The loader and kernel are two *separate* binaries built for two different targets, The loader and kernel are two *separate* binaries built for two different targets,
so everything they exchange must have an identically-defined memory layout. That's so everything they exchange must have an identically-defined memory layout. That's
what `src/root.zig` provides — imported by both as the `danos` module: what `system/danos.zig` provides — imported by both as the `danos` module:
- `BootInfo` — the top-level struct passed to the kernel (currently just the - `BootInfo` — the top-level struct passed to the kernel (currently just the
framebuffer; this is where future handoff data like the memory map will go). framebuffer; this is where future handoff data like the memory map will go).
@@ -170,10 +170,10 @@ power on
-> loadKernel (read danos ELF, load PT_LOAD segments to 0x100000) -> loadKernel (read danos ELF, load PT_LOAD segments to 0x100000)
-> exitBootServices (retry until the memory-map key holds) -> exitBootServices (retry until the memory-map key holds)
-> jump to e_entry, boot_info pointer in RDI -> jump to e_entry, boot_info pointer in RDI
-> kernel _start (src/kernel/main.zig: framebuffer console, then halt) -> kernel _start (system/kernel/main.zig: framebuffer console, then halt)
``` ```
Bottom line: **UEFI's job is to give us a CPU, memory, and a framebuffer, then Bottom line: **UEFI's job is to give us a CPU, memory, and a framebuffer, then
disappear.** `src/boot/efi.zig` is the thin bridge that collects those gifts into a disappear.** `boot/efi.zig` is the thin bridge that collects those gifts into a
`BootInfo`, tears down the firmware, and jumps into the kernel — after which we're `BootInfo`, tears down the firmware, and jumps into the kernel — after which we're
on our own. on our own.
+2 -2
View File
@@ -3,7 +3,7 @@
Once the kernel knows what RAM exists ([memory-map.md](memory-map.md)), it needs a Once the kernel knows what RAM exists ([memory-map.md](memory-map.md)), it needs a
way to *hand out* that RAM: give me a free page of physical memory, and later, way to *hand out* that RAM: give me a free page of physical memory, and later,
here's one back. That's the **physical frame allocator** (a "physical memory here's one back. That's the **physical frame allocator** (a "physical memory
manager", hence `src/kernel/pmm.zig`). It deals only in fixed 4 KiB **frames** — the manager", hence `system/kernel/pmm.zig`). It deals only in fixed 4 KiB **frames** — the
natural unit because that's the granularity the CPU's paging hardware maps — and natural unit because that's the granularity the CPU's paging hardware maps — and
it is the primitive everything above it stands on: page tables, the kernel heap, it is the primitive everything above it stands on: page tables, the kernel heap,
per-process memory all ultimately ask the frame allocator for pages. per-process memory all ultimately ask the frame allocator for pages.
@@ -33,7 +33,7 @@ RAM is 32768 frames — a **4 KiB bitmap, a single frame**. Even 64 GiB needs on
## How it works ## How it works
State lives in `src/kernel/pmm.zig`: the `bitmap` slice, `total_frames`, `used_frames`, State lives in `system/kernel/pmm.zig`: the `bitmap` slice, `total_frames`, `used_frames`,
and a `next_hint` marking where the next allocation scan should start. and a `next_hint` marking where the next allocation scan should start.
### init(map) — building it from the memory map ### init(map) — building it from the memory map
+2 -2
View File
@@ -9,10 +9,10 @@ write a 32-bit value to the right address, and a pixel changes color. That's
exactly what `Console.pixel` does: exactly what `Console.pixel` does:
```zig ```zig
self.rowPtr(y)[x] = color; // src/kernel/console.zig self.rowPtr(y)[x] = color; // system/kernel/console.zig
``` ```
Our `Framebuffer` struct (`src/root.zig`) is the four facts you need to Our `Framebuffer` struct (`system/danos.zig`) is the four facts you need to
address it: address it:
| Field | Meaning | | Field | Meaning |
+3 -3
View File
@@ -16,7 +16,7 @@ safely, until the machine is reset or powered off.
## The core of it: `hlt` ## The core of it: `hlt`
Everything comes down to one x86 instruction. It's CPU-specific, so it lives in Everything comes down to one x86 instruction. It's CPU-specific, so it lives in
the arch module, `src/kernel/arch/x86_64/cpu.zig` (see [arch.md](arch.md)), and the the arch module, `system/kernel/architecture/x86_64/cpu.zig` (see [arch.md](arch.md)), and the
generic kernel calls it as `arch.halt()`: generic kernel calls it as `arch.halt()`:
```zig ```zig
@@ -83,7 +83,7 @@ treats the call:
signature for a kernel entry point — the bootloader jumps in and nothing ever signature for a kernel entry point — the bootloader jumps in and nothing ever
jumps back out. jumps back out.
You can see the chain in `src/kernel/main.zig`: `_start` is `noreturn`, it calls You can see the chain in `system/kernel/main.zig`: `_start` is `noreturn`, it calls
`kmain` which is `noreturn`, which ends by calling `arch.halt()` which is `kmain` which is `noreturn`, which ends by calling `arch.halt()` which is
`noreturn`. The "never returns" property is threaded all the way down. `noreturn`. The "never returns" property is threaded all the way down.
@@ -106,7 +106,7 @@ There are three halt sites, and they're all the same idea:
`arch.halt()`. A panic is unrecoverable here, so stopping the machine — rather `arch.halt()`. A panic is unrecoverable here, so stopping the machine — rather
than limping on with corrupted state — is the safe response. than limping on with corrupted state — is the safe response.
3. **Bootloader failure** — in `src/boot/efi.zig`, if `boot()` fails *before* handing 3. **Bootloader failure** — in `boot/efi.zig`, if `boot()` fails *before* handing
off to the kernel, `main` logs the error and parks the machine with the same off to the kernel, `main` logs the error and parks the machine with the same
loop so the message stays on screen: loop so the message stays on screen:
+1 -1
View File
@@ -7,7 +7,7 @@ top of both to provide what the rest of the kernel actually wants: `alloc(n)` /
the thing that unlocks dynamic data structures — lists, hash maps, driver state, the thing that unlocks dynamic data structures — lists, hash maps, driver state,
eventually a process table. eventually a process table.
It's generic kernel code (`src/kernel/heap.zig`): the allocator logic is It's generic kernel code (`system/kernel/heap.zig`): the allocator logic is
architecture-neutral, using `arch.mapPage` and the frame allocator underneath. architecture-neutral, using `arch.mapPage` and the frame allocator underneath.
## A growable free-list allocator ## A growable free-list allocator
+5 -5
View File
@@ -9,7 +9,7 @@ reboot is miserable.
This is the machinery that catches those faults and prints what happened instead. This is the machinery that catches those faults and prints what happened instead.
It's all x86_64-specific, so it lives behind the [arch](arch.md) boundary in It's all x86_64-specific, so it lives behind the [arch](arch.md) boundary in
`src/kernel/arch/x86_64/`. Only the 32 CPU-defined exception vectors are wired up so far; `system/kernel/architecture/x86_64/`. Only the 32 CPU-defined exception vectors are wired up so far;
device interrupts (timer, keyboard, via the APIC) come later, on the same IDT. device interrupts (timer, keyboard, via the APIC) come later, on the same IDT.
## First the GDT ## First the GDT
@@ -20,7 +20,7 @@ IDT gate names a code-segment *selector* that must resolve in the current GDT. T
firmware left a GDT in place, but we don't control it, so we install our own with firmware left a GDT in place, but we don't control it, so we install our own with
known selectors: `0x08` kernel code, `0x10` kernel data. known selectors: `0x08` kernel code, `0x10` kernel data.
`src/kernel/arch/x86_64/gdt.zig` holds three flat descriptors — a required null entry, `system/kernel/architecture/x86_64/gdt.zig` holds three flat descriptors — a required null entry,
plus code and data — where the only bits that matter in long mode are the access plus code and data — where the only bits that matter in long mode are the access
byte and the code segment's long-mode (`L`) flag. Loading it (`gdt_flush` in byte and the code segment's long-mode (`L`) flag. Loading it (`gdt_flush` in
`isr.s`) does two things: `lgdt`, then reload the segment registers. The data `isr.s`) does two things: `lgdt`, then reload the segment registers. The data
@@ -33,7 +33,7 @@ into CS:RIP.
The **Interrupt Descriptor Table** maps each of 256 vectors to a handler. Each The **Interrupt Descriptor Table** maps each of 256 vectors to a handler. Each
entry is a 16-byte *gate* holding the handler's address (split across three entry is a 16-byte *gate* holding the handler's address (split across three
fields, a quirk of the format), the code selector (`0x08`), and flags: `0x8E` fields, a quirk of the format), the code selector (`0x08`), and flags: `0x8E`
means present, ring 0, 64-bit interrupt gate. `src/kernel/arch/x86_64/idt.zig` builds the means present, ring 0, 64-bit interrupt gate. `system/kernel/architecture/x86_64/idt.zig` builds the
table, points the first 32 vectors at their stubs, and loads it with `lidt` table, points the first 32 vectors at their stubs, and loads it with `lidt`
(`idt_flush`). (`idt_flush`).
@@ -49,7 +49,7 @@ hit a fault *while trying to deliver another fault* — very often because the
current stack pointer is bad, so pushing the exception frame itself faulted. If current stack pointer is bad, so pushing the exception frame itself faulted. If
the #DF handler then tried to push onto that same bad stack, it would fault a the #DF handler then tried to push onto that same bad stack, it would fault a
third time and **triple-fault** — an instant reset. So the #DF gate is pointed at third time and **triple-fault** — an instant reset. So the #DF gate is pointed at
**IST1**, a small dedicated stack (`src/kernel/arch/x86_64/tss.zig`) that's always valid. **IST1**, a small dedicated stack (`system/kernel/architecture/x86_64/tss.zig`) that's always valid.
Bringing it up: fill in the TSS's IST1 pointer, publish the TSS through a Bringing it up: fill in the TSS's IST1 pointer, publish the TSS through a
descriptor in the GDT (`gdt.setTss`), and load it into the task register with descriptor in the GDT (`gdt.setTss`), and load it into the task register with
@@ -60,7 +60,7 @@ which is why the GDT grew from three entries to five.
On an exception the CPU pushes a small frame (SS, RSP, RFLAGS, CS, RIP) and, for On an exception the CPU pushes a small frame (SS, RSP, RFLAGS, CS, RIP) and, for
*some* vectors, an **error code**. That inconsistency is a nuisance, so each stub *some* vectors, an **error code**. That inconsistency is a nuisance, so each stub
in `src/kernel/arch/x86_64/isr.s` normalises it: vectors that don't get a hardware error in `system/kernel/architecture/x86_64/isr.s` normalises it: vectors that don't get a hardware error
code push a dummy `0`, then every stub pushes its **vector number** and jumps to a code push a dummy `0`, then every stub pushes its **vector number** and jumps to a
shared tail, `isr_common`. The tail pushes all the general registers and calls the shared tail, `isr_common`. The tail pushes all the general registers and calls the
Zig handler with a pointer to the whole thing. Zig handler with a pointer to the whole thing.
+4 -4
View File
@@ -8,15 +8,15 @@ concern, not an afterthought.
There are two layers, built a milestone apart: There are two layers, built a milestone apart:
- **`src/kernel/ipc.zig`** — a bounded blocking channel between *kernel threads*, - **`system/kernel/ipc.zig`** — a bounded blocking channel between *kernel threads*,
described below. The primitive, and where the blocking discipline was worked out. described below. The primitive, and where the blocking discipline was worked out.
- **`src/kernel/ipc-synchronous.zig`** — synchronous call/reply between *processes*, across - **`system/kernel/ipc-synchronous.zig`** — synchronous call/reply between *processes*, across
address spaces. What user-space servers and drivers actually talk over. It's the address spaces. What user-space servers and drivers actually talk over. It's the
second half of this document. second half of this document.
## The channel ## The channel
The first form is a **bounded blocking channel** (`src/kernel/ipc.zig`): a fixed-size The first form is a **bounded blocking channel** (`system/kernel/ipc.zig`): a fixed-size
ring buffer of messages with a producer/consumer rendezvous, built on the ring buffer of messages with a producer/consumer rendezvous, built on the
scheduler's [wait queues](scheduling.md). scheduler's [wait queues](scheduling.md).
@@ -54,7 +54,7 @@ expected `5050`), and neither task busy-waits — they block and wake each other
A channel connects two kernel threads sharing one address space. Real servers are A channel connects two kernel threads sharing one address space. Real servers are
*processes*, so the payload has to cross an address-space boundary. That's *processes*, so the payload has to cross an address-space boundary. That's
`src/kernel/ipc-synchronous.zig`, and its shape is L4's: a synchronous **rendezvous** at an `system/kernel/ipc-synchronous.zig`, and its shape is L4's: a synchronous **rendezvous** at an
`Endpoint`, with the message copied directly from the sender's pages to the receiver's `Endpoint`, with the message copied directly from the sender's pages to the receiver's
(`copyAcross` walks both sets of page tables through the physmap — no CR3 switch, no (`copyAcross` walks both sets of page tables through the physmap — no CR3 switch, no
bounce buffer). bounce buffer).
+3 -3
View File
@@ -13,7 +13,7 @@ kernel follows.
## The log is multi-sink ## The log is multi-sink
`src/kernel/log.zig` is the diagnostic log. It fans a message out to a set of `system/kernel/log.zig` is the diagnostic log. It fans a message out to a set of
registered **sinks**, each best-effort and self-guarding: registered **sinks**, each best-effort and self-guarding:
```zig ```zig
@@ -36,7 +36,7 @@ Properties that matter:
## The framebuffer is *not* a log sink ## The framebuffer is *not* a log sink
The framebuffer is a general graphics surface, **not inherently a text terminal**. The framebuffer is a general graphics surface, **not inherently a text terminal**.
Today `src/kernel/console.zig` paints a text grid on it as a *bootstrap* console, but Today `system/kernel/console.zig` paints a text grid on it as a *bootstrap* console, but
that's a stop-gap: once the driver machinery exists the framebuffer becomes a proper that's a stop-gap: once the driver machinery exists the framebuffer becomes a proper
**graphics device driver**, and the text crutch goes away. So the log must not assume **graphics device driver**, and the text crutch goes away. So the log must not assume
it — routing the verbose log through a pixel console would bake in "the OS is text". it — routing the verbose log through a pixel console would bake in "the OS is text".
@@ -58,7 +58,7 @@ screen. `console.write` is a no-op when the firmware gave us no framebuffer.
A framebuffer is not guaranteed — a headless server exposes no UEFI Graphics Output A framebuffer is not guaranteed — a headless server exposes no UEFI Graphics Output
Protocol. That used to be *fatal* (the loader failed the boot). Now the loader hands Protocol. That used to be *fatal* (the loader failed the boot). Now the loader hands
over a "no framebuffer" descriptor (`base == 0`) rather than failing, and over a "no framebuffer" descriptor (`base == 0`) rather than failing, and
`Framebuffer.present()` (in `src/root.zig`) gates every on-screen path. A headless, `Framebuffer.present()` (in `system/danos.zig`) gates every on-screen path. A headless,
serial-less machine boots and runs correctly — it just goes quiet. serial-less machine boots and runs correctly — it just goes quiet.
## Last-resort channels (no text output at all) ## Last-resort channels (no text output at all)
+2 -2
View File
@@ -27,7 +27,7 @@ danos's own neutral format, and the kernel only ever sees that.**
## The neutral format ## The neutral format
Defined in `src/root.zig`, the shared loader↔kernel contract: Defined in `system/danos.zig`, the shared loader↔kernel contract:
```zig ```zig
pub const MemoryKind = enum(u32) { pub const MemoryKind = enum(u32) {
@@ -68,7 +68,7 @@ pub const BootInfo = extern struct {
## The loader side (UEFI) ## The loader side (UEFI)
Two functions in `src/boot/efi.zig`, called from `exitBootServices`: Two functions in `boot/efi.zig`, called from `exitBootServices`:
- **`classify`** maps each UEFI descriptor to a `MemoryKind`: - **`classify`** maps each UEFI descriptor to a `MemoryKind`:
`conventional_memory` **and** `boot_services_code`/`boot_services_data → usable`; `conventional_memory` **and** `boot_services_code`/`boot_services_data → usable`;
+2 -2
View File
@@ -7,7 +7,7 @@ which live in memory we'd like to reclaim and don't control), switches CR3 onto
them, and — crucially — maps with **real permissions**. them, and — crucially — maps with **real permissions**.
It's x86_64-specific (the 4-level table format is an Intel/AMD thing), so it lives It's x86_64-specific (the 4-level table format is an Intel/AMD thing), so it lives
behind the [arch](arch.md) boundary in `src/kernel/arch/x86_64/paging.zig`. behind the [arch](arch.md) boundary in `system/kernel/architecture/x86_64/paging.zig`.
## The format ## The format
@@ -27,7 +27,7 @@ address to the low load address in its bootstrap tables and jumps in). The entir
alongside a **physmap** — a straight window onto all of physical memory at alongside a **physmap** — a straight window onto all of physical memory at
`physmap_base + phys`. Wherever the kernel needs to touch a physical address (a `physmap_base + phys`. Wherever the kernel needs to touch a physical address (a
page-table frame, an ACPI table, a device register), it adds that constant: page-table frame, an ACPI table, a device register), it adds that constant:
`danos.physToVirt(phys)`. The layout constants live in `src/root.zig`: `danos.physToVirt(phys)`. The layout constants live in `system/danos.zig`:
| region | virtual base | PML4 slot | | region | virtual base | PML4 slot |
|--------|--------------|-----------| |--------|--------------|-----------|
+2 -2
View File
@@ -6,8 +6,8 @@ ready task always runs, and tasks at the same priority take turns. That model is
chosen for [real-time](vision.md) — it's predictable (you can reason about which chosen for [real-time](vision.md) — it's predictable (you can reason about which
task runs when) and its decisions are O(1), unlike a fair-share scheduler. task runs when) and its decisions are O(1), unlike a fair-share scheduler.
The scheduler proper (`src/kernel/sched.zig`) is generic; the context switch and new-task The scheduler proper (`system/kernel/sched.zig`) is generic; the context switch and new-task
stack setup are architecture-specific (`src/kernel/arch/x86_64/`, see [arch](arch.md)). stack setup are architecture-specific (`system/kernel/architecture/x86_64/`, see [arch](arch.md)).
## Tasks ## Tasks
+3 -3
View File
@@ -131,7 +131,7 @@ Whatever the top goal, the *sequence* is the same and seL4 validates starting si
model you already have (the interrupt-flag discipline in model you already have (the interrupt-flag discipline in
[scheduling.md](scheduling.md)) stay largely intact: one lock around kernel entry [scheduling.md](scheduling.md)) stay largely intact: one lock around kernel entry
instead of rethinking every critical section. **Done** — see instead of rethinking every critical section. **Done** — see
`src/kernel/sync.zig`. `system/kernel/sync.zig`.
4. **Later, if contention bites,** evolve toward **per-core run queues + explicit 4. **Later, if contention bites,** evolve toward **per-core run queues + explicit
affinity** (the Fiasco.OC direction) — also the more real-time-predictable model. affinity** (the Fiasco.OC direction) — also the more real-time-predictable model.
5. **Placement stays a user-space policy** — the kernel runs a thread on the core it's 5. **Placement stays a user-space policy** — the kernel runs a thread on the core it's
@@ -151,7 +151,7 @@ next lands.
- **Core enumeration** — the MADT parse records every usable Local APIC (with its - **Core enumeration** — the MADT parse records every usable Local APIC (with its
`apic_id`, which an AP wake targets); `platform.cpus()` returns the list. See `apic_id`, which an AP wake targets); `platform.cpus()` returns the list. See
[discovery.md](discovery.md). [discovery.md](discovery.md).
- **The big kernel lock** (`src/kernel/sync.zig`) — one coarse spinlock guarding the - **The big kernel lock** (`system/kernel/sync.zig`) — one coarse spinlock guarding the
scheduler queues and IPC, always held with local interrupts disabled. It is held scheduler queues and IPC, always held with local interrupts disabled. It is held
*across* a context switch and released by whichever task resumes (the hand-off *across* a context switch and released by whichever task resumes (the hand-off
rule); `task_trampoline` releases it for a freshly-spawned task. `scheduler.zig` and rule); `task_trampoline` releases it for a freshly-spawned task. `scheduler.zig` and
@@ -164,7 +164,7 @@ next lands.
the highest-priority ready task; per-core queues are a later optimisation. the highest-priority ready task; per-core queues are a later optimisation.
- **AP wake to long mode** — `arch.startSecondary` drives INIT–SIPI–SIPI (via the - **AP wake to long mode** — `arch.startSecondary` drives INIT–SIPI–SIPI (via the
LAPIC ICR) to wake each parked core one at a time. A woken core starts in 16-bit LAPIC ICR) to wake each parked core one at a time. A woken core starts in 16-bit
real mode at a low page and runs the [trampoline](../src/kernel/arch/x86_64/trampoline.s) real mode at a low page and runs the [trampoline](../system/kernel/architecture/x86_64/trampoline.s)
up through protected mode into 64-bit long mode, then lands in `smp.zig:apEntry`, up through protected mode into 64-bit long mode, then lands in `smp.zig:apEntry`,
publishes its per-CPU pointer, and reports in. Verified in QEMU with `-smp 4`: publishes its per-CPU pointer, and reports in. Verified in QEMU with `-smp 4`:
all four cores report `online`. all four cores report `online`.
+1 -1
View File
@@ -6,7 +6,7 @@ System calls (syscalls) are the bridge between your programs and the operating s
> `isr.s` does the `swapgs` + kernel-stack switch and reuses the interrupt > `isr.s` does the `swapgs` + kernel-stack switch and reuses the interrupt
> dispatcher). The `int 0x80` gate is kept alongside as a minimal test path. The > dispatcher). The `int 0x80` gate is kept alongside as a minimal test path. The
> current call set is still a placeholder — `0 = exit(code)`, `1 = ping`, > current call set is still a placeholder — `0 = exit(code)`, `1 = ping`,
> `2 = write(ptr, len)`, `3 = sleep(ms)` (see `src/kernel/process.zig`); the > `2 = write(ptr, len)`, `3 = sleep(ms)` (see `system/kernel/process.zig`); the
> handler dispatches on whether the caller is a scheduled process (its own address > handler dispatches on whether the caller is a scheduled process (its own address
> space) or a borrowed test thread. The microkernel set below (IPC_Call / > space) or a borrowed test thread. The microkernel set below (IPC_Call /
> IPC_ReplyWait / Yield) replaces it once a second user server exists. > IPC_ReplyWait / Yield) replaces it once a second user server exists.
+2 -2
View File
@@ -1,6 +1,6 @@
# SysV: the kernel's calling convention # SysV: the kernel's calling convention
Several places in danos say "the kernel is SysV" — most visibly `src/root.zig`: Several places in danos say "the kernel is SysV" — most visibly `system/danos.zig`:
```zig ```zig
pub const kernel_abi: std.builtin.CallingConvention = .{ .x86_64_sysv = .{} }; pub const kernel_abi: std.builtin.CallingConvention = .{ .x86_64_sysv = .{} };
@@ -52,7 +52,7 @@ argument arrives in **RCX**, not RDI.
danos's two binaries default to different conventions: danos's two binaries default to different conventions:
- `src/boot/efi.zig` is built for the UEFI target, so its default C convention is - `boot/efi.zig` is built for the UEFI target, so its default C convention is
Microsoft x64 (first argument → RCX). Microsoft x64 (first argument → RCX).
- The kernel is freestanding, so its convention is SysV (first argument → RDI). - The kernel is freestanding, so its convention is SysV (first argument → RDI).
+5 -5
View File
@@ -8,7 +8,7 @@ without a human staring at the screen.
There are two layers: There are two layers:
- **Host unit tests** (`zig build test`) — for pure, platform-independent logic in - **Host unit tests** (`zig build test`) — for pure, platform-independent logic in
the shared `danos` module (the handoff layout in `src/root.zig`). These compile the shared `danos` module (the handoff layout in `system/danos.zig`). These compile
for the host and run natively. for the host and run natively.
- **QEMU integration tests** (`python3 test/qemu_test.py`) — boot the real kernel - **QEMU integration tests** (`python3 test/qemu_test.py`) — boot the real kernel
and check its behaviour. This is the interesting part. and check its behaviour. This is the interesting part.
@@ -17,7 +17,7 @@ There are two layers:
The framebuffer console draws pixels, which a test can't read without The framebuffer console draws pixels, which a test can't read without
screen-scraping. So the kernel also writes everything to a **serial port** screen-scraping. So the kernel also writes everything to a **serial port**
(`src/kernel/arch/x86_64/serial.zig`, a 16550 UART on COM1). `Console.write` mirrors every (`system/kernel/architecture/x86_64/serial.zig`, a 16550 UART on COM1). `Console.write` mirrors every
byte to it, so all kernel output — boot log, memory summary, exception reports — byte to it, so all kernel output — boot log, memory summary, exception reports —
appears on serial as plain text. appears on serial as plain text.
@@ -29,7 +29,7 @@ a new architecture's UART is what makes the same tests run there.
## In-kernel test cases ## In-kernel test cases
Building with `-Dtest-case=<name>` makes the kernel, after normal bring-up, run one Building with `-Dtest-case=<name>` makes the kernel, after normal bring-up, run one
self-test from `src/kernel/tests.zig` instead of idling. Each case writes structured self-test from `system/kernel/tests.zig` instead of idling. Each case writes structured
markers to serial: markers to serial:
``` ```
@@ -108,7 +108,7 @@ firmware, boot method, serial device). The cases are architecture-neutral —
So bringing up a second architecture — an AArch64 Raspberry Pi is the motivating So bringing up a second architecture — an AArch64 Raspberry Pi is the motivating
one — means: one — means:
1. implement `src/kernel/arch/aarch64/` (CPU ops, its UART, exception vectors, page 1. implement `system/kernel/arch/aarch64/` (CPU ops, its UART, exception vectors, page
tables) behind the same `arch` interface, tables) behind the same `arch` interface,
2. add an `aarch64` entry to `ARCHES` with its `qemu-system-aarch64` invocation, 2. add an `aarch64` entry to `ARCHES` with its `qemu-system-aarch64` invocation,
@@ -118,7 +118,7 @@ architectures".
## Writing a new case ## Writing a new case
1. Add a function to `src/kernel/tests.zig` and dispatch it in `run` on its name. 1. Add a function to `system/kernel/tests.zig` and dispatch it in `run` on its name.
2. Emit `[PASS]/[FAIL]` lines and a `DANOS-TEST-RESULT:` line (non-faulting cases), 2. Emit `[PASS]/[FAIL]` lines and a `DANOS-TEST-RESULT:` line (non-faulting cases),
or trigger the condition and rely on the handler's output (faulting cases). or trigger the condition and rely on the handler's output (faulting cases).
3. Add an entry to `CASES` in `test/qemu_test.py` with the regex that proves it. 3. Add an entry to `CASES` in `test/qemu_test.py` with the regex that proves it.
+1 -1
View File
@@ -3,7 +3,7 @@
//! and Zig `std` containers share one heap. //! and Zig `std` containers share one heap.
//! //!
//! The algorithm is a straight port of the kernel's first-fit free list //! The algorithm is a straight port of the kernel's first-fit free list
//! (src/kernel/heap.zig): an address-ordered singly linked list of free blocks, //! (system/kernel/heap.zig): an address-ordered singly linked list of free blocks,
//! split on allocation and coalesced with neighbours on free. The only thing //! split on allocation and coalesced with neighbours on free. The only thing
//! that changes on this side of the system_call boundary is where memory comes from //! that changes on this side of the system_call boundary is where memory comes from
//! — `grow` asks the kernel for pages via `mmap` instead of mapping frames //! — `grow` asks the kernel for pages via `mmap` instead of mapping frames
@@ -15,7 +15,7 @@ pub const heap = @import("heap.zig");
pub const ipc = @import("ipc.zig"); pub const ipc = @import("ipc.zig");
pub const start = @import("start.zig"); pub const start = @import("start.zig");
/// The VFS wire protocol (shared with the VFS server). /// The VFS wire protocol (shared with the VFS server).
pub const vfs_protocol = @import("vfs-protocol.zig"); pub const vfs_protocol = @import("vfs-protocol");
/// POSIX-style file API: open/read/write/lseek/stat/close. /// POSIX-style file API: open/read/write/lseek/stat/close.
pub const unistd = @import("unistd.zig"); pub const unistd = @import("unistd.zig");
/// C stdio: fopen/fread/fwrite/fseek/ftell/fclose over unistd. /// C stdio: fopen/fread/fwrite/fseek/ftell/fclose over unistd.
+18 -18
View File
@@ -1,10 +1,10 @@
//! POSIX-style file API for user programs — the low level under C stdio. Files //! POSIX-style file API for user programs — the low level under C stdio. Files
//! are named objects served by the user-space VFS server (sbin/vfs.zig); each //! are named objects served by the user-space VFS server (system/services/vfs/vfs.zig); each
//! call marshals a request, IPC_Calls the VFS, and unmarshals the reply. The //! call marshals a request, IPC_Calls the VFS, and unmarshals the reply. The
//! kernel knows nothing of files or fds — the fd table lives here, per process. //! kernel knows nothing of files or fds — the fd table lives here, per process.
const std = @import("std"); const std = @import("std");
const protocol = @import("vfs-protocol.zig"); const protocol = @import("vfs-protocol");
const ipc = @import("ipc.zig"); const ipc = @import("ipc.zig");
const danos = @import("danos"); const danos = @import("danos");
@@ -42,15 +42,15 @@ const Result = struct { reply: protocol.Reply, payload: []u8 };
/// One request/reply round trip: [Request header][send payload] -> VFS -> /// One request/reply round trip: [Request header][send payload] -> VFS ->
/// [Reply header][receive payload]. The receive payload is written into `out`. /// [Reply header][receive payload]. The receive payload is written into `out`.
fn transact(req: protocol.Request, send: []const u8, out: []u8) ?Result { fn transact(request: protocol.Request, send: []const u8, out: []u8) ?Result {
const h = vfs() orelse return null; const h = vfs() orelse return null;
var message: [protocol.message_maximum]u8 = undefined; var message: [protocol.message_maximum]u8 = undefined;
@memcpy(message[0..protocol.req_size], std.mem.asBytes(&req)); @memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
const slen = @min(send.len, protocol.maximum_payload); const slen = @min(send.len, protocol.maximum_payload);
@memcpy(message[protocol.req_size..][0..slen], send[0..slen]); @memcpy(message[protocol.request_size..][0..slen], send[0..slen]);
var rbuf: [protocol.message_maximum]u8 = undefined; var rbuf: [protocol.message_maximum]u8 = undefined;
const n = ipc.call(h, message[0 .. protocol.req_size + slen], &rbuf) catch return null; const n = ipc.call(h, message[0 .. protocol.request_size + slen], &rbuf) catch return null;
if (n < protocol.reply_size) return null; if (n < protocol.reply_size) return null;
const reply = std.mem.bytesToValue(protocol.Reply, rbuf[0..protocol.reply_size]); const reply = std.mem.bytesToValue(protocol.Reply, rbuf[0..protocol.reply_size]);
const rpl = @min(n - protocol.reply_size, out.len); const rpl = @min(n - protocol.reply_size, out.len);
@@ -61,8 +61,8 @@ fn transact(req: protocol.Request, send: []const u8, out: []u8) ?Result {
/// Open (or create, with O_CREAT) `path`; returns an fd or -1. /// Open (or create, with O_CREAT) `path`; returns an fd or -1.
pub fn open(path: []const u8, flags: u32) i32 { pub fn open(path: []const u8, flags: u32) i32 {
const fd = allocFd() orelse return -1; const fd = allocFd() orelse return -1;
const req = protocol.Request{ .op = .open, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = flags }; const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(path.len), .flags = flags };
const r = transact(req, path, &.{}) orelse { const r = transact(request, path, &.{}) orelse {
fds[fd].used = false; fds[fd].used = false;
return -1; return -1;
}; };
@@ -84,8 +84,8 @@ fn fdPtr(fd: i32) ?*Fd {
pub fn read(fd: i32, buffer: []u8) isize { pub fn read(fd: i32, buffer: []u8) isize {
const f = fdPtr(fd) orelse return -1; const f = fdPtr(fd) orelse return -1;
const want: u32 = @intCast(@min(buffer.len, protocol.maximum_payload)); const want: u32 = @intCast(@min(buffer.len, protocol.maximum_payload));
const req = protocol.Request{ .op = .read, .node = f.node, .offset = f.offset, .len = want, .flags = 0 }; const request = protocol.Request{ .operation = .read, .node = f.node, .offset = f.offset, .len = want, .flags = 0 };
const r = transact(req, &.{}, buffer) orelse return -1; const r = transact(request, &.{}, buffer) orelse return -1;
if (r.reply.status != 0) return -1; if (r.reply.status != 0) return -1;
f.offset += r.reply.len; f.offset += r.reply.len;
return @intCast(r.reply.len); return @intCast(r.reply.len);
@@ -95,8 +95,8 @@ pub fn read(fd: i32, buffer: []u8) isize {
pub fn write(fd: i32, data: []const u8) isize { pub fn write(fd: i32, data: []const u8) isize {
const f = fdPtr(fd) orelse return -1; const f = fdPtr(fd) orelse return -1;
const want: u32 = @intCast(@min(data.len, protocol.maximum_payload)); const want: u32 = @intCast(@min(data.len, protocol.maximum_payload));
const req = protocol.Request{ .op = .write, .node = f.node, .offset = f.offset, .len = want, .flags = 0 }; const request = protocol.Request{ .operation = .write, .node = f.node, .offset = f.offset, .len = want, .flags = 0 };
const r = transact(req, data[0..want], &.{}) orelse return -1; const r = transact(request, data[0..want], &.{}) orelse return -1;
if (r.reply.status != 0) return -1; if (r.reply.status != 0) return -1;
f.offset += r.reply.len; f.offset += r.reply.len;
return @intCast(r.reply.len); return @intCast(r.reply.len);
@@ -110,9 +110,9 @@ pub fn lseek(fd: i32, off: i64, whence: u32) i64 {
SEEK_SET => 0, SEEK_SET => 0,
SEEK_CURRENT => @intCast(f.offset), SEEK_CURRENT => @intCast(f.offset),
SEEK_END => blk: { SEEK_END => blk: {
const req = protocol.Request{ .op = .stat, .node = f.node, .offset = 0, .len = 0, .flags = 0 }; const request = protocol.Request{ .operation = .stat, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
var sbuf: [@sizeOf(protocol.Stat)]u8 = undefined; var sbuf: [@sizeOf(protocol.Stat)]u8 = undefined;
const r = transact(req, &.{}, &sbuf) orelse return -1; const r = transact(request, &.{}, &sbuf) orelse return -1;
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.Stat)) return -1; if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.Stat)) return -1;
const st = std.mem.bytesToValue(protocol.Stat, sbuf[0..@sizeOf(protocol.Stat)]); const st = std.mem.bytesToValue(protocol.Stat, sbuf[0..@sizeOf(protocol.Stat)]);
break :blk @intCast(st.size); break :blk @intCast(st.size);
@@ -132,9 +132,9 @@ pub fn stat(path: []const u8, out: *protocol.Stat) i32 {
if (fd < 0) return -1; if (fd < 0) return -1;
defer close(fd); defer close(fd);
const f = fdPtr(fd).?; const f = fdPtr(fd).?;
const req = protocol.Request{ .op = .stat, .node = f.node, .offset = 0, .len = 0, .flags = 0 }; const request = protocol.Request{ .operation = .stat, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
var sbuf: [@sizeOf(protocol.Stat)]u8 = undefined; var sbuf: [@sizeOf(protocol.Stat)]u8 = undefined;
const r = transact(req, &.{}, &sbuf) orelse return -1; const r = transact(request, &.{}, &sbuf) orelse return -1;
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.Stat)) return -1; if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.Stat)) return -1;
out.* = std.mem.bytesToValue(protocol.Stat, sbuf[0..@sizeOf(protocol.Stat)]); out.* = std.mem.bytesToValue(protocol.Stat, sbuf[0..@sizeOf(protocol.Stat)]);
return 0; return 0;
@@ -143,7 +143,7 @@ pub fn stat(path: []const u8, out: *protocol.Stat) i32 {
/// Close an fd (best effort — tells the VFS to release the open file). /// Close an fd (best effort — tells the VFS to release the open file).
pub fn close(fd: i32) void { pub fn close(fd: i32) void {
const f = fdPtr(fd) orelse return; const f = fdPtr(fd) orelse return;
const req = protocol.Request{ .op = .close, .node = f.node, .offset = 0, .len = 0, .flags = 0 }; const request = protocol.Request{ .operation = .close, .node = f.node, .offset = 0, .len = 0, .flags = 0 };
_ = transact(req, &.{}, &.{}); _ = transact(request, &.{}, &.{});
f.used = false; f.used = false;
} }
+5 -5
View File
@@ -1,5 +1,5 @@
//! Shared definitions that form the contract between a bootloader //! Shared definitions that form the contract between a bootloader
//! (src/boot/, e.g. efi.zig built as BOOTX64.efi) and the kernel (src/kernel/main.zig). //! (boot/, e.g. efi.zig built as BOOTX64.efi) and the kernel (system/kernel/main.zig).
//! //!
//! Both binaries import this as the "danos" module, so the handoff layout is //! Both binaries import this as the "danos" module, so the handoff layout is
//! defined in exactly one place. //! defined in exactly one place.
@@ -59,7 +59,7 @@ pub const physmap_base: u64 = 0xFFFF_8800_0000_0000;
pub const kernel_virt_base: u64 = 0xFFFF_FFFF_8000_0000; pub const kernel_virt_base: u64 = 0xFFFF_FFFF_8000_0000;
/// The kernel system_call numbers — the single source of truth shared by the kernel /// The kernel system_call numbers — the single source of truth shared by the kernel
/// dispatcher (src/kernel/process.zig) and the user runtime library, so the two /// dispatcher (system/kernel/process.zig) and the user runtime library, so the two
/// can never drift. The set is deliberately microkernel-minimal: file/device I/O /// can never drift. The set is deliberately microkernel-minimal: file/device I/O
/// is not here — it lives in user-space servers reached through the IPC calls. /// is not here — it lives in user-space servers reached through the IPC calls.
/// The table grows one milestone at a time; see docs/syscall.md. /// The table grows one milestone at a time; see docs/syscall.md.
@@ -91,7 +91,7 @@ pub const SystemCall = enum(u64) {
/// event loop can't disagree about which bit means "the hardware spoke". /// event loop can't disagree about which bit means "the hardware spoke".
pub const notify_badge_bit: u64 = 1 << 63; pub const notify_badge_bit: u64 = 1 << 63;
/// A device class, mirroring src/device/device-model.zig's `DeviceClass` **in order** /// A device class, mirroring system/devices/device-model.zig's `DeviceClass` **in order**
/// (its `@intFromEnum` values cross the system_call boundary in `DeviceDescriptor.class`). /// (its `@intFromEnum` values cross the system_call boundary in `DeviceDescriptor.class`).
/// Keep the two in sync. /// Keep the two in sync.
pub const DeviceClass = enum(u32) { pub const DeviceClass = enum(u32) {
@@ -105,7 +105,7 @@ pub const DeviceClass = enum(u32) {
unknown, unknown,
}; };
/// A resource kind, mirroring src/device/device-model.zig's `ResourceKind` in order. /// A resource kind, mirroring system/devices/device-model.zig's `ResourceKind` in order.
pub const ResourceKind = enum(u32) { pub const ResourceKind = enum(u32) {
memory, memory,
io_port, io_port,
@@ -250,7 +250,7 @@ pub const BootInformation = extern struct {
init_len: u64 = 0, init_len: u64 = 0,
/// The initrd image (a bundle of extra user binaries — the VFS server and /// The initrd image (a bundle of extra user binaries — the VFS server and
/// device drivers), read off the boot volume into memory that survives the /// device drivers), read off the boot volume into memory that survives the
/// handoff, same as `init` above. 0/0 = no initrd. See src/user/protocol/initrd.zig. /// handoff, same as `init` above. 0/0 = no initrd. See system/initrd.zig.
initrd_base: u64 = 0, initrd_base: u64 = 0,
initrd_len: u64 = 0, initrd_len: u64 = 0,
}; };
@@ -64,7 +64,7 @@ fn findHpet(buffer: []device.DeviceDescriptor) ?Found {
for (buffer[0..n]) |d| { for (buffer[0..n]) |d| {
if (d.class != @intFromEnum(device.DeviceClass.timer)) continue; if (d.class != @intFromEnum(device.DeviceClass.timer)) continue;
// Skip comparator children a bus driver may have published below the block // Skip comparator children a bus driver may have published below the block
// (see sbin/busd.zig) — we want the register block itself. // (see system/drivers/busd/busd.zig) — we want the register block itself.
if (d.parent != device.no_parent) continue; if (d.parent != device.no_parent) continue;
var mmio: ?u64 = null; var mmio: ?u64 = null;
var irq: ?u64 = null; var irq: ?u64 = null;
@@ -373,7 +373,7 @@ pub fn irqRouteRaw(n: u32) u32 {
return ioapic.entryLow(n); return ioapic.entryLow(n);
} }
// --- device-IRQ plumbing, for src/kernel/irq.zig ----------------------------- // --- device-IRQ plumbing, for system/kernel/irq.zig -----------------------------
// //
// The generic IRQ layer speaks GSIs and vectors; everything below hides the fact // The generic IRQ layer speaks GSIs and vectors; everything below hides the fact
// that on x86_64 those mean "I/O APIC redirection entry" and "IDT gate". The // that on x86_64 those mean "I/O APIC redirection entry" and "IDT gate". The
@@ -4,7 +4,7 @@
//! //!
//! `init` maps the I/O APIC and **masks every input** — the correct quiescent state //! `init` maps the I/O APIC and **masks every input** — the correct quiescent state
//! on a legacy-free machine. Lines are then unmasked one at a time, as user-space //! on a legacy-free machine. Lines are then unmasked one at a time, as user-space
//! drivers bind them (`routeGsi`/`unmaskGsi`, driven by src/kernel/irq.zig). //! drivers bind them (`routeGsi`/`unmaskGsi`, driven by system/kernel/irq.zig).
//! //!
//! Two entry points, for two kinds of caller. `routeIrq` takes a legacy **ISA IRQ** //! Two entry points, for two kinds of caller. `routeIrq` takes a legacy **ISA IRQ**
//! and resolves it through the MADT overrides — for in-kernel use, and still without //! and resolves it through the MADT overrides — for in-kernel use, and still without
@@ -96,7 +96,7 @@ pub fn routeIrq(irq: u8, vector: u8, apic_id: u8) void {
// `routeIrq` above takes an *ISA IRQ* and resolves it through the MADT overrides. // `routeIrq` above takes an *ISA IRQ* and resolves it through the MADT overrides.
// A driver-bound interrupt is already a **GSI** (the device told us so, e.g. the // A driver-bound interrupt is already a **GSI** (the device told us so, e.g. the
// HPET's `Tn_INT_ROUTE_CAP`), so it needs no override lookup — just the redirection // HPET's `Tn_INT_ROUTE_CAP`), so it needs no override lookup — just the redirection
// entry. These three are what `src/kernel/irq.zig` drives. // entry. These three are what `system/kernel/irq.zig` drives.
// //
// Callers must serialise: the I/O APIC is reached through an index/data register // Callers must serialise: the I/O APIC is reached through an index/data register
// pair, so two cores interleaving `registerWrite` would corrupt each other. The kernel // pair, so two cores interleaving `registerWrite` would corrupt each other. The kernel
@@ -48,9 +48,9 @@ pub const ENOMEM: i64 = 6; // out of memory
/// A badge with this bit set is an asynchronous notification (e.g. an IRQ), not a /// A badge with this bit set is an asynchronous notification (e.g. an IRQ), not a
/// message from a client — there is no reply owed. The low bits carry the source /// message from a client — there is no reply owed. The low bits carry the source
/// (a GSI for IRQs). Posted by `notifyFromIsr`, from the ISR in src/kernel/irq.zig; /// (a GSI for IRQs). Posted by `notifyFromIsr`, from the ISR in system/kernel/irq.zig;
/// the message path uses a plain task-id badge with this bit clear. Defined in the /// the message path uses a plain task-id badge with this bit clear. Defined in the
/// shared contract (src/root.zig), because ring 3 has to test the same bit. /// shared contract (system/danos.zig), because ring 3 has to test the same bit.
pub const notify_badge_bit: u64 = danos.notify_badge_bit; pub const notify_badge_bit: u64 = danos.notify_badge_bit;
/// End of the user (low) canonical half — user buffers must lie below it. /// End of the user (low) canonical half — user buffers must lie below it.
+1 -1
View File
@@ -22,7 +22,7 @@
//! //!
//! Binding is capability-gated exactly like `mmio_map`: the caller must have //! Binding is capability-gated exactly like `mmio_map`: the caller must have
//! `device_claim`ed the device, and the GSI must come from one of that device's `irq` //! `device_claim`ed the device, and the GSI must come from one of that device's `irq`
//! resources in the discovered device table (src/kernel/device-service.zig). A driver can //! resources in the discovered device table (system/kernel/device-service.zig). A driver can
//! therefore never bind an interrupt it doesn't own — a raw-GSI system_call would let //! therefore never bind an interrupt it doesn't own — a raw-GSI system_call would let
//! any process steal the keyboard's line. //! any process steal the keyboard's line.
//! //!
@@ -291,7 +291,7 @@ fn ownedGsi(t: *scheduler.Task, device_id: u64, resource_index: u64) ?u32 {
/// irq_bind(device_id, resource_index, endpoint) -> 0/-1: deliver that device's IRQ to the /// irq_bind(device_id, resource_index, endpoint) -> 0/-1: deliver that device's IRQ to the
/// endpoint as an asynchronous IPC notification. The driver then blocks in /// endpoint as an asynchronous IPC notification. The driver then blocks in
/// IPC_ReplyWait and is woken by the ISR; see src/kernel/irq.zig for the cycle. /// IPC_ReplyWait and is woken by the ISR; see system/kernel/irq.zig for the cycle.
fn systemIrqBind(state: *architecture.CpuState) void { fn systemIrqBind(state: *architecture.CpuState) void {
const t = scheduler.current(); const t = scheduler.current();
if (t.aspace == 0) return fail(state); if (t.aspace == 0) return fail(state);
@@ -1,7 +1,7 @@
//! /sbin/init — the first user-space program, PID 1. Built as its own //! /sbin/init — the first user-space program, PID 1. Built as its own
//! freestanding binary (see build.zig), shipped on the boot volume at sbin/init, //! freestanding binary (see build.zig), shipped on the boot volume at sbin/init,
//! loaded by the bootloader, and started in ring 3 as a scheduled process by the //! loaded by the bootloader, and started in ring 3 as a scheduled process by the
//! kernel (src/kernel/process.zig). It links against the shared user runtime //! kernel (system/kernel/process.zig). It links against the shared user runtime
//! library `runtime` and talks to the kernel only through `runtime`'s system_call wrappers. //! library `runtime` and talks to the kernel only through `runtime`'s system_call wrappers.
//! //!
//! Today it proves the C-convention heap works, then settles into a heartbeat: //! Today it proves the C-convention heap works, then settles into a heartbeat:
@@ -5,9 +5,9 @@
//! a Stat). Everything fits in one IPC message (<= ipc MESSAGE_MAXIMUM = 256 bytes). //! a Stat). Everything fits in one IPC message (<= ipc MESSAGE_MAXIMUM = 256 bytes).
//! //!
//! This is user-space only — the kernel knows nothing of files or paths; it only //! This is user-space only — the kernel knows nothing of files or paths; it only
//! moves the bytes. Shared by lib/unistd.zig (client) and sbin/vfs.zig (server). //! moves the bytes. Shared by library/runtime/unistd.zig (client) and system/services/vfs/vfs.zig (server).
pub const Op = enum(u32) { pub const Operation = enum(u32) {
open, // open(path) -> node id open, // open(path) -> node id
close, // close(node) close, // close(node)
read, // read(node, offset, len) -> bytes read, // read(node, offset, len) -> bytes
@@ -19,7 +19,7 @@ pub const Op = enum(u32) {
/// for `open` the path is the payload and `len` is its length. `offset`/`len` /// for `open` the path is the payload and `len` is its length. `offset`/`len`
/// carry the read/write position and count. /// carry the read/write position and count.
pub const Request = extern struct { pub const Request = extern struct {
op: Op, operation: Operation,
node: u64, node: u64,
offset: u64, offset: u64,
len: u32, len: u32,
@@ -31,23 +31,23 @@ pub const Request = extern struct {
/// Stat size). /// Stat size).
pub const Reply = extern struct { pub const Reply = extern struct {
status: i32, status: i32,
_pad: u32 = 0, _padding: u32 = 0,
node: u64 = 0, node: u64 = 0,
len: u32 = 0, len: u32 = 0,
_pad2: u32 = 0, _padding2: u32 = 0,
}; };
pub const Stat = extern struct { pub const Stat = extern struct {
size: u64, size: u64,
kind: u32, kind: u32,
_pad: u32 = 0, _padding: u32 = 0,
}; };
pub const message_maximum: usize = 256; pub const message_maximum: usize = 256;
pub const req_size: usize = @sizeOf(Request); pub const request_size: usize = @sizeOf(Request);
pub const reply_size: usize = @sizeOf(Reply); pub const reply_size: usize = @sizeOf(Reply);
/// Largest inline payload that still fits one IPC message alongside a header. /// Largest inline payload that still fits one IPC message alongside a header.
pub const maximum_payload: usize = message_maximum - req_size; pub const maximum_payload: usize = message_maximum - request_size;
/// Open flags. /// Open flags.
pub const O_CREAT: u32 = 1; pub const O_CREAT: u32 = 1;
+13 -13
View File
@@ -67,13 +67,13 @@ fn fail(out: []u8) usize {
/// Handle one request; write the reply into `out`, return its length. /// Handle one request; write the reply into `out`, return its length.
fn handle(message: []const u8, out: []u8) usize { fn handle(message: []const u8, out: []u8) usize {
if (message.len < protocol.req_size) return fail(out); if (message.len < protocol.request_size) return fail(out);
const req = std.mem.bytesToValue(protocol.Request, message[0..protocol.req_size]); const request = std.mem.bytesToValue(protocol.Request, message[0..protocol.request_size]);
const payload = message[protocol.req_size..]; const payload = message[protocol.request_size..];
switch (req.op) { switch (request.operation) {
.open => { .open => {
const name = payload[0..@min(payload.len, req.len)]; const name = payload[0..@min(payload.len, request.len)];
const ni = findNode(name) orelse createNode(name) orelse return fail(out); const ni = findNode(name) orelse createNode(name) orelse return fail(out);
for (&opens, 0..) |*o, i| { for (&opens, 0..) |*o, i| {
if (!o.used) { if (!o.used) {
@@ -84,30 +84,30 @@ fn handle(message: []const u8, out: []u8) usize {
return fail(out); return fail(out);
}, },
.read => { .read => {
const of = openAt(req.node) orelse return fail(out); const of = openAt(request.node) orelse return fail(out);
const nd = &nodes[of.node]; const nd = &nodes[of.node];
const off: usize = @intCast(req.offset); const off: usize = @intCast(request.offset);
if (off >= nd.size) return writeReply(out, .{ .status = 0, .len = 0 }, &.{}); // EOF if (off >= nd.size) return writeReply(out, .{ .status = 0, .len = 0 }, &.{}); // EOF
const n = @min(@min(nd.size - off, req.len), protocol.maximum_payload); const n = @min(@min(nd.size - off, request.len), protocol.maximum_payload);
return writeReply(out, .{ .status = 0, .len = @intCast(n) }, nd.data[off .. off + n]); return writeReply(out, .{ .status = 0, .len = @intCast(n) }, nd.data[off .. off + n]);
}, },
.write => { .write => {
const of = openAt(req.node) orelse return fail(out); const of = openAt(request.node) orelse return fail(out);
const nd = &nodes[of.node]; const nd = &nodes[of.node];
const off: usize = @intCast(req.offset); const off: usize = @intCast(request.offset);
if (off > nd.data.len) return fail(out); if (off > nd.data.len) return fail(out);
const n = @min(@min(payload.len, req.len), nd.data.len - off); const n = @min(@min(payload.len, request.len), nd.data.len - off);
@memcpy(nd.data[off .. off + n], payload[0..n]); @memcpy(nd.data[off .. off + n], payload[0..n]);
if (off + n > nd.size) nd.size = off + n; if (off + n > nd.size) nd.size = off + n;
return writeReply(out, .{ .status = 0, .len = @intCast(n) }, &.{}); return writeReply(out, .{ .status = 0, .len = @intCast(n) }, &.{});
}, },
.stat => { .stat => {
const of = openAt(req.node) orelse return fail(out); const of = openAt(request.node) orelse return fail(out);
const st = protocol.Stat{ .size = nodes[of.node].size, .kind = 0 }; const st = protocol.Stat{ .size = nodes[of.node].size, .kind = 0 };
return writeReply(out, .{ .status = 0, .len = @sizeOf(protocol.Stat) }, std.mem.asBytes(&st)); return writeReply(out, .{ .status = 0, .len = @sizeOf(protocol.Stat) }, std.mem.asBytes(&st));
}, },
.close => { .close => {
if (req.node < opens.len) opens[@intCast(req.node)].used = false; if (request.node < opens.len) opens[@intCast(request.node)].used = false;
return writeReply(out, .{ .status = 0 }, &.{}); return writeReply(out, .{ .status = 0 }, &.{});
}, },
} }