Compare commits
25
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
52d6e372fd | ||
|
|
f023f1cfd6 | ||
|
|
b9cec7d1be | ||
|
|
6271278d4d | ||
|
|
37326c7664 | ||
|
|
23bcd77c58 | ||
|
|
dded46726b | ||
|
|
60f32ee9ff | ||
|
|
3e69712b97 | ||
|
|
11f567ee20 | ||
|
|
3a155cdc7d | ||
|
|
5b874fc756 | ||
|
|
7d540c4b2f | ||
|
|
ac2d102878 | ||
|
|
3c9475e33a | ||
|
|
44122bd44d | ||
|
|
9ef22d6e55 | ||
|
|
07c901c18c | ||
|
|
25cc4d610e | ||
|
|
f90dc6c121 | ||
|
|
1d1234c963 | ||
|
|
a7f0c1a450 | ||
|
|
f86f2987d5 | ||
|
|
794a8b5782 | ||
|
|
ea470afe84 |
@@ -52,7 +52,8 @@ zig build
|
||||
Produces a FHS-shaped `zig-out/` that *is* the danos filesystem and the boot volume:
|
||||
the UEFI bootloader at `zig-out/EFI/BOOT/BOOTX64.efi`, the kernel at
|
||||
`zig-out/system/kernel`, init at `zig-out/system/services/init`, drivers under
|
||||
`zig-out/system/drivers/`, and the initial-ramdisk at `zig-out/boot/`.
|
||||
`zig-out/system/drivers/`, the test fixtures under `zig-out/test/system/services/`,
|
||||
and the initial-ramdisk at `zig-out/boot/`.
|
||||
|
||||
## Release media
|
||||
|
||||
@@ -62,7 +63,7 @@ zig build release-x86-64
|
||||
|
||||
Produces `zig-out/danos-x86-64.iso`, a hybrid ISO that boots flashed raw to a
|
||||
USB stick (balenaEtcher, dd) or burned to optical media — see
|
||||
[docs/release-iso.md](docs/release-iso.md). `zig build check-iso-image`
|
||||
[docs/release-iso.md](docs/os-development-guide/release-iso.md). `zig build check-iso-image`
|
||||
validates it without booting.
|
||||
|
||||
## Run
|
||||
|
||||
+22
-14
@@ -16,11 +16,12 @@ const MemoryMapSlice = uefi.tables.MemoryMapSlice;
|
||||
/// The kernel image: /system/kernel.
|
||||
const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("system\\kernel");
|
||||
|
||||
/// The user binaries: everything under /system except the kernel itself. The
|
||||
/// loader walks this tree and packs it into the in-RAM initial_ramdisk image —
|
||||
/// the volume's file structure is the single source of truth (no packed image
|
||||
/// artifact on disk).
|
||||
/// The user binaries: everything under /system except the kernel itself, plus
|
||||
/// the test fixtures under /test. The loader walks both trees and packs them
|
||||
/// into the in-RAM initial_ramdisk image — the volume's file structure is the
|
||||
/// single source of truth (no packed image artifact on disk).
|
||||
const system_directory_name = std.unicode.utf8ToUtf16LeStringLiteral("system");
|
||||
const test_directory_name = std.unicode.utf8ToUtf16LeStringLiteral("test");
|
||||
|
||||
/// Physical page size, and the sentinel UEFI uses to seek to end-of-file.
|
||||
const page_size = 4096;
|
||||
@@ -364,16 +365,17 @@ fn handoff(cr3: u64, entry: usize, boot_information: *const BootInformation) nor
|
||||
unreachable;
|
||||
}
|
||||
|
||||
// --- the /system tree -> initial_ramdisk ------------------------------------
|
||||
// --- the /system and /test trees -> initial_ramdisk --------------------------
|
||||
|
||||
/// Cap on bundled binaries. Generous: the tree carries ~30 today.
|
||||
const maximum_bundled = 64;
|
||||
|
||||
/// How deep the walk goes below /system ("/system/services/x" is depth 1).
|
||||
/// How deep the walk goes below a tree root ("/system/services/x" is depth 1,
|
||||
/// "/test/system/services/x" is depth 2).
|
||||
const maximum_tree_depth = 3;
|
||||
|
||||
/// One binary discovered under /system: its FHS path (UTF-8, '/'-separated,
|
||||
/// NUL-free) and its contents in a transient pool buffer.
|
||||
/// One binary discovered under a walked tree: its FHS path (UTF-8,
|
||||
/// '/'-separated, NUL-free) and its contents in a transient pool buffer.
|
||||
const Bundled = struct {
|
||||
path: [initial_ramdisk.maximum_name]u8,
|
||||
path_len: usize,
|
||||
@@ -389,10 +391,10 @@ const Bundled = struct {
|
||||
/// 1. /system/manifest (written by the build): each listed path is opened BY
|
||||
/// NAME — the case-insensitive lookup every firmware FAT driver gets
|
||||
/// right, and the only file access the pre-tree loader ever used.
|
||||
/// 2. No manifest: ENUMERATE the /system tree. Portable in principle, but
|
||||
/// firmware differs in what names enumeration returns (bare 8.3 entries
|
||||
/// come back uppercase on some drivers), so this is the fallback for
|
||||
/// hand-assembled sticks, not the primary path.
|
||||
/// 2. No manifest: ENUMERATE the /system and /test trees. Portable in
|
||||
/// principle, but firmware differs in what names enumeration returns
|
||||
/// (bare 8.3 entries come back uppercase on some drivers), so this is
|
||||
/// the fallback for hand-assembled sticks, not the primary path.
|
||||
fn loadSystemTree(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
|
||||
return error.NoLoadedImage;
|
||||
@@ -426,6 +428,11 @@ fn loadSystemTree(bs: *uefi.tables.BootServices, boot_information: *BootInformat
|
||||
const system_directory = try root.open(system_directory_name, .read, .{});
|
||||
defer _ = system_directory.close() catch {};
|
||||
try walkDirectory(bs, system_directory, "/system", 0, &list, &count);
|
||||
// The /test tree is optional: a stick without fixtures still boots.
|
||||
if (root.open(test_directory_name, .read, .{})) |test_directory| {
|
||||
defer _ = test_directory.close() catch {};
|
||||
try walkDirectory(bs, test_directory, "/test", 0, &list, &count);
|
||||
} else |_| {}
|
||||
}
|
||||
if (count == 0) return error.NoBinaries;
|
||||
|
||||
@@ -452,7 +459,7 @@ fn loadSystemTree(bs: *uefi.tables.BootServices, boot_information: *BootInformat
|
||||
|
||||
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
||||
boot_information.initial_ramdisk_len = total;
|
||||
log("EFI: /system tree loaded, starting the kernel\r\n");
|
||||
log("EFI: boot tree loaded, starting the kernel\r\n");
|
||||
}
|
||||
|
||||
/// The boot capsule: the bundled binaries as one v2 initial_ramdisk image.
|
||||
@@ -522,7 +529,8 @@ fn loadByManifest(bs: *uefi.tables.BootServices, root: *uefi.protocol.File, list
|
||||
|
||||
/// Recursively collect the regular files below `directory` into `list`. Top-level
|
||||
/// files (depth 0) are skipped: the only one is /system/kernel, which loadKernel
|
||||
/// has already consumed and which is not a spawnable user binary.
|
||||
/// has already consumed and which is not a spawnable user binary (/test has no
|
||||
/// top-level files, so the skip is a no-op there).
|
||||
fn walkDirectory(
|
||||
bs: *uefi.tables.BootServices,
|
||||
directory: *uefi.protocol.File,
|
||||
|
||||
@@ -56,70 +56,63 @@ fn timestamp(b: *std.Build) []const u8 {
|
||||
/// are authoritative — the kernel's W^X user-ELF loader requires exact perms.
|
||||
///
|
||||
/// The compilation root is not the program's own file but the shared shim
|
||||
/// library/runtime/root.zig, which supplies the root declarations (`main`
|
||||
/// library/kernel/root.zig, which supplies the root declarations (`main`
|
||||
/// re-export, panic handler, `_start` pull) so a program only defines
|
||||
/// `pub fn main`. The program's file becomes the `program` module the shim
|
||||
/// imports; reach it through `programModule` to add per-binary imports.
|
||||
fn addUserBinary(
|
||||
b: *std.Build,
|
||||
target: std.Build.ResolvedTarget,
|
||||
runtime_module: *std.Build.Module,
|
||||
mmio_module: *std.Build.Module,
|
||||
xkeyboard_config_module: *std.Build.Module,
|
||||
acpi_ids_module: *std.Build.Module,
|
||||
default_imports: []const std.Build.Module.Import,
|
||||
name: []const u8,
|
||||
root: []const u8,
|
||||
) *std.Build.Step.Compile {
|
||||
return addUserBinaryImpl(b, target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, name, root, false);
|
||||
return addUserBinaryImpl(b, target, default_imports, name, root, false);
|
||||
}
|
||||
|
||||
/// As `addUserBinary`, but built multi-threaded (`single_threaded = false`) so real
|
||||
/// atomics/TLS work — required before a binary may call `runtime.Thread.spawn`
|
||||
/// atomics/TLS work — required before a binary may call `Thread.spawn`
|
||||
/// (docs/threading.md). Threads are a deliberate per-binary opt-in.
|
||||
fn addThreadedUserBinary(
|
||||
b: *std.Build,
|
||||
target: std.Build.ResolvedTarget,
|
||||
runtime_module: *std.Build.Module,
|
||||
mmio_module: *std.Build.Module,
|
||||
xkeyboard_config_module: *std.Build.Module,
|
||||
acpi_ids_module: *std.Build.Module,
|
||||
default_imports: []const std.Build.Module.Import,
|
||||
name: []const u8,
|
||||
root: []const u8,
|
||||
) *std.Build.Step.Compile {
|
||||
return addUserBinaryImpl(b, target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, name, root, true);
|
||||
return addUserBinaryImpl(b, target, default_imports, name, root, true);
|
||||
}
|
||||
|
||||
/// The module registered under `name` in `imports` — the root shim reaches the
|
||||
/// couple of concern modules it needs (start, logging) out of the default set.
|
||||
fn findImport(imports: []const std.Build.Module.Import, name: []const u8) *std.Build.Module {
|
||||
for (imports) |import| {
|
||||
if (std.mem.eql(u8, import.name, name)) return import.module;
|
||||
}
|
||||
@panic("default_imports is missing a module the root shim needs");
|
||||
}
|
||||
|
||||
fn addUserBinaryImpl(
|
||||
b: *std.Build,
|
||||
target: std.Build.ResolvedTarget,
|
||||
runtime_module: *std.Build.Module,
|
||||
mmio_module: *std.Build.Module,
|
||||
xkeyboard_config_module: *std.Build.Module,
|
||||
acpi_ids_module: *std.Build.Module,
|
||||
default_imports: []const std.Build.Module.Import,
|
||||
name: []const u8,
|
||||
root: []const u8,
|
||||
threaded: bool,
|
||||
) *std.Build.Step.Compile {
|
||||
// Settings (target, optimize, code model, ...) live on the root module only;
|
||||
// the program and runtime modules leave theirs null and inherit them.
|
||||
// Every user binary gets the same default set of importable modules — the library/kernel
|
||||
// concern modules (ipc, memory, process, time, logging, file-system, ...), the device/
|
||||
// service clients (driver, block, display, input), mmio, acpi-ids, and xkeyboard-config.
|
||||
// Per-binary extras go through programModule(exe).addImport. Settings (target, optimize,
|
||||
// code model, ...) live on the root module only; the program module inherits them.
|
||||
const program_module = b.createModule(.{
|
||||
.root_source_file = b.path(root),
|
||||
.imports = &.{
|
||||
.{ .name = "runtime", .module = runtime_module },
|
||||
// Typed volatile MMIO + memory barriers, for drivers. See library/mmio/.
|
||||
.{ .name = "mmio", .module = mmio_module },
|
||||
// Keyboard layouts (keycode + modifiers -> keysym/character), available
|
||||
// to any program that wants it. See library/xkeyboard-config/.
|
||||
.{ .name = "xkeyboard-config", .module = xkeyboard_config_module },
|
||||
// ACPI/PnP hardware-ID registry, so drivers name devices
|
||||
// (HardwareId.ps2_keyboard) instead of magic "_HID" strings.
|
||||
.{ .name = "acpi-ids", .module = acpi_ids_module },
|
||||
},
|
||||
.imports = default_imports,
|
||||
});
|
||||
const exe = b.addExecutable(.{
|
||||
.name = name,
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("library/runtime/root.zig"),
|
||||
.root_source_file = b.path("library/kernel/root.zig"),
|
||||
.target = target,
|
||||
.optimize = .ReleaseSmall,
|
||||
.code_model = .large,
|
||||
@@ -127,13 +120,16 @@ fn addUserBinaryImpl(
|
||||
.sanitize_c = .off,
|
||||
.stack_check = false,
|
||||
.stack_protector = false,
|
||||
// The root shim itself imports only start (_start + panic) and logging
|
||||
// (std_options); the program's own file reaches the full default set.
|
||||
.imports = &.{
|
||||
.{ .name = "runtime", .module = runtime_module },
|
||||
.{ .name = "start", .module = findImport(default_imports, "start") },
|
||||
.{ .name = "logging", .module = findImport(default_imports, "logging") },
|
||||
.{ .name = "program", .module = program_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
exe.setLinkerScript(b.path("library/runtime/user.ld"));
|
||||
exe.setLinkerScript(b.path("library/kernel/user.ld"));
|
||||
exe.entry = .{ .symbol_name = "_start" };
|
||||
exe.image_base = 0x7000_0000_0000;
|
||||
exe.use_llvm = true;
|
||||
@@ -278,14 +274,14 @@ pub fn build(b: *std.Build) void {
|
||||
});
|
||||
// The devices sub-project's public interface (the flat wire types), exposed as
|
||||
// its own module like vfs-protocol — importable by user space, unlike the
|
||||
// kernel-internal device model it also feeds (system/devices/device-model.zig).
|
||||
// kernel-internal device model it also feeds (system/kernel/device-model.zig).
|
||||
const device_abi_module = b.addModule("device-abi", .{
|
||||
.root_source_file = b.path("system/devices/device-abi.zig"),
|
||||
.root_source_file = b.path("library/device/model/device-abi.zig"),
|
||||
});
|
||||
// PCI class-code decoding (class/subclass/prog-IF -> names). Pure reference data,
|
||||
// shared by kernel discovery (the device-tree dump) and any user-space PCI tool.
|
||||
const pci_class_module = b.addModule("pci-class", .{
|
||||
.root_source_file = b.path("system/devices/pci-class.zig"),
|
||||
.root_source_file = b.path("library/device/pci/pci-class.zig"),
|
||||
});
|
||||
// ACPI/PnP hardware-ID (_HID) names — the flat analog of pci-class for acpi_device
|
||||
// nodes. Also shared reference data.
|
||||
@@ -293,11 +289,11 @@ pub fn build(b: *std.Build) void {
|
||||
// same parser the kernel does (docs/discovery.md — the shared AML module).
|
||||
// Pure Zig, no kernel imports — one source, two builds.
|
||||
const aml_module = b.addModule("aml", .{
|
||||
.root_source_file = b.path("system/devices/aml/aml.zig"),
|
||||
.root_source_file = b.path("library/device/acpi/aml/aml.zig"),
|
||||
});
|
||||
|
||||
const acpi_ids_module = b.addModule("acpi-ids", .{
|
||||
.root_source_file = b.path("system/devices/acpi-ids.zig"),
|
||||
.root_source_file = b.path("library/device/acpi/acpi-ids.zig"),
|
||||
});
|
||||
|
||||
// The USB device-framework wire ABI (chapter-9 set-up packets, standard +
|
||||
@@ -305,21 +301,21 @@ pub fn build(b: *std.Build) void {
|
||||
// reference the xHCI bus driver, the USB class drivers, and the device
|
||||
// manager's identity matcher all share. Pure data, like pci-class/acpi-ids.
|
||||
const usb_abi_module = b.addModule("usb-abi", .{
|
||||
.root_source_file = b.path("system/devices/usb-abi.zig"),
|
||||
.root_source_file = b.path("library/device/usb/usb-abi.zig"),
|
||||
});
|
||||
const usb_ids_module = b.addModule("usb-ids", .{
|
||||
.root_source_file = b.path("system/devices/usb-ids.zig"),
|
||||
.root_source_file = b.path("library/device/usb/usb-ids.zig"),
|
||||
});
|
||||
// The USB transfer protocol: what a USB class driver says to the xHCI bus
|
||||
// driver to drive its device (open / control / interrupt / bulk). A protocol
|
||||
// module like vfs-protocol, shared by the bus driver and every class driver.
|
||||
const usb_transfer_protocol_module = b.addModule("usb-transfer-protocol", .{
|
||||
.root_source_file = b.path("system/drivers/usb-xhci-bus/usb-transfer-protocol.zig"),
|
||||
.root_source_file = b.path("library/protocol/usb-transfer/usb-transfer-protocol.zig"),
|
||||
});
|
||||
// The block-device protocol: read/write of fixed-size blocks, spoken between a
|
||||
// filesystem and a block driver (usb-storage). A protocol module like the rest.
|
||||
const block_protocol_module = b.addModule("block-protocol", .{
|
||||
.root_source_file = b.path("system/services/block/protocol.zig"),
|
||||
.root_source_file = b.path("library/protocol/block/block-protocol.zig"),
|
||||
});
|
||||
|
||||
// Kernel tunables (maximum_cpus, stack sizes, tick rate). A dependency-free module of
|
||||
@@ -351,15 +347,13 @@ pub fn build(b: *std.Build) void {
|
||||
// "platform" and asks it to enumerate hardware into a backend-neutral device
|
||||
// tree, never naming ACPI (or, later, device-tree) — the same discipline the
|
||||
// architecture module applies to CPU code. The backend is selected at runtime from
|
||||
// the boot handoff (see system/devices/platform.zig).
|
||||
// the boot handoff (see system/kernel/platform.zig).
|
||||
const platform_module = b.addModule("platform", .{
|
||||
.root_source_file = b.path("system/devices/platform.zig"),
|
||||
.root_source_file = b.path("system/kernel/platform.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "boot-handoff", .module = boot_handoff_module }, // BootInformation (carries the ACPI RSDP), physicalToVirtual
|
||||
.{ .name = "abi", .module = abi_module }, // acpi.zig works in page_size units
|
||||
.{ .name = "device-abi", .module = device_abi_module }, // device-model's DeviceClass/ResourceKind live here
|
||||
.{ .name = "pci-class", .module = pci_class_module }, // decode PCI class codes in the device dump
|
||||
.{ .name = "acpi-ids", .module = acpi_ids_module }, // decode ACPI _HID names in the device dump
|
||||
.{ .name = "parameters", .module = parameters_module }, // maximum_cpus (the discovery pool)
|
||||
},
|
||||
});
|
||||
@@ -370,14 +364,14 @@ pub fn build(b: *std.Build) void {
|
||||
// is the first "protocol module" (see docs/driver-model.md); usb/block will
|
||||
// expose theirs the same way.
|
||||
const vfs_protocol_module = b.addModule("vfs-protocol", .{
|
||||
.root_source_file = b.path("system/vfs-protocol.zig"),
|
||||
.root_source_file = b.path("library/protocol/vfs/vfs-protocol.zig"),
|
||||
});
|
||||
|
||||
// The input wire protocol: the input service's public interface, exposed as its own
|
||||
// module the same way vfs-protocol is. Shared by the input service, the runtime's
|
||||
// `input` helper (subscribe/publish), and every source and subscriber.
|
||||
const input_protocol_module = b.addModule("input-protocol", .{
|
||||
.root_source_file = b.path("system/services/input/protocol.zig"),
|
||||
.root_source_file = b.path("library/protocol/input/input-protocol.zig"),
|
||||
});
|
||||
|
||||
// The danos-native user-space runtime: system_call wrappers, the C-convention
|
||||
@@ -389,53 +383,165 @@ pub fn build(b: *std.Build) void {
|
||||
// types its `device` helper wraps, and re-exports `vfs-protocol` for the VFS
|
||||
// server. It never touches `boot-handoff` — user space has no business with the
|
||||
// loader↔kernel handoff.
|
||||
const runtime_module = b.addModule("runtime", .{
|
||||
.root_source_file = b.path("library/runtime/runtime.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
.{ .name = "device-abi", .module = device_abi_module },
|
||||
.{ .name = "vfs-protocol", .module = vfs_protocol_module },
|
||||
.{ .name = "input-protocol", .module = input_protocol_module },
|
||||
},
|
||||
});
|
||||
|
||||
// The device-manager protocol: hello + (M18.2) tree reports, exposed as its
|
||||
// own module like the other protocol modules. Imported through the runtime.
|
||||
// Wire-protocol modules the kernel-library client wrappers (device-manager/block/display)
|
||||
// and the services speak. The `runtime` module itself is defined below, after the
|
||||
// library/kernel concern modules it shims over.
|
||||
const device_manager_protocol_module = b.addModule("device-manager-protocol", .{
|
||||
.root_source_file = b.path("system/services/device-manager/device-manager-protocol.zig"),
|
||||
.root_source_file = b.path("library/protocol/device-manager/device-manager-protocol.zig"),
|
||||
});
|
||||
runtime_module.addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
// The USB transfer protocol, so runtime.usb (the class-driver client) can speak
|
||||
// it, the way runtime.input speaks the input protocol.
|
||||
runtime_module.addImport("usb-transfer-protocol", usb_transfer_protocol_module);
|
||||
// The block protocol, so runtime.block (the block-device client) can speak it.
|
||||
runtime_module.addImport("block-protocol", block_protocol_module);
|
||||
|
||||
// The display protocol, so runtime.display (the compositor client) and the display
|
||||
// service both speak it through the runtime, like the other protocol modules.
|
||||
const display_protocol_module = b.addModule("display-protocol", .{
|
||||
.root_source_file = b.path("system/services/display/protocol.zig"),
|
||||
.root_source_file = b.path("library/protocol/display/display-protocol.zig"),
|
||||
});
|
||||
runtime_module.addImport("display-protocol", display_protocol_module);
|
||||
|
||||
// The scanout protocol: the compositor's outbound present channel to a native scanout
|
||||
// driver (virtio-gpu), separate from the client-facing display protocol (docs/display-v2.md).
|
||||
// No runtime client speaks it — imported directly by the compositor and the scanout driver.
|
||||
const scanout_protocol_module = b.addModule("scanout-protocol", .{
|
||||
.root_source_file = b.path("system/services/display/scanout-protocol.zig"),
|
||||
.root_source_file = b.path("library/protocol/scanout/scanout-protocol.zig"),
|
||||
});
|
||||
runtime_module.addImport("scanout-protocol", scanout_protocol_module);
|
||||
|
||||
// The power protocol: system power's domain-named surface (docs/power.md).
|
||||
// The power protocol: system power's domain-named surface (docs/power.md). No runtime
|
||||
// client speaks it — imported directly by init and the acpi discovery service.
|
||||
const power_protocol_module = b.addModule("power-protocol", .{
|
||||
.root_source_file = b.path("system/services/power/protocol.zig"),
|
||||
.root_source_file = b.path("library/protocol/power/power-protocol.zig"),
|
||||
});
|
||||
runtime_module.addImport("power-protocol", power_protocol_module);
|
||||
|
||||
// Typed volatile MMIO register access + memory-ordering barriers, for drivers on
|
||||
// top of an mmio_map grant. Depends only on `builtin` (arch-conditional barriers);
|
||||
// no target set, so it inherits each driver's. See library/mmio/mmio.zig.
|
||||
// no target set, so it inherits each driver's. See library/device/mmio/mmio.zig.
|
||||
const mmio_module = b.addModule("mmio", .{
|
||||
.root_source_file = b.path("library/mmio/mmio.zig"),
|
||||
.root_source_file = b.path("library/device/mmio/mmio.zig"),
|
||||
});
|
||||
|
||||
// --- library/kernel: the userspace private-ABI library (kernel32-style), split by
|
||||
// concern into directly-importable modules. `system.zig` (the old dumping ground) and
|
||||
// `runtime.zig` (the old aggregator) are compatibility shims re-exporting these until the
|
||||
// consumers migrate to direct imports (reorg C1–C5). The graph is a DAG: memory depends on
|
||||
// thread (heap needs Thread.Mutex), and thread does its own raw mmap so there is no cycle.
|
||||
const system_call_module = b.addModule("system-call", .{
|
||||
.root_source_file = b.path("library/kernel/system-call.zig"),
|
||||
.imports = &.{.{ .name = "abi", .module = abi_module }},
|
||||
});
|
||||
const ipc_module = b.addModule("ipc", .{
|
||||
.root_source_file = b.path("library/kernel/ipc.zig"),
|
||||
.imports = &.{ .{ .name = "abi", .module = abi_module }, .{ .name = "system-call", .module = system_call_module } },
|
||||
});
|
||||
const time_module = b.addModule("time", .{
|
||||
.root_source_file = b.path("library/kernel/time.zig"),
|
||||
.imports = &.{.{ .name = "system-call", .module = system_call_module }},
|
||||
});
|
||||
const thread_module = b.addModule("thread", .{
|
||||
.root_source_file = b.path("library/kernel/thread.zig"),
|
||||
.imports = &.{ .{ .name = "abi", .module = abi_module }, .{ .name = "system-call", .module = system_call_module } },
|
||||
});
|
||||
const logging_module = b.addModule("logging", .{
|
||||
.root_source_file = b.path("library/kernel/logging.zig"),
|
||||
.imports = &.{ .{ .name = "abi", .module = abi_module }, .{ .name = "system-call", .module = system_call_module } },
|
||||
});
|
||||
const process_module = b.addModule("process", .{
|
||||
.root_source_file = b.path("library/kernel/process.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
.{ .name = "system-call", .module = system_call_module },
|
||||
.{ .name = "ipc", .module = ipc_module },
|
||||
.{ .name = "time", .module = time_module },
|
||||
},
|
||||
});
|
||||
const file_system_module = b.addModule("file-system", .{
|
||||
.root_source_file = b.path("library/kernel/file-system.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
.{ .name = "system-call", .module = system_call_module },
|
||||
.{ .name = "ipc", .module = ipc_module },
|
||||
.{ .name = "vfs-protocol", .module = vfs_protocol_module },
|
||||
},
|
||||
});
|
||||
const memory_module = b.addModule("memory", .{
|
||||
.root_source_file = b.path("library/kernel/memory/memory.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
.{ .name = "system-call", .module = system_call_module },
|
||||
.{ .name = "ipc", .module = ipc_module },
|
||||
.{ .name = "thread", .module = thread_module },
|
||||
},
|
||||
});
|
||||
const service_module = b.addModule("service", .{
|
||||
.root_source_file = b.path("library/kernel/service.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
.{ .name = "ipc", .module = ipc_module },
|
||||
.{ .name = "process", .module = process_module },
|
||||
},
|
||||
});
|
||||
const start_module = b.addModule("start", .{
|
||||
.root_source_file = b.path("library/kernel/start.zig"),
|
||||
.imports = &.{ .{ .name = "process", .module = process_module }, .{ .name = "logging", .module = logging_module } },
|
||||
});
|
||||
// The driver author's interface (library/device/driver): device access + the
|
||||
// device-manager hello handshake, folded together.
|
||||
const driver_module = b.addModule("driver", .{
|
||||
.root_source_file = b.path("library/device/driver/driver.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi_module },
|
||||
.{ .name = "device-abi", .module = device_abi_module },
|
||||
.{ .name = "system-call", .module = system_call_module },
|
||||
.{ .name = "ipc", .module = ipc_module },
|
||||
.{ .name = "time", .module = time_module },
|
||||
.{ .name = "device-manager-protocol", .module = device_manager_protocol_module },
|
||||
},
|
||||
});
|
||||
// The block-device client — a device type, so library/device/block.
|
||||
const block_client_module = b.addModule("block", .{
|
||||
.root_source_file = b.path("library/device/block/block.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "ipc", .module = ipc_module },
|
||||
.{ .name = "time", .module = time_module },
|
||||
.{ .name = "block-protocol", .module = block_protocol_module },
|
||||
},
|
||||
});
|
||||
// Userspace-service clients live in library/client (they talk to services, not the kernel).
|
||||
const display_client_module = b.addModule("display", .{
|
||||
.root_source_file = b.path("library/client/display/display.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "ipc", .module = ipc_module },
|
||||
.{ .name = "time", .module = time_module },
|
||||
.{ .name = "display-protocol", .module = display_protocol_module },
|
||||
},
|
||||
});
|
||||
const input_client_module = b.addModule("input", .{
|
||||
.root_source_file = b.path("library/client/input/input.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "ipc", .module = ipc_module },
|
||||
.{ .name = "time", .module = time_module },
|
||||
.{ .name = "input-protocol", .module = input_protocol_module },
|
||||
},
|
||||
});
|
||||
// A device driver's view of its claimed PCI function: config-space header fields, BAR
|
||||
// decode + map, and the capability walk (library/device/pci/pci.zig). The generic PCI
|
||||
// mechanics every leaf PCI driver used to re-derive inline. Imports the driver (device
|
||||
// access) client + mmio + the pci-class data module (config-space layout constants).
|
||||
const pci_module = b.addModule("pci", .{
|
||||
.root_source_file = b.path("library/device/pci/pci.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "driver", .module = driver_module },
|
||||
.{ .name = "mmio", .module = mmio_module },
|
||||
.{ .name = "pci-class", .module = pci_class_module },
|
||||
},
|
||||
});
|
||||
|
||||
// The USB class-driver transfer client (library/device/usb/usb.zig): open a device on
|
||||
// the xHCI bus and drive it (control / interrupt / bulk). Bus-family logic a class
|
||||
// driver imports directly — over ipc + time. Re-exports usb-abi / usb-ids as
|
||||
// usb.abi / usb.ids for a single USB import.
|
||||
const usb_module = b.addModule("usb", .{
|
||||
.root_source_file = b.path("library/device/usb/usb.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "ipc", .module = ipc_module },
|
||||
.{ .name = "time", .module = time_module },
|
||||
.{ .name = "usb-transfer-protocol", .module = usb_transfer_protocol_module },
|
||||
.{ .name = "usb-abi", .module = usb_abi_module },
|
||||
.{ .name = "usb-ids", .module = usb_ids_module },
|
||||
},
|
||||
});
|
||||
|
||||
// Keyboard layouts compiled from the X11 xkeyboard-config database into native Zig
|
||||
@@ -504,10 +610,34 @@ pub fn build(b: *std.Build) void {
|
||||
b.getInstallStep().dependOn(&kernel_install.step);
|
||||
|
||||
// --- init: the first user-space program (a system service) ---
|
||||
// The default module set every user binary can import directly: the library/kernel
|
||||
// concern modules, the device/service clients, mmio, the keyboard layouts, and the ACPI
|
||||
// id registry. Per-binary extras are added with programModule(exe).addImport.
|
||||
const default_imports = [_]std.Build.Module.Import{
|
||||
.{ .name = "mmio", .module = mmio_module },
|
||||
.{ .name = "xkeyboard-config", .module = xkeyboard_config_module },
|
||||
.{ .name = "acpi-ids", .module = acpi_ids_module },
|
||||
.{ .name = "system-call", .module = system_call_module },
|
||||
.{ .name = "ipc", .module = ipc_module },
|
||||
.{ .name = "memory", .module = memory_module },
|
||||
.{ .name = "process", .module = process_module },
|
||||
.{ .name = "thread", .module = thread_module },
|
||||
.{ .name = "time", .module = time_module },
|
||||
.{ .name = "logging", .module = logging_module },
|
||||
.{ .name = "file-system", .module = file_system_module },
|
||||
.{ .name = "service", .module = service_module },
|
||||
.{ .name = "start", .module = start_module },
|
||||
.{ .name = "driver", .module = driver_module },
|
||||
.{ .name = "block", .module = block_client_module },
|
||||
.{ .name = "display", .module = display_client_module },
|
||||
.{ .name = "input", .module = input_client_module },
|
||||
};
|
||||
|
||||
// Built by the shared user-binary recipe (see addUserBinary): freestanding,
|
||||
// linked into the kernel's user region against the `runtime` runtime library, and
|
||||
// linked into the kernel's user region against the library/kernel modules, and
|
||||
// started in ring 3 by the kernel's user-ELF loader.
|
||||
const init_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "init", "system/services/init/init.zig");
|
||||
const init_exe = addUserBinary(b, kernel_target, &default_imports, "init", "system/services/init/init.zig");
|
||||
programModule(init_exe).addImport("power-protocol", power_protocol_module);
|
||||
// init reads the same `serial` flag the kernel does: its liveness heartbeat is a
|
||||
// serial/test-build diagnostic (the QEMU harness's init tests assert on it, and
|
||||
// -Dserial images emit it), so a flashable image runs a purely event-driven PID 1
|
||||
@@ -518,15 +648,18 @@ pub fn build(b: *std.Build) void {
|
||||
init_options.addOption(bool, "diagnose", diagnose);
|
||||
programModule(init_exe).addImport("build_options", init_options.createModule());
|
||||
|
||||
// --- the rest of the /system tree: services, drivers, test fixtures ---
|
||||
// --- the rest of the boot tree: /system services and drivers, /test fixtures ---
|
||||
// Each is built by the same user-binary recipe and laid out at its FHS path on
|
||||
// the boot volume (see `bundled` below). The EFI loader walks the tree at boot
|
||||
// and hands the kernel an in-RAM initial_ramdisk of it (system/initial-ramdisk.zig).
|
||||
const vfstest_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "vfs-test", "system/services/vfs-test/vfs-test.zig");
|
||||
const ps2_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-bus", "system/drivers/ps2-bus/ps2-bus.zig");
|
||||
const ps2_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig");
|
||||
const ps2_mouse_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig");
|
||||
const usb_xhci_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-xhci-bus", "system/drivers/usb-xhci-bus/usb-xhci-bus.zig");
|
||||
const vfstest_exe = addUserBinary(b, kernel_target, &default_imports, "vfs-test", "test/system/services/vfs-test/vfs-test.zig");
|
||||
const ps2_bus_exe = addUserBinary(b, kernel_target, &default_imports, "ps2-bus", "system/drivers/ps2-bus/ps2-bus.zig");
|
||||
const ps2_keyboard_exe = addUserBinary(b, kernel_target, &default_imports, "ps2-keyboard", "system/drivers/ps2-bus/keyboard.zig");
|
||||
programModule(ps2_keyboard_exe).addImport("input-protocol", input_protocol_module);
|
||||
const ps2_mouse_exe = addUserBinary(b, kernel_target, &default_imports, "ps2-mouse", "system/drivers/ps2-bus/mouse.zig");
|
||||
programModule(ps2_mouse_exe).addImport("input-protocol", input_protocol_module);
|
||||
const usb_xhci_bus_exe = addUserBinary(b, kernel_target, &default_imports, "usb-xhci-bus", "system/drivers/usb-xhci-bus/usb-xhci-bus.zig");
|
||||
programModule(usb_xhci_bus_exe).addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
// The xHCI bus driver builds chapter-9 requests and decodes descriptors from
|
||||
// usb-abi, and reports each interface's (class,subclass,protocol) identity via
|
||||
// usb-ids.packTriple.
|
||||
@@ -536,33 +669,47 @@ pub fn build(b: *std.Build) void {
|
||||
// The USB HID class drivers: keyboard and mouse. They own no hardware — each
|
||||
// opens its device through runtime.usb (the transfer protocol) and publishes to
|
||||
// the input service. They build chapter-9 class requests from usb-abi.
|
||||
const usb_hid_keyboard_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-hid-keyboard", "system/drivers/usb-hid/keyboard.zig");
|
||||
const usb_hid_keyboard_exe = addUserBinary(b, kernel_target, &default_imports, "usb-hid-keyboard", "system/drivers/usb-hid/keyboard.zig");
|
||||
programModule(usb_hid_keyboard_exe).addImport("usb", usb_module);
|
||||
programModule(usb_hid_keyboard_exe).addImport("usb-abi", usb_abi_module);
|
||||
const usb_hid_mouse_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-hid-mouse", "system/drivers/usb-hid/mouse.zig");
|
||||
programModule(usb_hid_keyboard_exe).addImport("input-protocol", input_protocol_module);
|
||||
const usb_hid_mouse_exe = addUserBinary(b, kernel_target, &default_imports, "usb-hid-mouse", "system/drivers/usb-hid/mouse.zig");
|
||||
programModule(usb_hid_mouse_exe).addImport("usb", usb_module);
|
||||
programModule(usb_hid_mouse_exe).addImport("usb-abi", usb_abi_module);
|
||||
programModule(usb_hid_mouse_exe).addImport("input-protocol", input_protocol_module);
|
||||
// The USB mass-storage class driver: opens its device via runtime.usb, drives it
|
||||
// with Bulk-Only Transport + SCSI, and serves the block protocol under `.block`.
|
||||
const usb_storage_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "usb-storage", "system/drivers/usb-storage/usb-storage.zig");
|
||||
const usb_storage_exe = addUserBinary(b, kernel_target, &default_imports, "usb-storage", "system/drivers/usb-storage/usb-storage.zig");
|
||||
programModule(usb_storage_exe).addImport("usb", usb_module);
|
||||
programModule(usb_storage_exe).addImport("block-protocol", block_protocol_module);
|
||||
// The FAT filesystem server: mounts the block device and serves it into the VFS
|
||||
// at /mnt/usb. Its engine (engine.zig / on-disk.zig) is imported relatively.
|
||||
const fat_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat", "system/services/fat/fat.zig");
|
||||
const fat_exe = addUserBinary(b, kernel_target, &default_imports, "fat", "system/services/fat/fat.zig");
|
||||
programModule(fat_exe).addImport("vfs-protocol", vfs_protocol_module);
|
||||
// Threaded: the display runs a mouse-listener thread alongside its compositor loop
|
||||
// (docs/threading.md, docs/display.md), so it opts into real atomics/TLS.
|
||||
const display_exe = addThreadedUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "display", "system/services/display/display.zig");
|
||||
const display_demo_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "display-demo", "system/services/display-demo/display-demo.zig");
|
||||
const virtio_gpu_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "virtio-gpu", "system/drivers/virtio-gpu/virtio-gpu.zig");
|
||||
const shared_memory_server_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "shared-memory-server", "system/services/shared-memory-server/shared-memory-server.zig");
|
||||
const shared_memory_client_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "shared-memory-client", "system/services/shared-memory-client/shared-memory-client.zig");
|
||||
const fat_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "fat-test", "system/services/fat/fat-test.zig");
|
||||
const pci_bus_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "pci-bus", "system/drivers/pci-bus/pci-bus.zig");
|
||||
const display_exe = addThreadedUserBinary(b, kernel_target, &default_imports, "display", "system/services/display/display.zig");
|
||||
programModule(display_exe).addImport("display-protocol", display_protocol_module);
|
||||
programModule(display_exe).addImport("scanout-protocol", scanout_protocol_module);
|
||||
const display_demo_exe = addUserBinary(b, kernel_target, &default_imports, "display-demo", "system/services/display-demo/display-demo.zig");
|
||||
const virtio_gpu_exe = addUserBinary(b, kernel_target, &default_imports, "virtio-gpu", "system/drivers/virtio-gpu/virtio-gpu.zig");
|
||||
programModule(virtio_gpu_exe).addImport("pci", pci_module); // library/device/pci — the claimed-function view
|
||||
programModule(virtio_gpu_exe).addImport("display-protocol", display_protocol_module);
|
||||
programModule(virtio_gpu_exe).addImport("scanout-protocol", scanout_protocol_module);
|
||||
const shared_memory_server_exe = addUserBinary(b, kernel_target, &default_imports, "shared-memory-server", "test/system/services/shared-memory-server/shared-memory-server.zig");
|
||||
const shared_memory_client_exe = addUserBinary(b, kernel_target, &default_imports, "shared-memory-client", "test/system/services/shared-memory-client/shared-memory-client.zig");
|
||||
const fat_test_exe = addUserBinary(b, kernel_target, &default_imports, "fat-test", "test/system/services/fat-test/fat-test.zig");
|
||||
const pci_bus_exe = addUserBinary(b, kernel_target, &default_imports, "pci-bus", "system/drivers/pci-bus/pci-bus.zig");
|
||||
programModule(pci_bus_exe).addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
// The PCI bus driver decodes each function's class triple to human names in its
|
||||
// boot log (class/subclass/prog-IF), so pull in the shared pci-class reference.
|
||||
programModule(pci_bus_exe).addImport("pci-class", pci_class_module);
|
||||
// A test fixture, not a real driver: hellos to the device manager, then faults —
|
||||
// what the driver-restart scenario drives the crash-loop cap with.
|
||||
const crash_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "crash-test", "system/services/crash-test/crash-test.zig");
|
||||
const device_list_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-list", "system/services/device-list/device-list.zig");
|
||||
const crash_test_exe = addUserBinary(b, kernel_target, &default_imports, "crash-test", "test/system/services/crash-test/crash-test.zig");
|
||||
programModule(crash_test_exe).addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
const device_list_exe = addUserBinary(b, kernel_target, &default_imports, "device-list", "test/system/services/device-list/device-list.zig");
|
||||
programModule(device_list_exe).addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
// The discovery service: one swappable process per firmware
|
||||
// (docs/discovery.md), bundled under the neutral ramdisk name
|
||||
// "discovery" so the device manager never learns which firmware it is on.
|
||||
@@ -576,32 +723,37 @@ pub fn build(b: *std.Build) void {
|
||||
.acpi => "system/services/acpi/acpi.zig",
|
||||
.fdt => "system/services/fdt/fdt.zig",
|
||||
};
|
||||
const discovery_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "discovery", discovery_source);
|
||||
const discovery_exe = addUserBinary(b, kernel_target, &default_imports, "discovery", discovery_source);
|
||||
if (discovery == .acpi) programModule(discovery_exe).addImport("aml", aml_module);
|
||||
const device_manager_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "device-manager", "system/services/device-manager/device-manager.zig");
|
||||
if (discovery == .acpi) programModule(discovery_exe).addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
if (discovery == .acpi) programModule(discovery_exe).addImport("power-protocol", power_protocol_module);
|
||||
const device_manager_exe = addUserBinary(b, kernel_target, &default_imports, "device-manager", "system/services/device-manager/device-manager.zig");
|
||||
// Names the xHCI PCI class triple from the shared taxonomy instead of a bare 0x0C0330.
|
||||
programModule(device_manager_exe).addImport("pci-class", pci_class_module);
|
||||
programModule(device_manager_exe).addImport("device-manager-protocol", device_manager_protocol_module);
|
||||
// The manager matches reported USB interfaces by their (class,subclass,protocol)
|
||||
// triple (usbDriverForIdentity), built from the named usb-ids codes.
|
||||
programModule(device_manager_exe).addImport("usb-ids", usb_ids_module);
|
||||
// The input service and its exercisers: the fan-out server, a hardware-free synthetic
|
||||
// source, and a subscriber that doubles as the `input` test's oracle. See docs/input.md.
|
||||
const input_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input", "system/services/input/input.zig");
|
||||
const input_source_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-source", "system/services/input-source/input-source.zig");
|
||||
const input_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "input-test", "system/services/input-test/input-test.zig");
|
||||
const args_echo_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "args-echo", "system/services/args-echo/args-echo.zig");
|
||||
const process_test_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "process-test", "system/services/process-test/process-test.zig");
|
||||
const logger_exe = addUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "logger", "system/services/logger/logger.zig");
|
||||
const input_exe = addUserBinary(b, kernel_target, &default_imports, "input", "system/services/input/input.zig");
|
||||
programModule(input_exe).addImport("input-protocol", input_protocol_module);
|
||||
const input_source_exe = addUserBinary(b, kernel_target, &default_imports, "input-source", "test/system/services/input-source/input-source.zig");
|
||||
const input_test_exe = addUserBinary(b, kernel_target, &default_imports, "input-test", "test/system/services/input-test/input-test.zig");
|
||||
const args_echo_exe = addUserBinary(b, kernel_target, &default_imports, "args-echo", "test/system/services/args-echo/args-echo.zig");
|
||||
const process_test_exe = addUserBinary(b, kernel_target, &default_imports, "process-test", "test/system/services/process-test/process-test.zig");
|
||||
const logger_exe = addUserBinary(b, kernel_target, &default_imports, "logger", "system/services/logger/logger.zig");
|
||||
// The first multi-threaded binary: exercises runtime.Thread over the thread ABI
|
||||
// (docs/threading.md). Built threaded so its shared-memory poll is real.
|
||||
const thread_test_exe = addThreadedUserBinary(b, kernel_target, runtime_module, mmio_module, xkeyboard_config_module, acpi_ids_module, "thread-test", "system/services/thread-test/thread-test.zig");
|
||||
const thread_test_exe = addThreadedUserBinary(b, kernel_target, &default_imports, "thread-test", "test/system/services/thread-test/thread-test.zig");
|
||||
|
||||
// Every user binary and its FHS home on the boot volume. There is no packed
|
||||
// ramdisk artifact any more: make-fat-image.py lays each binary out at this
|
||||
// path on the image, and the EFI loader walks /system at boot and builds the
|
||||
// in-RAM initial_ramdisk table from the tree — the volume's file structure is
|
||||
// the single source of truth. Entry names (and hence argv[0] and task names)
|
||||
// are these paths with a leading slash.
|
||||
// path on the image, and the EFI loader walks /system and /test at boot and
|
||||
// builds the in-RAM initial_ramdisk table from the trees — the volume's file
|
||||
// structure is the single source of truth. Entry names (and hence argv[0] and
|
||||
// task names) are these paths with a leading slash. Test fixtures mirror their
|
||||
// repo home: test/system/services/<name> in the source tree IS the boot path.
|
||||
const bundled = [_]BundledBinary{
|
||||
.{ .path = "system/services/init", .binary = init_exe.getEmittedBin() },
|
||||
.{ .path = "system/services/fat", .binary = fat_exe.getEmittedBin() },
|
||||
@@ -620,17 +772,17 @@ pub fn build(b: *std.Build) void {
|
||||
.{ .path = "system/drivers/usb-storage", .binary = usb_storage_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/virtio-gpu", .binary = virtio_gpu_exe.getEmittedBin() },
|
||||
.{ .path = "system/drivers/pci-bus", .binary = pci_bus_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/vfs-test", .binary = vfstest_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/fat-test", .binary = fat_test_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/shared-memory-server", .binary = shared_memory_server_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/shared-memory-client", .binary = shared_memory_client_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/crash-test", .binary = crash_test_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/device-list", .binary = device_list_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/input-source", .binary = input_source_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/input-test", .binary = input_test_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/args-echo", .binary = args_echo_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/process-test", .binary = process_test_exe.getEmittedBin() },
|
||||
.{ .path = "system/tests/thread-test", .binary = thread_test_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/vfs-test", .binary = vfstest_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/fat-test", .binary = fat_test_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/shared-memory-server", .binary = shared_memory_server_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/shared-memory-client", .binary = shared_memory_client_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/crash-test", .binary = crash_test_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/device-list", .binary = device_list_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/input-source", .binary = input_source_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/input-test", .binary = input_test_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/args-echo", .binary = args_echo_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/process-test", .binary = process_test_exe.getEmittedBin() },
|
||||
.{ .path = "test/system/services/thread-test", .binary = thread_test_exe.getEmittedBin() },
|
||||
};
|
||||
|
||||
// The boot manifest: the FHS path of every bundled binary, one per line. The
|
||||
@@ -919,23 +1071,23 @@ pub fn build(b: *std.Build) void {
|
||||
"system/boot-handoff.zig",
|
||||
"system/abi.zig",
|
||||
"system/initial-ramdisk.zig", // v2 path-named entries: find/basename/magic
|
||||
"system/devices/device-abi.zig",
|
||||
"system/devices/pci-class.zig", // class/subclass/prog-IF name decoding
|
||||
"system/devices/acpi-ids.zig", // _HID name decoding
|
||||
"system/devices/aml/aml.zig", // AML parse + interpret, incl. Notify dispatch (M21)
|
||||
"system/devices/usb-abi.zig", // wire sizes + bit packings + set-up packet encodings
|
||||
"system/devices/usb-ids.zig", // class/subclass/protocol code assignments
|
||||
"library/mmio/mmio.zig", // barriers assemble + registers round-trip
|
||||
"library/device/model/device-abi.zig",
|
||||
"library/device/pci/pci-class.zig", // class/subclass/prog-IF name decoding
|
||||
"library/device/acpi/acpi-ids.zig", // _HID name decoding
|
||||
"library/device/acpi/aml/aml.zig", // AML parse + interpret, incl. Notify dispatch (M21)
|
||||
"library/device/usb/usb-abi.zig", // wire sizes + bit packings + set-up packet encodings
|
||||
"library/device/usb/usb-ids.zig", // class/subclass/protocol code assignments
|
||||
"library/device/mmio/mmio.zig", // barriers assemble + registers round-trip
|
||||
"system/drivers/ps2-bus/scancode.zig", // set-2 decode + keyboard state machine
|
||||
"system/drivers/ps2-bus/mouse-packet.zig", // 3-byte mouse packet assembly
|
||||
"system/drivers/usb-hid/hid-report.zig", // HID boot-report keyboard/mouse decode
|
||||
"system/drivers/usb-storage/bulk-only-transport.zig", // CBW/CSW wrapper sizes
|
||||
"system/drivers/usb-storage/scsi.zig", // SCSI CDB encodings (big-endian)
|
||||
"system/vfs-protocol.zig", // NodeKind / DirectoryEntry sizes + op values
|
||||
"library/protocol/vfs/vfs-protocol.zig", // NodeKind / DirectoryEntry sizes + op values
|
||||
"system/services/fat/on-disk.zig", // FAT on-disk struct sizes + type detection
|
||||
"system/services/fat/engine.zig", // FAT read/write over a RAM-backed image
|
||||
"system/services/display/compositor.zig", // Rect math + fill/composite/blit-tile
|
||||
"system/services/display/protocol.zig", // pack(): native pixel encoding per format
|
||||
"library/protocol/display/display-protocol.zig", // pack(): native pixel encoding per format
|
||||
"system/drivers/virtio-gpu/virtio-gpu-protocol.zig", // virtio-gpu command struct sizes
|
||||
"system/drivers/virtio-gpu/virtio-pci.zig", // virtio 1.0 PCI transport struct sizes
|
||||
}) |root| {
|
||||
@@ -984,7 +1136,7 @@ pub fn build(b: *std.Build) void {
|
||||
// loop above.
|
||||
const time_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("library/runtime/time.zig"),
|
||||
.root_source_file = b.path("library/kernel/time.zig"),
|
||||
.target = target,
|
||||
.optimize = optimize,
|
||||
.imports = &.{
|
||||
@@ -1000,7 +1152,7 @@ pub fn build(b: *std.Build) void {
|
||||
// system.zig (syscall wrappers), which needs the `abi` module.
|
||||
const thread_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("library/runtime/thread.zig"),
|
||||
.root_source_file = b.path("library/kernel/thread.zig"),
|
||||
.target = target,
|
||||
.optimize = optimize,
|
||||
.imports = &.{
|
||||
|
||||
+120
-91
@@ -3,102 +3,102 @@
|
||||
Notes on how danos boots and draws, written to explain the *why* behind the code
|
||||
rather than restate it. Roughly in the order things happen at runtime:
|
||||
|
||||
1. **[efi.md](efi.md) — EFI / the boot process.** How UEFI firmware finds and
|
||||
1. **[efi.md](os-development-guide/efi.md) — EFI / the boot process.** How UEFI firmware finds and
|
||||
runs the bootloader, what the loader gathers before `ExitBootServices`, how it
|
||||
loads the kernel ELF, and the ABI contract for the jump into the kernel. Start
|
||||
here.
|
||||
2. **[system-image.md](system-image.md) — system.img, the boot capsule.** The
|
||||
2. **[system-image.md](os-development-guide/system-image.md) — system.img, the boot capsule.** The
|
||||
bundled user binaries packed into one file in the initial-ramdisk wire
|
||||
format, because one open + one sequential read is the only file I/O shape
|
||||
firmware is fast at. The trivial container format, the three artifacts one
|
||||
build list derives (tree, manifest, capsule), the loader's three-strategy
|
||||
fallback chain, and the capsule's kernel-side life as both the spawn table
|
||||
and the read-only `/system` mount.
|
||||
3. **[gop.md](gop.md) — the Graphics Output Protocol.** How UEFI exposes graphics
|
||||
3. **[gop.md](os-development-guide/gop.md) — the Graphics Output Protocol.** How UEFI exposes graphics
|
||||
modes (unlike fixed VGA modes), how we detect the monitor's native resolution
|
||||
from EDID and switch to it, and the pixel formats we accept or reject.
|
||||
4. **[framebuffer.md](framebuffer.md) — the framebuffer.** What the linear
|
||||
4. **[framebuffer.md](os-development-guide/framebuffer.md) — the framebuffer.** What the linear
|
||||
framebuffer the loader hands over actually is, and what **pitch** (stride)
|
||||
means versus width — the detail you have to get right to avoid a skewed image.
|
||||
5. **[memory-map.md](memory-map.md) — the memory map.** How the loader learns what
|
||||
5. **[memory-map.md](os-development-guide/memory-map.md) — the memory map.** How the loader learns what
|
||||
physical RAM exists and hands it to the kernel in danos's own neutral format,
|
||||
rather than leaking UEFI's memory descriptors across the boundary.
|
||||
6. **[frame-allocator.md](frame-allocator.md) — the physical frame allocator.** The
|
||||
6. **[frame-allocator.md](os-development-guide/frame-allocator.md) — the physical frame allocator.** The
|
||||
bitmap allocator that hands out and reclaims 4 KiB physical frames from that
|
||||
map — the primitive page tables and the heap are built on.
|
||||
7. **[interrupts.md](interrupts.md) — interrupts and exceptions.** The GDT, IDT and
|
||||
7. **[interrupts.md](os-development-guide/interrupts.md) — interrupts and exceptions.** The GDT, IDT and
|
||||
TSS, the exception stubs, and the handler that reports a CPU fault in red instead
|
||||
of letting it triple-fault into a silent reset.
|
||||
8. **[paging.md](paging.md) — the kernel's page tables.** Building our own 4-level
|
||||
8. **[paging.md](os-development-guide/paging.md) — the kernel's page tables.** Building our own 4-level
|
||||
page tables, identity-mapping the low 4 GiB, and switching CR3 off the firmware's
|
||||
tables onto ours.
|
||||
9. **[device-interrupts.md](device-interrupts.md) — device interrupts.** The Local
|
||||
9. **[device-interrupts.md](device-driver-development-guide/device-interrupts.md) — device interrupts.** The Local
|
||||
APIC and its timer — the kernel's first interrupt that is *handled and returned
|
||||
from*, giving it a heartbeat.
|
||||
10. **[heap.md](heap.md) — the kernel heap.** A growable free-list allocator built on
|
||||
10. **[heap.md](os-development-guide/heap.md) — the kernel heap.** A growable free-list allocator built on
|
||||
the VMM, exposed as a `std.mem.Allocator` so std containers work — dynamic
|
||||
allocation for the kernel.
|
||||
11. **[scheduling.md](scheduling.md) — the scheduler.** Fixed-priority preemptive
|
||||
11. **[scheduling.md](os-development-guide/scheduling.md) — the scheduler.** Fixed-priority preemptive
|
||||
multitasking: kernel threads, the context switch, O(1) priority selection, and
|
||||
blocking (sleep, wait queues) — the leap to a running system.
|
||||
12. **[ipc.md](ipc.md) — inter-process communication.** Bounded blocking
|
||||
12. **[ipc.md](device-driver-development-guide/ipc.md) — inter-process communication.** Bounded blocking
|
||||
message-passing channels, then synchronous call/reply between *processes* over
|
||||
endpoints — the backbone the microkernel's isolated servers talk over.
|
||||
13. **[syscall.md](syscall.md) — system calls.** How ring 3 asks the kernel for
|
||||
13. **[syscall.md](os-development-guide/syscall.md) — system calls.** How ring 3 asks the kernel for
|
||||
something: the `syscall`/`sysret` fast path, the trap frame, and why the table is
|
||||
deliberately tiny. The numbers are a **private** ABI — [vdso.md](vdso.md) designs
|
||||
deliberately tiny. The numbers are a **private** ABI — [vdso.md](os-development-guide/vdso.md) designs
|
||||
the public boundary that will hide them.
|
||||
14. **[vfs-protocol.md](vfs-protocol.md) — the VFS wire protocol.** The language-neutral
|
||||
14. **[vfs-protocol.md](file-system-development/vfs-protocol.md) — the VFS wire protocol.** The language-neutral
|
||||
byte-level spec of the file protocol spoken over IPC: request/reply headers,
|
||||
the operation table, mount routing, and the append-only evolution rules — the
|
||||
first IPC protocol documented as public ABI.
|
||||
15. **[drivers.md](drivers.md) — writing a driver.** The payoff: a driver is an
|
||||
15. **[drivers.md](device-driver-development-guide/drivers.md) — writing a driver.** The payoff: a driver is an
|
||||
ordinary ring-3 process that claims a device, maps its registers, and **sleeps
|
||||
until its hardware interrupts it**. The claim is the capability; `irq_ack` is the
|
||||
unmask.
|
||||
16. **[driver-model.md](driver-model.md) — buses, classes and host controllers.** How
|
||||
16. **[driver-model.md](device-driver-development-guide/driver-model.md) — buses, classes and host controllers.** How
|
||||
real driver stacks factor into three shapes and how families share code. The
|
||||
three primitives it proposed are long since built (M13 capability passing,
|
||||
M14 DMA + barriers, M15 MSI), and the driver *contract* on top of them —
|
||||
hello, supervision, restart — is built too (device-manager.md, M18).
|
||||
17. **[usb-hub.md](usb-hub.md) — USB hubs.** Built (M22): why hub topology is handled
|
||||
17. **[usb-hub.md](device-driver-development-guide/usb-hub.md) — USB hubs.** Built (M22): why hub topology is handled
|
||||
*inside* the `usb-xhci-bus` driver rather than a separate hub class driver — a
|
||||
device behind a hub is reached by the **controller**, programmed with a route
|
||||
string in its slot context — plus the compound-hub reality (a USB 3.0 hub is
|
||||
physically two hubs) and detection via the hub's status-change interrupt endpoint.
|
||||
18. **[process-management.md](process-management.md) — process management.** The
|
||||
18. **[process-management.md](os-development-guide/process-management.md) — process management.** The
|
||||
microkernel's `ps`/`kill`/SIGCHLD: enumerate as a table snapshot, the
|
||||
supervision link as the kill authority, and child-exit notifications over the
|
||||
same endpoints IRQs arrive on.
|
||||
19. **[process-lifecycle.md](process-lifecycle.md) — the process lifecycle.** Built
|
||||
19. **[process-lifecycle.md](os-development-guide/process-lifecycle.md) — the process lifecycle.** Built
|
||||
(M17): signals over IPC as the one lifecycle vocabulary every process speaks — the
|
||||
POSIX.1-1990 words with message delivery instead of stack hijack, the stable
|
||||
`runtime.process` interface, exit reasons, published exit events any stateful
|
||||
`process` module interface, exit reasons, published exit events any stateful
|
||||
service can subscribe to (the VFS releasing dead clients' handles), and the two
|
||||
iron rules (cleanup is the kernel's job; kill is not a signal).
|
||||
20. **[device-manager.md](device-manager.md) — the device manager.** Built (M18,
|
||||
20. **[device-manager.md](device-driver-development-guide/device-manager.md) — the device manager.** Built (M18,
|
||||
through the app surface): the
|
||||
tree, the matcher, and the supervisor. Tree structure lives in the manager,
|
||||
authority stays in the kernel; bus drivers report what they see; drivers are
|
||||
restarted through the lifecycle vocabulary — the plan that turns
|
||||
[resilience.md](resilience.md)'s restart goal into increments.
|
||||
21. **[input.md](input.md) — the input module.** Broadcasting input events (keyboard,
|
||||
[resilience.md](os-development-guide/resilience.md)'s restart goal into increments.
|
||||
21. **[input.md](device-driver-development-guide/input.md) — the input module.** Broadcasting input events (keyboard,
|
||||
mouse, joystick): why a synchronous rendezvous can't fan out to many listeners, the
|
||||
asynchronous `ipc_send` primitive built to fix it, and the per-device subscribe/publish
|
||||
service layered on top.
|
||||
22. **[display.md](display.md) — the display service.** The display half of the GUI
|
||||
22. **[display.md](device-driver-development-guide/display.md) — the display service.** The display half of the GUI
|
||||
track: a user-space compositor that owns the framebuffer, composes a layer stack into
|
||||
a double buffer, and presents it. Why GOP and the PCI display device are two views of
|
||||
one controller, the device-node + write-combining handoff, and what flicker-free buys
|
||||
that tear-free doesn't. Plan: [display-plan.md](display-plan.md). **v2** (complete) makes
|
||||
that tear-free doesn't. Plan: [display-plan.md](device-driver-development-guide/display-plan.md). **v2** (complete) makes
|
||||
scanout a pluggable backend — GOP floor + a native virtio-gpu driver, hot-attached, with
|
||||
runtime mode-set, EDID, fenced vsync presents, and restart re-attach:
|
||||
[display-v2.md](display-v2.md), plan [display-v2-plan.md](display-v2-plan.md). Looking
|
||||
[display-v2.md](device-driver-development-guide/display-v2.md), plan [display-v2-plan.md](device-driver-development-guide/display-v2-plan.md). Looking
|
||||
further out, three research snapshots survey what a *native* driver for real GPU silicon
|
||||
would take as another `.scanout` backend: [nvidia-gpus.md](nvidia-gpus.md) (RTX 3060 /
|
||||
Ampere), [amd-gpus.md](amd-gpus.md) (RX 6600 / RDNA2), and [intel-igpu.md](intel-igpu.md)
|
||||
would take as another `.scanout` backend: [nvidia-gpus.md](device-driver-development-guide/nvidia-gpus.md) (RTX 3060 /
|
||||
Ampere), [amd-gpus.md](device-driver-development-guide/amd-gpus.md) (RX 6600 / RDNA2), and [intel-igpu.md](device-driver-development-guide/intel-igpu.md)
|
||||
(Intel iGPU).
|
||||
23. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
||||
23. **[halting.md](os-development-guide/halting.md) — halting.** Why a kernel can't just "exit", and
|
||||
how `while (true) hlt` parks the CPU safely once there's nothing left to do.
|
||||
|
||||
Start with the north star:
|
||||
@@ -108,28 +108,28 @@ Start with the north star:
|
||||
**resilience** (restartable components). Win condition: runs on the author's PC and
|
||||
both Raspberry Pis, ideally with a GUI. Real-time is an option to explore, not a
|
||||
requirement. The *why* that shapes everything below.
|
||||
- **[resilience.md](resilience.md) — resilience.** A design note (not built yet) on
|
||||
- **[resilience.md](os-development-guide/resilience.md) — resilience.** A design note (not built yet) on
|
||||
fault isolation + live restart — the reincarnation-server + capability model that
|
||||
makes "if I break it, I can restart it" real. danos's core motivation.
|
||||
- **[zig-self-hosting.md](zig-self-hosting.md) — running Zig on danos.** A design note
|
||||
(not built yet) on making danos a real Zig target (`-target x86_64-danos`) and
|
||||
eventually running the compiler on it. The key realisation: Zig 0.16 reduces an OS
|
||||
port to **one seam** (`std.os.danos`), so we build `runtime.os` (→ that seam) plus a
|
||||
thin `runtime.fs`, retire the `posix` shim, and follow a phased path to
|
||||
port to **one seam** (`std.os.danos`), so we build an `os` seam module (→ that seam) plus
|
||||
the thin `file-system` module, retire the `posix` shim, and follow a phased path to
|
||||
`zig build-exe hello.zig` running on danos — **not** Linux-ABI emulation.
|
||||
- **[threading.md](threading.md) — threads, the std-shaped way.** **Built** (M1–M6):
|
||||
`runtime.Thread` mirrors `std.Thread`'s API (spawn/join/detach, Mutex/Condition/
|
||||
- **[threading.md](os-development-guide/threading.md) — threads, the std-shaped way.** **Built** (M1–M6):
|
||||
the `thread` module's `Thread` mirrors `std.Thread`'s API (spawn/join/detach, Mutex/Condition/
|
||||
Semaphore) over a **private** thread ABI — several tasks sharing one address space via
|
||||
a `thread_spawn` syscall, futex-backed blocking, address-space refcounting. Why it's the
|
||||
native type and not literal `std.Thread` (the [private ABI](syscall.md)), and why
|
||||
threads stay a narrow opt-in against the [resilience](resilience.md) default. Build
|
||||
plan + gates: [threading-plan.md](threading-plan.md).
|
||||
- **[vdso.md](vdso.md) — the vDSO, the public system-call boundary.** A design note
|
||||
native type and not literal `std.Thread` (the [private ABI](os-development-guide/syscall.md)), and why
|
||||
threads stay a narrow opt-in against the [resilience](os-development-guide/resilience.md) default. Build
|
||||
plan + gates: [threading-plan.md](os-development-guide/threading-plan.md).
|
||||
- **[vdso.md](os-development-guide/vdso.md) — the vDSO, the public system-call boundary.** A design note
|
||||
(not built yet) on keeping `abi.zig` genuinely private: a kernel-supplied, C-ABI
|
||||
entry blob mapped into every process as the *only* way into the kernel — so the
|
||||
syscall numbers can be renumbered or randomised at will, and Rust/C binaries get a
|
||||
stable boundary without danos growing a dynamic linker. danos's public ABI = the
|
||||
vDSO + the documented IPC wire protocols ([vfs-protocol.md](vfs-protocol.md) first).
|
||||
vDSO + the documented IPC wire protocols ([vfs-protocol.md](file-system-development/vfs-protocol.md) first).
|
||||
|
||||
Cutting across all of these:
|
||||
|
||||
@@ -137,69 +137,69 @@ Cutting across all of these:
|
||||
hardware needed to run danos: minimum specs (UEFI x86-64, ACPI, PCIe ECAM,
|
||||
xHCI, ~128 MiB RAM) grounded in what the boot path actually assumes, plus a
|
||||
plain-language guide matching Intel/AMD CPU generations by name.
|
||||
- **[release-iso.md](release-iso.md) — the release ISO.** The flashable boot
|
||||
- **[release-iso.md](os-development-guide/release-iso.md) — the release ISO.** The flashable boot
|
||||
media: `zig build release-x86-64` wraps the FAT32 boot volume in a hybrid ISO
|
||||
(MBR ESP partition + El Torito EFI entry, one embedded image) that Etcher/dd
|
||||
flash to USB or a burner writes to disc — built by an in-repo pure-Python
|
||||
tool, like the FAT image itself.
|
||||
- **[architecture.md](architecture.md) — the architecture split.** How CPU-specific code is kept
|
||||
- **[architecture.md](os-development-guide/architecture.md) — the architecture split.** How CPU-specific code is kept
|
||||
behind a build-time `arch` module so the generic kernel never names x86_64,
|
||||
leaving room for other systems (e.g. an AArch64 Raspberry Pi) later.
|
||||
- **[arm.md](arm.md) — ARM targets.** The Raspberry Pi landscape the arch split is
|
||||
- **[arm.md](os-development-guide/arm.md) — ARM targets.** The Raspberry Pi landscape the arch split is
|
||||
aiming at: `arm` (32-bit, Pi Zero W) vs `aarch64` (64-bit, Pi 3-5), UEFI vs
|
||||
device-tree boot, and what each layer needs.
|
||||
- **[discovery.md](discovery.md) — device discovery.** A design note on learning what
|
||||
- **[discovery.md](os-development-guide/discovery.md) — device discovery.** A design note on learning what
|
||||
hardware exists via ACPI (x86) or device tree (ARM) behind one neutral device model —
|
||||
when to build it, and how to keep it architecture-agnostic.
|
||||
- **[acpi.md](acpi.md) — finding the ACPI tables.** The concrete x86 locator chain:
|
||||
- **[acpi.md](os-development-guide/acpi.md) — finding the ACPI tables.** The concrete x86 locator chain:
|
||||
how the loader captures the **RSDP**, hands its physical address across in `BootInformation`,
|
||||
and how the platform derives the **RSDT/XSDT** from it and walks the SDTs — plus the
|
||||
live event side (the SCI, the power button, GPE/Notify) the ring-3 acpi service runs.
|
||||
- **[power.md](power.md) — the power service.** System power as a domain-named
|
||||
- **[power.md](os-development-guide/power.md) — the power service.** System power as a domain-named
|
||||
service: button/lid/battery events published to subscribers, and init's orderly
|
||||
shutdown composing the [lifecycle](process-lifecycle.md) stop sequence with an ACPI
|
||||
shutdown composing the [lifecycle](os-development-guide/process-lifecycle.md) stop sequence with an ACPI
|
||||
S5 write. Firmware-neutral — a PSCI backend drops in on ARM.
|
||||
- **[timers.md](timers.md) — timers and time.** The ring-3 surface for reading the
|
||||
- **[timers.md](os-development-guide/timers.md) — timers and time.** The ring-3 surface for reading the
|
||||
clock and waiting: why `now()` is a syscall rather than a service, and the one-shot
|
||||
timer notification (`timer_bind`) that gives supervisors a timed wait — built on the
|
||||
LAPIC heartbeat and calibrated TSC of [device-interrupts.md](device-interrupts.md).
|
||||
- **[smp.md](smp.md) — multiple cores.** A design/research note on how microkernels
|
||||
LAPIC heartbeat and calibrated TSC of [device-interrupts.md](device-driver-development-guide/device-interrupts.md).
|
||||
- **[smp.md](os-development-guide/smp.md) — multiple cores.** A design/research note on how microkernels
|
||||
(L4, seL4) handle SMP — big kernel lock vs per-CPU vs multikernel — and how the
|
||||
right choice depends on whether danos is chasing real-time or resilience.
|
||||
- **[coding-standards.md](coding-standards.md) — coding standards.** The naming rule the
|
||||
tree follows: non-acronyms are spelled out in full (`message`, not `msg`), files are
|
||||
`kebab-case`, code follows Zig's case conventions, and the handful of exceptions
|
||||
(POSIX/C ABI names, `init`/`len`/`ptr`, acronyms).
|
||||
- **[sysv.md](sysv.md) — the calling convention.** What "the kernel is SysV" means,
|
||||
- **[sysv.md](os-development-guide/sysv.md) — the calling convention.** What "the kernel is SysV" means,
|
||||
and why the loader→kernel boundary has to pin it (the RDI-vs-RCX handoff).
|
||||
- **[testing.md](testing.md) — testing.** How the kernel is tested by booting it in
|
||||
QEMU and asserting on its serial output — reproducibly, and structured so the
|
||||
same tests run across architectures.
|
||||
- **[logging.md](logging.md) — logging.** The multi-sink diagnostic log (serial,
|
||||
- **[logging.md](os-development-guide/logging.md) — logging.** The multi-sink diagnostic log (serial,
|
||||
0xE9 debugcon, file later) kept separate from the framebuffer display, plus the
|
||||
robustness path: optional framebuffer, POST-code checkpoints, and a persistent
|
||||
panic breadcrumb so the kernel survives — and can be diagnosed — with no output.
|
||||
|
||||
## How the pieces relate
|
||||
|
||||
The boot flow ties them together: UEFI runs the loader ([efi.md](efi.md)), which
|
||||
queries the **GOP** to pick a graphics mode ([gop.md](gop.md)), hands the kernel a
|
||||
**framebuffer** to draw into ([framebuffer.md](framebuffer.md)) and a **memory
|
||||
map** of physical RAM ([memory-map.md](memory-map.md)); the kernel turns that map
|
||||
into a **frame allocator** ([frame-allocator.md](frame-allocator.md)), installs
|
||||
its **descriptor tables** so CPU faults are caught ([interrupts.md](interrupts.md)),
|
||||
builds its own **page tables** and switches onto them ([paging.md](paging.md)),
|
||||
brings up the **heap** for dynamic allocation ([heap.md](heap.md)), starts the
|
||||
**scheduler** ([scheduling.md](scheduling.md)) and the **timer** that preempts it
|
||||
([device-interrupts.md](device-interrupts.md)) — with tasks blocking, sleeping and
|
||||
passing messages over **[IPC](ipc.md)** channels — runs, its CPU-specific bits
|
||||
behind the [architecture](architecture.md) boundary, and when idle, or on a panic, it **halts**
|
||||
([halting.md](halting.md)).
|
||||
The boot flow ties them together: UEFI runs the loader ([efi.md](os-development-guide/efi.md)), which
|
||||
queries the **GOP** to pick a graphics mode ([gop.md](os-development-guide/gop.md)), hands the kernel a
|
||||
**framebuffer** to draw into ([framebuffer.md](os-development-guide/framebuffer.md)) and a **memory
|
||||
map** of physical RAM ([memory-map.md](os-development-guide/memory-map.md)); the kernel turns that map
|
||||
into a **frame allocator** ([frame-allocator.md](os-development-guide/frame-allocator.md)), installs
|
||||
its **descriptor tables** so CPU faults are caught ([interrupts.md](os-development-guide/interrupts.md)),
|
||||
builds its own **page tables** and switches onto them ([paging.md](os-development-guide/paging.md)),
|
||||
brings up the **heap** for dynamic allocation ([heap.md](os-development-guide/heap.md)), starts the
|
||||
**scheduler** ([scheduling.md](os-development-guide/scheduling.md)) and the **timer** that preempts it
|
||||
([device-interrupts.md](device-driver-development-guide/device-interrupts.md)) — with tasks blocking, sleeping and
|
||||
passing messages over **[IPC](device-driver-development-guide/ipc.md)** channels — runs, its CPU-specific bits
|
||||
behind the [architecture](os-development-guide/architecture.md) boundary, and when idle, or on a panic, it **halts**
|
||||
([halting.md](os-development-guide/halting.md)).
|
||||
|
||||
Above that line the microkernel proper begins: **discovery** ([discovery.md](discovery.md),
|
||||
[acpi.md](acpi.md)) learns what hardware exists, ring-3 processes ask the kernel for
|
||||
things through the small **[syscall](syscall.md)** table, isolated servers reach each
|
||||
other over IPC **endpoints** ([ipc.md](ipc.md)), and a **[driver](drivers.md)** claims
|
||||
Above that line the microkernel proper begins: **discovery** ([discovery.md](os-development-guide/discovery.md),
|
||||
[acpi.md](os-development-guide/acpi.md)) learns what hardware exists, ring-3 processes ask the kernel for
|
||||
things through the small **[syscall](os-development-guide/syscall.md)** table, isolated servers reach each
|
||||
other over IPC **endpoints** ([ipc.md](device-driver-development-guide/ipc.md)), and a **[driver](device-driver-development-guide/drivers.md)** claims
|
||||
a device, maps its registers, and sleeps until the hardware interrupts it — which is
|
||||
the whole reason for the arrangement ([vision.md](vision.md)).
|
||||
|
||||
@@ -208,7 +208,7 @@ the whole reason for the arrangement ([vision.md](vision.md)).
|
||||
danos is a **monorepo of sub-projects**. Each service or driver is a directory that is
|
||||
its own Zig module — it can hold as many files as it needs, and other sub-projects
|
||||
reach it *by module name*, never by a path into its files. The source tree deliberately
|
||||
**mirrors the runtime FHS** ([danos-file-system-hierarchy-FSH.md](danos-file-system-hierarchy-FSH.md)):
|
||||
**mirrors the runtime FHS** ([danos-file-system-hierarchy-FSH.md](file-system-development/danos-file-system-hierarchy-FSH.md)):
|
||||
what you see under `system/` in the source is what a running danos represents under
|
||||
`/system`.
|
||||
|
||||
@@ -220,7 +220,8 @@ addressed as **`system/services/init`** — the repeated leaf resolves away:
|
||||
|----------------------------------------|--------------------------------------------|
|
||||
| `system/services/init/init.zig` | `system/services/init` → `/system/services/init` |
|
||||
| `system/drivers/ps2-bus/ps2-bus.zig` | `system/drivers/ps2-bus` → `/system/drivers/ps2-bus` |
|
||||
| `library/runtime/runtime.zig` | `library/runtime` (the `runtime` module) |
|
||||
| `test/system/services/vfs-test/vfs-test.zig` | `test/system/services/vfs-test` → `/test/system/services/vfs-test` |
|
||||
| `library/device/pci/pci.zig` | `library/device/pci` (the `pci` module) |
|
||||
|
||||
In **source**, a sub-project is a directory so it can hold many files — the entry is
|
||||
`fat/fat.zig`, beside it `fat/engine.zig`, `fat/on-disk.zig`, and so on. When
|
||||
@@ -234,32 +235,58 @@ A sub-project's extra files are reached through the module, never as separate pa
|
||||
```
|
||||
system/ → /system danos's own internals (the self-representation)
|
||||
boot-handoff.zig the loader↔kernel contract (the `boot-handoff` module)
|
||||
abi.zig the private kernel↔runtime syscall ABI (the `abi` module)
|
||||
parameters.zig initial-ramdisk.zig vfs-protocol.zig shared contracts
|
||||
abi.zig the private kernel↔userspace syscall ABI (the `abi` module)
|
||||
parameters.zig initial-ramdisk.zig shared contracts
|
||||
kernel/ IPC, memory, scheduling, the VFS root, the private syscall dispatch
|
||||
architecture/x86_64/ the `architecture` module (never named by generic code)
|
||||
devices/ the device model /system/devices reflects (+ aml/)
|
||||
device-abi.zig the device wire types (the `device-abi` module)
|
||||
devices-broker.zig the syscall-facing device table
|
||||
platform.zig acpi.zig fdt.zig device-model.zig firmware discovery + the kernel's
|
||||
device model — the implementation of what /system/devices reflects
|
||||
drivers/ pci-bus/ ps2-bus/ usb-xhci-bus/ one sub-project per driver → /system/drivers
|
||||
services/ init/ fat/ device-manager/ system servers → /system/services (fat/ holds
|
||||
fat.zig, engine.zig, on-disk.zig)
|
||||
library/ → /lib libraries, one sub-directory each
|
||||
runtime/ the danos-native runtime + file API (fs) — the stable application ABI
|
||||
kernel/ the danos-native system library (kernel32-style): the syscall
|
||||
surface split by concern — ipc, memory (heap/dma/shared-memory),
|
||||
process, time, logging, file-system, thread, service, plus the
|
||||
system-call stubs and the start/root entry shim
|
||||
device/ device code by domain — mmio/ model/ pci/ usb/ acpi/ driver/
|
||||
block/ — each a shareable data module (device-abi, pci-class,
|
||||
usb-abi/ids, acpi-ids) plus a logic module (mmio, pci, usb, aml,
|
||||
driver — the device-access + device-manager-hello client)
|
||||
client/ userspace service clients (display, input) — a program's view of
|
||||
a service, layered over that service's protocol
|
||||
protocol/ driver↔service wire contracts (vfs block display scanout input
|
||||
power device-manager usb-transfer), one module per directory
|
||||
boot/ → /boot the loaders
|
||||
tools/ test/ host-side build + QEMU test harness
|
||||
test/ → /test the test tree: the QEMU harness (qemu_test.py, host-side)
|
||||
system/services/ beside the on-image test fixtures — vfs-test/ thread-test/
|
||||
crash-test/ … — whose repo path IS their boot-volume path
|
||||
(/test/system/services/<name>)
|
||||
tools/ host-side build scripts
|
||||
```
|
||||
|
||||
A sub-project exposes its **public interface as a module**: the `usb-xhci-bus` driver
|
||||
owns the USB transfer protocol (`usb-transfer-protocol.zig`, the `usb-transfer-protocol`
|
||||
module), which `runtime.usb` imports by name — the USB class drivers reach the
|
||||
protocol through that wrapper; `block` exposes its protocol (`block-protocol`) the
|
||||
same way. The VFS wire protocol is the one that outgrew its
|
||||
sub-project: the VFS root moved into the kernel (`system/kernel/vfs.zig`), so the
|
||||
protocol lives as a shared contract at `system/vfs-protocol.zig` (the `vfs-protocol`
|
||||
module), which the runtime's file API (`runtime.fs`) imports by name.
|
||||
**Wire protocols live in `library/protocol/`**, one module per directory
|
||||
(`library/protocol/vfs/vfs-protocol.zig` is the `vfs-protocol` module), imported by module
|
||||
name. A protocol is the seam between a low-level driver and the higher-level service it
|
||||
serves — block ↔ the filesystem, a scanout driver ↔ the compositor — so both sides depend
|
||||
on the contract, not on each other, and the contract belongs to neither sub-project. A
|
||||
client module may *wrap* one for application convenience (the `file-system` module over
|
||||
`vfs-protocol`, and the `block`, `display`, `input` clients over theirs), but the protocol
|
||||
module is the boundary — a client re-exports no protocol, it imports it by name. A driver's
|
||||
private wire to its *hardware* (virtio-gpu's command set) is not a service seam and stays a
|
||||
driver-private file, beside the transport that reaches the same device.
|
||||
|
||||
**Device code lives in `library/device/<domain>/`**, grouped by what it is about (pci, usb,
|
||||
acpi, and the cross-cutting device model) and split by dependency weight: a data module of
|
||||
enums and wire types that is `std`-only and cheap for anyone to import, and a logic module
|
||||
that needs `mmio` or IPC. This is what keeps the microkernel out of device business — it
|
||||
imports exactly one `library/` module, `device-abi` (the descriptor types its broker
|
||||
marshals across the syscall boundary), and nothing with logic or a taxonomy in it. That
|
||||
lone pure-data import is the only edge from `system/kernel/` into `library/`.
|
||||
|
||||
There is **no POSIX/C compatibility layer today**: danos programs do file I/O through the
|
||||
danos-native `runtime.fs` (open/read/write/list over the VFS). A hand-rolled POSIX shim
|
||||
danos-native `file-system` module (open/read/write/list over the VFS). A hand-rolled POSIX shim
|
||||
(`library/posix/`) was retired as premature — the real POSIX/C surface will come later
|
||||
from the `std.os.danos` seam (and, eventually, musl) when danos becomes a Zig target (see
|
||||
[zig-self-hosting.md](zig-self-hosting.md)). When it does, the foreign-ABI naming
|
||||
@@ -272,8 +299,8 @@ exception in [coding-standards.md](coding-standards.md) applies to that seam.
|
||||
| Boot methods (one per way of booting the kernel) | `boot/` — `efi.zig` (UEFI) → `BOOTX64.efi` |
|
||||
| Kernel entry, panic, bring-up | `system/kernel/kernel.zig` |
|
||||
| Loader↔kernel handoff (`BootInformation`, `Framebuffer`, `MemoryMap`, VM layout) | `system/boot-handoff.zig` |
|
||||
| Private kernel↔runtime syscall ABI (`SystemCall`, mmap prot flags, `page_size`) — the runtime speaks it, not apps | `system/abi.zig` |
|
||||
| Device wire types (`DeviceDescriptor`, `DeviceClass`, …) | `system/devices/device-abi.zig` |
|
||||
| Private kernel↔userspace syscall ABI (`SystemCall`, mmap prot flags, `page_size`) — the system library speaks it, not apps | `system/abi.zig` |
|
||||
| Device wire types (`DeviceDescriptor`, `DeviceClass`, …) | `library/device/model/device-abi.zig` |
|
||||
| Physical frame allocator | `system/kernel/pmm.zig` |
|
||||
| Kernel heap (`std.mem.Allocator`) | `system/kernel/heap.zig` |
|
||||
| Scheduler (fixed-priority preemptive; blocking, wait queues) | `system/kernel/scheduler.zig` |
|
||||
@@ -281,15 +308,17 @@ exception in [coding-standards.md](coding-standards.md) applies to that seam.
|
||||
| IPC channels between kernel threads (message passing) | `system/kernel/ipc.zig` |
|
||||
| IPC endpoints: cross-address-space call/reply, handles, notifications | `system/kernel/ipc-synchronous.zig` |
|
||||
| User processes: ELF loading, address spaces, the syscall table | `system/kernel/process.zig` |
|
||||
| VFS root: mount table + kernel-served nodes (`fs_resolve`/`fs_node`); wire protocol in `system/vfs-protocol.zig` | `system/kernel/vfs.zig` |
|
||||
| VFS root: mount table + kernel-served nodes (`fs_resolve`/`fs_node`); wire protocol in `library/protocol/vfs/vfs-protocol.zig` | `system/kernel/vfs.zig` |
|
||||
| Device tree + claim capability + `device_register` containment | `system/kernel/devices-broker.zig` |
|
||||
| IRQ-as-IPC: routing a device interrupt to a driver's endpoint | `system/kernel/irq.zig` |
|
||||
| Hardware discovery (ACPI/device tree) behind one neutral device model | `system/devices/` |
|
||||
| Hardware discovery (ACPI/device tree) behind one neutral device model | `system/kernel/platform.zig` |
|
||||
| Framebuffer text console (mirrors to serial) | `system/kernel/console.zig` |
|
||||
| In-kernel test cases | `system/kernel/tests.zig` |
|
||||
| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/IO-APIC/timer, serial, linker script) | `system/kernel/architecture/x86_64/` |
|
||||
| danos-native runtime (`runtime`): syscall wrappers, heap, IPC, device access, the file API (`fs`) — the stable application ABI | `library/runtime/` |
|
||||
| danos-native system library (kernel32-style): the syscall surface by concern — `ipc`, `memory`, `process`, `time`, `logging`, `file-system`, `thread`, `service` — the stable application ABI | `library/kernel/` |
|
||||
| Service clients (a program's view of a service) and device clients | `library/client/` (display, input), `library/device/driver` |
|
||||
| System services (init, the `fat` filesystem, the device-manager) | `system/services/` |
|
||||
| Device drivers, one sub-project each (`pci-bus`, `ps2-bus`, `usb-xhci-bus` bus drivers) | `system/drivers/` |
|
||||
| On-image test fixtures for the QEMU cases (`vfs-test`, `crash-test`, `thread-test`, …) → `/test/system/services` | `test/system/services/` |
|
||||
| Build + `run-x86-64` (QEMU/OVMF) + `release-x86-64` (the flashable ISO) | `build.zig` |
|
||||
| QEMU integration test harness | `test/qemu_test.py` |
|
||||
|
||||
@@ -67,7 +67,7 @@ Three, and only three.
|
||||
|
||||
**This exception is scoped to a file that *is* a foreign ABI, and nothing else.**
|
||||
danos has no such file today: the old `library/posix/` compatibility shim was retired
|
||||
once its callers moved to the danos-native `runtime.fs`, since a hand-rolled POSIX
|
||||
once its callers moved to the danos-native `file_system`, since a hand-rolled POSIX
|
||||
layer is premature until danos actually needs it (see
|
||||
[zig-self-hosting.md](zig-self-hosting.md)). The exception will apply again to the
|
||||
`std.os.danos` seam when danos becomes a real Zig target — that module *is* the C-ABI
|
||||
@@ -97,8 +97,9 @@ Three, and only three.
|
||||
|
||||
That's all — no Unix-abbreviation exception. The source directories are full words
|
||||
(`system`, `library`, not `src`/`lib`), and there is no daemon `d` suffix: a driver
|
||||
lives in `system/drivers/` and a service in `system/services/`, so the *location*
|
||||
already says what it is. Encoding the role in the name too (`busd`, `fatd`) is
|
||||
lives in `system/drivers/`, a service in `system/services/`, and a test fixture in
|
||||
`test/system/services/` (the repo path *is* its path on the boot volume), so the
|
||||
*location* already says what it is. Encoding the role in the name too (`busd`, `fatd`) is
|
||||
redundant — the program is just `ps2-bus`, `fat`. Don't put in a name what its directory
|
||||
already tells you.
|
||||
|
||||
@@ -135,15 +136,16 @@ Within those spelling rules, follow Zig's own conventions:
|
||||
`notify_badge_bit`.
|
||||
|
||||
**File names are `kebab-case`.** A file named for a multi-word thing hyphenates it:
|
||||
`device-tree.zig`, `ipc-synchronous.zig`, `vfs-protocol.zig`, `devices-broker.zig`. A
|
||||
`device-model.zig`, `ipc-synchronous.zig`, `vfs-protocol.zig`, `devices-broker.zig`. A
|
||||
single word or acronym needs no hyphen: `scheduler.zig`, `paging.zig`, `apic.zig`,
|
||||
`idt.zig`. (The module *alias* a file is imported under still follows the code
|
||||
conventions above — `snake_case` — because it's an identifier, not a filename.)
|
||||
|
||||
**A sub-project's entry point repeats its directory's name** — `init/init.zig`,
|
||||
`runtime/runtime.zig`, `ps2-bus/ps2-bus.zig` — and the sub-project is addressed by the
|
||||
*directory* (`system/services/init`, `library/runtime`), with the repeated leaf
|
||||
resolving away. See the repository-layout section of [README.md](README.md).
|
||||
`pci/pci.zig`, `ps2-bus/ps2-bus.zig` — and the sub-project is addressed by the
|
||||
*directory* (`system/services/init`, `library/device/pci`,
|
||||
`test/system/services/vfs-test`), with the repeated leaf resolving away. See the
|
||||
repository-layout section of [README.md](README.md).
|
||||
|
||||
## Named values, not magic numbers
|
||||
|
||||
|
||||
+10
-10
@@ -1,6 +1,6 @@
|
||||
# Device interrupts
|
||||
|
||||
CPU exceptions ([interrupts.md](interrupts.md)) are the kernel reacting to its own
|
||||
CPU exceptions ([interrupts.md](../os-development-guide/interrupts.md)) are the kernel reacting to its own
|
||||
mistakes. **Device interrupts** are the opposite: hardware asking for attention —
|
||||
a timer firing, a key pressed, a packet arriving. They share the IDT, but differ
|
||||
in one fundamental way: an exception here is terminal (we report and halt), while a
|
||||
@@ -10,7 +10,7 @@ back — the same mechanism a scheduler will later use to preempt tasks.
|
||||
|
||||
The first device we bring up is the **timer**, because it's the simplest: it lives
|
||||
entirely on the CPU's local interrupt controller, needing no external routing.
|
||||
It's all x86_64-specific, behind the [architecture](architecture.md) boundary.
|
||||
It's all x86_64-specific, behind the [architecture](../os-development-guide/architecture.md) boundary.
|
||||
|
||||
## The APIC, not the PIC
|
||||
|
||||
@@ -40,7 +40,7 @@ count that becomes the reload value. From then on it fires vector 32 repeatedly,
|
||||
its own, forever.
|
||||
|
||||
The reload count isn't picked arbitrarily — it's **calibrated to real time**,
|
||||
which the [real-time](vision.md) scheduling guarantees depend on. Since the LAPIC
|
||||
which the [real-time](../vision.md) scheduling guarantees depend on. Since the LAPIC
|
||||
timer's raw rate is bus-clock dependent and unknown up front, `calibrate` runs the
|
||||
LAPIC timer one-shot from its maximum count while a **reference clock** counts out a
|
||||
known 10 ms, then sees how far the LAPIC got — its counts-per-millisecond, from which
|
||||
@@ -53,7 +53,7 @@ a missing PIT would hang the boot):
|
||||
|
||||
1. **CPUID leaf 0x15** — the CPU's TSC frequency directly, needing no external timer
|
||||
at all (the LAPIC is then measured against the TSC).
|
||||
2. The **HPET**, discovered via ACPI (see [discovery](discovery.md) / [acpi](acpi.md)).
|
||||
2. The **HPET**, discovered via ACPI (see [discovery](../os-development-guide/discovery.md) / [acpi](../os-development-guide/acpi.md)).
|
||||
3. The **ACPI PM timer** (a fixed 3.579545 MHz counter from the FADT).
|
||||
4. The **PIT** (legacy 8254, 1.193182 MHz) — last resort, and bounded so it can't hang.
|
||||
|
||||
@@ -100,7 +100,7 @@ values (a second socket, some firmware), so a thread migrating from a core readi
|
||||
check** as each application processor comes online (`checkWarpSource`, adapted from
|
||||
Linux's): the waking core and the BSP hammer a shared "highest seen" TSC under a lock,
|
||||
and if either ever reads below it, the cores' TSCs are skewed. It's pairwise because APs
|
||||
come up one at a time ([smp.md](smp.md)).
|
||||
come up one at a time ([smp.md](../os-development-guide/smp.md)).
|
||||
|
||||
**The fallback.** When the TSC fails either test — non-invariant (a bare VM such as the
|
||||
default qemu64), or warped between cores — danos moves the monotonic clock onto the
|
||||
@@ -160,7 +160,7 @@ A device handler is a plain `fn () void` — a timer or keyboard handler doesn't
|
||||
the interrupted registers. (The stubs originally didn't save the SSE/vector
|
||||
registers, so a handler couldn't use them; `isr_common` now does an
|
||||
`fxsave`/`fxrstor` of the full SSE/x87 state around dispatch — see
|
||||
[interrupts.md](interrupts.md).)
|
||||
[interrupts.md](../os-development-guide/interrupts.md).)
|
||||
|
||||
## Turning them on
|
||||
|
||||
@@ -168,11 +168,11 @@ Exceptions can't be masked, which is why they worked all along. Maskable device
|
||||
interrupts don't fire until the CPU's interrupt flag is set — so the final step is
|
||||
`sti` (`arch.enableInterrupts()`), after the APIC and timer are configured. From
|
||||
that instant the kernel has a heartbeat, and its idle `hlt` loop
|
||||
([halting.md](halting.md)) wakes on every tick and dozes off again.
|
||||
([halting.md](../os-development-guide/halting.md)) wakes on every tick and dozes off again.
|
||||
|
||||
## Verifying it
|
||||
|
||||
The `timer` test (see [testing.md](testing.md)) is the proof that an interrupt both
|
||||
The `timer` test (see [testing.md](../testing.md)) is the proof that an interrupt both
|
||||
*fires* and *returns*: it records the tick count, busy-waits, and checks the count
|
||||
advanced on its own.
|
||||
|
||||
@@ -188,13 +188,13 @@ spinning in unrelated code — is the whole mechanism working end to end.
|
||||
## Since (done elsewhere)
|
||||
|
||||
- **Preemption**: the timer handler is where the scheduler decides to switch — the
|
||||
reason a *returning* interrupt matters. See [scheduling.md](scheduling.md).
|
||||
reason a *returning* interrupt matters. See [scheduling.md](../os-development-guide/scheduling.md).
|
||||
- **`sleep()` / timeouts** built on the calibrated clock.
|
||||
- **The I/O APIC, routed**: external device lines now reach a vector, and the
|
||||
interrupt is delivered onward to a *user-space* driver as an IPC message. See
|
||||
[drivers.md](drivers.md).
|
||||
- **Uncacheable MMIO**: device grants are mapped `PCD|PWT` (strong-uncacheable) for
|
||||
user drivers — see [paging.md](paging.md).
|
||||
user drivers — see [paging.md](../os-development-guide/paging.md).
|
||||
|
||||
## What's next (partly done since)
|
||||
|
||||
@@ -10,19 +10,19 @@ mirrors them and prunes a dead reporter's children, and the `usb-report`
|
||||
scenario proves report → prune → respawn → re-report. The application surface is built (M18.3, 2026-07-13):
|
||||
`enumerate` and `subscribe` over IPC, with `device-list` as the first client —
|
||||
the manager is now the one answer to "what devices exist" for applications.
|
||||
The primitives underneath are real ([process-management.md](process-management.md):
|
||||
The primitives underneath are real ([process-management.md](../os-development-guide/process-management.md):
|
||||
spawn/supervise/kill/exit-notification; [driver-model.md](driver-model.md): the device
|
||||
table as a capability system; [drivers.md](drivers.md): claim/map/IRQ), and the first
|
||||
per-device driver spawn works (the device manager matches the xHCI controller by PCI
|
||||
class and spawns `usb-xhci-bus` with the device id as argv[1]). This document designs
|
||||
the rest: the device manager as **the tree, the matcher, and the supervisor** — the
|
||||
policy process that turns [resilience.md](resilience.md)'s restart goal into practice
|
||||
policy process that turns [resilience.md](../os-development-guide/resilience.md)'s restart goal into practice
|
||||
for drivers.
|
||||
|
||||
How processes stop, reload, and report their deaths is deliberately **not** in this
|
||||
document: that is the universal lifecycle every danos process speaks —
|
||||
[process-lifecycle.md](process-lifecycle.md), signals over IPC and the stable
|
||||
`runtime.process` interface. The device manager is that design's first serious
|
||||
[process-lifecycle.md](../os-development-guide/process-lifecycle.md), signals over IPC and the stable
|
||||
`process` interface. The device manager is that design's first serious
|
||||
customer, not its owner. Its own protocol contains nothing lifecycle-shaped; a
|
||||
driver is stopped, health-checked, and buried exactly like any other process.
|
||||
|
||||
@@ -35,7 +35,7 @@ The device tree is two things fused: *information* (what exists, how it nests) a
|
||||
claims, resource containment on `device_register`, the
|
||||
`mmio_map`/`irq_bind`/`msi_bind` gates — and **cleans all of it up when a process
|
||||
dies** (settled; it is increment 1 of
|
||||
[process-lifecycle.md](process-lifecycle.md)). The three invariants in
|
||||
[process-lifecycle.md](../os-development-guide/process-lifecycle.md)). The three invariants in
|
||||
[driver-model.md](driver-model.md) stay exactly where they are. A device manager
|
||||
that could mint MMIO mappings by its own say-so would be a second kernel, and a
|
||||
buggy one would un-earn everything the microkernel bought.
|
||||
@@ -51,7 +51,7 @@ enumeration is a **pci-bus driver**: the manager spawns it against the host brid
|
||||
like any bus reports children. ACPI becomes an **acpi service** that interprets the
|
||||
tables and reports the namespace. The manager only orchestrates and merges. Moving
|
||||
AML interpretation out of ring 0 is its own project on its own track; nothing here
|
||||
depends on when it lands. (It landed: [discovery.md](discovery.md), M19–M20.)
|
||||
depends on when it lands. (It landed: [discovery.md](../os-development-guide/discovery.md), M19–M20.)
|
||||
|
||||
`device_register` is **idempotent on exact match**: a re-registration with an
|
||||
identical (parent, class, identity, resources) tuple returns the existing id
|
||||
@@ -82,7 +82,7 @@ one world.
|
||||
deadline means wrong binary, wrong protocol version, or wedged before main — apply
|
||||
the stop sequence and the restart policy. Everything else lifecycle-shaped
|
||||
(terminate, the common `ping` liveness call, exit reasons) arrives through
|
||||
[process-lifecycle.md](process-lifecycle.md)'s vocabulary, not this protocol.
|
||||
[process-lifecycle.md](../os-development-guide/process-lifecycle.md)'s vocabulary, not this protocol.
|
||||
|
||||
Assignment stays argv (`usb-xhci-bus <device id>`) for now — simple, and it works.
|
||||
The step after `hello` exists is delegation: the manager claims (or is granted) the
|
||||
@@ -97,7 +97,7 @@ from usb-ids.zig — each bus's native language, decoded by the shared ids modul
|
||||
Every driver is spawned with the manager's exit endpoint (`spawnSupervised` — built).
|
||||
On a death notification:
|
||||
|
||||
1. **Read the reason** ([process-lifecycle.md](process-lifecycle.md) increment 2).
|
||||
1. **Read the reason** ([process-lifecycle.md](../os-development-guide/process-lifecycle.md) increment 2).
|
||||
Clean exit → it meant to; don't restart. Fault or missed `hello` deadline →
|
||||
restart with **backoff**, and a crash-loop cap (three fast deaths → mark failed,
|
||||
stop respawning, log loudly; a later `reload` to the manager can retry).
|
||||
@@ -136,8 +136,8 @@ way.
|
||||
## Increments
|
||||
|
||||
Increments 1–4 are the lifecycle prerequisites and live in
|
||||
[process-lifecycle.md](process-lifecycle.md) (claim cleanup on death, exit reasons,
|
||||
published exit events, signals + `runtime.process`). On top of those:
|
||||
[process-lifecycle.md](../os-development-guide/process-lifecycle.md) (claim cleanup on death, exit reasons,
|
||||
published exit events, signals + `process`). On top of those:
|
||||
|
||||
5. **device-manager-protocol**: `hello`, supervised spawn with restart policy;
|
||||
usb-xhci-bus becomes the first conforming driver.
|
||||
@@ -147,7 +147,7 @@ published exit events, signals + `runtime.process`). On top of those:
|
||||
to a manager-internal seam.
|
||||
8. **Discovery migration** — DONE (M19–M20, 2026-07-13): enumeration moved to
|
||||
ring 3 as swappable per-firmware discoverers — the pci-bus driver (M19) then
|
||||
the acpi service (M20), see [discovery.md](discovery.md); of the enumerable
|
||||
the acpi service (M20), see [discovery.md](../os-development-guide/discovery.md); of the enumerable
|
||||
devices, the kernel seeds only the host bridge and the acpi-tables node (the
|
||||
non-enumerable platform nodes — processors, interrupt controllers, the HPET,
|
||||
the loader's framebuffer — stay kernel-seeded too). Matching moved with it:
|
||||
@@ -18,9 +18,9 @@ Read [display.md](display.md) first for the *why*; this is the *what* and the *o
|
||||
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](coding-standards.md): spell out non-acronym abbreviations in
|
||||
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym abbreviations in
|
||||
full, kebab-case file names, no `Co-Authored-By` trailers on commits. New user binaries
|
||||
go through `addUserBinary` in [build.zig](../build.zig) and get packed into the
|
||||
go through `addUserBinary` in [build.zig](../../build.zig) and get packed into the
|
||||
initial-ramdisk; protocols are `b.addModule("…-protocol", …)` and imported into the
|
||||
`runtime` module.
|
||||
|
||||
@@ -29,7 +29,7 @@ initial-ramdisk; protocols are `b.addModule("…-protocol", …)` and imported i
|
||||
- `zig build test` — host unit tests (compositor math: layer clipping, damage merge,
|
||||
pitch/format blits are all host-testable with a fake framebuffer).
|
||||
- `python3 test/qemu_test.py <case>` — boots the real kernel in QEMU; assert on the
|
||||
serial log ([tests.zig](../system/kernel/tests.zig) is the registry).
|
||||
serial log ([tests.zig](../../system/kernel/tests.zig) is the registry).
|
||||
- The `run-efi` target renders to QEMU's display (`-device VGA,edid=on,xres=1280,yres=720`)
|
||||
— a screenshot confirms pixels for the milestones whose gate is visual.
|
||||
|
||||
@@ -39,22 +39,22 @@ initial-ramdisk; protocols are `b.addModule("…-protocol", …)` and imported i
|
||||
|
||||
Make the boot framebuffer reachable and mappable **write-combining** from user space.
|
||||
|
||||
- [x] [device-abi.zig](../system/devices/device-abi.zig): added `DeviceClass.display`; a
|
||||
- [x] [device-abi.zig](../../library/device/model/device-abi.zig): added `DeviceClass.display`; a
|
||||
`DisplayInfo{ width, height, pitch, format }` carried on the descriptor; a
|
||||
`flags` field on `ResourceDescriptor` + `resource_flag_write_combining`.
|
||||
- [x] [devices-broker.zig](../system/kernel/devices-broker.zig): `seedDisplay(base, w, h,
|
||||
- [x] [devices-broker.zig](../../system/kernel/devices-broker.zig): `seedDisplay(base, w, h,
|
||||
pitch, format)` publishes a root-level `display` node with one WC-flagged `memory`
|
||||
resource `[base, height*pitch]` + the `DisplayInfo`; `displayDevice()` /
|
||||
`displayClaimed()`. Seeded from `kmain` after `devices_broker.init`.
|
||||
- [x] [process.zig](../system/kernel/process.zig) `systemMmioMap` + paging
|
||||
- [x] [process.zig](../../system/kernel/process.zig) `systemMmioMap` + paging
|
||||
(`mapUserDeviceInto` gains a `write_combining` bool): a resource's WC flag maps it
|
||||
through the WC PAT slot (`setupPat`) instead of strong-uncacheable.
|
||||
- [x] [console.zig](../system/kernel/console.zig): `setSuppressed` quiesces `write` while
|
||||
- [x] [console.zig](../../system/kernel/console.zig): `setSuppressed` quiesces `write` while
|
||||
the display device is claimed (driven from `systemDeviceClaim` / release); the
|
||||
terminal panic + exception paths clear it first so a dying machine still draws.
|
||||
|
||||
**Gate (met, automated):** the `display` kernel test (`python3 test/qemu_test.py display`,
|
||||
`displayTest` in [tests.zig](../system/kernel/tests.zig)) asserts the seeded node's shape
|
||||
`displayTest` in [tests.zig](../../system/kernel/tests.zig)) asserts the seeded node's shape
|
||||
and geometry, then walks the real claim + `mmio_map` path into a throwaway address space
|
||||
and verifies the leaf is **write-combining** (PAT entry 4: PAT bit set, PCD/PWT clear) —
|
||||
with an uncacheable-still-uncacheable regression guard. Chosen over the original
|
||||
@@ -67,10 +67,10 @@ Regression-checked: `discovery`, `ioport`, `claim-release`, `supervision`, `devi
|
||||
|
||||
Stand up the named service and the double-buffer, no layers yet.
|
||||
|
||||
- [x] `system/services/display/protocol.zig`: `Operation{ info, create_layer,
|
||||
- [x] `library/protocol/display/display-protocol.zig`: `Operation{ info, create_layer,
|
||||
configure_layer, destroy_layer, fill_rect, blit_tile, damage, present }`; `extern`
|
||||
`Request`/`Reply`; size + `maximum_payload` consts. (Model: block/protocol.zig.)
|
||||
- [x] [abi.zig](../system/abi.zig): `ServiceId.display = 9`.
|
||||
- [x] [abi.zig](../../system/abi.zig): `ServiceId.display = 9`.
|
||||
- [x] `system/services/display/display.zig`: `main` → enumerate + claim + WC-map the LFB
|
||||
(front) → `mmap` a cacheable back buffer of `height*pitch` → `runtime.service.run`.
|
||||
`info` and a whole-screen `present` (back → front) are live; layer ops fail-stub
|
||||
@@ -78,8 +78,8 @@ Stand up the named service and the double-buffer, no layers yet.
|
||||
- [x] [library/runtime/display.zig](../library/runtime/runtime.zig) (+ barrel export of
|
||||
`display` and `display_protocol`): `info()` and `present()`, cached `.display`
|
||||
lookup with retry (model: block.zig).
|
||||
- [x] [init.zig](../system/services/init/init.zig): `"display"` added to `boot_services`.
|
||||
- [x] [build.zig](../build.zig): `display-protocol` module on the runtime; `display` exe
|
||||
- [x] [init.zig](../../system/services/init/init.zig): `"display"` added to `boot_services`.
|
||||
- [x] [build.zig](../../build.zig): `display-protocol` module on the runtime; `display` exe
|
||||
via `addUserBinary`; packed into the initial-ramdisk; installed to
|
||||
`/system/services/display`.
|
||||
- [x] **Kernel fix the back buffer surfaced:** `mmap` was capped at 256 pages (1 MiB) by
|
||||
@@ -105,7 +105,7 @@ The heart: composite an ordered layer stack, present only what changed.
|
||||
- [x] `create_layer` / `configure_layer` (damages old + new footprints) / `destroy_layer`,
|
||||
`fill_rect`, `blit_tile` (reads the inline tile from the IPC payload, unaligned-safe),
|
||||
`damage`, `present`.
|
||||
- [x] Pure, host-tested [compositor.zig](../system/services/display/compositor.zig): `Rect`
|
||||
- [x] Pure, host-tested [compositor.zig](../../system/services/display/compositor.zig): `Rect`
|
||||
(intersect/unite), `Surface`, `fillRect`, `composite` (opaque, clipped to a damage
|
||||
rect), `blitTile`. `present` clears the damaged region to the wallpaper, paints the
|
||||
visible layers bottom-to-top (z-sorted), and flushes just that rect back → front (WC).
|
||||
@@ -148,10 +148,10 @@ still pass, and the default `zig build` is clean.
|
||||
- [x] The three integration cases exist and pass: `display` (D1 handoff, kernel),
|
||||
`display-service` (D2/D3 compositor + self-check), and `display-demo` (D4 full
|
||||
pipeline: spawn `display` + `display-demo`, match `display-demo: ok`) —
|
||||
[tests.zig](../system/kernel/tests.zig) + [qemu_test.py](../test/qemu_test.py). Plus
|
||||
[tests.zig](../../system/kernel/tests.zig) + [qemu_test.py](../../test/qemu_test.py). Plus
|
||||
the pure host tests (`zig build test`).
|
||||
- [x] [display.md](display.md) updated to the built state (the "Verifying it" section names
|
||||
the real cases); [README index](README.md) entry present (#19); the `display-track`
|
||||
the real cases); [README index](../README.md) entry present (#19); the `display-track`
|
||||
memory marked DONE with the commits.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py display display-service display-demo` all pass,
|
||||
@@ -16,11 +16,11 @@ lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run,
|
||||
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](coding-standards.md): spell out non-acronym abbreviations,
|
||||
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym abbreviations,
|
||||
kebab-case file names, no `Co-Authored-By` trailers. New user binaries go through
|
||||
`addUserBinary` and get packed into the initial-ramdisk; protocols are
|
||||
`b.addModule("…-protocol", …)` imported into `runtime`; new syscalls extend
|
||||
[abi.zig](../system/abi.zig) `SystemCall` + a `library/runtime` wrapper.
|
||||
[abi.zig](../../system/abi.zig) `SystemCall` + a `library/runtime` wrapper.
|
||||
|
||||
## How to verify along the way
|
||||
|
||||
@@ -59,7 +59,7 @@ is the only backend), and `zig build test` stays green.
|
||||
|
||||
## V2 — The shared-memory cross-process capability (kernel) ✅
|
||||
|
||||
- [x] [abi.zig](../system/abi.zig): `shared_memory_create` (34) / `shared_memory_map` (35) syscalls + a
|
||||
- [x] [abi.zig](../../system/abi.zig): `shared_memory_create` (34) / `shared_memory_map` (35) syscalls + a
|
||||
`shared_memory_test` service id. Handlers in process.zig: `shared_memory_create(len)` allocates contiguous,
|
||||
zeroed, **cacheable** frames, wraps them in a refcounted object, installs a capability
|
||||
handle, maps them into the caller's shared-memory arena → returns virtual_address + handle; `shared_memory_map(cap)`
|
||||
@@ -144,4 +144,4 @@ path in VMs**, where danos development happens. The framebuffer floor never goes
|
||||
- [display.md](display.md) — v1: the compositor, the GOP-vs-device split, the WC discipline.
|
||||
- [display-v2-plan.md](display-v2-plan.md) — the ordered build-out.
|
||||
- [driver-model.md](driver-model.md) — claim / `mmio_map` / MSI / capability passing (M13).
|
||||
- [resilience.md](resilience.md) — the restart machinery the hot-attach leans on.
|
||||
- [resilience.md](../os-development-guide/resilience.md) — the restart machinery the hot-attach leans on.
|
||||
@@ -1,12 +1,12 @@
|
||||
# The display service: a framebuffer compositor
|
||||
|
||||
The [framebuffer](framebuffer.md) the loader hands over is a flat block of pixel
|
||||
memory, and the kernel's [bootstrap console](../system/kernel/console.zig) draws text
|
||||
The [framebuffer](../os-development-guide/framebuffer.md) the loader hands over is a flat block of pixel
|
||||
memory, and the kernel's [bootstrap console](../../system/kernel/console.zig) draws text
|
||||
into it directly. That console is a stop-gap. The **display service**
|
||||
(`system/services/display/`) is the real thing: an ordinary ring-3 process that *owns*
|
||||
the framebuffer, composes a stack of **layers** into an off-screen back buffer, and
|
||||
**presents** finished frames to the screen — the display half of the GUI track
|
||||
([vision.md](vision.md)), the sibling of the [input service](input.md).
|
||||
([vision.md](../vision.md)), the sibling of the [input service](input.md).
|
||||
|
||||
This note is the architecture and the reasoning behind it. The concrete build order
|
||||
lives in [display-plan.md](display-plan.md).
|
||||
@@ -20,20 +20,20 @@ which one you're holding decides what you can do.
|
||||
|
||||
- **GOP is firmware's *temporary* driver** for the display controller. It gives you a
|
||||
linear framebuffer pointer and can set video modes — but only until
|
||||
`ExitBootServices`. The loader already leans on this: [`queryFramebuffer`](../boot/efi.zig)
|
||||
`ExitBootServices`. The loader already leans on this: [`queryFramebuffer`](../../boot/efi.zig)
|
||||
reads the monitor's EDID, picks the native mode, and calls `set_mode` **before**
|
||||
exiting ([gop.md](gop.md)). Once the kernel runs, GOP is **gone** — no `set_mode`, no
|
||||
exiting ([gop.md](../os-development-guide/gop.md)). Once the kernel runs, GOP is **gone** — no `set_mode`, no
|
||||
mode list, no EDID. What survives is the frozen snapshot in
|
||||
[`BootInformation.framebuffer`](../system/boot-handoff.zig): `{base, width, height,
|
||||
[`BootInformation.framebuffer`](../../system/boot-handoff.zig): `{base, width, height,
|
||||
pitch, format, refresh_hz}`, and nothing more.
|
||||
|
||||
- **The PCI class-0x03 device is the raw controller** — BARs, config space, registers,
|
||||
IO ports. It is what you actually *own* after boot. On QEMU's emulated adapter
|
||||
([`-device VGA,edid=on`](../build.zig), the Bochs VBE/DISPI model) the `base` GOP handed
|
||||
([`-device VGA,edid=on`](../../build.zig), the Bochs VBE/DISPI model) the `base` GOP handed
|
||||
you *is* that device's linear-framebuffer BAR — the same physical memory, seen through
|
||||
a different door. On a real discrete GPU, GOP's `base` is an aperture inside the GPU's
|
||||
VRAM BAR. danos already decodes this device
|
||||
([pci-class.zig](../system/devices/pci-class.zig) has the full `display` namespace, and
|
||||
([pci-class.zig](../../library/device/pci/pci-class.zig) has the full `display` namespace, and
|
||||
`pci-bus` already reports it to the [device manager](device-manager.md) with its class
|
||||
triple) — but nothing binds it yet.
|
||||
|
||||
@@ -63,16 +63,16 @@ rest of the system hasn't had to face:
|
||||
|
||||
1. **The framebuffer is kernel-only today.** It arrives through the boot handoff, is
|
||||
mapped into the kernel's physmap, and is touched only by
|
||||
[`console.zig`](../system/kernel/console.zig). It is *not* a
|
||||
[devices-broker](../system/kernel/devices-broker.zig) node, so
|
||||
[`console.zig`](../../system/kernel/console.zig). It is *not* a
|
||||
[devices-broker](../../system/kernel/devices-broker.zig) node, so
|
||||
`device.claim`/`mmio_map` cannot reach it, and there is no framebuffer
|
||||
[syscall](syscall.md). A user-space display service needs a **new mechanism just to
|
||||
[syscall](../os-development-guide/syscall.md). A user-space display service needs a **new mechanism just to
|
||||
touch the pixels**. (See "The handoff" below — this is built.)
|
||||
|
||||
2. **danos had no cross-process shared memory.** At v1 the memory syscalls were `mmap`
|
||||
(private, zeroed), `mmio_map` (a *claimed device's* MMIO), and `dma_alloc` (new
|
||||
pinned physical). The block driver's "pass a buffer by physical address" trick
|
||||
([block/protocol.zig](../system/services/block/protocol.zig)) works *only because its
|
||||
([block/protocol.zig](../../library/protocol/block/block-protocol.zig)) works *only because its
|
||||
consumer is DMA hardware*. A compositor that CPU-reads and blends client layers can't
|
||||
use it — it would have to *map* another process's memory, which nothing allowed. v1
|
||||
sidesteps it entirely (see "What v1 does not do"); v2 has since built the primitive
|
||||
@@ -96,17 +96,17 @@ rest of the system hasn't had to face:
|
||||
▼ reached by name (ipc_lookup); clients drive it over the display protocol
|
||||
┌────────────────────────────────────┬──────────────────────────────────────┐
|
||||
drawing clients (v1) surface clients (deferred)
|
||||
runtime.display commands: runtime.display surfaces:
|
||||
display commands: display surfaces:
|
||||
create_layer / configure_layer shared_memory_create → pass as a capability →
|
||||
fill_rect / blit_tile / damage the compositor maps & composites the
|
||||
present client-rendered bitmap directly
|
||||
```
|
||||
|
||||
The bring-up sequence mirrors a hardware driver's — it is the
|
||||
[`usb-xhci-bus` `initialise`](../system/drivers/usb-xhci-bus/usb-xhci-bus.zig) shape
|
||||
[`usb-xhci-bus` `initialise`](../../system/drivers/usb-xhci-bus/usb-xhci-bus.zig) shape
|
||||
(claim → `mmio_map` → run loop) — and the request/reply service shell is the
|
||||
[FAT](../system/services/fat/fat.zig) / [input](../system/services/input/input.zig) shape
|
||||
([`runtime.service.run`](../library/runtime/service.zig) with a `protocol.zig` of
|
||||
[FAT](../../system/services/fat/fat.zig) / [input](../../system/services/input/input.zig) shape
|
||||
([`service.run`](../../library/kernel/service.zig) with a `protocol.zig` of
|
||||
`extern struct` messages and an `Operation` tag).
|
||||
|
||||
**One process, for now.** v1 is a *single* service that both owns the framebuffer and
|
||||
@@ -119,12 +119,12 @@ second backend or a second monitor appears; until then it is complexity with no
|
||||
|
||||
The framebuffer crosses into user space through the machinery that already exists for
|
||||
every other device, rather than a bespoke syscall — so it inherits ownership,
|
||||
release-on-death, and re-claim-on-restart for free (the [resilience](resilience.md)
|
||||
release-on-death, and re-claim-on-restart for free (the [resilience](../os-development-guide/resilience.md)
|
||||
story: a crashed display service returns the LFB to the kernel, and its restart
|
||||
re-claims it).
|
||||
|
||||
- The kernel seeds a synthetic **display-class** node into the
|
||||
[devices-broker](../system/kernel/devices-broker.zig) at init (`seedDisplay`), from
|
||||
[devices-broker](../../system/kernel/devices-broker.zig) at init (`seedDisplay`), from
|
||||
`BootInformation.framebuffer`: one `ResourceKind.memory` resource spanning
|
||||
`[base, height*pitch]`, tagged **write-combining**, plus a small
|
||||
`DisplayInfo{width, height, pitch, format, refresh_hz}` (the memory resource says *where*
|
||||
@@ -135,7 +135,7 @@ re-claims it).
|
||||
- The service `device.claim`s it and `mmio_map`s the resource. The map is
|
||||
**write-combining**, not the strong-uncacheable that `mmio_map` uses for register
|
||||
MMIO. The kernel already programs a WC PAT slot for its own console
|
||||
([`setupPat`](../system/kernel/architecture/x86_64/paging.zig)); this reaches it from
|
||||
([`setupPat`](../../system/kernel/architecture/x86_64/paging.zig)); this reaches it from
|
||||
the user mapping path. **This matters:** an uncacheable framebuffer makes the
|
||||
back→front blit unusably slow.
|
||||
- On `claim`, the kernel's bootstrap console goes quiet, so the two never fight over the
|
||||
@@ -143,7 +143,7 @@ re-claims it).
|
||||
panic on screen wins.
|
||||
|
||||
The display service is a **named boot service**: `init` spawns it by name alongside
|
||||
`input`/`device-manager`/`fat` ([init.zig](../system/services/init/init.zig)), and it
|
||||
`input`/`device-manager`/`fat` ([init.zig](../../system/services/init/init.zig)), and it
|
||||
self-discovers the display node with `device.enumerate` (matching on `DeviceClass.display`). The [device manager](device-manager.md)
|
||||
matching path (PCI class 0x03 → a driver) is reserved for the future *native* backend, not
|
||||
this singleton synthetic node.
|
||||
@@ -160,8 +160,8 @@ Two buffers, with deliberately different memory types:
|
||||
|
||||
So a frame is: compose every dirty layer into the cacheable back buffer, then **present**
|
||||
— copy the changed regions back→front in sequential, WC-friendly writes. Two details the
|
||||
[framebuffer](framebuffer.md) note already establishes carry over: step rows by `pitch`,
|
||||
not `width*4`; and handle both `rgbx` and `bgrx` [pixel formats](gop.md).
|
||||
[framebuffer](../os-development-guide/framebuffer.md) note already establishes carry over: step rows by `pitch`,
|
||||
not `width*4`; and handle both `rgbx` and `bgrx` [pixel formats](../os-development-guide/gop.md).
|
||||
|
||||
## Flicker vs. tearing — what double buffering does and doesn't buy
|
||||
|
||||
@@ -188,10 +188,10 @@ The compositor holds an **ordered stack of layers**. Each layer has a rectangle,
|
||||
z-order, a visibility flag, and a surface. Presenting walks the stack bottom-to-top,
|
||||
painting each dirty layer into the back buffer, then flushes the damage to the front.
|
||||
Damage is tracked by one of two interchangeable trackers behind a compile-time
|
||||
`damage_mode` A/B switch ([display.zig](../system/services/display/display.zig)): a
|
||||
`damage_mode` A/B switch ([display.zig](../../system/services/display/display.zig)): a
|
||||
free-form dirty-rectangle **list** (tight bounds, heuristic merging) or a fixed 64-px
|
||||
**tile grid** (exact O(1) merging, tile-quantized repaints) — the grid is the default;
|
||||
[compositor.zig](../system/services/display/compositor.zig) has both, with the trade-off
|
||||
[compositor.zig](../../system/services/display/compositor.zig) has both, with the trade-off
|
||||
discussion.
|
||||
|
||||
In v1 the surfaces are **server-owned**, and clients draw into them with a small
|
||||
@@ -210,7 +210,7 @@ shell, a terminal, a cursor, and a wallpaper:
|
||||
| `present` | request a repaint: composited at the next frame-clock tick |
|
||||
|
||||
Text is intentionally *not* an operation — a client renders glyphs by blitting tiles
|
||||
(the [PSF font](../system/kernel/font.psf) path the console already uses can move into a
|
||||
(the [PSF font](../../system/kernel/font.psf) path the console already uses can move into a
|
||||
client). Keeping the protocol to rectangles and tiles keeps the compositor small and the
|
||||
policy in the client.
|
||||
|
||||
@@ -222,19 +222,19 @@ pacing on backends that have none (all of them today; see
|
||||
[display-v2.md](display-v2.md), "Fenced is not vsync"). Bring-up paths that must put
|
||||
pixels on screen synchronously (initialisation, the self-checks) bypass the clock.
|
||||
|
||||
## `runtime.display`
|
||||
## `display`
|
||||
|
||||
Clients speak the protocol through a new [`library/runtime/display.zig`](../library/runtime/runtime.zig),
|
||||
the [`runtime.block`](../library/runtime/block.zig) shape (a cached `.display` lookup
|
||||
Clients speak the protocol through a new [`library/client/display/display.zig`](../../library/client/display/display.zig),
|
||||
the [`block`](../../library/device/block/block.zig) shape (a cached `.display` lookup
|
||||
with a boot-race retry): `display.info()`, a `Layer` handle with `fill` / `blitTile` /
|
||||
`damage`, and `present()`. Application code never issues the raw syscalls — it calls the
|
||||
runtime, as with every other danos service.
|
||||
client module, as with every other danos service.
|
||||
|
||||
## The cursor: a mouse-listener thread feeding the compositor
|
||||
|
||||
The compositor is the single owner of the framebuffer — only the main `service.run` loop
|
||||
touches the backend and the layer stack. Tracking the mouse without breaking that
|
||||
ownership is the display's first use of [threads](threading.md): the service is built
|
||||
ownership is the display's first use of [threads](../os-development-guide/threading.md): the service is built
|
||||
multi-threaded (`addThreadedUserBinary`) and, at startup, spawns a **mouse-listener
|
||||
thread** beside the compositor loop.
|
||||
|
||||
@@ -242,9 +242,9 @@ thread** beside the compositor loop.
|
||||
(`input.subscribeMouse()`), accumulates the relative `dx`/`dy` motion into an absolute
|
||||
cursor position clamped to the screen, and hands it to the compositor. It never touches
|
||||
the compositor — so no lock guards the framebuffer. A parked `next()` leaves its core
|
||||
free to halt ([halting.md](halting.md)).
|
||||
free to halt ([halting.md](../os-development-guide/halting.md)).
|
||||
- **The channel.** A single-slot *latest-value* cell (`CursorChannel`) guarded by a
|
||||
`runtime.Thread.Mutex`: the renderer wants where the cursor *is now*, not a replay of
|
||||
`Thread.Mutex`: the renderer wants where the cursor *is now*, not a replay of
|
||||
every delta, so a new position overwrites the old. The listener also **pokes** the
|
||||
compositor awake — the main loop is parked in `replyWait`, so the listener posts a
|
||||
zero-payload `ipc.send` to the compositor's endpoint, which arrives as a
|
||||
@@ -254,13 +254,13 @@ thread** beside the compositor loop.
|
||||
which is just a top-z compositor layer — with the existing `configure` + `present` path
|
||||
(it damages the old and new footprints, so only those two rectangles repaint).
|
||||
|
||||
Two threading facts shape this (both in [threading.md](threading.md)). IPC **handles do
|
||||
Two threading facts shape this (both in [threading.md](../os-development-guide/threading.md)). IPC **handles do
|
||||
not cross threads**, so the listener can't reuse the main loop's endpoint handle — it
|
||||
`ipc.lookup(.display)`s its *own* handle to the same endpoint to poke through. And a
|
||||
multi-threaded service doing concurrent IPC is why the kernel's endpoint-create / register
|
||||
/ lookup syscalls now serialize under the big kernel lock. Shared fate applies: a fault in
|
||||
the listener takes the whole display down, and the supervisor restarts the process
|
||||
([resilience.md](resilience.md)).
|
||||
([resilience.md](../os-development-guide/resilience.md)).
|
||||
|
||||
## What v1 does not do (and why that's fine)
|
||||
|
||||
@@ -284,7 +284,7 @@ both are clean additions behind the interfaces v1 establishes.
|
||||
|
||||
## Verifying it
|
||||
|
||||
Four QEMU test cases ([tests.zig](../system/kernel/tests.zig), `python3
|
||||
Four QEMU test cases ([tests.zig](../../system/kernel/tests.zig), `python3
|
||||
test/qemu_test.py <case>`), each layering on the last:
|
||||
|
||||
- **`display`** — the kernel handoff: the seeded `display` device is shaped correctly and
|
||||
@@ -297,7 +297,7 @@ test/qemu_test.py <case>`), each layering on the last:
|
||||
layer — logging `display: compositor self-check ok`.
|
||||
- **`display-demo`** — the full pipeline from a separate process: the hardware-free
|
||||
[`display-demo`](../system/services/display-demo/) client (the
|
||||
[`input-source`](../system/services/input-source/) analog) drives layers — a wallpaper and
|
||||
[`input-source`](../test/system/services/input-source/) analog) drives layers — a wallpaper and
|
||||
a sliding rectangle — through the layer client API and heartbeats
|
||||
`display-demo: ok`, proving a frame travelled client → compositor → screen, exactly as
|
||||
the [input test](input.md) proves an event travels source → service → subscriber. It draws
|
||||
@@ -317,8 +317,8 @@ packing are additionally covered by pure host unit tests under `zig build test`.
|
||||
|
||||
## See also
|
||||
|
||||
- [framebuffer.md](framebuffer.md) — the linear framebuffer, pitch vs. width, `volatile`.
|
||||
- [gop.md](gop.md) — GOP, and why only linear RGBX/BGRX modes are paintable.
|
||||
- [framebuffer.md](../os-development-guide/framebuffer.md) — the linear framebuffer, pitch vs. width, `volatile`.
|
||||
- [gop.md](../os-development-guide/gop.md) — GOP, and why only linear RGBX/BGRX modes are paintable.
|
||||
- [input.md](input.md) — the sibling service; the async `ipc_send` fan-out.
|
||||
- [driver-model.md](driver-model.md) — claim / `mmio_map`, capability passing, the trust model.
|
||||
- [device-manager.md](device-manager.md) — matching and supervision (the native backend's route).
|
||||
@@ -34,7 +34,7 @@ plain bus driver with no controller — a USB hub — is also a real thing.
|
||||
|
||||
danos already has the right central structure. `system/kernel/devices-broker.zig` holds a table of
|
||||
`DeviceDescriptor`, each with a parent, a class, and a set of resources. Firmware discovery
|
||||
seeds it ([discovery.md](discovery.md)); `device_register` grows it.
|
||||
seeds it ([discovery.md](../os-development-guide/discovery.md)); `device_register` grows it.
|
||||
|
||||
Three invariants make it a capability system rather than a directory:
|
||||
|
||||
@@ -95,40 +95,62 @@ A "family" is two modules, not one:
|
||||
- **A protocol module** — the IPC message types that let a class driver talk to
|
||||
*whatever* published its device. This is the part that makes class drivers portable.
|
||||
|
||||
danos already has one of each: `library/runtime/device.zig` is a logic module,
|
||||
[`system/vfs-protocol.zig`](system/vfs-protocol.zig) is a protocol module shared by the
|
||||
mount backends (today the fat server) and their clients. (The user-space VFS server it
|
||||
was originally written against has since retired — path routing moved into the kernel,
|
||||
`system/kernel/vfs.zig`'s `fs_resolve` — but the protocol module outlived it, which is
|
||||
rather the point.) The pattern generalises directly:
|
||||
danos already has one of each: `library/device/pci/pci.zig` is a logic module (the
|
||||
`Function` view of a claimed PCI function),
|
||||
[`library/protocol/vfs/vfs-protocol.zig`](../../library/protocol/vfs/vfs-protocol.zig) is a
|
||||
protocol module shared by the mount backends (today the fat server) and their clients.
|
||||
(The user-space VFS server it was originally written against has since retired — path
|
||||
routing moved into the kernel, `system/kernel/vfs.zig`'s `fs_resolve` — but the protocol
|
||||
module outlived it, which is rather the point.) The pattern generalises directly:
|
||||
|
||||
```
|
||||
library/
|
||||
runtime/ module "runtime" — syscalls, heap, ipc, device, stdio
|
||||
mmio/ module "mmio" — volatile register access + barriers [M14]
|
||||
bus/
|
||||
pci/ module "pci" — ECAM, BAR decode, capability walk
|
||||
usb/ module "usb" — descriptors, control transfers, hubs
|
||||
proto/
|
||||
vfs/ module "vfs-protocol" (today: system/vfs-protocol.zig)
|
||||
block/ module "block-protocol"
|
||||
hid/ module "hid-protocol"
|
||||
kernel/ the system library (kernel32-style): the syscall surface split by concern
|
||||
— ipc, memory (heap/dma/shared-memory), process, time, logging,
|
||||
file-system, thread, service, plus system-call stubs + start/root
|
||||
device/ device code grouped by domain; each domain splits into a shareable
|
||||
data module (enums/wire types, std-only) and a logic module (mmio/IPC)
|
||||
mmio/ module "mmio" — typed volatile register access + barriers [M14]
|
||||
model/ module "device-abi" — DeviceDescriptor, DeviceClass, ResourceKind
|
||||
pci/ "pci-class" (data) + "pci" — config/BAR/capability walk (Function)
|
||||
usb/ "usb-abi" + "usb-ids" (data) + "usb" — descriptors, control/interrupt/bulk client
|
||||
acpi/ "acpi-ids" (data) + "aml" — _HID names, the AML interpreter
|
||||
driver/ module "driver" — device-access syscalls + device-manager hello
|
||||
block/ module "block" — the block-device client (a device type)
|
||||
client/ userspace service clients — display, input (a program's view of a service)
|
||||
protocol/ driver <-> service wire contracts, one module per directory
|
||||
vfs/ block/ display/ scanout/ input/ power/ device-manager/ usb-transfer/
|
||||
|
||||
system/drivers/ one sub-project each → /system/drivers (no `d` suffix)
|
||||
xhci/ HCD + bus driver imports runtime, pci, usb, mmio
|
||||
usb-hid/ class driver imports runtime, usb, hid-protocol
|
||||
block/ class driver imports runtime, block-protocol
|
||||
usb-xhci-bus/ HCD + bus driver imports usb, mmio, usb-transfer-protocol (+ kernel modules)
|
||||
usb-hid/ class driver imports usb, input-protocol (+ kernel modules)
|
||||
virtio-gpu/ scanout driver imports pci, mmio, display-/scanout-protocol (+ kernel modules)
|
||||
```
|
||||
|
||||
The split by *dependency weight* is what lets the microkernel stay out of device
|
||||
business: it imports only the `device-abi` data module (the descriptor types its broker
|
||||
marshals across the syscall boundary) — never a logic module, never a taxonomy. That one
|
||||
pure-data import is the only edge from `system/kernel/` into `library/`; decoding a class
|
||||
code or `_HID` to a name is user space's job (the device manager owns those taxonomies).
|
||||
|
||||
A protocol lives in `library/protocol/` when it is the seam between a low-level driver and
|
||||
a higher-level service (block ↔ filesystem, a scanout driver ↔ the compositor). A driver's
|
||||
private wire to its *hardware* — virtio-gpu's command set — is not that; it stays a
|
||||
driver-private file, like the virtio-pci transport beside it.
|
||||
|
||||
The build side of this has since landed: [`addUserBinary`](build.zig) injects the
|
||||
default modules (`runtime`, `mmio`, `xkeyboard-config`, `acpi-ids`) into every user
|
||||
default modules — the library/kernel concern modules (`ipc`, `memory`, `process`, `time`,
|
||||
`logging`, `file-system`, `thread`, `service`), the device/service clients (`driver`,
|
||||
`block`, `display`, `input`), plus `mmio`, `xkeyboard-config`, `acpi-ids` — into every user
|
||||
binary, and per-binary extras — protocol modules, bus logic — are added with
|
||||
`programModule(exe).addImport(...)`. That's the *entire* mechanism — Zig modules
|
||||
already give you everything else.
|
||||
|
||||
The discipline that makes this work: **a class driver must not import a bus's logic
|
||||
module.** `usbhid` imports `proto.hid` and `usb` (for descriptor types), never `pci`.
|
||||
If a class driver needs `mmio`, it has become an HCD and should be one.
|
||||
The discipline that makes this work: **a class driver must not import a bus's *hardware*
|
||||
logic module.** `usb-hid` imports `usb` (the transfer client) and `input-protocol`, never
|
||||
`pci` and never `mmio`. If a class driver needs `mmio`, it has become an HCD and should be
|
||||
one. The domain data modules (`usb-abi`, `usb-ids`, `pci-class`) carry no such weight — a
|
||||
class driver, the device manager, or the kernel may share them freely.
|
||||
|
||||
## What exists today
|
||||
|
||||
@@ -141,16 +163,16 @@ If a class driver needs `mmio`, it has become an HCD and should be one.
|
||||
and a `received_cap` return (r8): an endpoint travels with a message, installed into
|
||||
the receiver's handle table (shared, refcount-bumped — a copy, not a move). A full
|
||||
table fails `-ENOSPC` and does not half-deliver. This is the "open" primitive — a bus
|
||||
driver mints a per-device endpoint and hands it to a class driver. The runtime exposes
|
||||
`callCap` and `replyWait(..., send_cap)`, and class drivers consume them now: the
|
||||
PS/2 keyboard and mouse drivers attach to ps2-bus this way, and `runtime.usb` /
|
||||
`runtime.input` open their per-device and subscription channels with `callCap`.
|
||||
- **M14** — DMA memory + the memory-ordering layer. `/lib/mmio` gives drivers typed
|
||||
volatile access and `mb`/`rmb`/`wmb` (per-arch); `dma_alloc`/`dma_free` grant
|
||||
driver mints a per-device endpoint and hands it to a class driver. The `ipc` module
|
||||
exposes `callCap` and `replyWait(..., send_cap)`, and class drivers consume them now: the
|
||||
PS/2 keyboard and mouse drivers attach to ps2-bus this way, and the `usb` / `input`
|
||||
client modules open their per-device and subscription channels with `callCap`.
|
||||
- **M14** — DMA memory + the memory-ordering layer. `/lib/device/mmio` gives drivers typed
|
||||
volatile access and `memoryBarrier`/`readMemoryBarrier`/`writeMemoryBarrier` (per-arch); `dma_alloc`/`dma_free` grant
|
||||
physically-contiguous, pinned, uncacheable, reclaim-on-teardown buffers with the
|
||||
physical address exposed (`pmm.allocContiguous`, a DMA arena, `mapUserDmaInto`).
|
||||
`dma_below_4g` caps the address for legacy engines; `dma_write_combining` is accepted
|
||||
but falls back to coherent until PAT is programmed. The bus drivers use `/lib/mmio`,
|
||||
but falls back to coherent until PAT is programmed. The bus drivers use `/lib/device/mmio`,
|
||||
and `dma_alloc` has real consumers now: the xHCI driver's rings and contexts,
|
||||
usb-storage's command/status wrappers, virtio-gpu's virtqueue, and the fat
|
||||
service's bounce buffer.
|
||||
@@ -177,7 +199,7 @@ If a class driver needs `mmio`, it has become an HCD and should be one.
|
||||
`system_spawn(name, arguments)` loads a binary bundled in the initial-ramdisk as a
|
||||
fresh ring-3 process; `name` becomes the child's argv[0] and the optional
|
||||
NUL-separated `arguments` blob its argv[1..], delivered on a SysV entry stack
|
||||
([sysv.md](sysv.md)). This is what
|
||||
([sysv.md](../os-development-guide/sysv.md)). This is what
|
||||
turned the device manager from "log the match" into "run the driver": the kernel now
|
||||
spawns only `init`, `init` spawns the services, and the **device-manager** discovers
|
||||
the hardware and spawns each driver ([drivers.md](drivers.md)). Ungated for now — a
|
||||
@@ -237,7 +259,7 @@ const dev_ep = ipc.callCap(h, // ... mint a per-device endpoint,
|
||||
|
||||
## M14 — DMA memory and the memory-ordering contract, for HCDs ✅ done
|
||||
|
||||
*Implemented: `/lib/mmio` (typed volatile access + `mb`/`rmb`/`wmb`, per-arch) and
|
||||
*Implemented: `/lib/device/mmio` (typed volatile access + `memoryBarrier`/`readMemoryBarrier`/`writeMemoryBarrier`, per-arch) and
|
||||
`dma_alloc`/`dma_free` (contiguous, pinned, uncacheable, reclaim-on-teardown, physical
|
||||
address exposed). `dma_write_combining` still falls back to coherent — real WC needs
|
||||
PAT, a small follow-up. The rest of this section is the original design note.*
|
||||
@@ -279,31 +301,31 @@ doorbell.* = i; // volatile store to UC MMIO
|
||||
// nothing stops the compiler reordering these; the device reads a stale descriptor
|
||||
```
|
||||
|
||||
So the rules, which belong in `library/mmio.zig` and behind `arch`:
|
||||
So the rules, which belong in `library/device/mmio/mmio.zig` and behind `arch`:
|
||||
|
||||
| Situation | Required |
|
||||
|---|---|
|
||||
| MMIO register read/write | `mmio.read` / `mmio.write` (volatile) |
|
||||
| Fill DMA descriptor, then ring doorbell | `wmb()` between them |
|
||||
| Woken by IRQ, then read what the device wrote | `rmb()` before the read |
|
||||
| MMIO write that must complete before the next read | `mb()` |
|
||||
| Fill DMA descriptor, then ring doorbell | `writeMemoryBarrier()` between them |
|
||||
| Woken by IRQ, then read what the device wrote | `readMemoryBarrier()` before the read |
|
||||
| MMIO write that must complete before the next read | `memoryBarrier()` |
|
||||
|
||||
And the per-arch lowering — the reason this must be an `arch` primitive and not a
|
||||
sprinkling of `asm volatile`:
|
||||
|
||||
| | x86_64 | aarch64 |
|
||||
|---|---|---|
|
||||
| `mb()` | `mfence` | `dsb sy` |
|
||||
| `rmb()` | `lfence` | `dsb ld` |
|
||||
| `wmb()` | `sfence` | `dsb st` |
|
||||
| `memoryBarrier()` | `mfence` | `dsb sy` |
|
||||
| `readMemoryBarrier()` | `lfence` | `dsb ld` |
|
||||
| `writeMemoryBarrier()` | `sfence` | `dsb st` |
|
||||
| DMA cache coherency | coherent; nothing to do | **not guaranteed**; needs non-cacheable buffers or cache maintenance |
|
||||
|
||||
x86 is forgiving here — TSO plus strong-uncacheable MMIO means you usually get away
|
||||
with a compiler barrier alone. ARM is not, and [vision.md](vision.md) makes ARM the win
|
||||
with a compiler barrier alone. ARM is not, and [vision.md](../vision.md) makes ARM the win
|
||||
condition. Build the abstraction while there is one caller to fix.
|
||||
|
||||
(Zig note: `@fence` was **removed in 0.16**. Use `@atomicRmw(..., .seq_cst)` for a full
|
||||
barrier, or per-arch inline asm — which is what `library/mmio.zig` should hide.)
|
||||
barrier, or per-arch inline asm — which is what `library/device/mmio/mmio.zig` should hide.)
|
||||
|
||||
## M15 — interrupts for PCI devices ✅ done (MSI)
|
||||
|
||||
@@ -376,6 +398,6 @@ from hand-rolling `*volatile` and getting ARM wrong.
|
||||
## See also
|
||||
|
||||
- [drivers.md](drivers.md) — how to write one, concretely.
|
||||
- [discovery.md](discovery.md) / [acpi.md](acpi.md) — where the device table comes from.
|
||||
- [discovery.md](../os-development-guide/discovery.md) / [acpi.md](../os-development-guide/acpi.md) — where the device table comes from.
|
||||
- [ipc.md](ipc.md) — endpoints, badges, and the notification path an IRQ arrives on.
|
||||
- [resilience.md](resilience.md) — restart, the reason any of this is worth the trouble.
|
||||
- [resilience.md](../os-development-guide/resilience.md) — restart, the reason any of this is worth the trouble.
|
||||
@@ -4,13 +4,13 @@ In a monolithic kernel a driver is a function call away from everything: it runs
|
||||
ring 0, dereferences any physical address, and its interrupt handler *is* the ISR. In
|
||||
danos a driver is **an ordinary ring-3 process**. It has its own address space, it
|
||||
can crash without taking the kernel with it, and — the point of this document — it
|
||||
can be restarted ([resilience](resilience.md)).
|
||||
can be restarted ([resilience](../os-development-guide/resilience.md)).
|
||||
|
||||
That leaves three questions the kernel has to answer, because a process can't answer
|
||||
them for itself:
|
||||
|
||||
1. **What hardware exists?** → `device_enumerate`, over the device table discovery built
|
||||
([discovery](discovery.md), [acpi](acpi.md)).
|
||||
([discovery](../os-development-guide/discovery.md), [acpi](../os-development-guide/acpi.md)).
|
||||
2. **How do I touch its registers?** → `device_claim` + `mmio_map`: the kernel maps the
|
||||
device's physical MMIO window into your address space, and from then on it's plain
|
||||
memory. No syscall per register access.
|
||||
@@ -51,7 +51,7 @@ the optional arguments its argv[1..], on a SysV entry stack, see sysv.md). Every
|
||||
is the **driver supervisor**. It does the three steps a monolithic kernel would do in
|
||||
its probe path, entirely from ring 3:
|
||||
1. **Discover** — `device_enumerate` snapshots the device table the kernel built from
|
||||
ACPI/PCI ([discovery](discovery.md)).
|
||||
ACPI/PCI ([discovery](../os-development-guide/discovery.md)).
|
||||
2. **Match** — for each device it looks up a driver. The match policy is code, a few
|
||||
small per-bus tables: from the boot snapshot only the PCI host bridge matches
|
||||
(→ `pci-bus`); everything else arrives later as bus reports and matches on
|
||||
@@ -75,7 +75,7 @@ capability yet.
|
||||
## The capability: claim before touch
|
||||
|
||||
The driver syscall numbers (`system/abi.zig`) with the device types they carry
|
||||
(`system/devices/device-abi.zig`), dispatched in `system/kernel/process.zig`:
|
||||
(`library/device/model/device-abi.zig`), dispatched in `system/kernel/process.zig`:
|
||||
|
||||
| # | Call | Meaning |
|
||||
|---|------|---------|
|
||||
@@ -269,7 +269,7 @@ already owns.
|
||||
A device with **no resources** is legal and common. A USB device is reached through its
|
||||
controller, not by MMIO, so it gets `resource_count = 0`.
|
||||
|
||||
See [`system/drivers/pci-bus/pci-bus.zig`](../system/drivers/pci-bus/pci-bus.zig) for a
|
||||
See [`system/drivers/pci-bus/pci-bus.zig`](../../system/drivers/pci-bus/pci-bus.zig) for a
|
||||
real one — it claims a PCI host bridge, maps its ECAM window, and publishes each function
|
||||
it finds as a child — and [driver-model.md](driver-model.md) for how bus drivers, class
|
||||
drivers and host controller drivers fit together.
|
||||
@@ -290,10 +290,10 @@ Several things this list used to warn about are now available (see
|
||||
[driver-model.md](driver-model.md)): **port I/O** (`io_read`/`io_write`, claim-gated by
|
||||
the device's `io_port` resource — direct ring-3 `in`/`out` is still a #GP, so a PS/2 or
|
||||
16550 driver goes through these), **DMA memory** (`dma_alloc`: contiguous, pinned,
|
||||
uncacheable, physical address exposed), **memory barriers** (`library/mmio`'s
|
||||
`mb`/`rmb`/`wmb`, imported as the `mmio` module), **fault isolation** (a ring-3 fault kills only the faulting
|
||||
uncacheable, physical address exposed), **memory barriers** (`library/device/mmio`'s
|
||||
`memoryBarrier`/`readMemoryBarrier`/`writeMemoryBarrier`, imported as the `mmio` module), **fault isolation** (a ring-3 fault kills only the faulting
|
||||
process — `killCurrentProcess` — and the machine keeps running,
|
||||
[resilience](resilience.md)), and **reclaim + restart on death** (every path out of a
|
||||
[resilience](../os-development-guide/resilience.md)), and **reclaim + restart on death** (every path out of a
|
||||
process releases its claims and IRQ/MSI bindings — `releaseAllOwnedBy`,
|
||||
`irq.releaseOwner` — and the device manager respawns the driver with backoff,
|
||||
[device-manager.md](device-manager.md)). What remains:
|
||||
@@ -391,10 +391,10 @@ the first DMA driver to protect and test against) and these smaller items:
|
||||
Claiming and mapping is half of being a danos driver; the other half is the
|
||||
**lifecycle and protocol contract**, and the runtime makes it nearly free:
|
||||
|
||||
- Build on `runtime.service.run` — one replyWait loop folding protocol
|
||||
- Build on `service.run` — one replyWait loop folding protocol
|
||||
requests, signals, and notifications into callbacks. The harness answers the
|
||||
universal zero-length ping and turns `terminate` into a clean exit for you
|
||||
([process-lifecycle.md](process-lifecycle.md)).
|
||||
([process-lifecycle.md](../os-development-guide/process-lifecycle.md)).
|
||||
- A driver spawned with an assignment (its device id as argv[1]) sends the
|
||||
versioned `hello` to the device manager inside the deadline, and a **bus**
|
||||
driver reports what it discovers with `child_added`
|
||||
@@ -5,14 +5,14 @@ window server, a logger. None of them owns the hardware, and the driver should n
|
||||
who is listening. So between the drivers and the listeners sits the **input service**
|
||||
(`system/services/input/`): drivers **publish** events to it, programs **subscribe**, and
|
||||
it fans each event out to every interested subscriber. It is an ordinary ring-3 process
|
||||
reached over IPC, like the [FAT server](../system/services/fat/fat.zig) — no kernel knows
|
||||
reached over IPC, like the [FAT server](../../system/services/fat/fat.zig) — no kernel knows
|
||||
what a key is.
|
||||
|
||||
## One service, several device classes
|
||||
|
||||
The service carries three device classes today — **keyboard**, **mouse**, and
|
||||
**joystick/gamepad** — and is built to take more
|
||||
([protocol.zig](../system/services/input/protocol.zig)). Each class has its own typed
|
||||
([protocol.zig](../../library/protocol/input/input-protocol.zig)). Each class has its own typed
|
||||
event:
|
||||
|
||||
- `KeyEvent` — `key_down`/`key_up` (physical make/break) and `key_press` (a character was
|
||||
@@ -45,7 +45,7 @@ consequences decide the whole design:
|
||||
`ipc_call`, it would block until each subscriber replied. `ipc_call` has no timeout, and
|
||||
a subscriber's endpoint is an *unregistered* capability the kernel's death path cannot
|
||||
reach (since display v2's V6, `killOwnedEndpointsLocked` in
|
||||
[ipc-synchronous.zig](../system/kernel/ipc-synchronous.zig) marks a dead owner's
|
||||
[ipc-synchronous.zig](../../system/kernel/ipc-synchronous.zig) marks a dead owner's
|
||||
*registered* endpoints dead and wakes parked callers with `-EPEER` — but unregistered
|
||||
ones just drop with the task's handle table). One subscriber that exits mid-delivery
|
||||
would wedge input for everyone. That is the opposite of the resilience the microkernel
|
||||
@@ -66,7 +66,7 @@ the badge (distinguishing it from a bare IRQ/child-exit notification), the sende
|
||||
in the low bits, and the payload in the receive buffer, with no reply owed. The queue holds
|
||||
16 messages per endpoint; a full queue **drops the oldest**, because a buffered message is
|
||||
discrete data, not a coalescing "level" like an interrupt. See
|
||||
[ipc-synchronous.zig](../system/kernel/ipc-synchronous.zig) (`sendLocked`, `popPost`, and
|
||||
[ipc-synchronous.zig](../../system/kernel/ipc-synchronous.zig) (`sendLocked`, `popPost`, and
|
||||
the `replyWait` receive loop).
|
||||
|
||||
This is the async counterpart of `ipc_call`, and the input service is its first consumer.
|
||||
@@ -89,7 +89,7 @@ This is the async counterpart of `ipc_call`, and the input service is its first
|
||||
- A **subscriber** calls `input.subscribe(mask)` — or a typed helper: `subscribeKeyboard()`,
|
||||
`subscribeMouse()`, `subscribeJoystick()` (one class, `next()` returns the decoded event),
|
||||
or `subscribeAll()` (every class, `next()` returns a tagged `InputEvent`)
|
||||
([library/runtime/input.zig](../library/runtime/input.zig)). It creates its own endpoint
|
||||
([library/client/input/input.zig](../../library/client/input/input.zig)). It creates its own endpoint
|
||||
and hands it to the service as a **capability** (M13 capability passing — the input
|
||||
service is that feature's first real user), along with its `device_mask`. Then it loops on
|
||||
`next()`, a `replyWait` on that endpoint returning each pushed event.
|
||||
@@ -98,7 +98,7 @@ This is the async counterpart of `ipc_call`, and the input service is its first
|
||||
`publishJoystickEvent`. Publishing is a short synchronous `ipc_call` the service answers at
|
||||
once; the service's own fan-out is asynchronous, so publishing never blocks on a slow
|
||||
subscriber.
|
||||
- The **service** ([input.zig](../system/services/input/input.zig)) keeps a small subscriber
|
||||
- The **service** ([input.zig](../../system/services/input/input.zig)) keeps a small subscriber
|
||||
table (endpoint handle + owning task id + `device_mask`). On `publish` it `ipc_send`s the
|
||||
event to every subscriber whose mask includes the event's device class. On `subscribe` it
|
||||
stores the passed capability and mask and, as housekeeping, prunes any slot whose owning
|
||||
@@ -113,18 +113,18 @@ the service delivers to its endpoint, which only the same thread could receive).
|
||||
|
||||
- **The keyboard is real.** The `ps2-bus` driver owns PNP0303, which carries *both* the
|
||||
0x60/0x64 ports and IRQ1, so reading the hardware lives in the bus, not in
|
||||
[keyboard.zig](../system/drivers/ps2-bus/keyboard.zig): the bus binds IRQ1 and, on each
|
||||
[keyboard.zig](../../system/drivers/ps2-bus/keyboard.zig): the bus binds IRQ1 and, on each
|
||||
interrupt, drains port 0x60, routing every byte by the status register's
|
||||
auxiliary-output bit to whichever child driver **attached** for that device (an
|
||||
`AttachRequest` to the well-known `ps2_bus` service, carrying the child's endpoint as a
|
||||
capability; the bytes then arrive as asynchronous `ForwardedByte` messages, so the IRQ
|
||||
path never blocks on a child). The keyboard driver decodes the stream — scancode **set 2**,
|
||||
what the keyboard sends with the 8042's legacy translation off, decoded by
|
||||
[scancode.zig](../system/drivers/ps2-bus/scancode.zig) into USB HID usage keycodes with
|
||||
[scancode.zig](../../system/drivers/ps2-bus/scancode.zig) into USB HID usage keycodes with
|
||||
make/break, typematic-repeat, and modifier tracking (host-tested under `zig build test`) —
|
||||
and publishes real `key_down`/`key_press`/`key_up` events.
|
||||
- **Keycode → character** is wired in: the keyboard driver fills a `key_press` event's
|
||||
`character` through [`library/xkeyboard-config`](../library/xkeyboard-config/README.md)
|
||||
`character` through [`library/xkeyboard-config`](../../library/xkeyboard-config/README.md)
|
||||
(`xkb.map(layout, keycode, mods)` → keysym + Unicode character), synthesizing the ASCII
|
||||
control characters for Enter/Tab/Backspace/Escape, whose keysyms map to no Unicode. The
|
||||
layout defaults to `us`; the bus can pass another as the driver's argv[2] — the seam for
|
||||
@@ -132,9 +132,9 @@ the service delivers to its endpoint, which only the same thread could receive).
|
||||
- **The mouse is real too.** IRQ12 is enumerated on the auxiliary device's own ACPI node
|
||||
(PNP0F13), so the bus claims that node alongside the controller and routes both IRQs to
|
||||
its one endpoint, acking whichever line the notification's badge names.
|
||||
[mouse.zig](../system/drivers/ps2-bus/mouse.zig) attaches the way the keyboard does and
|
||||
[mouse.zig](../../system/drivers/ps2-bus/mouse.zig) attaches the way the keyboard does and
|
||||
assembles the forwarded bytes with
|
||||
[mouse-packet.zig](../system/drivers/ps2-bus/mouse-packet.zig) (three-byte stream-mode
|
||||
[mouse-packet.zig](../../system/drivers/ps2-bus/mouse-packet.zig) (three-byte stream-mode
|
||||
packets: sync/overflow handling, nine-bit movement, screen-convention `dy` — host-tested
|
||||
under `zig build test`) into `button_down`/`button_up` transitions and `motion` events.
|
||||
**Follow-up:** the IntelliMouse magic-knock for a scroll wheel (four-byte packets) and
|
||||
@@ -150,7 +150,7 @@ the service delivers to its endpoint, which only the same thread could receive).
|
||||
## Verifying it
|
||||
|
||||
The `input` case (`python3 test/qemu_test.py input`, in
|
||||
[tests.zig](../system/kernel/tests.zig) `inputTest`) boots the real kernel and spawns the
|
||||
[tests.zig](../../system/kernel/tests.zig) `inputTest`) boots the real kernel and spawns the
|
||||
service, the synthetic source (which cycles keyboard, mouse, and joystick events), and a
|
||||
subscriber that took all three classes. It passes only when the subscriber heartbeats
|
||||
`input-test: ok` — proof that an event travelled source → service → subscriber over IPC,
|
||||
@@ -160,5 +160,5 @@ serial line names the class received, so the log shows all three arriving on one
|
||||
## See also
|
||||
|
||||
- [ipc.md](ipc.md) — the synchronous rendezvous and the notification path `ipc_send` extends.
|
||||
- [syscall.md](syscall.md) — the system-call surface, including `ipc_send`.
|
||||
- [syscall.md](../os-development-guide/syscall.md) — the system-call surface, including `ipc_send`.
|
||||
- [driver-model.md](driver-model.md) — class drivers, capability passing (M13), the trust model.
|
||||
@@ -1,7 +1,7 @@
|
||||
# IPC: message-passing channels
|
||||
|
||||
Inter-process communication is the **backbone of a microkernel**. Once drivers and
|
||||
services run isolated in their own address spaces ([vision](vision.md)), they can't
|
||||
services run isolated in their own address spaces ([vision](../vision.md)), they can't
|
||||
just call each other — a request becomes a **message**. In a microkernel, whatever
|
||||
was a function call across a monolithic kernel is IPC, so it's a first-class
|
||||
concern, not an afterthought.
|
||||
@@ -18,7 +18,7 @@ There are two layers, built a milestone apart:
|
||||
|
||||
The first form is a **bounded blocking channel** (`system/kernel/ipc.zig`): a fixed-size
|
||||
ring buffer of messages with a producer/consumer rendezvous, built on the
|
||||
scheduler's [wait queues](scheduling.md).
|
||||
scheduler's [wait queues](../os-development-guide/scheduling.md).
|
||||
|
||||
`Channel(T, capacity)` is generic over the message type and buffer size. It holds a
|
||||
ring buffer, a count, and two wait queues:
|
||||
@@ -37,7 +37,7 @@ Two details make it correct:
|
||||
rather than assuming the slot is still available — another waiter may have taken
|
||||
it first. This is the standard guard against spurious or racing wakeups.
|
||||
- **One critical section.** `send`/`receive` run under the [big kernel
|
||||
lock](smp.md) (`sync.enter` / `sync.leave`), which disables interrupts on this
|
||||
lock](../os-development-guide/smp.md) (`sync.enter` / `sync.leave`), which disables interrupts on this
|
||||
core *and* takes the kernel's one spinlock — since SMP, the interrupt flag alone
|
||||
is not atomicity, because `cli` on one core does nothing to another. So checking
|
||||
the condition and committing the block/enqueue happen atomically both with respect
|
||||
@@ -47,7 +47,7 @@ Two details make it correct:
|
||||
|
||||
## Verifying it
|
||||
|
||||
The `ipc` test (see [testing.md](testing.md)) runs a producer and a consumer passing
|
||||
The `ipc` test (see [testing.md](../testing.md)) runs a producer and a consumer passing
|
||||
**100 messages through a 4-slot channel**. The small buffer means the channel goes
|
||||
full and empty over and over, so both the blocking-send and blocking-receive paths are
|
||||
exercised heavily. The messages arrive intact and in order (their sum is the
|
||||
@@ -114,19 +114,19 @@ This is what makes a user-space driver possible at all, and it's the subject of
|
||||
|
||||
## Lifecycle conventions over IPC (M17)
|
||||
|
||||
Three conventions from [process-lifecycle.md](process-lifecycle.md) ride the
|
||||
Three conventions from [process-lifecycle.md](../os-development-guide/process-lifecycle.md) ride the
|
||||
notification mechanism:
|
||||
|
||||
- **Signals** arrive as notifications on the endpoint a process nominated with
|
||||
`signal_bind` (`runtime.process.bindSignals`): badge = the signal bit plus the
|
||||
coalesced pending mask (`runtime.process.signalsFrom` decodes). Statements,
|
||||
`signal_bind` (`process.bindSignals`): badge = the signal bit plus the
|
||||
coalesced pending mask (`process.signalsFrom` decodes). Statements,
|
||||
never questions; no payload, no reply.
|
||||
- **One-shot timers** (`timer_bind`, `runtime.system.timerOnce`) land as a
|
||||
- **One-shot timers** (`timer_bind`, `time.timerOnce`) land as a
|
||||
timer-bit notification — the timed wait: a service arms a deadline and keeps
|
||||
serving, instead of blocking in sleep.
|
||||
- **The universal ping**: a **zero-length request is the liveness probe**,
|
||||
answered with a zero-length reply by the service harness itself
|
||||
(`runtime.service.run`). No protocol's requests start at length zero, so the
|
||||
(`service.run`). No protocol's requests start at length zero, so the
|
||||
encoding cannot collide, and a wedged service simply fails to answer — which
|
||||
is the diagnosis. Deep health ("can I reach my hardware?") stays a per-service
|
||||
protocol message.
|
||||
+8
-5
@@ -20,6 +20,8 @@ Most modern Unix and Unix-like operating systems follow the FHS. DanOS has its o
|
||||
| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/pci-bus, /system/drivers/ps2-bus) |
|
||||
| /system/services | system-service binaries — init, the FAT server, and other user-mode servers (e.g. /system/services/init, /system/services/fat) |
|
||||
| /system/kernel | the kernel image |
|
||||
| /test | Test fixtures for the QEMU integration suite. Read-only and initrd-backed like /system, and its layout likewise mirrors the source tree (the repo's test/ directory). Present on development and test images; a volume without it still boots. |
|
||||
| /test/system/services | test-fixture binaries (e.g. /test/system/services/vfs-test, /test/system/services/thread-test) — the same path in the repo source tree and on the boot volume |
|
||||
| /tmp | Directory for temporary files (see also /var/tmp). Often not preserved between system reboots and may be severely size-restricted. |
|
||||
| /usr | Secondary hierarchy for read-only user data; contains the majority of (multi-)user utilities and applications. Should be shareable and read-only. |
|
||||
| /var | Variable files: files whose content is expected to continually change during normal operation of the system, such as logs, spool files, and temporary e-mail files. |
|
||||
@@ -47,13 +49,14 @@ addressed by device id. `/dev` is the much smaller set of devices that have a dr
|
||||
willing to serve them, addressed by name.
|
||||
|
||||
A device node is not a file the VFS can read. The bytes live in a driver process
|
||||
([drivers.md](drivers.md)), so opening a `/dev` name has to resolve to that driver's
|
||||
([drivers.md](../device-driver-development-guide/drivers.md)), so opening a `/dev` name has to resolve to that driver's
|
||||
IPC endpoint, and subsequent reads and writes are calls against it. Resolve-to-endpoint
|
||||
is exactly what the kernel's `fs_resolve` already does for any mounted backend, and
|
||||
`FileStatus.kind` is the field that marks a device node; **what is not implemented today
|
||||
is `/dev` itself** — no service mounts it. (The flat eight-node ramfs this section once
|
||||
described is retired: the kernel-resident VFS root in `system/kernel/vfs.zig` serves the
|
||||
read-only `/system` initrd mount with real directories and node kinds, and filesystem
|
||||
described is retired: the kernel-resident VFS root in `system/kernel/vfs.zig` serves a
|
||||
read-only initrd mount per top-level tree — `/system`, and `/test` on images that carry
|
||||
the fixtures — with real directories and node kinds, and filesystem
|
||||
backends such as the FAT server mount the rest.) The three sections below describe the
|
||||
intended shape, and are honest about which parts the kernel can already support.
|
||||
|
||||
@@ -87,9 +90,9 @@ A block driver is now **writable, but not yet memory-safe.** Every storage contr
|
||||
worth naming is a bus master: it is programmed by handing it the physical address of a
|
||||
descriptor ring and left to read and write memory on its own. That ring is exactly what
|
||||
**`dma_alloc`** now provides — physically contiguous, pinned, uncacheable, with its
|
||||
physical address disclosed — and **`/lib/mmio`**'s barriers order the descriptor writes
|
||||
physical address disclosed — and **`/lib/device/mmio`**'s barriers order the descriptor writes
|
||||
against the doorbell, and **`msi_bind`** delivers completions. So an AHCI or NVMe driver
|
||||
can be written today (the M14/M15 work in [driver-model.md](driver-model.md); the earlier
|
||||
can be written today (the M14/M15 work in [driver-model.md](../device-driver-development-guide/driver-model.md); the earlier
|
||||
"cannot host a block driver at all" is no longer true).
|
||||
|
||||
What is *not* yet true is that it is safe. A device programmed with an arbitrary physical
|
||||
@@ -1,15 +1,15 @@
|
||||
# The VFS wire protocol
|
||||
|
||||
> **Status:** built and spoken today between `runtime.fs` (the client) and the
|
||||
> **Status:** built and spoken today between `file_system` (the client) and the
|
||||
> filesystem BACKENDS (the FAT server). The mount router lives in the
|
||||
> **kernel** (`system/kernel/vfs.zig`): `fs_resolve` routes a path and either
|
||||
> serves it directly (the read-only /system initrd mount, via `fs_node`) or
|
||||
> redirects the caller to the owning backend's endpoint plus the rewritten
|
||||
> mount-relative path — after which the client speaks THIS protocol to the
|
||||
> backend, unchanged. The Zig source of truth is `system/vfs-protocol.zig`
|
||||
> backend, unchanged. The Zig source of truth is `library/protocol/vfs/vfs-protocol.zig`
|
||||
> (the `vfs-protocol` module), whose unit test pins a sample of the sizes
|
||||
> and values below. This page is the **language-neutral wire specification**
|
||||
> of that contract — what a Rust or C client implements ([vdso.md](vdso.md)
|
||||
> of that contract — what a Rust or C client implements ([vdso.md](../os-development-guide/vdso.md)
|
||||
> explains why the IPC protocols, not the syscall numbers, are danos's
|
||||
> public ABI).
|
||||
|
||||
@@ -113,7 +113,7 @@ Notes per operation:
|
||||
prefix maps a mount into the backend's namespace (fat serves `/mnt/usb`
|
||||
from its volume root and `/var` from its `/var` subtree).
|
||||
- **rename** — same-directory rename only: the backend compares the old and
|
||||
new parent paths and refuses a mismatch. The client (`runtime.fs`) refuses
|
||||
new parent paths and refuses a mismatch. The client (`file_system`) refuses
|
||||
earlier when the two paths resolve to different backend endpoints, but that
|
||||
check is coarser than "one mount" — one endpoint can serve several mounts
|
||||
(fat serves `/mnt/usb` and `/var`), so a cross-mount rename reaches the
|
||||
@@ -182,7 +182,7 @@ What a non-Zig implementation may rely on, and what it must not:
|
||||
|
||||
- Operation values, flag bits, `NodeKind` values, and struct layouts are
|
||||
**append-only and frozen once shipped**. The unit test in
|
||||
`system/vfs-protocol.zig` pins a sample of them (the `DirectoryEntry`
|
||||
`library/protocol/vfs/vfs-protocol.zig` pins a sample of them (the `DirectoryEntry`
|
||||
size, `NodeKind` 0–1, `Operation` values 0, 4 and 5); this page is the
|
||||
full record of the frozen values.
|
||||
- The 256-byte message ceiling is a property of the current IPC transport,
|
||||
@@ -19,10 +19,10 @@ UEFI configuration table
|
||||
BootInformation.acpi_rsdp (u64, in the loader↔kernel handoff) system/boot-handoff.zig
|
||||
│ the kernel forwards the whole BootInformation
|
||||
▼
|
||||
platform.discover(boot_information, …) system/devices/platform.zig
|
||||
platform.discover(boot_information, …) system/kernel/platform.zig
|
||||
│ reads boot_information.acpi_rsdp, hands it to the ACPI backend
|
||||
▼
|
||||
acpi.discover(rsdp_phys, …) system/devices/acpi.zig
|
||||
acpi.discover(rsdp_phys, …) system/kernel/acpi.zig
|
||||
│ dereferences the RSDP, reads the pointer it contains
|
||||
▼
|
||||
RSDP ──(a field in the struct)──► RSDT / XSDT ──► SDTs (MADT, MCFG, FADT, HPET, DSDT…)
|
||||
@@ -75,9 +75,9 @@ There are really two independent questions, and it's worth not conflating them:
|
||||
- **`system/kernel/architecture/x86_64/paging.zig`** — the kernel's page tables and address-space
|
||||
management (see [paging.md](paging.md)).
|
||||
- **`system/kernel/architecture/x86_64/apic.zig`** / **`ioapic.zig`** — the Local APIC, its timer,
|
||||
and the I/O APIC for device interrupts (see [device-interrupts.md](device-interrupts.md)).
|
||||
and the I/O APIC for device interrupts (see [device-interrupts.md](../device-driver-development-guide/device-interrupts.md)).
|
||||
- **`system/kernel/architecture/x86_64/serial.zig`** / **`io.zig`** — the COM1 UART (the kernel's
|
||||
machine-readable log channel, see [testing.md](testing.md)) and the shared port-I/O + MSR primitives.
|
||||
machine-readable log channel, see [testing.md](../testing.md)) and the shared port-I/O + MSR primitives.
|
||||
- **`system/kernel/architecture/x86_64/smp.zig`** / **`per-cpu.zig`** — application-processor bring-up
|
||||
and per-CPU state (GS base, system-call entry point, see [scheduling.md](scheduling.md)).
|
||||
- **`system/kernel/architecture/x86_64/isr.s`** — the exception stubs, the `lgdt`/`lidt`/`ltr` load
|
||||
@@ -87,7 +87,7 @@ The Pi is not a "standard" ARM platform — expect Broadcom-specific peripherals
|
||||
Pi 5. Everything below is an offset from it.
|
||||
- **UART**: a **PL011** (at base + `0x20_1000`) plus a mini-UART; on some boards the
|
||||
PL011 is wired to Bluetooth, so which one is the console varies. This is the
|
||||
`aarch64`/`arm` equivalent of our x86 [COM1 serial](testing.md).
|
||||
`aarch64`/`arm` equivalent of our x86 [COM1 serial](../testing.md).
|
||||
- **Interrupt controller**: *not* a standard ARM GIC on the older parts — the Zero W
|
||||
and Pi 3 use Broadcom's own ARMCTRL controller (Pi 3 adds a per-core "local"
|
||||
controller for timers/mailboxes). The **Pi 4 and 5 do have a GIC-400**. So the
|
||||
@@ -122,4 +122,4 @@ Two routes, mirroring how we test x86-64 with OVMF:
|
||||
- [architecture.md](architecture.md) — the arch-module boundary these targets plug into, and the
|
||||
CPU-arch vs boot-protocol "two axes".
|
||||
- [efi.md](efi.md) — the UEFI loader that carries over to aarch64-UEFI.
|
||||
- [vision.md](vision.md) — why isolated, portable-across-architectures is the goal.
|
||||
- [vision.md](../vision.md) — why isolated, portable-across-architectures is the goal.
|
||||
@@ -36,7 +36,7 @@ Two things drive the need, and they set the timing:
|
||||
**aarch64 it's required to boot at all**. The [aarch64 port](arm.md) is what
|
||||
forces the issue.
|
||||
|
||||
2. **Isolated user-space drivers need it.** In the [microkernel vision](vision.md),
|
||||
2. **Isolated user-space drivers need it.** In the [microkernel vision](../vision.md),
|
||||
drivers live in user space — but something has to enumerate the hardware and hand
|
||||
each driver its MMIO regions and IRQs. That enumeration *is* device discovery. So
|
||||
discovery is a prerequisite for real drivers, **not** for user mode itself.
|
||||
@@ -132,7 +132,7 @@ when*:
|
||||
- **User-space enumeration: a device-manager server.** Everything else — PCI devices,
|
||||
peripherals — is parsed (or queried from the kernel's parse) by a privileged
|
||||
user-space server that hands each driver process its MMIO regions and IRQ rights
|
||||
over [IPC](ipc.md). Combined with **interrupts-as-messages** (an IRQ delivered to a
|
||||
over [IPC](../device-driver-development-guide/ipc.md). Combined with **interrupts-as-messages** (an IRQ delivered to a
|
||||
driver as a message on a channel — a natural extension of the wait queues and
|
||||
channels already built), that's what makes drivers genuinely isolated.
|
||||
|
||||
@@ -144,7 +144,7 @@ slice is unavoidably in-kernel.
|
||||
On ARMv8 the generic timer exposes its frequency directly via the `CNTFRQ` register —
|
||||
no calibration needed. That's cleaner than the x86 side, where we measure the LAPIC
|
||||
and TSC against the PIT because nothing tells us their frequency (see
|
||||
[device-interrupts.md](device-interrupts.md)). Discovery on ARM hands you more for
|
||||
[device-interrupts.md](../device-driver-development-guide/device-interrupts.md)). Discovery on ARM hands you more for
|
||||
free; discovery on x86 is partly about *finding* what ARM just tells you.
|
||||
|
||||
## Suggested ordering
|
||||
@@ -166,18 +166,18 @@ free; discovery on x86 is partly about *finding* what ARM just tells you.
|
||||
- [arm.md](arm.md) — the aarch64 target that forces genuine discovery (DTB, GIC).
|
||||
- [memory-map.md](memory-map.md) — the same loader-captures / kernel-consumes seam,
|
||||
and the note about grabbing the RSDP before `ExitBootServices`.
|
||||
- [device-interrupts.md](device-interrupts.md) — the LAPIC/timer bring-up that
|
||||
- [device-interrupts.md](../device-driver-development-guide/device-interrupts.md) — the LAPIC/timer bring-up that
|
||||
discovery will eventually feed (IOAPIC, real IRQ routing).
|
||||
- [ipc.md](ipc.md) — the channels that interrupts-as-messages and the device manager
|
||||
- [ipc.md](../device-driver-development-guide/ipc.md) — the channels that interrupts-as-messages and the device manager
|
||||
will ride on.
|
||||
- [vision.md](vision.md) — why drivers belong in isolated user space at all.
|
||||
- [vision.md](../vision.md) — why drivers belong in isolated user space at all.
|
||||
|
||||
## Update (M19.3, 2026-07-13): PCI enumeration left the kernel
|
||||
|
||||
The kernel now seeds only the `pci_host_bridge` node (ECAM window, MMIO
|
||||
apertures derived from the memory map's holes, bus range, and the 16-bit I/O
|
||||
window). The per-function walk moved to the ring-3 `pci-bus` driver
|
||||
([device-manager.md](device-manager.md)): it claims the bridge, repeats the
|
||||
([device-manager.md](../device-driver-development-guide/device-manager.md)): it claims the bridge, repeats the
|
||||
ECAM scan through its mmio grant, and `device_register`s what it finds, which
|
||||
the device manager mirrors and matches. The ACPI namespace walk follows in M20;
|
||||
the static tables (MADT, HPET, MCFG, FADT + `\\_S5`) stay kernel-side.
|
||||
@@ -190,7 +190,7 @@ for the host bridge, FADT); at this point it also still built the AML namespace
|
||||
but only to read the `\\_S5` sleep type for poweroff. (That remnant is gone too:
|
||||
the kernel now runs no AML at all — soft-off belongs to the acpi service, and the
|
||||
kernel keeps only the AML-free reboot path.) Device discovery is the ring-3 **acpi
|
||||
service** ([device-manager.md](device-manager.md)): it claims the `acpi-tables`
|
||||
service** ([device-manager.md](../device-driver-development-guide/device-manager.md)): it claims the `acpi-tables`
|
||||
node the kernel publishes (the AML blobs, a broad io_port grant, the SCI),
|
||||
re-parses the same blobs with the shared AML module, evaluates `_STA`/`_CRS`,
|
||||
and registers + reports each `_HID` device — the device manager matches drivers
|
||||
@@ -206,7 +206,7 @@ ring 0.)
|
||||
Moving PCI and ACPI enumeration out of ring 0 was not just a relocation — it
|
||||
made discovery **firmware-neutral by construction**, which is the whole reason
|
||||
to do it before the second architecture rather than after. Everything at and
|
||||
above the [device-manager](device-manager.md) protocol — descriptors,
|
||||
above the [device-manager](../device-driver-development-guide/device-manager.md) protocol — descriptors,
|
||||
containment, reports, matching, supervision — is generic and may never become
|
||||
x86-specific. Discovery is the single firmware-specific piece, and it is
|
||||
isolated as **one swappable process per firmware**:
|
||||
@@ -241,7 +241,7 @@ Two consequences of neutrality bind on later work:
|
||||
Two supporting decisions keep the kernel's remaining slice honest:
|
||||
|
||||
- **The AML interpreter is a single build module**
|
||||
(`system/devices/aml/aml.zig`) — one source, no fork. During the ring-3 move
|
||||
(`library/device/acpi/aml/aml.zig`) — one source, no fork. During the ring-3 move
|
||||
it was compiled into both the kernel (which linked it just for the `\_S5`
|
||||
poweroff evaluation) and the acpi service, with the `acpi-parse` test
|
||||
asserting the two produce the same device count. Since soft-off followed
|
||||
@@ -24,7 +24,7 @@ EFI/BOOT/BOOTX64.efi <- the "removable media" default for x86-64
|
||||
```
|
||||
|
||||
The boot volume is **FHS-shaped** (see the repository-layout note in
|
||||
[README.md](README.md)): `build.zig` installs `boot/efi.zig` (built for the `uefi`
|
||||
[README.md](../README.md)): `build.zig` installs `boot/efi.zig` (built for the `uefi`
|
||||
target) at `EFI/BOOT/BOOTX64.efi` — the one path UEFI firmware fixes — and lays
|
||||
the rest out by FHS path: the kernel at `system/kernel`, init at
|
||||
`system/services/init`, the pre-packed boot capsule at `boot/system.img`
|
||||
@@ -67,8 +67,8 @@ captures the **ACPI RSDP** from the UEFI configuration table (while boot
|
||||
services are still up), loads the system binaries into an in-RAM
|
||||
**initial ramdisk** (`loadSystemTree` — normally a single read of the pre-packed
|
||||
`boot\system.img` capsule, which already *is* the ramdisk wire format; it falls
|
||||
back to opening each manifest-listed path, and walks the `/system` tree only as
|
||||
a last resort for hand-assembled sticks — the capsule's format, builder, and
|
||||
back to opening each manifest-listed path, and walks the `/system` and `/test`
|
||||
trees only as a last resort for hand-assembled sticks — the capsule's format, builder, and
|
||||
fallback chain are documented in [system-image.md](system-image.md). Best-effort either way — a kernel-only
|
||||
volume still boots), and builds the **bootstrap page tables** the kernel starts
|
||||
life on (`buildBootstrapTables`), all before the jump:
|
||||
@@ -173,7 +173,7 @@ The loader and kernel are two *separate* binaries built for two different target
|
||||
so everything they exchange must have an identically-defined memory layout. That's
|
||||
what `system/boot-handoff.zig` provides — imported by both as the `boot-handoff` module.
|
||||
It is *only* the handoff: the kernel↔user ABI (`system/abi.zig`) and the device types
|
||||
(`system/devices/device-abi.zig`) are separate contracts the bootloader never sees.
|
||||
(`library/device/model/device-abi.zig`) are separate contracts the bootloader never sees.
|
||||
|
||||
- `BootInformation` — the top-level struct passed to the kernel: the
|
||||
framebuffer, the memory map, the kernel's own `PT_LOAD` segments
|
||||
@@ -196,7 +196,7 @@ power on
|
||||
-> queryFramebuffer (via GOP: EDID native res, setMode, describe fb)
|
||||
-> loadKernel (read system/kernel ELF, load PT_LOAD segments low, .text at 0x100000)
|
||||
-> loadSystemTree (read the boot\system.img capsule as the in-RAM initial ramdisk;
|
||||
fallbacks: manifest-listed paths, then a /system tree walk)
|
||||
fallbacks: manifest-listed paths, then a /system + /test tree walk)
|
||||
-> buildBootstrapTables (identity + physmap + higher-half kernel mappings)
|
||||
-> exitBootServices (retry until the memory-map key holds)
|
||||
-> handoff: load bootstrap CR3, jump to e_entry, boot_information pointer in RDI
|
||||
@@ -51,7 +51,7 @@ rest — works directly on the kernel heap, no bespoke containers required.
|
||||
|
||||
## Verifying it
|
||||
|
||||
The `heap` test (see [testing.md](testing.md)) exercises the allocator end to end:
|
||||
The `heap` test (see [testing.md](../testing.md)) exercises the allocator end to end:
|
||||
|
||||
```
|
||||
[PASS] alloc 4096 bytes
|
||||
@@ -133,10 +133,10 @@ TSS/IST is wired up: the handler survived a completely broken stack.
|
||||
|
||||
Both items originally deferred here have landed:
|
||||
|
||||
- **The IO-APIC**: [ioapic.zig](../system/kernel/architecture/x86_64/ioapic.zig)
|
||||
- **The IO-APIC**: [ioapic.zig](../../system/kernel/architecture/x86_64/ioapic.zig)
|
||||
routes external device lines onto vectors — discovered via ACPI's MADT, every
|
||||
input masked at init, lines unmasked one at a time as user-space drivers bind
|
||||
them (see [device-interrupts.md](device-interrupts.md)). The keyboard followed
|
||||
them (see [device-interrupts.md](../device-driver-development-guide/device-interrupts.md)). The keyboard followed
|
||||
exactly as predicted: the PS/2 bus driver (`system/drivers/ps2-bus/`) claims
|
||||
the 8042 controller and binds its IRQ 1 (and the aux mouse's IRQ 12) through
|
||||
this routing. USB HID keyboards arrive over xHCI instead, which interrupts via
|
||||
@@ -21,9 +21,9 @@ kernel log.print ─┘ │
|
||||
|
||||
1. **Emit.** A program calls `std.log.info("mounted {s}", .{path})` — the
|
||||
runtime's `logFn` (installed for every binary by the root shim,
|
||||
`library/runtime/log.zig`) formats one line and issues one `debug_write`
|
||||
`library/kernel/logging.zig`) formats one line and issues one `debug_write`
|
||||
carrying the level. The payload does NOT contain the process's name.
|
||||
`runtime.system.write` remains as the raw/bring-up path (panics, test
|
||||
`logging.write` remains as the raw/bring-up path (panics, test
|
||||
fixtures); raw bytes ride the same ring, attributed all the same.
|
||||
|
||||
2. **Stamp.** The kernel wraps every payload LINE in a record stamped with the
|
||||
@@ -112,7 +112,7 @@ kernel heap will build on to map pages on demand.
|
||||
|
||||
## Verifying it
|
||||
|
||||
Four tests (see [testing.md](testing.md)) pin down the guarantees:
|
||||
Four tests (see [testing.md](../testing.md)) pin down the guarantees:
|
||||
|
||||
- **`vmm`** — map a fresh frame at an unused virtual address, write and read it
|
||||
back. Proves `map` works end to end.
|
||||
@@ -138,8 +138,8 @@ Four tests (see [testing.md](testing.md)) pin down the guarantees:
|
||||
processes own the low half.
|
||||
- **Per-address-space tables** — done: each user process gets its own root with
|
||||
the kernel half shared, and refcounted shared-memory mappings exist
|
||||
([ipc.md](ipc.md)). Copy-on-write remains unbuilt — nothing has needed it yet.
|
||||
([ipc.md](../device-driver-development-guide/ipc.md)). Copy-on-write remains unbuilt — nothing has needed it yet.
|
||||
- **Uncacheable MMIO** — half done: user-space device and DMA mappings are
|
||||
strong-uncacheable and the framebuffer is write-combining via the PAT, but the
|
||||
kernel's own `mapMmio` path is still writeback — the LAPIC included (see
|
||||
[device-interrupts.md](device-interrupts.md)).
|
||||
[device-interrupts.md](../device-driver-development-guide/device-interrupts.md)).
|
||||
@@ -7,7 +7,7 @@ them owns the hardware that reported the event, and the reporter should not know
|
||||
who is listening. So system power is a **service**: an event source **publishes**
|
||||
button/lid/battery/AC events, interested processes **subscribe**, and one
|
||||
privileged caller — init — can ask it to power the machine off. It is the same
|
||||
publish/subscribe shape as the [input service](input.md), applied to power.
|
||||
publish/subscribe shape as the [input service](../device-driver-development-guide/input.md), applied to power.
|
||||
|
||||
## Why a service, and why it is named for the domain, not the firmware
|
||||
|
||||
@@ -17,7 +17,7 @@ What subscribers want is not: *the lid closed* means the same thing regardless o
|
||||
who noticed. So the surface is **domain-named**. There is a `power-protocol`
|
||||
module and a well-known `ServiceId.power = 5`; on x86 the **acpi service**
|
||||
registers it, and on ARM a PSCI/mailbox service will register the *same* id.
|
||||
Subscribers call `runtime.ipc.lookup(.power)` and never learn which firmware they
|
||||
Subscribers call `ipc.lookup(.power)` and never learn which firmware they
|
||||
are on — the neutrality the whole [discovery](discovery.md) migration exists to
|
||||
preserve, carried one layer up into a running-system surface.
|
||||
|
||||
@@ -27,7 +27,7 @@ unchanged.
|
||||
|
||||
## The protocol
|
||||
|
||||
The `power-protocol` module ([system/services/power/protocol.zig](../system/services/power/protocol.zig))
|
||||
The `power-protocol` module ([library/protocol/power/power-protocol.zig](../../library/protocol/power/power-protocol.zig))
|
||||
follows the vfs-protocol pattern — extern-struct messages, a version, reserved
|
||||
fields. Three operations:
|
||||
|
||||
@@ -75,7 +75,7 @@ notifications, the lifecycle **signals** it can receive (`terminate`), and the
|
||||
On a `power_button` event or a `terminate` signal, init:
|
||||
|
||||
1. logs that it is shutting down,
|
||||
2. runs the standard stop sequence — `runtime.process.stop(child, deadline,
|
||||
2. runs the standard stop sequence — `process.stop(child, deadline,
|
||||
endpoint)` — over its children **in reverse spawn order**, so the VFS stops
|
||||
last (other services may flush through it), each child getting the
|
||||
*terminate → deadline → kill* escalation from
|
||||
@@ -92,9 +92,9 @@ fails rather than hangs.
|
||||
`acpi-tables` node ([discovery.md](discovery.md)) already put the PM1 control
|
||||
ports in the acpi service's hands, so writing S5 from ring 3 is something it
|
||||
could physically already do; formalizing it as a protocol operation added a
|
||||
contract, not authority. The kernel's `system/devices/power.zig` keeps only
|
||||
**reboot** (the FADT reset register plus the legacy fallbacks, which need no AML);
|
||||
it has no poweroff path at all — S5 is not a kernel operation.
|
||||
contract, not authority. The kernel keeps only **reboot** (`acpi.reboot` in
|
||||
`system/kernel/acpi.zig` — the FADT reset register plus the legacy fallbacks, which
|
||||
need no AML); it has no poweroff path at all — S5 is not a kernel operation.
|
||||
|
||||
## Verifying it
|
||||
|
||||
@@ -126,5 +126,5 @@ until laptop sleep), and thermal zones.
|
||||
firmware neutrality that makes a PSCI backend drop-in on ARM.
|
||||
- [process-lifecycle.md](process-lifecycle.md) — the stop sequence
|
||||
(`terminate → deadline → kill`) and signals init composes into shutdown.
|
||||
- [device-manager.md](device-manager.md) — the supervision model init mirrors for
|
||||
- [device-manager.md](../device-driver-development-guide/device-manager.md) — the supervision model init mirrors for
|
||||
its own children.
|
||||
@@ -6,10 +6,10 @@ harness are all in — the interface below is as-built. The primitives underneat
|
||||
predate this design ([process-management.md](process-management.md):
|
||||
spawn, the supervision link, kill, child-exit notifications); this document designs
|
||||
the layer above them — the standard vocabulary a danos process speaks about its own
|
||||
life, and the stable `runtime.process` interface that carries it. Nothing here is
|
||||
life, and the stable `process` interface that carries it. Nothing here is
|
||||
device- or driver-specific: a driver, the VFS, and a user application all stop,
|
||||
reload, and die the same way. The device manager is simply this design's first
|
||||
serious customer ([device-manager.md](device-manager.md)).
|
||||
serious customer ([device-manager.md](../device-driver-development-guide/device-manager.md)).
|
||||
|
||||
**"POSIX" in this document means the concepts, never the letter of the standard.**
|
||||
danos borrows the ideas and the hard-won lessons (what SIGTERM *means*, why SIGPIPE
|
||||
@@ -19,7 +19,7 @@ rule is danos's own and it is strict: plain words that communicate intent
|
||||
(`bind`, `subscribe`, `publish`, `endpoint`) — never `SIG*`, never a second word for
|
||||
a concept that already has one. Literal POSIX arrives later and lives elsewhere: the
|
||||
`std.os.danos` seam that makes danos a Zig target, and eventually a **musl-based C
|
||||
layer** on the same native surface (see [zig-self-hosting.md](zig-self-hosting.md)) —
|
||||
layer** on the same native surface (see [zig-self-hosting.md](../zig-self-hosting.md)) —
|
||||
musl's syscall surface retargeted at danos system calls and IPC protocols (files onto
|
||||
the VFS protocol, `sigaction`/`wait` onto this lifecycle, sockets onto whatever
|
||||
networking becomes). Ported programs see POSIX; the system underneath never does.
|
||||
@@ -172,7 +172,7 @@ zombie state or privileged snooping:
|
||||
the server's reply with `-EPEER`; a server that dies fails its waiting clients
|
||||
the same way. This covers the *synchronous* case only.
|
||||
3. **The subscribers** — the new piece, and it is the input service's
|
||||
publish/subscribe shape ([input.md](input.md)) applied to exits. A stateful
|
||||
publish/subscribe shape ([input.md](../device-driver-development-guide/input.md)) applied to exits. A stateful
|
||||
service accumulates per-client state across many requests: a filesystem server
|
||||
(FAT today) holds a dead client's open file handles, the input service holds
|
||||
its subscriptions, a future network stack holds its sockets. None of these
|
||||
@@ -184,7 +184,7 @@ zombie state or privileged snooping:
|
||||
same encoding supervisors already decode, the IRQ-as-IPC pattern once more). The
|
||||
subscriber filters for ids it holds state for and releases what the dead client
|
||||
held. Correlating is free of bookkeeping: an IPC sender's badge already *is* its
|
||||
task id (`runtime.ipc.Received`), so the id a service has been keying client
|
||||
task id (`ipc.Received`), so the id a service has been keying client
|
||||
state by all along is the id the exit event carries.
|
||||
|
||||
Subscription, not broadcast-to-everyone: only processes that asked receive
|
||||
@@ -200,9 +200,9 @@ its clients cleaning up after themselves.** Handle release on client death is th
|
||||
service's job, triggered by the published exit event — never by a courtesy
|
||||
"closing now" message that a crashed client will never send.
|
||||
|
||||
## The stable interface: `runtime.process`
|
||||
## The stable interface: `process`
|
||||
|
||||
`runtime.process` already owns what a process receives at birth (`Init`, the
|
||||
`process` already owns what a process receives at birth (`Init`, the
|
||||
argv contract). It grows to own the other end of life.
|
||||
|
||||
**The runtime is the stable interface; the numbers are not.** danos applications do
|
||||
@@ -283,7 +283,7 @@ callbacks (`on_terminate`, `on_reload`) for programs that want defaults.
|
||||
|
||||
### The service harness
|
||||
|
||||
`runtime.service` owns the `replyWait` loop and folds every event source — signals,
|
||||
`service` owns the `replyWait` loop and folds every event source — signals,
|
||||
child exits, protocol messages — into callbacks, with the vocabulary's defaults:
|
||||
`terminate` returns from the loop (clean exit), the common `ping` is answered automatically,
|
||||
`reload` is ignored unless overridden. One loop, no locking, nothing reentrant. A
|
||||
@@ -312,15 +312,15 @@ get POSIX; danos-native programs never pay for it.
|
||||
kill a claiming driver, spawn it again, the claim succeeds.
|
||||
2. **Exit reason in the death notification** (`ExitReason` above).
|
||||
3. **Exit events**: `process_subscribe` in the kernel (bounded subscriber table,
|
||||
publishes on every death), `runtime.process.subscribeExits`; the userspace VFS
|
||||
publishes on every death), `process.subscribeExits`; the userspace VFS
|
||||
router was the first subscriber — releasing a dead client's handles was its
|
||||
proof test — and the FAT server inherited the role when the router moved into
|
||||
the kernel (clients now hold the filesystem server's node ids directly).
|
||||
4. **Signals**: `signal_bind` + `process_signal` + the pending mask in the kernel;
|
||||
`runtime.process` grows the interface above; the service harness handles
|
||||
`process` grows the interface above; the service harness handles
|
||||
`terminate` and answers the common `ping`; `stop()` for supervisors.
|
||||
|
||||
[device-manager.md](device-manager.md) builds directly on all four.
|
||||
[device-manager.md](../device-driver-development-guide/device-manager.md) builds directly on all four.
|
||||
|
||||
## Settled questions (2026-07-12)
|
||||
|
||||
@@ -114,7 +114,7 @@ the architecture layer calls up into `tick`.
|
||||
- ~~There is no exit status in the notification~~ Closed (M17.2): the kernel
|
||||
records how every process ends — exited, a fault class, or killed — before it
|
||||
posts the exit notification, and the supervisor reads it with
|
||||
`process_exit_reason` (`runtime.process.exitReason`). This is the input to
|
||||
`process_exit_reason` (`process.exitReason`). This is the input to
|
||||
restart policy ([process-lifecycle.md](process-lifecycle.md)); an exit *code*
|
||||
for the clean case can still ride alongside later.
|
||||
- Enumerate writes through the caller's raw pointer under the bring-up trust
|
||||
@@ -44,7 +44,7 @@ tables of contents, both pointing at the same embedded FAT image:
|
||||
same `BOOTX64.efi` off it.
|
||||
|
||||
Neither path involves the legacy BIOS boot-sector machinery: danos is
|
||||
UEFI-only ([system-requirements.md](system-requirements.md)), so the MBR holds
|
||||
UEFI-only ([system-requirements.md](../system-requirements.md)), so the MBR holds
|
||||
no boot code, just the partition entry, and the El Torito entry is EFI-class,
|
||||
not floppy emulation.
|
||||
|
||||
@@ -57,7 +57,7 @@ allocated from the front) either way. The USB path has no such cap.
|
||||
## The builder
|
||||
|
||||
`tools/make-iso-image.py` follows the house rule of
|
||||
[make-fat-image.py](../tools/make-fat-image.py): pure Python 3 standard
|
||||
[make-fat-image.py](../../tools/make-fat-image.py): pure Python 3 standard
|
||||
library, no external tools (no xorriso, mkisofs, or isohybrid), with a
|
||||
`--verify` mode the `check-iso-image` step runs — it checks that the MBR
|
||||
partition and the El Torito catalog agree on where the FAT image lives and
|
||||
@@ -5,7 +5,7 @@ isolation; fault → kill the process → keep the core (`onException`; the
|
||||
`fault-recovery` test); the supervisor notification **with exit reasons**
|
||||
([process-lifecycle.md](process-lifecycle.md) — clean exit, fault class, or
|
||||
killed, recorded before the notice posts); and the **restart policy itself**
|
||||
([device-manager.md](device-manager.md)): the device manager supervises every
|
||||
([device-manager.md](../device-driver-development-guide/device-manager.md)): the device manager supervises every
|
||||
driver, restarts crashes with backoff, caps crash loops, and re-claims work
|
||||
because the kernel releases a dead process's claims. The `driver-restart` and
|
||||
`usb-report` scenarios prove kill → release → respawn → re-claim → re-report
|
||||
@@ -14,9 +14,9 @@ more of the system moved into restartable processes (the discovery migration,
|
||||
[discovery.md](discovery.md), is the next rung). This is the property danos is really chasing:
|
||||
**if a part of the OS breaks, isolate it, and re-initialise it — without rebooting.**
|
||||
A crashed driver gets restarted; a wedged service gets killed and brought back. It's
|
||||
the reason the [microkernel](vision.md) shape was chosen, and it's a *separate* goal
|
||||
the reason the [microkernel](../vision.md) shape was chosen, and it's a *separate* goal
|
||||
from [real-time](smp.md#does-the-right-choice-depend-on-real-time-vs-resilience) —
|
||||
one that's less pervasive to build (see [vision.md](vision.md)).
|
||||
one that's less pervasive to build (see [vision.md](../vision.md)).
|
||||
|
||||
## The idea: "let it crash" + supervision
|
||||
|
||||
@@ -44,7 +44,7 @@ down. **Keeping the kernel minimal is a resilience strategy, not just an aesthet
|
||||
## The building blocks
|
||||
|
||||
1. **Address-space isolation.** A fault in one component can't corrupt another or the
|
||||
kernel. This is the [user-mode milestone](vision.md) (ring 3, per-process page
|
||||
kernel. This is the [user-mode milestone](../vision.md) (ring 3, per-process page
|
||||
tables) — the shared prerequisite for *any* of this, and it's needed regardless.
|
||||
2. **Fault detection** — how the system notices a component is dead or sick:
|
||||
- **Crash**: a CPU fault in a user process (page fault, illegal instruction) traps
|
||||
@@ -81,7 +81,7 @@ Detecting and killing is the easy half. The genuinely tricky questions are about
|
||||
- **In-flight IPC**: messages sent to the dead component, or replies its clients are
|
||||
blocked waiting for. The channel has to break cleanly and unblock the waiters with
|
||||
an error rather than hang them forever (a design constraint that reaches back into
|
||||
[ipc.md](ipc.md) — channels need a "peer died" outcome).
|
||||
[ipc.md](../device-driver-development-guide/ipc.md) — channels need a "peer died" outcome).
|
||||
- **Clients**: how does a client discover the service it was talking to is gone and
|
||||
has been replaced? Options: capability revocation makes stale handles fail; or a
|
||||
**name server** re-binds clients to the new instance; or clients retry through a
|
||||
@@ -137,7 +137,7 @@ Honest boundaries:
|
||||
|
||||
Resilience needs **structural** features (isolation + supervision + a resource
|
||||
model); real-time needs a **pervasive** timing invariant. They're separable, and
|
||||
resilience is the lighter commitment (see [smp.md](smp.md) and [vision.md](vision.md)).
|
||||
resilience is the lighter commitment (see [smp.md](smp.md) and [vision.md](../vision.md)).
|
||||
Note the overlap, though: **preemptive scheduling** and **priorities** — already
|
||||
built — serve resilience too (you can preempt and kill a misbehaving component, and
|
||||
run the supervisor at high priority). So danos keeps the useful *mechanisms* of the
|
||||
@@ -156,10 +156,10 @@ real-time work without owing anyone a timing *guarantee*.
|
||||
|
||||
## Related
|
||||
|
||||
- [vision.md](vision.md) — the goals this serves (learning by doing; resilience over
|
||||
- [vision.md](../vision.md) — the goals this serves (learning by doing; resilience over
|
||||
hard real-time).
|
||||
- [scheduling.md](scheduling.md) — preemption, which makes runaway components killable.
|
||||
- [ipc.md](ipc.md) — channels that need a "peer died" outcome for clean restart.
|
||||
- [ipc.md](../device-driver-development-guide/ipc.md) — channels that need a "peer died" outcome for clean restart.
|
||||
- [interrupts.md](interrupts.md) — fault reporting that user mode turns into "kill and
|
||||
restart" instead of "halt".
|
||||
- [smp.md](smp.md) — the real-time-vs-resilience fork, in the SMP context.
|
||||
@@ -3,7 +3,7 @@
|
||||
The scheduler turns danos from a linear "boot then halt" kernel into a **running
|
||||
multitasking system**. It's **fixed-priority preemptive**: the highest-priority
|
||||
ready task always runs, and tasks at the same priority take turns. That model is
|
||||
chosen for [real-time](vision.md) — it's predictable (you can reason about which
|
||||
chosen for [real-time](../vision.md) — it's predictable (you can reason about which
|
||||
task runs when) and its decisions are O(1), unlike a fair-share scheduler.
|
||||
|
||||
The scheduler proper (`system/kernel/scheduler.zig`) is generic; the context switch and new-task
|
||||
@@ -37,7 +37,7 @@ down a return address pointing at `task_trampoline` and zeroed callee-saved slot
|
||||
`schedule()` — pick the best task and switch — runs from two places:
|
||||
|
||||
- **`yield()`** — a task voluntarily gives up the CPU.
|
||||
- **`tick()`** — the 1000 Hz [timer](device-interrupts.md) preempts the running
|
||||
- **`tick()`** — the 1000 Hz [timer](../device-driver-development-guide/device-interrupts.md) preempts the running
|
||||
task. This is what lets a task that never yields still share the CPU.
|
||||
|
||||
The subtlety in mixing them is the **interrupt flag (IF)**. The rule: `switch_context`
|
||||
@@ -97,7 +97,7 @@ marks the task blocked with a wake deadline and switches away. On every tick the
|
||||
timer wakes any task whose deadline has passed (a bounded scan, so it stays
|
||||
deterministic), which makes it ready again; the scheduler then runs it when its
|
||||
priority comes up. `sleep` measures its deadline on the [calibrated
|
||||
clock](device-interrupts.md), so it's real time.
|
||||
clock](../device-driver-development-guide/device-interrupts.md), so it's real time.
|
||||
|
||||
When *every* task is blocked, something still has to run — so there's an **idle
|
||||
task** at the lowest priority that just `hlt`s until the next interrupt (see
|
||||
@@ -111,7 +111,7 @@ The other form of blocking is waiting for an **event** rather than a duration. A
|
||||
the caller on it, `wake(wq)` moves the highest-priority waiter back to ready
|
||||
(preempting if it now outranks the running task). A task links into a wait queue
|
||||
through the same field the ready queues use — it's in exactly one queue at a time.
|
||||
These are the primitives locks, semaphores and [IPC](ipc.md) are built on.
|
||||
These are the primitives locks, semaphores and [IPC](../device-driver-development-guide/ipc.md) are built on.
|
||||
|
||||
Blocking safely needs **composable critical sections**. A blanket `cli`/`sti` pair
|
||||
doesn't nest: an IPC channel that `cli`s and then calls `wait` would have `wait`'s
|
||||
@@ -124,7 +124,7 @@ caller's state.
|
||||
|
||||
## Verifying it
|
||||
|
||||
Three tests (see [testing.md](testing.md)) prove the guarantees:
|
||||
Three tests (see [testing.md](../testing.md)) prove the guarantees:
|
||||
|
||||
- **`sched`** spawns three tasks that busy-loop *without ever yielding*. They all
|
||||
make progress — which can only happen if the timer is **preempting** between them
|
||||
@@ -140,7 +140,7 @@ Three tests (see [testing.md](testing.md)) prove the guarantees:
|
||||
|
||||
- **Priority inheritance** — still open. Tasks now do block on shared resources
|
||||
(IPC rendezvous, the big kernel lock), and nothing yet bounds priority
|
||||
inversion — a [real-time](vision.md) requirement.
|
||||
inversion — a [real-time](../vision.md) requirement.
|
||||
- **Task exit / a reaper** — done. A dying task goes on its core's reap list in a
|
||||
`.reaping` state; the timer tick drains the list, frees the stack back to the
|
||||
heap, and recycles the task-table slot.
|
||||
@@ -307,4 +307,4 @@ refcount, and no group-kill special case is needed at all.
|
||||
Then update [threading.md](threading.md) (the shared-fate gap note),
|
||||
[process-lifecycle.md](process-lifecycle.md),
|
||||
[process-management.md](process-management.md), and
|
||||
[ipc.md](ipc.md)/[drivers.md](drivers.md) mentions.
|
||||
[ipc.md](../device-driver-development-guide/ipc.md)/[drivers.md](../device-driver-development-guide/drivers.md) mentions.
|
||||
@@ -112,7 +112,7 @@ Yes — and this is the branch that matters for danos right now.
|
||||
|
||||
These pull in different directions, so **picking the primary goal comes before
|
||||
picking the SMP design.** (danos's founding assumption was real-time; that's under
|
||||
active reconsideration in favour of resilience — see [vision.md](vision.md).)
|
||||
active reconsideration in favour of resilience — see [vision.md](../vision.md).)
|
||||
|
||||
## What this would mean for danos
|
||||
|
||||
@@ -169,7 +169,7 @@ next lands.
|
||||
the highest-priority ready task; per-core queues are a later optimisation.
|
||||
- **AP wake to long mode** — `architecture.startSecondary` drives INIT–SIPI–SIPI (via the
|
||||
LAPIC ICR) to wake each parked core one at a time. A woken core starts in 16-bit
|
||||
real mode at a low page and runs the [trampoline](../system/kernel/architecture/x86_64/trampoline.s)
|
||||
real mode at a low page and runs the [trampoline](../../system/kernel/architecture/x86_64/trampoline.s)
|
||||
up through protected mode into 64-bit long mode, then lands in `smp.zig:apEntry`,
|
||||
publishes its per-CPU pointer, and reports in. Verified in QEMU with `-smp 4`:
|
||||
all four cores report `online`.
|
||||
@@ -270,6 +270,6 @@ next lands.
|
||||
|
||||
- [scheduling.md](scheduling.md) — the single-core scheduler SMP would extend.
|
||||
- [discovery.md](discovery.md) — enumerating cores is a device-discovery problem.
|
||||
- [ipc.md](ipc.md) — the message passing cross-core coordination rides on.
|
||||
- [vision.md](vision.md) — the goals question (real-time vs resilience) this note
|
||||
- [ipc.md](../device-driver-development-guide/ipc.md) — the message passing cross-core coordination rides on.
|
||||
- [vision.md](../vision.md) — the goals question (real-time vs resilience) this note
|
||||
keeps bumping into.
|
||||
@@ -51,7 +51,7 @@ Everything else---including`read()`,`write()`,`malloc()`, and`fork()`---will run
|
||||
3. **`Yield()`/`Thread_Ctrl()`**
|
||||
- **What it does:**Allows a thread to voluntarily give up its CPU time slice, or allows a root task to spawn/kill threads.
|
||||
4. **`ipc_send(endpoint, message_buffer)`(Asynchronous Send)**
|
||||
- **What it does:**Posts a small payload to an endpoint's bounded queue and returns *without* blocking — no rendezvous, no reply. The receiver picks it up through the same `IPC_ReplyWait`, as a buffered message. It is the async counterpart of `IPC_Call`, for one-to-many broadcasts where a synchronous rendezvous would let one dead or slow receiver hang the sender. The [input service](input.md) — keyboard-event fan-out — is its first user. A full queue drops the oldest message (a buffered message is discrete data, unlike a coalescing interrupt notification).
|
||||
- **What it does:**Posts a small payload to an endpoint's bounded queue and returns *without* blocking — no rendezvous, no reply. The receiver picks it up through the same `IPC_ReplyWait`, as a buffered message. It is the async counterpart of `IPC_Call`, for one-to-many broadcasts where a synchronous rendezvous would let one dead or slow receiver hang the sender. The [input service](../device-driver-development-guide/input.md) — keyboard-event fan-out — is its first user. A full queue drops the oldest message (a buffered message is discrete data, unlike a coalescing interrupt notification).
|
||||
|
||||
* * * * *
|
||||
|
||||
@@ -10,8 +10,8 @@ v2), written to disk ahead of time. The EFI loader reads it in a single
|
||||
sequential pass and hands the bytes to the kernel unmodified.
|
||||
|
||||
The capsule is a *performance artifact*, not a source of truth. The boot
|
||||
volume's `/system` file tree remains the canonical layout (see
|
||||
[danos-file-system-hierarchy-FSH.md](danos-file-system-hierarchy-FSH.md));
|
||||
volume's `/system` and `/test` file trees remain the canonical layout (see
|
||||
[danos-file-system-hierarchy-FSH.md](../file-system-development/danos-file-system-hierarchy-FSH.md));
|
||||
the capsule is a pre-baked snapshot of the same binaries, derived from the same
|
||||
build graph, so the running system is identical whether the loader read the
|
||||
capsule or walked the tree.
|
||||
@@ -58,8 +58,9 @@ blobs... each entry's file bytes, at its offset within the image
|
||||
Three artifacts are derived from that same list, in the same build graph, so
|
||||
they cannot drift apart:
|
||||
|
||||
1. **The tree**: each binary installed at its FHS path (`zig-out/system/...`,
|
||||
mirrored onto the FAT boot volume by `tools/make-fat-image.py`).
|
||||
1. **The tree**: each binary installed at its FHS path (`zig-out/system/...`
|
||||
and `zig-out/test/...`, mirrored onto the FAT boot volume by
|
||||
`tools/make-fat-image.py`).
|
||||
2. **The manifest** (`system/manifest`): the FHS path of every bundled binary,
|
||||
one per line — the loader's per-file fallback input.
|
||||
3. **The capsule**: `tools/pack-system-image.py` packs the same binaries into
|
||||
@@ -81,7 +82,8 @@ same in-RAM ramdisk image:
|
||||
2. **The manifest** — read `system\manifest` and open each listed path *by
|
||||
name*. FAT name lookup is case-insensitive and firmware-portable, unlike
|
||||
directory enumeration. The loader assembles the v2 image in RAM itself.
|
||||
3. **The tree walk** — enumerate `/system` recursively. Last resort for
|
||||
3. **The tree walk** — enumerate `/system` and `/test` recursively (`/test`
|
||||
is optional: a stick without fixtures still boots). Last resort for
|
||||
hand-assembled sticks with neither file: some firmware FAT drivers return
|
||||
bare 8.3 names uppercase from enumeration, which is why this is the
|
||||
fallback and not the primary path.
|
||||
@@ -105,10 +107,11 @@ consumers:
|
||||
pre-path callers — and loads them as fresh ring-3 processes. The stored path
|
||||
becomes the task's name.
|
||||
- **The VFS root** (`vfs.zig`, `setInitialRamdisk`): the image is mounted as
|
||||
the kernel-backed, read-only `/system` mount. Directory nodes are derived
|
||||
from the entry paths (the unique parents), so `/system` is listable and its
|
||||
files readable over the normal VFS protocol — the FHS boot tree every
|
||||
process sees comes straight out of the capsule bytes.
|
||||
kernel-backed, read-only mounts — one per top-level tree named by the entry
|
||||
paths, so `/system` and, when the fixtures are bundled, `/test`. Directory
|
||||
nodes are derived from the entry paths (the unique parents), so the trees
|
||||
are listable and their files readable over the normal VFS protocol — the
|
||||
FHS boot tree every process sees comes straight out of the capsule bytes.
|
||||
|
||||
The image is never copied after the handoff and never mutated: the initrd is
|
||||
immutable, which is what makes the VFS's node serving lock-free.
|
||||
@@ -59,9 +59,9 @@ danos's two binaries default to different conventions:
|
||||
When the loader jumps to the kernel passing the `BootInformation` pointer, both sides have
|
||||
to agree *which register that pointer lands in*. Left to their defaults, the loader
|
||||
would place it in RCX while the kernel looked in RDI — and the kernel would read
|
||||
garbage. So both sides reference the same `system.kernel_abi` (SysV): the loader's
|
||||
garbage. So both sides reference the same `boot_handoff.kernel_abi` (SysV): the loader's
|
||||
function-pointer type and the kernel's `_start` both carry
|
||||
`callconv(system.kernel_abi)`, and the pointer reliably arrives in RDI. That is the
|
||||
`callconv(boot_handoff.kernel_abi)`, and the pointer reliably arrives in RDI. That is the
|
||||
whole reason `kernel_abi` lives in the shared contract — see [efi.md](efi.md) for
|
||||
the handoff it governs.
|
||||
|
||||
@@ -90,9 +90,9 @@ process) instead of silently corrupting the image
|
||||
(`buildEntryStack` in `system/kernel/process.zig`); `argv[0]` is always the path
|
||||
or initial-ramdisk name the process was spawned as, and `system_spawn`'s optional
|
||||
argument blob becomes `argv[1..]`. The runtime's `_start`
|
||||
(`library/runtime/start.zig`) hands the block to `rt_start`, which builds a
|
||||
`runtime.process.Init` from it and passes that to the program's `main`
|
||||
(`pub fn main(init: runtime.process.Init)`; a parameterless `main()` is also
|
||||
(`library/kernel/start.zig`) hands the block to `rt_start`, which builds a
|
||||
`process.Init` from it and passes that to the program's `main`
|
||||
(`pub fn main(init: process.Init)`; a parameterless `main()` is also
|
||||
accepted). A C runtime's `crt0` would walk
|
||||
the identical layout unmodified — that's the compatibility being bought. The
|
||||
`args` test proves the round trip.
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
The ordered, checkpointable build-out for [threading.md](threading.md). Each milestone
|
||||
lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run, like
|
||||
[display-v2-plan.md](display-v2-plan.md). Read threading.md first for the *why*.
|
||||
[display-v2-plan.md](../device-driver-development-guide/display-v2-plan.md). Read threading.md first for the *why*.
|
||||
|
||||
## Locked decisions (do not relitigate)
|
||||
|
||||
@@ -13,7 +13,7 @@ lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run,
|
||||
`single_threaded = false`.
|
||||
- **Blocking is futex-backed, never spin-backed** — waiters park in the kernel so an
|
||||
idle core still halts ([halting.md](halting.md)).
|
||||
- **New syscalls are private**: extend [abi.zig](../system/abi.zig) `SystemCall` after
|
||||
- **New syscalls are private**: extend [abi.zig](../../system/abi.zig) `SystemCall` after
|
||||
`shared_memory_physical = 36` (`thread_spawn = 37`, `thread_exit = 38`, `current_core = 39`,
|
||||
`futex_wait = 40`, `futex_wake = 41`) + a `library/runtime` wrapper; user code never names a number.
|
||||
- **Restart granularity stays the process** — a faulting thread kills its process; the
|
||||
@@ -21,10 +21,10 @@ lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run,
|
||||
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](coding-standards.md): spell out non-acronym abbreviations,
|
||||
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym abbreviations,
|
||||
kebab-case file names, no `Co-Authored-By` trailers. New user binaries go through
|
||||
`addUserBinary` (with the new `threaded` flag where a binary spawns threads) and get
|
||||
packed into the initial-ramdisk; new syscalls extend [abi.zig](../system/abi.zig)
|
||||
packed into the initial-ramdisk; new syscalls extend [abi.zig](../../system/abi.zig)
|
||||
`SystemCall` + a `library/runtime` wrapper; test services live beside the code they
|
||||
exercise and register a `ServiceId` if they must be looked up.
|
||||
|
||||
@@ -58,7 +58,7 @@ fixed in *Locked decisions*; the checkboxes are the only state. A loop iteration
|
||||
5) so progress is backed up remotely; **do not push `main`** — merging Phase 2 into
|
||||
`main` stays a human step.
|
||||
3. **Implement** every unchecked item in that milestone, including adding its
|
||||
`-Dtest-case` to `CASES` in [test/qemu_test.py](../test/qemu_test.py) (with
|
||||
`-Dtest-case` to `CASES` in [test/qemu_test.py](../../test/qemu_test.py) (with
|
||||
`smp: true` / a `mem` bump where noted) so the gate is runnable.
|
||||
4. **Run the gate**: `python3 test/qemu_test.py <case>`, then the full **guardrail
|
||||
set**, then `zig build` (clean) and `zig build test` (green).
|
||||
@@ -67,7 +67,7 @@ fixed in *Locked decisions*; the checkboxes are the only state. A loop iteration
|
||||
the whole guardrail set passes, `zig build` is clean, and host tests are green.
|
||||
→ tick this milestone's boxes **and** its `**Gate:**`-referenced case, `git commit`
|
||||
(`threads(M<n>): <summary>`, no `Co-Authored-By` trailer per
|
||||
[coding-standards.md](coding-standards.md)), then **`git push` the working branch to
|
||||
[coding-standards.md](../coding-standards.md)), then **`git push` the working branch to
|
||||
`origin`** (use `-u` on the first push to set upstream). Continue to the next
|
||||
milestone in the same iteration if budget remains; otherwise let the loop re-fire.
|
||||
- **Red** = anything above fails. Diagnose from the captured serial log
|
||||
@@ -111,7 +111,7 @@ task's exit; make destruction happen on the **last** exit.
|
||||
- [x] A refcount keyed by the address-space root, held in `scheduler.zig`
|
||||
(`address_space_refs`): `retainAddressSpace` takes a reference in `spawnUserLocked` (on the
|
||||
success path, after the slot + stack are secured), all under the big kernel lock.
|
||||
- [x] Both task-teardown paths ([scheduler.zig](../system/kernel/scheduler.zig):
|
||||
- [x] Both task-teardown paths ([scheduler.zig](../../system/kernel/scheduler.zig):
|
||||
`exitUserLocked` and `destroyTaskLocked`) call `releaseAddressSpace`, which decrements
|
||||
and only `destroyAddressSpace`s at **zero**; an unretained space (hand-built test
|
||||
spaces) is destroyed directly, preserving prior behaviour.
|
||||
@@ -133,7 +133,7 @@ full guardrail set passes unchanged — 13/13 (`smoke`, `sched`, `priority`, `sm
|
||||
Spawn only — no join yet. Prove a second task executes in the **caller's** address
|
||||
space and exits cleanly.
|
||||
|
||||
- [x] [abi.zig](../system/abi.zig): `thread_spawn = 37`, `thread_exit = 38`. Handlers in
|
||||
- [x] [abi.zig](../../system/abi.zig): `thread_spawn = 37`, `thread_exit = 38`. Handlers in
|
||||
process.zig; `thread_spawn` calls `scheduler.spawnThread` (today, after M3, the
|
||||
handler goes `spawnThreadSupervised` → `scheduler.spawnUserLocked`; shares the caller's
|
||||
address space, `retainAddressSpace`); `thread_exit` ends the task like a process `exit(0)`
|
||||
@@ -195,7 +195,7 @@ plus `address-space-refcount`/`thread-spawn`; `zig build` clean, `zig build test
|
||||
|
||||
## M4 — Futex: the one blocking primitive ✅
|
||||
|
||||
- [x] [abi.zig](../system/abi.zig): `futex_wait = 40`, `futex_wake = 41`. A waiter is a
|
||||
- [x] [abi.zig](../../system/abi.zig): `futex_wait = 40`, `futex_wake = 41`. A waiter is a
|
||||
`.blocked` task tagged with `Task.futex_addr` (no queue linkage);
|
||||
`futex_wait(addr, expected, timeout_ns)` reads the user word under the big lock,
|
||||
parks iff `*addr == expected`, and returns on wake or timeout; `futex_wake(addr,
|
||||
@@ -261,7 +261,7 @@ green.
|
||||
restores it.
|
||||
- [x] `RwLock` / `WaitGroup` deferred (no consumer yet); they slot onto the same
|
||||
`Futex`/`Mutex`/`Condition` when wanted.
|
||||
- [x] All `thread-*` cases wired into [test/qemu_test.py](../test/qemu_test.py)
|
||||
- [x] All `thread-*` cases wired into [test/qemu_test.py](../../test/qemu_test.py)
|
||||
(`thread-spawn`/`-join`/`-futex`/`-mutex`/`-id`); threading.md + docs/README.md
|
||||
status updated to **built**; the worked example is threading.md's win-condition.
|
||||
- [x] `-Dtest-case=thread-id` (`smp: 4`): two workers read `getCurrentId`; the main
|
||||
@@ -405,7 +405,7 @@ guardrail 26/26 (incl. `process-kill`, `supervision`, `fault-recovery`, `task-re
|
||||
### M10 — Per-thread TLS: the thread-pointer mechanism ✅
|
||||
|
||||
Give each thread its own thread pointer and private TLS storage — the foundation
|
||||
self-hosting Zig ([zig-self-hosting.md](zig-self-hosting.md)) will build `threadlocal` on.
|
||||
self-hosting Zig ([zig-self-hosting.md](../zig-self-hosting.md)) will build `threadlocal` on.
|
||||
|
||||
- [x] **Kernel** stores `thread_pointer` on `Task` and restores it on every context switch
|
||||
**only when it changes** (the same conditional-load discipline as CR3;
|
||||
@@ -459,7 +459,7 @@ clean.
|
||||
## Deferred (explicitly not in this plan)
|
||||
|
||||
- **Cross-process shared-memory futex** — the `(address_space, virtual_address)` key can become a
|
||||
physical-address key so two processes share a futex through a [shared-memory](display-v2.md)
|
||||
physical-address key so two processes share a futex through a [shared-memory](../device-driver-development-guide/display-v2.md)
|
||||
region. Not needed for intra-process threads.
|
||||
- **Per-thread priorities / affinity distinct from the process** — threads inherit the
|
||||
process priority ([scheduling.md](scheduling.md)); revisit only if it earns its keep.
|
||||
@@ -467,5 +467,5 @@ clean.
|
||||
([process-lifecycle.md](process-lifecycle.md)).
|
||||
- **A `pthread`/POSIX surface** — the API is `std.Thread`-shaped Zig, nothing more.
|
||||
- **A real `std.Thread` backend** — arrives with self-hosting
|
||||
([zig-self-hosting.md](zig-self-hosting.md)); it sits on these same primitives, so it
|
||||
([zig-self-hosting.md](../zig-self-hosting.md)); it sits on these same primitives, so it
|
||||
swaps the impl under `runtime.Thread`, not the call sites.
|
||||
@@ -1,8 +1,8 @@
|
||||
# Threading: `runtime.Thread`, a std-shaped API over a private thread ABI
|
||||
# Threading: `Thread`, a std-shaped API over a private thread ABI
|
||||
|
||||
A note on danos **threads** — several tasks sharing one address space — provided by a
|
||||
`runtime.Thread` type that mirrors the shape of Zig's `std.Thread` while keeping every
|
||||
kernel entry behind the [runtime](../library/runtime). **Built** (M1–M11, see
|
||||
`Thread` type that mirrors the shape of Zig's `std.Thread` while keeping every
|
||||
kernel entry behind the [runtime](../../library/kernel). **Built** (M1–M11, see
|
||||
[threading-plan.md](threading-plan.md)): `spawn`/`join`/`detach`, cross-core parallelism,
|
||||
a futex, `Mutex`/`Condition`/`Semaphore`/`RwLock`/`WaitGroup`, `getCurrentId`/`currentCore`,
|
||||
per-thread thread-pointer TLS, thread-safe allocation, and a task reaper that reclaims dead
|
||||
@@ -17,13 +17,13 @@ treat upstream shapes as "0.16.x."
|
||||
A danos service can write
|
||||
|
||||
```zig
|
||||
const t = try runtime.Thread.spawn(.{}, worker, .{ctx});
|
||||
const t = try Thread.spawn(.{}, worker, .{ctx});
|
||||
// ... do other work concurrently ...
|
||||
t.join();
|
||||
```
|
||||
|
||||
and get real parallelism across cores — with `runtime.Thread.Mutex`,
|
||||
`runtime.Thread.Condition`, and `runtime.Thread.Semaphore` available for
|
||||
and get real parallelism across cores — with `Thread.Mutex`,
|
||||
`Thread.Condition`, and `Thread.Semaphore` available for
|
||||
coordination — **without any code path reaching the kernel except through the
|
||||
runtime**. The call sites read exactly like `std.Thread`, so the day danos becomes a
|
||||
real Zig target (see [self-hosting](#the-self-hosting-endgame)) we swap the
|
||||
@@ -31,18 +31,18 @@ implementation underneath, not the API above.
|
||||
|
||||
## Locked decisions (do not relitigate)
|
||||
|
||||
- **We build `runtime.Thread`, not literal `std.Thread`.** It mirrors std's *API and
|
||||
- **We build `Thread`, not literal `std.Thread`.** It mirrors std's *API and
|
||||
features*; the implementation underneath is danos-native. See
|
||||
[Why not literal std.Thread](#why-not-literal-stdthread).
|
||||
- **Threads are a narrow, opt-in capability — not the default concurrency tool.** The
|
||||
default for resilience stays **process + IPC** ([resilience.md](resilience.md),
|
||||
[ipc.md](ipc.md)). See [Where threads fit](#where-threads-fit-the-resilience-tension).
|
||||
[ipc.md](../device-driver-development-guide/ipc.md)). See [Where threads fit](#where-threads-fit-the-resilience-tension).
|
||||
- **Blocking synchronization is futex-backed, never spin-backed.** Waiters sleep in
|
||||
the kernel so an idle core still halts ([halting.md](halting.md)).
|
||||
- **Per-binary opt-in to multi-threaded codegen.** Only a service that asks for
|
||||
threads is built `single_threaded = false`; the rest stay lean and single-threaded.
|
||||
- **The thread ABI is private.** New syscalls extend [abi.zig](../system/abi.zig)
|
||||
`SystemCall` and are reached only through `library/runtime` wrappers, exactly like
|
||||
- **The thread ABI is private.** New syscalls extend [abi.zig](../../system/abi.zig)
|
||||
`SystemCall` and are reached only through `library/kernel` wrappers, exactly like
|
||||
every other danos syscall ([syscall.md](syscall.md)) — numbers stay renumberable.
|
||||
|
||||
## Why not literal `std.Thread`
|
||||
@@ -56,12 +56,12 @@ runtime — rebuilt in lockstep — knows the mapping.
|
||||
`std.Thread` is incompatible with that invariant on two counts:
|
||||
|
||||
1. **It selects its backend from `builtin.os.tag`, and issues syscalls directly.**
|
||||
danos targets `.os_tag = .freestanding` ([build.zig](../build.zig)), for which
|
||||
danos targets `.os_tag = .freestanding` ([build.zig](../../build.zig)), for which
|
||||
`std.Thread` resolves to an unsupported stub that `@compileError`s. Adding a real
|
||||
backend would either bake danos syscall numbers into std (breaking ABI privacy and
|
||||
renumbering) or fork std to route back through the runtime — a permanent rebase
|
||||
cost that buys nothing the native type doesn't.
|
||||
2. **Our user binaries are built `single_threaded = true`** ([build.zig](../build.zig)
|
||||
2. **Our user binaries are built `single_threaded = true`** ([build.zig](../../build.zig)
|
||||
`addUserBinary`), which compiles threading out entirely and makes atomics and TLS
|
||||
single-threaded. Threads need this flipped per binary regardless.
|
||||
|
||||
@@ -99,7 +99,7 @@ processes. The isolation boundary stays at process granularity.
|
||||
|
||||
## The API surface (mirrors `std.Thread`)
|
||||
|
||||
Lives in `library/runtime/thread.zig`, re-exported as `runtime.Thread`.
|
||||
Lives in `library/kernel/thread.zig`, re-exported as `Thread`.
|
||||
|
||||
```zig
|
||||
pub const Thread = struct {
|
||||
@@ -129,14 +129,14 @@ Deviations from `std.Thread`, called out honestly:
|
||||
`void`). Return data through shared state or a `Semaphore`/`Condition`, not the
|
||||
return.
|
||||
- No `getCpuCount()` (a service rarely needs it) and no `Thread.yield()` — `yield`
|
||||
lives in `runtime.system`. Instead `currentCore()` exposes the calling core's dense
|
||||
lives in the `process` module. Instead `currentCore()` exposes the calling core's dense
|
||||
index ([smp.md](smp.md)), used to observe genuine cross-core parallelism.
|
||||
|
||||
## Kernel primitives (new private syscalls)
|
||||
|
||||
Five core entries extend [abi.zig](../system/abi.zig) `SystemCall` after
|
||||
Five core entries extend [abi.zig](../../system/abi.zig) `SystemCall` after
|
||||
`shared_memory_physical = 36` (plus small helpers `current_core`, `thread_self`, and
|
||||
`set_thread_pointer`), each with a `library/runtime` wrapper:
|
||||
`set_thread_pointer`), each with a `library/kernel` wrapper:
|
||||
|
||||
| Syscall | Signature | Purpose |
|
||||
|---|---|---|
|
||||
@@ -155,12 +155,12 @@ Plus one invariant change with no new syscall: **address-space reference countin
|
||||
Before this work an address space was 1:1 with a task: `spawnUserLocked` records
|
||||
`address_space` on the Task (as it still does), and teardown did
|
||||
`destroyAddressSpace(t.address_space)` when **any** user task exited
|
||||
([scheduler.zig](../system/kernel/scheduler.zig)). With threads, several tasks share
|
||||
([scheduler.zig](../../system/kernel/scheduler.zig)). With threads, several tasks share
|
||||
one `address_space`, so the first to exit would rip the address space out from under its
|
||||
siblings.
|
||||
|
||||
Fix: a small refcount keyed by the address-space root, kept in
|
||||
[scheduler.zig](../system/kernel/scheduler.zig): `retainAddressSpace` takes a
|
||||
[scheduler.zig](../../system/kernel/scheduler.zig): `retainAddressSpace` takes a
|
||||
reference for every user task `spawnUserLocked` starts (count 1 on the first take, so
|
||||
a thread sharing the caller's space increments it); task teardown calls
|
||||
`releaseAddressSpace`, which only calls `destroyAddressSpace` at **zero**. All of
|
||||
@@ -172,7 +172,7 @@ that must land and be proven before anything shares an address space.
|
||||
The scheduler already accepts an arbitrary `address_space` and does **not** smuggle
|
||||
values through scratch registers — `startUserTask` reads the entry/stack (and the
|
||||
thread's closure arg, delivered in `rdi` via `jumpToUserArg`) from the Task
|
||||
([scheduler.zig](../system/kernel/scheduler.zig)). That makes the thread path clean:
|
||||
([scheduler.zig](../../system/kernel/scheduler.zig)). That makes the thread path clean:
|
||||
|
||||
1. The runtime's `spawn` `mmap`s a stack (syscall `4`) and writes the closure —
|
||||
`{ tls_base, args }`, the std "Instance" pattern — at the **top of the new stack
|
||||
@@ -214,7 +214,7 @@ Keying: threads share an address space, so a **virtual address within that addre
|
||||
identifies a futex uniquely; the kernel keys its wait queue by `(address_space_root, virtual_address)`.
|
||||
Keying by the **physical** address instead (translate `virtual_address -> physical_address` on entry) is a
|
||||
deliberate forward door: it lets two *processes* share a futex through an
|
||||
[shared-memory](display-v2.md) region later, without changing the API. We start with the
|
||||
[shared-memory](../device-driver-development-guide/display-v2.md) region later, without changing the API. We start with the
|
||||
private-per-address-space key and note the physical-key upgrade.
|
||||
|
||||
No spinning: a contended lock parks the task in the kernel and the core is free to run
|
||||
@@ -239,7 +239,7 @@ see the intro). Two scoped pieces, as built:
|
||||
A binary opts in by being added with `addThreadedUserBinary` — as `addUserBinary`,
|
||||
but the shared implementation builds it `single_threaded = false` — so atomics and
|
||||
(later) TLS are real. Threads and atomics are unsound in a `single_threaded` image,
|
||||
so a binary must opt in **before** it may call `runtime.Thread.spawn`. Everyone else
|
||||
so a binary must opt in **before** it may call `Thread.spawn`. Everyone else
|
||||
stays single-threaded and lean.
|
||||
|
||||
## Interaction with the rest of the kernel
|
||||
@@ -264,9 +264,9 @@ stays single-threaded and lean.
|
||||
**process**, which respawns its threads from a known-good state — restart
|
||||
granularity stays the process. The leader's recorded exit reason carries the fault
|
||||
class even when a worker faulted, so restart policy is unchanged.
|
||||
- **IPC — two consequences threads forced ([ipc.md](ipc.md)):**
|
||||
- **IPC — two consequences threads forced ([ipc.md](../device-driver-development-guide/ipc.md)):**
|
||||
- *Handles do not cross threads.* The handle table lives on the `Task`
|
||||
([scheduler.zig](../system/kernel/scheduler.zig)), so a handle number is meaningful
|
||||
([scheduler.zig](../../system/kernel/scheduler.zig)), so a handle number is meaningful
|
||||
only to the thread that created it — thread A's endpoint handle `3` is not thread B's.
|
||||
A thread that needs to reach an endpoint another thread owns looks it up
|
||||
(`ipc.lookup(service)`) to install its **own** handle to the same underlying endpoint.
|
||||
@@ -284,7 +284,7 @@ stays single-threaded and lean.
|
||||
|
||||
The ordered, `/loop`-runnable milestones live in
|
||||
**[threading-plan.md](threading-plan.md)** (shaped like
|
||||
[display-v2-plan.md](display-v2-plan.md)): every milestone lands on its own and ends in
|
||||
[display-v2-plan.md](../device-driver-development-guide/display-v2-plan.md)): every milestone lands on its own and ends in
|
||||
a verifiable gate (`python3 test/qemu_test.py <case>`, asserting serial markers;
|
||||
`zig build test` for host unit tests). The stages below are the shape it expands.
|
||||
|
||||
@@ -307,15 +307,15 @@ a verifiable gate (`python3 test/qemu_test.py <case>`, asserting serial markers;
|
||||
the consumer blocked, e.g. via a low idle tick count).
|
||||
- **Stage 3 — polish.** Per-thread TLS / thread pointer and `threadlocal` (only if a
|
||||
consumer needs it), `RwLock`/`WaitGroup` as demanded, and this doc's cases wired
|
||||
into [test/qemu_test.py](../test/qemu_test.py).
|
||||
into [test/qemu_test.py](../../test/qemu_test.py).
|
||||
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](coding-standards.md): spell out non-acronym
|
||||
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym
|
||||
abbreviations, kebab-case file names, no `Co-Authored-By` trailers. New syscalls
|
||||
extend [abi.zig](../system/abi.zig) `SystemCall` + a `library/runtime` wrapper
|
||||
([syscall.md](syscall.md)). `runtime.Thread` is a first-class runtime module, the same
|
||||
way `runtime.process` ([process-lifecycle.md](process-lifecycle.md)) and `runtime.ipc`
|
||||
extend [abi.zig](../../system/abi.zig) `SystemCall` + a `library/kernel` wrapper
|
||||
([syscall.md](syscall.md)). `Thread` is a first-class runtime module, the same
|
||||
way `process` ([process-lifecycle.md](process-lifecycle.md)) and `ipc`
|
||||
are — user code never names a syscall.
|
||||
|
||||
## Non-goals
|
||||
@@ -331,19 +331,19 @@ are — user code never names a syscall.
|
||||
## The self-hosting endgame
|
||||
|
||||
When danos becomes a real Zig target and we (eventually) add a danos backend to std
|
||||
([zig-self-hosting.md](zig-self-hosting.md)), `std.Thread` can sit *on top of* these
|
||||
([zig-self-hosting.md](../zig-self-hosting.md)), `std.Thread` can sit *on top of* these
|
||||
same kernel primitives — the danos `std.Thread.Impl` would call the very
|
||||
`thread_spawn`/`futex_*` wrappers `runtime.Thread` already uses. Because
|
||||
`runtime.Thread` was built API-compatible from day one, that transition swaps the
|
||||
`thread_spawn`/`futex_*` wrappers `Thread` already uses. Because
|
||||
`Thread` was built API-compatible from day one, that transition swaps the
|
||||
implementation, not a single call site. Designing to the std shape now is what makes
|
||||
the later self-hosting lift cheap.
|
||||
|
||||
## Further reading
|
||||
|
||||
- [scheduling.md](scheduling.md), [smp.md](smp.md) — the task model these threads join.
|
||||
- [resilience.md](resilience.md), [vision.md](vision.md) — why isolation is the default
|
||||
- [resilience.md](resilience.md), [vision.md](../vision.md) — why isolation is the default
|
||||
and threads are the exception.
|
||||
- [syscall.md](syscall.md), [ipc.md](ipc.md) — the private ABI and the messaging model
|
||||
- [syscall.md](syscall.md), [ipc.md](../device-driver-development-guide/ipc.md) — the private ABI and the messaging model
|
||||
threads sit beside.
|
||||
- [halting.md](halting.md) — the idle/halt property futex-backed blocking preserves.
|
||||
- [zig-self-hosting.md](zig-self-hosting.md) — the target this bends toward.
|
||||
- [zig-self-hosting.md](../zig-self-hosting.md) — the target this bends toward.
|
||||
@@ -7,7 +7,7 @@ Two different needs hide under the word "timer", and danos keeps them apart:
|
||||
|
||||
Both are answered by the **kernel**, because the kernel already owns a timer: it has
|
||||
to, to preempt tasks. The LAPIC heartbeat and the calibrated TSC that back all of this
|
||||
are built in [device-interrupts.md](device-interrupts.md); the scheduler's blocking and
|
||||
are built in [device-interrupts.md](../device-driver-development-guide/device-interrupts.md); the scheduler's blocking and
|
||||
wait queues are in [scheduling.md](scheduling.md). This page is about the surface a
|
||||
ring-3 program actually uses, and one deliberate absence: **there is no user-space time
|
||||
service.**
|
||||
@@ -32,11 +32,11 @@ danos checks both — the invariant-TSC CPUID bit (`0x80000007` EDX[8], set on I
|
||||
AMD), and a cross-core "warp" check as the cores come up — and falls back to the HPET
|
||||
counter when either fails. So `now()` stays accurate on a real Intel box, a real AMD box,
|
||||
and inside a VM alike; only the source behind it differs. The mechanism is in
|
||||
[device-interrupts.md](device-interrupts.md).
|
||||
[device-interrupts.md](../device-driver-development-guide/device-interrupts.md).
|
||||
|
||||
So the timer hardware lives in the kernel, and there is **no `hpet` driver and no time
|
||||
server** to consume. (An earlier HPET driver existed only to *demonstrate* the driver
|
||||
model; that role now lives in [drivers.md](drivers.md), as documentation.) The one place
|
||||
model; that role now lives in [drivers.md](../device-driver-development-guide/drivers.md), as documentation.) The one place
|
||||
a user-space time service *is* justified — **wall-clock / calendar time** — is discussed
|
||||
at the end; it is deliberately not built yet.
|
||||
|
||||
@@ -55,20 +55,20 @@ Time and waiting are three entries in the small syscall table ([syscall.md](sysc
|
||||
service can keep answering messages on the same endpoint while a deadline is pending.
|
||||
This is the timed wait that stop-sequence escalation, hello deadlines, and restart
|
||||
backoff are built from ([process-lifecycle.md](process-lifecycle.md),
|
||||
[device-manager.md](device-manager.md)).
|
||||
[device-manager.md](../device-driver-development-guide/device-manager.md)).
|
||||
|
||||
The kernel's own scheduling timer (the LAPIC, vector 32) is never exposed to user space;
|
||||
programs read the TSC through `clock` and get timed wakeups through `sleep`/`timer_bind`,
|
||||
both riding the scheduler tick.
|
||||
|
||||
## `runtime.time` — the generic interface
|
||||
## `time` — the generic interface
|
||||
|
||||
Applications don't call the syscalls directly; they use `runtime.time`
|
||||
(`library/runtime/time.zig`), a thin `Instant`/`Duration` layer over them — an ergonomic
|
||||
Applications don't call the syscalls directly; they use `time`
|
||||
(`library/kernel/time.zig`), a thin `Instant`/`Duration` layer over them — an ergonomic
|
||||
front door, not new mechanism.
|
||||
|
||||
```zig
|
||||
const time = @import("runtime").time;
|
||||
const time = @import("time");
|
||||
|
||||
const start = time.now(); // Instant — monotonic
|
||||
doWork();
|
||||
@@ -91,8 +91,9 @@ _ = time.after(endpoint, time.Duration.fromMillis(200));
|
||||
- `sleep(d)` wraps `sleep`; `spin(d)` busy-polls `now()` for the sub-millisecond delays
|
||||
the millisecond tick can't express; `after(endpoint, d)` wraps `timer_bind`.
|
||||
|
||||
The raw wrappers (`system.clock`, `system.sleep`, `system.timerOnce`) stay in
|
||||
`library/runtime/system.zig`; `runtime.time` is the layer meant for everyday use.
|
||||
The raw wrappers (`clock`, `sleepMillis`, `timerOnce`) and the ergonomic
|
||||
`Instant`/`Duration` layer both live in the `time` module
|
||||
(`library/kernel/time.zig`); the latter is what everyday code uses.
|
||||
|
||||
## Wall-clock time (not built)
|
||||
|
||||
@@ -106,10 +107,10 @@ owns covers every current use.
|
||||
|
||||
## Verifying it
|
||||
|
||||
`runtime.time`'s `Instant`/`Duration` arithmetic has unit tests that run on the host:
|
||||
`time`'s `Instant`/`Duration` arithmetic has unit tests that run on the host:
|
||||
|
||||
```
|
||||
$ zig build test # includes library/runtime/time.zig
|
||||
$ zig build test # includes library/kernel/time.zig
|
||||
```
|
||||
|
||||
End to end, the proof the clock is real is that it *advances*: read `now()`, `sleep` a
|
||||
@@ -1,7 +1,7 @@
|
||||
# The vDSO — the public system-call boundary
|
||||
|
||||
> **Status:** design note, not built. The runtime today issues raw `syscall`
|
||||
> instructions from `library/runtime/system-call.zig` using the numbers in
|
||||
> instructions from `library/kernel/system-call.zig` using the numbers in
|
||||
> `system/abi.zig`. This note designs the layer that replaces that arrangement:
|
||||
> a **kernel-supplied, C-ABI entry library** mapped into every process — the
|
||||
> only supported way into the kernel — so the raw numbers can stay private,
|
||||
@@ -25,8 +25,8 @@ ourselves:
|
||||
this mistake: it issued XNU syscalls directly instead of going through
|
||||
libSystem, and macOS updates repeatedly broke every Go binary until Go
|
||||
switched to the library like everyone else.
|
||||
2. **Not everything is Zig.** A Rust or C program can't import the `runtime`
|
||||
module. The public boundary has to be expressible in the one calling
|
||||
2. **Not everything is Zig.** A Rust or C program can't import the danos Zig
|
||||
modules. The public boundary has to be expressible in the one calling
|
||||
convention every language speaks: the C ABI.
|
||||
3. **Randomised syscall numbers** — a hardening option we want open — only
|
||||
work if no user binary anywhere knows a number at build time. The binding
|
||||
@@ -49,10 +49,10 @@ The public danos ABI then has exactly two layers, neither of which is
|
||||
|
||||
| Layer | Contract | Spoken by |
|
||||
|-------|----------|-----------|
|
||||
| **vDSO** | C-ABI functions, this note | every language's thin shim (`runtime.system` for Zig, a `-sys` crate for Rust, a header for C) |
|
||||
| **IPC wire protocols** | byte layouts over `ipc_call` ([vfs-protocol.md](vfs-protocol.md) is the first one documented) | any client that can lay out bytes |
|
||||
| **vDSO** | C-ABI functions, this note | every language's thin shim (the `system-call` module for Zig, a `-sys` crate for Rust, a header for C) |
|
||||
| **IPC wire protocols** | byte layouts over `ipc_call` ([vfs-protocol.md](../file-system-development/vfs-protocol.md) is the first one documented) | any client that can lay out bytes |
|
||||
|
||||
Everything above those — the heap, `runtime.fs`, the service harness — is
|
||||
Everything above those — the heap, `file_system`, the service harness — is
|
||||
per-language convenience, compiled into each binary from source, exactly as
|
||||
today. Nothing about the Zig runtime's shape changes; it just stops being the
|
||||
*only* door.
|
||||
@@ -107,7 +107,7 @@ convenience, not a requirement.)
|
||||
|
||||
The kernel already builds a System V entry block — argc, argv, envp
|
||||
terminator, **auxiliary vector** — on every new process's stack
|
||||
(`buildEntryStack`, read by `runtime.start`). The vDSO base rides in a new
|
||||
(`buildEntryStack`, read by the `start` module). The vDSO base rides in a new
|
||||
auxv entry, exactly Linux's `AT_SYSINFO_EHDR` move. No new syscall, no magic
|
||||
address, and a language shim finds it the same portable way on every
|
||||
architecture.
|
||||
@@ -183,11 +183,11 @@ second — but the design should never be sold as more than that.
|
||||
Phased so every step ships alone (the M-milestone discipline):
|
||||
|
||||
1. **The blob + the table.** Build the vDSO, map it at spawn, deliver the
|
||||
base via auxv. `runtime.system-call.zig` binds through the table when the
|
||||
base via auxv. `library/kernel/system-call.zig` binds through the table when the
|
||||
auxv entry is present, falls back to raw `syscall` when absent — the whole
|
||||
tree keeps booting during the transition.
|
||||
2. **Cut the runtime over.** Delete the raw stubs; `runtime` no longer
|
||||
imports the `SystemCall` numbers at all (`abi.zig`'s enum becomes
|
||||
2. **Cut the system library over.** Delete the raw stubs; the `system-call`
|
||||
module no longer imports the `SystemCall` numbers at all (`abi.zig`'s enum becomes
|
||||
kernel-internal). The QEMU suite passing proves the table carries the
|
||||
whole system.
|
||||
3. **Enforce + randomise.** Add the `rip`-range check, then per-boot number
|
||||
@@ -115,12 +115,13 @@ hypervisor configured for UEFI firmware and an xHCI USB controller.
|
||||
(`efi.zig:790`, `boot-handoff.zig:149`)
|
||||
- **Required ACPI tables:** MADT (interrupt topology), MCFG (PCIe ECAM base),
|
||||
FADT (power / PM timer). Optionally consumed: HPET, DMAR, SPCR.
|
||||
(`system/devices/acpi.zig:3`)
|
||||
(`system/kernel/acpi.zig:3`)
|
||||
- The loader reads `/system/kernel` off the FAT boot volume, then loads user
|
||||
space: a prebuilt `boot\system.img` capsule
|
||||
([system-image.md](system-image.md)) when present, otherwise it walks
|
||||
the volume's `/system` tree (init included) into the initial ramdisk. The
|
||||
kernel can boot "kernel-only" without either. (`efi.zig:16`, `efi.zig:68`)
|
||||
([system-image.md](os-development-guide/system-image.md)) when present, otherwise it walks
|
||||
the volume's `/system` and optional `/test` trees (init included) into the
|
||||
initial ramdisk. The kernel can boot "kernel-only" without either.
|
||||
(`efi.zig:16`, `efi.zig:68`)
|
||||
|
||||
## Interrupt controller
|
||||
|
||||
@@ -221,7 +222,7 @@ named for reporting only; internal SATA / NVMe / IDE disks have no driver.
|
||||
first VT-d DRHD unit and its capabilities are exposed via `PlatformInformation`
|
||||
(`iommu_present`, `iommu_base`, `iommu_version`). No DMA-remapping tables are
|
||||
programmed and no translation is enforced. An IOMMU is therefore **not required**
|
||||
and does not currently constrain devices. (`system/devices/acpi.zig:96`)
|
||||
and does not currently constrain devices. (`system/kernel/acpi.zig:96`)
|
||||
|
||||
## What is explicitly NOT supported
|
||||
|
||||
|
||||
+3
-3
@@ -9,7 +9,7 @@ There are two layers:
|
||||
|
||||
- **Host unit tests** (`zig build test`) — for pure, platform-independent logic.
|
||||
What began as the three shared contracts (`system/boot-handoff.zig`,
|
||||
`system/abi.zig`, `system/devices/device-abi.zig`) now spans ~26 modules:
|
||||
`system/abi.zig`, `library/device/model/device-abi.zig`) now spans ~26 modules:
|
||||
protocol and on-wire definitions (VFS, USB, virtio-gpu), the FAT engine, the
|
||||
display compositor, PS/2 and HID decoding, the kernel log ring, and the
|
||||
runtime's `time`/`thread` — the full list is the test step in `build.zig`.
|
||||
@@ -27,7 +27,7 @@ boot log, memory summary, exception reports — appears on serial as plain text.
|
||||
|
||||
QEMU captures that with `-serial file:serial.log`, giving a machine-readable
|
||||
transcript. Serial is per-architecture (x86 uses port I/O; an ARM board uses a
|
||||
memory-mapped UART), so it lives behind the [architecture](architecture.md) boundary — and adding
|
||||
memory-mapped UART), so it lives behind the [architecture](os-development-guide/architecture.md) boundary — and adding
|
||||
a new architecture's UART is what makes the same tests run there.
|
||||
|
||||
The serial log sink is **compiled in only under `-Dserial`** (off by default).
|
||||
@@ -88,7 +88,7 @@ table in `test/qemu_test.py`):
|
||||
| `fault-recovery` | a ring-3 process that faults is killed and reaped while init keeps heartbeating — the OS survives | `DANOS-TEST-RESULT: PASS` |
|
||||
|
||||
The faulting cases don't print a result line — they deliberately raise a CPU
|
||||
exception, and the harness asserts on the [exception report](interrupts.md) the
|
||||
exception, and the harness asserts on the [exception report](os-development-guide/interrupts.md) the
|
||||
handler prints (which also reaches serial). This reuses the real fault path as the
|
||||
test oracle: if the IDT/TSS weren't wired up, `fault-df` would triple-fault and the
|
||||
marker would never appear.
|
||||
|
||||
-118
@@ -1,118 +0,0 @@
|
||||
# Vision: a microkernel, built to learn
|
||||
|
||||
danos exists first and foremost as a **learning-by-doing project**: the point is to
|
||||
build a real operating system, bump into the hard constraints for real, and research
|
||||
them from a position of having actually hit them. The docs in this folder are part of
|
||||
that — they're where a constraint gets understood once it's been met.
|
||||
|
||||
That framing sets the priorities. danos is not chasing a spec or a product; it's
|
||||
chasing understanding, with a concrete, motivating **win condition** to aim at.
|
||||
|
||||
## The win condition
|
||||
|
||||
danos is a "win" when it:
|
||||
|
||||
- **boots and runs on real hardware** — the author's **PC** (x86-64) and **both
|
||||
Raspberry Pis**: the **Zero 2 W** and the **Pi 5** (both `aarch64`, one backend —
|
||||
see [arm.md](arm.md)),
|
||||
- **has a graphical user interface**, ideally — building on the framebuffer it
|
||||
already draws to.
|
||||
|
||||
Everything below serves that, or serves the curiosity that the project runs on.
|
||||
|
||||
## Why a microkernel: resilience
|
||||
|
||||
The kernel stays **minimal** — only what genuinely must run privileged:
|
||||
|
||||
- scheduling,
|
||||
- inter-process communication (IPC),
|
||||
- memory management (address spaces, page tables),
|
||||
- low-level interrupt dispatch.
|
||||
|
||||
Everything else — device drivers, filesystems, the GUI, the network stack — runs as
|
||||
an **isolated user-space server**, each in its own address space with only the
|
||||
privileges it needs.
|
||||
|
||||
The reason for this shape is **resilience**: the ability to **re-initialise parts of
|
||||
the OS while it runs**. A driver bug can't corrupt the kernel or another driver; a
|
||||
crashed or wedged component is contained, killed, and **restarted** — "if I break
|
||||
something, I can just fix it," without rebooting. Keeping the kernel tiny is part of
|
||||
that strategy: the one thing that *can't* be restarted is the trusted base, so the
|
||||
less code in it, the less that can take the whole system down. This is the project's
|
||||
real motivation, and it has its own design note: [resilience.md](resilience.md).
|
||||
|
||||
The cost is that **IPC becomes the backbone**: what used to be a function call inside
|
||||
a monolithic kernel is now a message between address spaces. In a microkernel, IPC
|
||||
performance essentially *is* system performance (the lesson of L4), so it's a
|
||||
first-class concern. Hardware interrupts become IPC too: the kernel turns an IRQ into
|
||||
a message to the driver that owns the device.
|
||||
|
||||
## On real-time: an option, not a commitment
|
||||
|
||||
danos was originally framed as a hard **real-time** OS. That's now held as **one
|
||||
interesting constraint to explore, not a requirement** — because real-time is a
|
||||
*pervasive* invariant (every operation must be provably time-bounded, everywhere)
|
||||
that would slow every milestone, whereas resilience is a set of *structural* features
|
||||
that's lighter to build and is what the project actually wants. The trade-off is
|
||||
written up in [smp.md](smp.md#does-the-right-choice-depend-on-real-time-vs-resilience).
|
||||
|
||||
What danos keeps from the real-time direction, because it's cheap and useful anyway:
|
||||
|
||||
- **Fixed-priority preemptive scheduling** — the highest-priority ready task runs, and
|
||||
preemption lets a runaway component be interrupted and killed (which *serves
|
||||
resilience*). Already built ([scheduling.md](scheduling.md)).
|
||||
- **A calibrated, deterministic clock** — already built ([device-interrupts.md](device-interrupts.md)).
|
||||
|
||||
What danos does *not* owe anyone unless it deliberately chooses real-time later:
|
||||
timing *guarantees*, priority inheritance, bounded allocators, tickless timers, MCS
|
||||
scheduling contexts. Concretely, the current [heap](heap.md) is a first-fit free list
|
||||
with unbounded allocation time — fine here, and only a problem *if* a hard-real-time
|
||||
path is ever added. Note that **QNX is both** a real-time and a restartable
|
||||
microkernel, so choosing resilience now doesn't close the real-time door — it just
|
||||
doesn't pay the tax yet.
|
||||
|
||||
## The roadmap — tracks, not a strict line
|
||||
|
||||
Because the driver is curiosity plus the win condition, the roadmap is a set of
|
||||
**tracks** with dependencies, not a rigid sequence. Pick by interest; mind the
|
||||
prerequisites.
|
||||
|
||||
**Done:** UEFI boot, framebuffer + [serial](testing.md), [physical frames](frame-allocator.md)
|
||||
(with boot-services memory reclaimed), [paging](paging.md) with W^X, [exceptions and
|
||||
interrupts](interrupts.md), a [calibrated timer + ns clock](device-interrupts.md), a
|
||||
[heap](heap.md), a [fixed-priority preemptive scheduler](scheduling.md) with blocking,
|
||||
in-kernel [IPC channels](ipc.md), SMP (all cores scheduling, with affinity), a
|
||||
**higher-half kernel** with a physmap, and **user space**: per-process address
|
||||
spaces, `syscall`/`sysret` with the `swapgs` discipline, a user-ELF loader, and
|
||||
`/system/services/init` — a real user ELF built from `system/services/init/`, running at CPL 3 as PID 1 on its
|
||||
own page tables — plus a [test harness](testing.md).
|
||||
|
||||
- **Isolation track** — **user mode + address-space isolation**. *Done: a
|
||||
higher-half kernel with a physmap (the low half is user space), per-process
|
||||
address spaces with CR3 switched on context switch, the `swapgs` discipline,
|
||||
`syscall`/`sysret`, a user-ELF loader, an address-space/stack reaper for exited
|
||||
tasks, and `/system/services/init` running as a real preemptive ring-3 process
|
||||
(PID 1). Remaining polish: SMAP + fault-recovering copy-in/out, and TLB shootdown
|
||||
once a process has more than one thread. (The real IPC syscalls —
|
||||
`ipc_call`/`ipc_reply_wait` — have since been built and are the backbone every
|
||||
driver and service speaks; see [ipc.md](ipc.md).)*
|
||||
- **Resilience track** — fault → kill → notify, a supervisor/reincarnation server,
|
||||
resource cleanup on death, then a restartable driver as proof. Needs isolation.
|
||||
See [resilience.md](resilience.md).
|
||||
- **ARM track** — the `aarch64` port so danos runs on the Zero 2 W and Pi 5. Largely
|
||||
independent of the others (it's the [architecture layer](architecture.md)); directly serves the win
|
||||
condition. Likely via aarch64-UEFI first (QEMU `virt` + AAVMF), then real boards.
|
||||
See [arm.md](arm.md), and [discovery.md](discovery.md) for the device tree it needs.
|
||||
- **GUI track** — a framebuffer-based windowing/compositor, and the input + display
|
||||
drivers under it. Builds on the neutral framebuffer (so it's arch-independent), and
|
||||
on the driver model from the isolation/resilience tracks. The visible payoff.
|
||||
|
||||
The natural spine is **isolation → (resilience + drivers) → GUI**, with the **ARM
|
||||
track** pursued alongside whenever the itch to see it boot on a Pi wins out.
|
||||
|
||||
## How to use this page
|
||||
|
||||
Read it before adding anything structural. When a design decision comes up, the
|
||||
question is: does it serve the **win condition** (runs on the three machines, with a
|
||||
GUI), or the **learning** (a constraint worth meeting)? If it serves neither — e.g.
|
||||
paying the full real-time tax with no payoff in sight — it can wait.
|
||||
@@ -108,7 +108,7 @@ localised (below).
|
||||
## The architecture decision: `runtime.os` + `runtime.fs`, and retire `posix`
|
||||
|
||||
danos already has the right split ([the private-ABI boundary](../README.md)): the
|
||||
kernel exposes a minimal syscall ABI ([syscall.md](syscall.md)); the **`runtime`**
|
||||
kernel exposes a minimal syscall ABI ([syscall.md](os-development-guide/syscall.md)); the **`runtime`**
|
||||
library is the stable, danos-native application ABI. What this roadmap adds:
|
||||
|
||||
- **`runtime.os` — the seam.** A C-ABI-shaped module of the ~30 operations
|
||||
@@ -165,7 +165,7 @@ What the seam needs, and what danos already provides:
|
||||
| mmap / munmap | native syscalls ([abi.zig](../system/abi.zig)) | none |
|
||||
| page allocator | over `mmap`, via `root.os.heap.page_allocator` override | ~30-line hook |
|
||||
| monotonic clock | `clock` syscall | none |
|
||||
| args / argv | SysV entry stack ([sysv.md](sysv.md)), `runtime.process.Init` | none |
|
||||
| args / argv | SysV entry stack ([sysv.md](os-development-guide/sysv.md)), `runtime.process.Init` | none |
|
||||
| stdout / stderr | `debug_write` today | wire fd 1/2 to a console **byte** stream |
|
||||
| mkdir / unlink / rename / truncate | done — engine + VFS + `runtime.fs` (Phase 2) | — |
|
||||
| stat fields | `{size, kind, mtime}` | **mode / inode** still missing (cache validity) |
|
||||
@@ -209,7 +209,7 @@ build); point danos's `build.zig`/CI at the resulting binary. Four localised pat
|
||||
plan9/serenity;
|
||||
- add `danos` to the freestanding/other **no-op `_start` list** in `std`'s `start.zig`,
|
||||
so std does *not* emit its own System-V `_start` — danos keeps owning the entry shim
|
||||
and `Init`/argv construction it already builds ([sysv.md](sysv.md));
|
||||
and `Init`/argv construction it already builds ([sysv.md](os-development-guide/sysv.md));
|
||||
- wire the `system` selector `.danos => std.os.danos` in `std.posix`;
|
||||
- add `std/os/danos.zig` — **the seam itself**, promoted near-verbatim from the
|
||||
`runtime.os` developed first in Phase 1 (against the stock toolchain, so the fork is
|
||||
@@ -243,7 +243,7 @@ readdir/isatty/args/exit) exists. Those are downstream and out of scope here.
|
||||
danos's biggest genuine gap, and the correctness-critical one:
|
||||
|
||||
- Add **mkdir / unlink / rename / truncate** to *both* the VFS wire protocol
|
||||
([vfs-protocol.zig](../system/vfs-protocol.zig)) and the FAT engine
|
||||
([vfs-protocol.zig](../library/protocol/vfs/vfs-protocol.zig)) and the FAT engine
|
||||
([engine.zig](../system/services/fat/engine.zig)), then expose them via `runtime.os`.
|
||||
- Extend `stat` beyond `{size, kind}` to carry **mtime + inode + mode** — `std`'s file
|
||||
stat needs them for build-cache validity — which in turn needs **wall-clock** time
|
||||
@@ -344,10 +344,10 @@ Two current decisions fall out of this roadmap:
|
||||
## Related
|
||||
|
||||
- [vision.md](vision.md) — the north star this serves.
|
||||
- [syscall.md](syscall.md) — the kernel↔runtime ABI `runtime.os` is built on.
|
||||
- [sysv.md](sysv.md) — the entry stack (`argc/argv/envp/auxv`) danos already constructs.
|
||||
- [ipc.md](ipc.md) — the IPC the VFS/FAT operations travel over.
|
||||
- [danos-file-system-hierarchy-FSH.md](danos-file-system-hierarchy-FSH.md) — the
|
||||
- [syscall.md](os-development-guide/syscall.md) — the kernel↔runtime ABI `runtime.os` is built on.
|
||||
- [sysv.md](os-development-guide/sysv.md) — the entry stack (`argc/argv/envp/auxv`) danos already constructs.
|
||||
- [ipc.md](device-driver-development-guide/ipc.md) — the IPC the VFS/FAT operations travel over.
|
||||
- [danos-file-system-hierarchy-FSH.md](file-system-development/danos-file-system-hierarchy-FSH.md) — the
|
||||
filesystem layout the file surface serves.
|
||||
- [coding-standards.md](coding-standards.md) — danos naming (why the compat spellings
|
||||
are confined, and now retired).
|
||||
|
||||
@@ -4,9 +4,9 @@
|
||||
//! reply marshalling. See system/services/display/ and docs/display.md.
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const protocol = @import("display-protocol");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const display_protocol = @import("display-protocol");
|
||||
|
||||
/// The display's current mode, as `info()` reports it.
|
||||
pub const Info = struct {
|
||||
@@ -29,50 +29,50 @@ fn service() ?ipc.Handle {
|
||||
handle = h;
|
||||
return h;
|
||||
}
|
||||
system.sleep(50);
|
||||
time.sleepMillis(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Send one request, receive its reply; true on a zero status. `out` receives the reply
|
||||
/// so callers can read `info`/`layer` fields on success.
|
||||
fn transact(request: protocol.Request, out: *protocol.Reply) bool {
|
||||
fn transact(request: display_protocol.Request, out: *display_protocol.Reply) bool {
|
||||
const h = service() orelse return false;
|
||||
var req = request;
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
var reply: [display_protocol.reply_size]u8 = undefined;
|
||||
const len = ipc.call(h, std.mem.asBytes(&req), &reply) catch return false;
|
||||
if (len < protocol.reply_size) return false;
|
||||
out.* = std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]);
|
||||
if (len < display_protocol.reply_size) return false;
|
||||
out.* = std.mem.bytesToValue(display_protocol.Reply, reply[0..display_protocol.reply_size]);
|
||||
return out.status == 0;
|
||||
}
|
||||
|
||||
/// The display's current mode, or null if the service never came up.
|
||||
pub fn info() ?Info {
|
||||
var reply: protocol.Reply = undefined;
|
||||
if (!transact(.{ .operation = @intFromEnum(protocol.Operation.info) }, &reply)) return null;
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
if (!transact(.{ .operation = @intFromEnum(display_protocol.Operation.info) }, &reply)) return null;
|
||||
return .{ .width = reply.width, .height = reply.height, .pitch = reply.pitch, .format = reply.format };
|
||||
}
|
||||
|
||||
/// Composite the dirty layers and flush the frame to the screen.
|
||||
pub fn present() bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
return transact(.{ .operation = @intFromEnum(protocol.Operation.present) }, &reply);
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
return transact(.{ .operation = @intFromEnum(display_protocol.Operation.present) }, &reply);
|
||||
}
|
||||
|
||||
/// One selectable display mode.
|
||||
pub const Mode = protocol.Mode;
|
||||
pub const Mode = display_protocol.Mode;
|
||||
|
||||
/// Fill `out` with the resolutions the display can switch to; returns how many were written
|
||||
/// (zero on the GOP floor, or if the service never came up).
|
||||
pub fn modes(out: []Mode) usize {
|
||||
const h = service() orelse return 0;
|
||||
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.get_modes) };
|
||||
var reply: [protocol.modes_reply_size]u8 = undefined;
|
||||
var request = display_protocol.Request{ .operation = @intFromEnum(display_protocol.Operation.get_modes) };
|
||||
var reply: [display_protocol.modes_reply_size]u8 = undefined;
|
||||
const len = ipc.call(h, std.mem.asBytes(&request), &reply) catch return 0;
|
||||
if (len < protocol.modes_reply_size) return 0;
|
||||
const answer = std.mem.bytesToValue(protocol.ModesReply, reply[0..protocol.modes_reply_size]);
|
||||
if (len < display_protocol.modes_reply_size) return 0;
|
||||
const answer = std.mem.bytesToValue(display_protocol.ModesReply, reply[0..display_protocol.modes_reply_size]);
|
||||
if (answer.status != 0) return 0;
|
||||
const count = @min(@min(answer.count, protocol.max_modes), out.len);
|
||||
const count = @min(@min(answer.count, display_protocol.max_modes), out.len);
|
||||
for (0..count) |i| out[i] = answer.modes[i];
|
||||
return count;
|
||||
}
|
||||
@@ -80,8 +80,8 @@ pub fn modes(out: []Mode) usize {
|
||||
/// Change the display resolution. Only a native backend that supports mode-setting honours it
|
||||
/// (on the GOP floor it returns false); on success the display's `info()` reports the new mode.
|
||||
pub fn setMode(width: u32, height: u32) bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
const changed = transact(.{ .operation = @intFromEnum(protocol.Operation.set_mode), .width = width, .height = height }, &reply);
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
const changed = transact(.{ .operation = @intFromEnum(display_protocol.Operation.set_mode), .width = width, .height = height }, &reply);
|
||||
if (changed) mode = null; // the cached mode is stale now
|
||||
return changed;
|
||||
}
|
||||
@@ -100,7 +100,7 @@ fn cachedInfo() ?Info {
|
||||
/// client packs colours through this so it never has to know the byte order itself.
|
||||
pub fn color(r: u8, g: u8, b: u8) u32 {
|
||||
const format = if (cachedInfo()) |i| i.format else 0;
|
||||
return protocol.pack(format, r, g, b);
|
||||
return display_protocol.pack(format, r, g, b);
|
||||
}
|
||||
|
||||
/// A handle to a server-owned layer: a positioned, z-ordered surface the client draws
|
||||
@@ -111,9 +111,9 @@ pub const Layer = struct {
|
||||
|
||||
/// Fill a rectangle of this layer (layer-local coordinates) with a native `colour`.
|
||||
pub fn fill(self: Layer, x: i32, y: i32, w: u32, h: u32, colour: u32) bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
return transact(.{
|
||||
.operation = @intFromEnum(protocol.Operation.fill_rect),
|
||||
.operation = @intFromEnum(display_protocol.Operation.fill_rect),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
@@ -125,10 +125,10 @@ pub const Layer = struct {
|
||||
|
||||
/// Copy a `w`×`h` tile of native pixels (row-major, little-endian bytes) into this
|
||||
/// layer at (`x`, `y`). The tile rides inline in the request, so `w*h*4` must fit
|
||||
/// `protocol.maximum_payload`.
|
||||
/// `display_protocol.maximum_payload`.
|
||||
pub fn blitTile(self: Layer, x: i32, y: i32, w: u32, h: u32, pixels: []const u8) bool {
|
||||
var request = protocol.Request{
|
||||
.operation = @intFromEnum(protocol.Operation.blit_tile),
|
||||
var request = display_protocol.Request{
|
||||
.operation = @intFromEnum(display_protocol.Operation.blit_tile),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
@@ -136,22 +136,22 @@ pub const Layer = struct {
|
||||
.height = h,
|
||||
};
|
||||
const header = std.mem.asBytes(&request);
|
||||
if (header.len + pixels.len > protocol.message_maximum) return false;
|
||||
var buffer: [protocol.message_maximum]u8 = undefined;
|
||||
if (header.len + pixels.len > display_protocol.message_maximum) return false;
|
||||
var buffer: [display_protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(buffer[0..header.len], header);
|
||||
@memcpy(buffer[header.len..][0..pixels.len], pixels);
|
||||
const h_svc = service() orelse return false;
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
var reply: [display_protocol.reply_size]u8 = undefined;
|
||||
const len = ipc.call(h_svc, buffer[0 .. header.len + pixels.len], &reply) catch return false;
|
||||
if (len < protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status == 0;
|
||||
if (len < display_protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(display_protocol.Reply, reply[0..display_protocol.reply_size]).status == 0;
|
||||
}
|
||||
|
||||
/// Move / restack / show or hide the layer.
|
||||
pub fn configure(self: Layer, x: i32, y: i32, z: u32, visible: bool) bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
return transact(.{
|
||||
.operation = @intFromEnum(protocol.Operation.configure_layer),
|
||||
.operation = @intFromEnum(display_protocol.Operation.configure_layer),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
@@ -163,9 +163,9 @@ pub const Layer = struct {
|
||||
/// Mark a rectangle of this layer (layer-local) dirty for the next present — for when
|
||||
/// the layer's pixels changed without a drawing call the compositor already tracked.
|
||||
pub fn damage(self: Layer, x: i32, y: i32, w: u32, h: u32) bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
return transact(.{
|
||||
.operation = @intFromEnum(protocol.Operation.damage),
|
||||
.operation = @intFromEnum(display_protocol.Operation.damage),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
@@ -176,17 +176,17 @@ pub const Layer = struct {
|
||||
|
||||
/// Release the layer and its surface.
|
||||
pub fn destroy(self: Layer) bool {
|
||||
var reply: protocol.Reply = undefined;
|
||||
return transact(.{ .operation = @intFromEnum(protocol.Operation.destroy_layer), .layer = self.id }, &reply);
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
return transact(.{ .operation = @intFromEnum(display_protocol.Operation.destroy_layer), .layer = self.id }, &reply);
|
||||
}
|
||||
};
|
||||
|
||||
/// Create a server-owned layer of `w`×`h` pixels at screen (`x`, `y`) with stacking order
|
||||
/// `z` (higher is nearer the front), initially visible. Returns a handle, or null.
|
||||
pub fn createLayer(x: i32, y: i32, w: u32, h: u32, z: u32) ?Layer {
|
||||
var reply: protocol.Reply = undefined;
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
if (!transact(.{
|
||||
.operation = @intFromEnum(protocol.Operation.create_layer),
|
||||
.operation = @intFromEnum(display_protocol.Operation.create_layer),
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
.width = w,
|
||||
@@ -23,26 +23,26 @@
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const protocol = @import("input-protocol");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const input_protocol = @import("input-protocol");
|
||||
|
||||
pub const DeviceKind = protocol.DeviceKind;
|
||||
pub const InputEvent = protocol.InputEvent;
|
||||
pub const KeyEvent = protocol.KeyEvent;
|
||||
pub const MouseEvent = protocol.MouseEvent;
|
||||
pub const JoystickEvent = protocol.JoystickEvent;
|
||||
pub const EventKind = protocol.EventKind;
|
||||
pub const MouseEventKind = protocol.MouseEventKind;
|
||||
pub const JoystickEventKind = protocol.JoystickEventKind;
|
||||
pub const Keycode = protocol.Keycode;
|
||||
pub const DeviceKind = input_protocol.DeviceKind;
|
||||
pub const InputEvent = input_protocol.InputEvent;
|
||||
pub const KeyEvent = input_protocol.KeyEvent;
|
||||
pub const MouseEvent = input_protocol.MouseEvent;
|
||||
pub const JoystickEvent = input_protocol.JoystickEvent;
|
||||
pub const EventKind = input_protocol.EventKind;
|
||||
pub const MouseEventKind = input_protocol.MouseEventKind;
|
||||
pub const JoystickEventKind = input_protocol.JoystickEventKind;
|
||||
pub const Keycode = input_protocol.Keycode;
|
||||
|
||||
/// Interest masks re-exported so a caller can `subscribe(input.device_keyboard |
|
||||
/// input.device_mouse)`.
|
||||
pub const device_keyboard = protocol.device_keyboard;
|
||||
pub const device_mouse = protocol.device_mouse;
|
||||
pub const device_joystick = protocol.device_joystick;
|
||||
pub const device_all = protocol.device_all;
|
||||
pub const device_keyboard = input_protocol.device_keyboard;
|
||||
pub const device_mouse = input_protocol.device_mouse;
|
||||
pub const device_joystick = input_protocol.device_joystick;
|
||||
pub const device_all = input_protocol.device_all;
|
||||
|
||||
/// Look up the input service, retrying while it is still coming up. Both a subscriber and
|
||||
/// a source race the service's registration at boot, so both wait for it here rather than
|
||||
@@ -51,7 +51,7 @@ fn lookupService() ?ipc.Handle {
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 100) : (attempts += 1) {
|
||||
if (ipc.lookup(.input)) |handle| return handle;
|
||||
system.sleep(50);
|
||||
time.sleepMillis(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
@@ -66,7 +66,7 @@ pub const Subscriber = struct {
|
||||
/// The endpoint the service delivers events to (created and owned by us; its handle
|
||||
/// was handed to the service as a capability at subscribe time).
|
||||
endpoint: ipc.Handle,
|
||||
receive: [protocol.event_size]u8 = undefined,
|
||||
receive: [input_protocol.event_size]u8 = undefined,
|
||||
|
||||
/// Block until the next event is pushed, and return it. Events arrive as asynchronous
|
||||
/// buffered messages (`ipc_send` from the service), so nothing is owed in reply — the
|
||||
@@ -74,8 +74,8 @@ pub const Subscriber = struct {
|
||||
/// (there should be none), so callers can loop.
|
||||
pub fn next(self: *Subscriber) ?InputEvent {
|
||||
const got = ipc.replyWait(self.endpoint, &.{}, &self.receive, null);
|
||||
if (!got.isMessage() or got.len < protocol.event_size) return null;
|
||||
return std.mem.bytesToValue(InputEvent, self.receive[0..protocol.event_size]);
|
||||
if (!got.isMessage() or got.len < input_protocol.event_size) return null;
|
||||
return std.mem.bytesToValue(InputEvent, self.receive[0..input_protocol.event_size]);
|
||||
}
|
||||
};
|
||||
|
||||
@@ -86,11 +86,11 @@ pub fn subscribe(device_mask: u32) ?Subscriber {
|
||||
const service = lookupService() orelse return null;
|
||||
const endpoint = ipc.createIpcEndpoint() orelse return null;
|
||||
|
||||
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.subscribe), .device_mask = device_mask };
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
var request = input_protocol.Request{ .operation = @intFromEnum(input_protocol.Operation.subscribe), .device_mask = device_mask };
|
||||
var reply: [input_protocol.reply_size]u8 = undefined;
|
||||
const result = ipc.callCap(service, std.mem.asBytes(&request), &reply, endpoint) catch return null;
|
||||
if (result.len < protocol.reply_size) return null;
|
||||
if (std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status != 0) return null;
|
||||
if (result.len < input_protocol.reply_size) return null;
|
||||
if (std.mem.bytesToValue(input_protocol.Reply, reply[0..input_protocol.reply_size]).status != 0) return null;
|
||||
return .{ .endpoint = endpoint };
|
||||
}
|
||||
|
||||
@@ -149,11 +149,11 @@ pub const Publisher = struct {
|
||||
service: ipc.Handle,
|
||||
|
||||
fn publish(self: Publisher, event: InputEvent) bool {
|
||||
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.publish), .event = event };
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
var request = input_protocol.Request{ .operation = @intFromEnum(input_protocol.Operation.publish), .event = event };
|
||||
var reply: [input_protocol.reply_size]u8 = undefined;
|
||||
const len = ipc.call(self.service, std.mem.asBytes(&request), &reply) catch return false;
|
||||
if (len < protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status == 0;
|
||||
if (len < input_protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(input_protocol.Reply, reply[0..input_protocol.reply_size]).status == 0;
|
||||
}
|
||||
|
||||
/// Broadcast a keyboard event to every subscriber that took keyboard events.
|
||||
@@ -204,8 +204,8 @@ pub fn syntheticKeyEvent(step: usize) KeyEvent {
|
||||
pub fn syntheticMouseEvent(step: usize) MouseEvent {
|
||||
return switch (step % 3) {
|
||||
0 => .{ .kind = @intFromEnum(MouseEventKind.motion), .button = 0, .dx = 1, .dy = 1, .scroll_x = 0, .scroll_y = 0, .buttons = 0 },
|
||||
1 => .{ .kind = @intFromEnum(MouseEventKind.button_down), .button = protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = protocol.mouse_button_left },
|
||||
else => .{ .kind = @intFromEnum(MouseEventKind.button_up), .button = protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = 0 },
|
||||
1 => .{ .kind = @intFromEnum(MouseEventKind.button_down), .button = input_protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = input_protocol.mouse_button_left },
|
||||
else => .{ .kind = @intFromEnum(MouseEventKind.button_up), .button = input_protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = 0 },
|
||||
};
|
||||
}
|
||||
|
||||
@@ -8,9 +8,9 @@
|
||||
//! limit — the same handoff usb-storage uses toward the controller.
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const protocol = @import("block-protocol");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const block_protocol = @import("block-protocol");
|
||||
|
||||
pub const Geometry = struct { block_size: u32, block_count: u64 };
|
||||
|
||||
@@ -19,11 +19,11 @@ pub const Device = struct {
|
||||
|
||||
/// The device's block size and total block count.
|
||||
pub fn geometry(self: Device) ?Geometry {
|
||||
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.geometry), .lba = 0, .count = 0, .physical = 0 };
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
var request = block_protocol.Request{ .operation = @intFromEnum(block_protocol.Operation.geometry), .lba = 0, .count = 0, .physical = 0 };
|
||||
var reply: [block_protocol.reply_size]u8 = undefined;
|
||||
const n = ipc.call(self.endpoint, std.mem.asBytes(&request), &reply) catch return null;
|
||||
if (n < protocol.reply_size) return null;
|
||||
const result = std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]);
|
||||
if (n < block_protocol.reply_size) return null;
|
||||
const result = std.mem.bytesToValue(block_protocol.Reply, reply[0..block_protocol.reply_size]);
|
||||
if (result.status != 0) return null;
|
||||
return .{ .block_size = result.block_size, .block_count = result.block_count };
|
||||
}
|
||||
@@ -45,12 +45,12 @@ pub const Device = struct {
|
||||
return self.transfer(.flush, 0, 0, 0);
|
||||
}
|
||||
|
||||
fn transfer(self: Device, operation: protocol.Operation, lba: u64, count: u32, physical: u64) bool {
|
||||
var request = protocol.Request{ .operation = @intFromEnum(operation), .lba = lba, .count = count, .physical = physical };
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
fn transfer(self: Device, operation: block_protocol.Operation, lba: u64, count: u32, physical: u64) bool {
|
||||
var request = block_protocol.Request{ .operation = @intFromEnum(operation), .lba = lba, .count = count, .physical = physical };
|
||||
var reply: [block_protocol.reply_size]u8 = undefined;
|
||||
const n = ipc.call(self.endpoint, std.mem.asBytes(&request), &reply) catch return false;
|
||||
if (n < protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status == 0;
|
||||
if (n < block_protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(block_protocol.Reply, reply[0..block_protocol.reply_size]).status == 0;
|
||||
}
|
||||
};
|
||||
|
||||
@@ -73,7 +73,7 @@ pub fn open() ?Device {
|
||||
// not sit a further minute pretending otherwise.
|
||||
while (attempts < 600) : (attempts += 1) {
|
||||
if (ipc.lookup(.block)) |handle| return .{ .endpoint = handle };
|
||||
system.sleep(50);
|
||||
time.sleepMillis(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
@@ -1,12 +1,15 @@
|
||||
//! User-space device access: enumerate the kernel's device table, claim a device,
|
||||
//! map its MMIO, and bind its interrupt. A driver uses these to find and take
|
||||
//! ownership of its hardware; the claim is the capability the kernel checks before
|
||||
//! mapping registers or routing an IRQ.
|
||||
//! library/device/driver — the driver author's interface: enumerate the kernel's device
|
||||
//! table, claim a device, map its MMIO, bind its interrupt (the claim is the capability the
|
||||
//! kernel checks before mapping registers or routing an IRQ), and say `hello` to the device
|
||||
//! manager at startup. The whole kernel + manager surface a driver needs, in one import.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const device_abi = @import("device-abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const sc = @import("system-call");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const device_manager_protocol = @import("device-manager-protocol");
|
||||
|
||||
pub const DeviceDescriptor = device_abi.DeviceDescriptor;
|
||||
pub const ResourceDescriptor = device_abi.ResourceDescriptor;
|
||||
@@ -129,3 +132,42 @@ pub fn findDeviceDescriptorByHid(buffer: []DeviceDescriptor, hid_needle: []const
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- device-manager handshake (folded in from the former device-manager.zig) ---
|
||||
|
||||
/// What kind of driver is announcing itself (a bus that reports children, or a leaf
|
||||
/// device). Re-exported so callers name it without importing the protocol.
|
||||
pub const Role = device_manager_protocol.Role;
|
||||
|
||||
const lookup_attempts: u32 = 100;
|
||||
const lookup_pause_ms: u64 = 20;
|
||||
|
||||
/// Say hello to the device manager and return its endpoint, or null if there is no manager
|
||||
/// (best-effort standalone bring-up) or it refused the handshake. Bus drivers keep the handle
|
||||
/// to report children through; a driver that runs fine unsupervised discards it with `_ =`,
|
||||
/// and one that requires supervision bails on null. Logs the outcome itself.
|
||||
pub fn hello(role: Role, device_id: u64) ?ipc.Handle {
|
||||
var attempts: u32 = 0;
|
||||
const manager = while (attempts < lookup_attempts) : (attempts += 1) {
|
||||
if (ipc.lookup(.device_manager)) |handle| break handle;
|
||||
time.sleepMillis(lookup_pause_ms);
|
||||
} else {
|
||||
std.log.info("no device manager to hello", .{});
|
||||
return null;
|
||||
};
|
||||
|
||||
const message = device_manager_protocol.Hello{ .role = @intFromEnum(role), .device_id = device_id };
|
||||
var reply: [device_manager_protocol.reply_size]u8 = undefined;
|
||||
const length = ipc.call(manager, std.mem.asBytes(&message), &reply) catch {
|
||||
std.log.info("hello call failed", .{});
|
||||
return null;
|
||||
};
|
||||
if (length < device_manager_protocol.reply_size or
|
||||
std.mem.bytesToValue(device_manager_protocol.HelloReply, reply[0..device_manager_protocol.reply_size]).status != 0)
|
||||
{
|
||||
std.log.info("hello refused", .{});
|
||||
return null;
|
||||
}
|
||||
std.log.info("hello acknowledged", .{});
|
||||
return manager;
|
||||
}
|
||||
@@ -1,4 +1,4 @@
|
||||
//! /lib/mmio — typed volatile MMIO register access, plus the memory-ordering
|
||||
//! /lib/device/mmio — typed volatile MMIO register access, plus the memory-ordering
|
||||
//! barriers a device driver needs. Used by drivers on top of an `mmio_map` grant.
|
||||
//!
|
||||
//! **`volatile` is not a barrier.** In Zig it means only: don't elide this access, and
|
||||
@@ -11,13 +11,13 @@
|
||||
//! doorbell.* = i; // volatile store to UC MMIO
|
||||
//! // nothing orders these; the device can read a stale descriptor
|
||||
//!
|
||||
//! Put a `wmb()` between them. The barriers lower per-architecture — which is the whole
|
||||
//! reason they are a named primitive and not scattered `asm volatile`:
|
||||
//! Put a `writeMemoryBarrier()` between them. The barriers lower per-architecture — which
|
||||
//! is the whole reason they are a named primitive and not scattered `asm volatile`:
|
||||
//!
|
||||
//! x86_64 aarch64
|
||||
//! mb() mfence dsb sy
|
||||
//! rmb() lfence dsb ld
|
||||
//! wmb() sfence dsb st
|
||||
//! x86_64 aarch64
|
||||
//! memoryBarrier() mfence dsb sy
|
||||
//! readMemoryBarrier() lfence dsb ld
|
||||
//! writeMemoryBarrier() sfence dsb st
|
||||
//!
|
||||
//! x86 is forgiving (TSO + strong-uncacheable MMIO), so a compiler barrier usually
|
||||
//! suffices; ARM is not, and ARM is the win condition (docs/vision.md) — so the
|
||||
@@ -29,52 +29,52 @@ const builtin = @import("builtin");
|
||||
/// Read a register of type `T` at absolute virtual address `addr` — a location inside
|
||||
/// a device's `mmio_map` grant. `volatile`: never elided, never reordered against
|
||||
/// another volatile access.
|
||||
pub inline fn read(comptime T: type, addr: usize) T {
|
||||
pub inline fn readRegister(comptime T: type, addr: usize) T {
|
||||
return @as(*const volatile T, @ptrFromInt(addr)).*;
|
||||
}
|
||||
|
||||
/// Write `value` of type `T` to the register at absolute virtual address `addr`.
|
||||
pub inline fn write(comptime T: type, addr: usize, value: T) void {
|
||||
pub inline fn writeRegister(comptime T: type, addr: usize, value: T) void {
|
||||
@as(*volatile T, @ptrFromInt(addr)).* = value;
|
||||
}
|
||||
|
||||
/// Full barrier: all loads and stores before it are globally visible before any after
|
||||
/// it. Use when an MMIO write must complete before a following read.
|
||||
pub inline fn mb() void {
|
||||
/// Full memory barrier: all loads and stores before it are globally visible before any
|
||||
/// after it. Use when an MMIO write must complete before a following read.
|
||||
pub inline fn memoryBarrier() void {
|
||||
switch (builtin.target.cpu.arch) {
|
||||
.x86_64 => asm volatile ("mfence" ::: .{ .memory = true }),
|
||||
.aarch64 => asm volatile ("dsb sy" ::: .{ .memory = true }),
|
||||
else => @compileError("mmio.mb: unsupported architecture"),
|
||||
else => @compileError("mmio.memoryBarrier: unsupported architecture"),
|
||||
}
|
||||
}
|
||||
|
||||
/// Read barrier: loads before it complete before loads after it. Use after an IRQ
|
||||
/// Read memory barrier: loads before it complete before loads after it. Use after an IRQ
|
||||
/// wake, before reading what the device wrote to shared memory.
|
||||
pub inline fn rmb() void {
|
||||
pub inline fn readMemoryBarrier() void {
|
||||
switch (builtin.target.cpu.arch) {
|
||||
.x86_64 => asm volatile ("lfence" ::: .{ .memory = true }),
|
||||
.aarch64 => asm volatile ("dsb ld" ::: .{ .memory = true }),
|
||||
else => @compileError("mmio.rmb: unsupported architecture"),
|
||||
else => @compileError("mmio.readMemoryBarrier: unsupported architecture"),
|
||||
}
|
||||
}
|
||||
|
||||
/// Write barrier: stores before it become visible before stores after it. Use between
|
||||
/// filling a DMA descriptor in RAM and ringing the device's doorbell.
|
||||
pub inline fn wmb() void {
|
||||
/// Write memory barrier: stores before it become visible before stores after it. Use
|
||||
/// between filling a DMA descriptor in RAM and ringing the device's doorbell.
|
||||
pub inline fn writeMemoryBarrier() void {
|
||||
switch (builtin.target.cpu.arch) {
|
||||
.x86_64 => asm volatile ("sfence" ::: .{ .memory = true }),
|
||||
.aarch64 => asm volatile ("dsb st" ::: .{ .memory = true }),
|
||||
else => @compileError("mmio.wmb: unsupported architecture"),
|
||||
else => @compileError("mmio.writeMemoryBarrier: unsupported architecture"),
|
||||
}
|
||||
}
|
||||
|
||||
test "barriers emit and registers round-trip through a RAM cell" {
|
||||
// The barriers must at least assemble for the host arch; ordering can't be unit
|
||||
// tested, but a missing/mistyped mnemonic is caught here.
|
||||
wmb();
|
||||
rmb();
|
||||
mb();
|
||||
writeMemoryBarrier();
|
||||
readMemoryBarrier();
|
||||
memoryBarrier();
|
||||
var cell: u64 = 0;
|
||||
write(u64, @intFromPtr(&cell), 0xDEAD_BEEF);
|
||||
try @import("std").testing.expectEqual(@as(u64, 0xDEAD_BEEF), read(u64, @intFromPtr(&cell)));
|
||||
writeRegister(u64, @intFromPtr(&cell), 0xDEAD_BEEF);
|
||||
try @import("std").testing.expectEqual(@as(u64, 0xDEAD_BEEF), readRegister(u64, @intFromPtr(&cell)));
|
||||
}
|
||||
@@ -6,7 +6,7 @@
|
||||
//! by name, and neither reaches into the other's files.
|
||||
//!
|
||||
//! It is also the **single source of truth** for `DeviceClass` and `ResourceKind`:
|
||||
//! the kernel's rich, pointer-based device tree (system/devices/device-model.zig,
|
||||
//! the kernel's rich, pointer-based device tree (system/kernel/device-model.zig,
|
||||
//! which user space must never import) re-exports these, so the enum that a driver
|
||||
//! matches on and the enum the kernel classifies with are the *same* type — no
|
||||
//! hand-kept "mirror in order" to drift. The core kernel↔user ABI is [[abi]]; the
|
||||
@@ -41,6 +41,35 @@ pub const ClassCode = struct {
|
||||
}
|
||||
};
|
||||
|
||||
// --- Configuration-space layout ---------------------------------------------------------
|
||||
// The offsets and bit layouts of the PCI configuration header (PCI spec; see
|
||||
// https://wiki.osdev.org/PCI). Pure data — named here so both a device driver's view of
|
||||
// its own claimed function (library/device/pci/pci.zig) and the bus enumerator name the
|
||||
// same bytes instead of scattering bare 0x04/0x34/0xFFFF_FFF0 magic across the tree.
|
||||
|
||||
/// Header field offsets (byte offsets into the 256-byte configuration space).
|
||||
pub const config_vendor_id: usize = 0x00;
|
||||
pub const config_device_id: usize = 0x02;
|
||||
pub const config_command: usize = 0x04;
|
||||
pub const config_status: usize = 0x06;
|
||||
pub const config_capabilities_pointer: usize = 0x34;
|
||||
pub const config_bar0: usize = 0x10; // BAR0; BAR n is at config_bar0 + n*4
|
||||
|
||||
/// Command register: Memory-Space enable (bit 1) | Bus-Master enable (bit 2).
|
||||
pub const command_memory_and_bus_master: u16 = 0x06;
|
||||
/// Status register bit 4: a capability list is present at config_capabilities_pointer.
|
||||
pub const status_capabilities_list: u16 = 0x10;
|
||||
/// Capability pointers are dword-aligned; the low two bits are reserved.
|
||||
pub const capability_pointer_mask: u8 = 0xFC;
|
||||
|
||||
/// BAR bit layout: bit 0 selects I/O (1) vs memory (0) space; for a memory BAR, bits 2:1
|
||||
/// give the type (00 = 32-bit, 10 = 64-bit spanning the next BAR), and the base address is
|
||||
/// the dword with the low 4 flag bits masked off.
|
||||
pub const bar_io_space: u32 = 0x1;
|
||||
pub const bar_type_mask: u32 = 0x6;
|
||||
pub const bar_type_64bit: u32 = 0x4;
|
||||
pub const bar_memory_base_mask: u32 = 0xFFFF_FFF0;
|
||||
|
||||
/// Base class (config byte 0x0B). Non-exhaustive: an unlisted code is a real but
|
||||
/// unnamed class, decoded as "Unknown" rather than rejected.
|
||||
pub const BaseClass = enum(u8) {
|
||||
@@ -0,0 +1,108 @@
|
||||
//! library/device/pci/pci.zig — a device driver's view of the ONE PCI function it has
|
||||
//! claimed. Config space is mapped as resource 0; this gives header-field accessors, BAR
|
||||
//! decode + map, and a capability-list iterator, so a driver never re-derives the
|
||||
//! config-space layout by hand.
|
||||
//!
|
||||
//! This is the *device-owned* view: read my own function's live config, map my own BARs.
|
||||
//! The bus enumerator's view — probing arbitrary, not-yet-claimed functions and sizing
|
||||
//! their BARs — is a different mechanism and lives in the pci-bus driver. The pure
|
||||
//! config-space layout both need (offsets, BAR bit fields) is named once in the `pci-class`
|
||||
//! data module; this logic module adds the parts that need `mmio` + the `driver` client.
|
||||
|
||||
const std = @import("std");
|
||||
const mmio = @import("mmio");
|
||||
const pci_class = @import("pci-class");
|
||||
const device = @import("driver");
|
||||
|
||||
/// A claimed PCI function whose configuration space is mapped (resource 0). `descriptor`
|
||||
/// must outlive the Function — the driver's `device.enumerate` buffer does, for the whole
|
||||
/// bring-up. Header reads and the capability walk hit live config space; `mapBar` caches.
|
||||
pub const Function = struct {
|
||||
device_id: u64,
|
||||
descriptor: *const device.DeviceDescriptor,
|
||||
config: usize, // virtual base of mapped resource 0
|
||||
bar_virtual: [6]usize = .{ 0, 0, 0, 0, 0, 0 }, // per-BAR mmio_map cache
|
||||
|
||||
/// Map config space (resource 0) of the already-claimed `device_id`. null if the map
|
||||
/// fails (not claimed, or no config resource).
|
||||
pub fn map(device_id: u64, descriptor: *const device.DeviceDescriptor) ?Function {
|
||||
const base = device.mmioMap(device_id, 0) orelse return null;
|
||||
return .{ .device_id = device_id, .descriptor = descriptor, .config = base };
|
||||
}
|
||||
|
||||
pub fn vendorId(self: *const Function) u16 {
|
||||
return mmio.readRegister(u16, self.config + pci_class.config_vendor_id);
|
||||
}
|
||||
pub fn deviceId(self: *const Function) u16 {
|
||||
return mmio.readRegister(u16, self.config + pci_class.config_device_id);
|
||||
}
|
||||
pub fn command(self: *const Function) u16 {
|
||||
return mmio.readRegister(u16, self.config + pci_class.config_command);
|
||||
}
|
||||
pub fn status(self: *const Function) u16 {
|
||||
return mmio.readRegister(u16, self.config + pci_class.config_status);
|
||||
}
|
||||
|
||||
/// Set Memory-Space + Bus-Master enable in the command register. Firmware often leaves
|
||||
/// a secondary display's decode off; a bus-mastering device must enable both.
|
||||
pub fn enableMemoryAndBusMaster(self: *const Function) void {
|
||||
const at = self.config + pci_class.config_command;
|
||||
mmio.writeRegister(u16, at, mmio.readRegister(u16, at) | pci_class.command_memory_and_bus_master);
|
||||
}
|
||||
|
||||
/// Decode BAR `bar` (0..5) and map it: read the BAR register, reject I/O-space BARs,
|
||||
/// combine the high dword for a 64-bit BAR, mask the base, then correlate that physical
|
||||
/// base with one of the descriptor's memory resources and `mmio_map` it — a BAR names a
|
||||
/// *number*, while `mmio_map` takes a *resource index*, and gaps/config-space shift the
|
||||
/// numbering. Cached per BAR. null if the BAR is I/O-space or is not a mapped resource.
|
||||
pub fn mapBar(self: *Function, bar: u8) ?usize {
|
||||
if (bar >= 6) return null;
|
||||
if (self.bar_virtual[bar] != 0) return self.bar_virtual[bar];
|
||||
|
||||
const low = mmio.readRegister(u32, self.config + pci_class.config_bar0 + @as(usize, bar) * 4);
|
||||
if (low & pci_class.bar_io_space != 0) return null; // an I/O-space BAR
|
||||
var base: u64 = low & pci_class.bar_memory_base_mask;
|
||||
if ((low & pci_class.bar_type_mask) == pci_class.bar_type_64bit) { // 64-bit: high half is the next dword
|
||||
const high = mmio.readRegister(u32, self.config + pci_class.config_bar0 + (@as(usize, bar) + 1) * 4);
|
||||
base |= @as(u64, high) << 32;
|
||||
}
|
||||
|
||||
for (self.descriptor.resources[0..@intCast(self.descriptor.resource_count)], 0..) |resource, index| {
|
||||
if (resource.kind == @intFromEnum(device.ResourceKind.memory) and resource.start == base) {
|
||||
const v = device.mmioMap(self.device_id, index) orelse return null;
|
||||
self.bar_virtual[bar] = v;
|
||||
return v;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Iterate the capability list. Empty when the function advertises none.
|
||||
pub fn capabilities(self: *const Function) CapabilityIterator {
|
||||
const present = self.status() & pci_class.status_capabilities_list != 0;
|
||||
const first = if (present)
|
||||
mmio.readRegister(u8, self.config + pci_class.config_capabilities_pointer) & pci_class.capability_pointer_mask
|
||||
else
|
||||
0;
|
||||
return .{ .config = self.config, .cursor = first };
|
||||
}
|
||||
};
|
||||
|
||||
/// One capability header. `offset` is the ABSOLUTE virtual address of the header, so the
|
||||
/// caller reads its body with `mmio.readRegister(T, cap.offset + n)`.
|
||||
pub const Capability = struct { id: u8, offset: usize };
|
||||
|
||||
pub const CapabilityIterator = struct {
|
||||
config: usize,
|
||||
cursor: u8,
|
||||
guard: u32 = 0, // bounds a malformed/looping chain (48 = the 256-byte space in dwords)
|
||||
|
||||
pub fn next(self: *CapabilityIterator) ?Capability {
|
||||
if (self.cursor == 0 or self.guard >= 48) return null;
|
||||
self.guard += 1;
|
||||
const at = self.config + self.cursor;
|
||||
const id = mmio.readRegister(u8, at + 0);
|
||||
self.cursor = mmio.readRegister(u8, at + 1) & pci_class.capability_pointer_mask;
|
||||
return .{ .id = id, .offset = at };
|
||||
}
|
||||
};
|
||||
@@ -5,7 +5,7 @@
|
||||
//! service and `device.zig` over the raw device calls.
|
||||
//!
|
||||
//! A class driver, spawned with its interface's assigned device id as argv[1]:
|
||||
//! if (!usb.helloManager(id)) return; // meet the spawn deadline
|
||||
//! if (device_manager.hello(.device, id) == null) return; // meet the spawn deadline
|
||||
//! var device = usb.open(id) orelse return; // open + get its endpoints
|
||||
//! _ = device.controlOut(usb_abi.setProtocol(...));// class requests, descriptors
|
||||
//! _ = device.subscribeInterrupt(address, length); // reports arrive asynchronously
|
||||
@@ -16,14 +16,18 @@
|
||||
//! the service harness drops buffered-message payloads — see service.zig).
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const protocol = @import("usb-transfer-protocol");
|
||||
const device_manager = @import("device-manager-protocol");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const usb_transfer_protocol = @import("usb-transfer-protocol");
|
||||
|
||||
pub const Endpoint = protocol.Endpoint;
|
||||
pub const InterruptReport = protocol.InterruptReport;
|
||||
pub const max_report_data = protocol.max_report_data;
|
||||
/// The USB chapter-9 wire ABI and the class taxonomy, re-exported so a class driver reaches
|
||||
/// the whole USB domain through its one `usb` import (`usb.abi.getDescriptor`, `usb.ids.Class`).
|
||||
pub const abi = @import("usb-abi");
|
||||
pub const ids = @import("usb-ids");
|
||||
|
||||
pub const Endpoint = usb_transfer_protocol.Endpoint;
|
||||
pub const InterruptReport = usb_transfer_protocol.InterruptReport;
|
||||
pub const max_report_data = usb_transfer_protocol.max_report_data;
|
||||
|
||||
// Endpoint transfer types (EndpointDescriptor attributes), for `findEndpoint`.
|
||||
pub const transfer_type_bulk: u8 = 2;
|
||||
@@ -41,7 +45,7 @@ pub const Device = struct {
|
||||
protocol_code: u8,
|
||||
interface_number: u8,
|
||||
endpoint_count: usize = 0,
|
||||
endpoints: [protocol.max_reported_endpoints]Endpoint = undefined,
|
||||
endpoints: [usb_transfer_protocol.max_reported_endpoints]Endpoint = undefined,
|
||||
|
||||
/// The interface's first endpoint of the given transfer type and direction
|
||||
/// (`transfer_type_bulk` / `transfer_type_interrupt`), or null.
|
||||
@@ -53,17 +57,17 @@ pub const Device = struct {
|
||||
}
|
||||
|
||||
fn controlTransfer(self: *Device, setup: [8]u8, direction_in: bool, data: []u8) ?usize {
|
||||
var request = protocol.ControlRequest{
|
||||
var request = usb_transfer_protocol.ControlRequest{
|
||||
.device_token = self.token,
|
||||
.setup = setup,
|
||||
.direction_in = @intFromBool(direction_in),
|
||||
.data_length = @intCast(data.len),
|
||||
};
|
||||
if (!direction_in and data.len > 0) @memcpy(request.data[0..data.len], data);
|
||||
var reply: [@sizeOf(protocol.ControlReply)]u8 = undefined;
|
||||
var reply: [@sizeOf(usb_transfer_protocol.ControlReply)]u8 = undefined;
|
||||
const length = ipc.call(self.bus, std.mem.asBytes(&request), &reply) catch return null;
|
||||
if (length < @sizeOf(protocol.ControlReply)) return null;
|
||||
const control_reply = std.mem.bytesToValue(protocol.ControlReply, reply[0..@sizeOf(protocol.ControlReply)]);
|
||||
if (length < @sizeOf(usb_transfer_protocol.ControlReply)) return null;
|
||||
const control_reply = std.mem.bytesToValue(usb_transfer_protocol.ControlReply, reply[0..@sizeOf(usb_transfer_protocol.ControlReply)]);
|
||||
if (control_reply.status != 0) return null;
|
||||
const actual = @min(control_reply.actual_length, data.len);
|
||||
if (direction_in and actual > 0) @memcpy(data[0..actual], control_reply.data[0..actual]);
|
||||
@@ -84,30 +88,30 @@ pub const Device = struct {
|
||||
/// Begin periodic IN polling of an interrupt endpoint; reports flow back to
|
||||
/// `self.endpoint` as asynchronous `InterruptReport` messages.
|
||||
pub fn subscribeInterrupt(self: *Device, endpoint_address: u8, max_length: u16) bool {
|
||||
var request = protocol.InterruptSubscribeRequest{
|
||||
var request = usb_transfer_protocol.InterruptSubscribeRequest{
|
||||
.device_token = self.token,
|
||||
.endpoint_address = endpoint_address,
|
||||
.max_length = max_length,
|
||||
};
|
||||
var reply: [@sizeOf(protocol.InterruptSubscribeReply)]u8 = undefined;
|
||||
var reply: [@sizeOf(usb_transfer_protocol.InterruptSubscribeReply)]u8 = undefined;
|
||||
const length = ipc.call(self.bus, std.mem.asBytes(&request), &reply) catch return false;
|
||||
if (length < @sizeOf(protocol.InterruptSubscribeReply)) return false;
|
||||
return std.mem.bytesToValue(protocol.InterruptSubscribeReply, reply[0..@sizeOf(protocol.InterruptSubscribeReply)]).status == 0;
|
||||
if (length < @sizeOf(usb_transfer_protocol.InterruptSubscribeReply)) return false;
|
||||
return std.mem.bytesToValue(usb_transfer_protocol.InterruptSubscribeReply, reply[0..@sizeOf(usb_transfer_protocol.InterruptSubscribeReply)]).status == 0;
|
||||
}
|
||||
|
||||
/// One bulk transfer (IN or OUT per `endpoint_address`'s direction bit) to or
|
||||
/// from the caller's own DMA buffer at `physical`. Returns the bytes moved.
|
||||
pub fn bulk(self: *Device, endpoint_address: u8, physical: u64, length: u32) ?u32 {
|
||||
var request = protocol.BulkRequest{
|
||||
var request = usb_transfer_protocol.BulkRequest{
|
||||
.device_token = self.token,
|
||||
.physical_address = physical,
|
||||
.length = length,
|
||||
.endpoint_address = endpoint_address,
|
||||
};
|
||||
var reply: [@sizeOf(protocol.BulkReply)]u8 = undefined;
|
||||
var reply: [@sizeOf(usb_transfer_protocol.BulkReply)]u8 = undefined;
|
||||
const replied = ipc.call(self.bus, std.mem.asBytes(&request), &reply) catch return null;
|
||||
if (replied < @sizeOf(protocol.BulkReply)) return null;
|
||||
const bulk_reply = std.mem.bytesToValue(protocol.BulkReply, reply[0..@sizeOf(protocol.BulkReply)]);
|
||||
if (replied < @sizeOf(usb_transfer_protocol.BulkReply)) return null;
|
||||
const bulk_reply = std.mem.bytesToValue(usb_transfer_protocol.BulkReply, reply[0..@sizeOf(usb_transfer_protocol.BulkReply)]);
|
||||
if (bulk_reply.status != 0) return null;
|
||||
return bulk_reply.actual_length;
|
||||
}
|
||||
@@ -120,15 +124,15 @@ pub fn open(device_id: u64) ?Device {
|
||||
var attempts: usize = 0;
|
||||
const bus = while (attempts < 100) : (attempts += 1) {
|
||||
if (ipc.lookup(.usb_bus)) |handle| break handle;
|
||||
system.sleep(20);
|
||||
time.sleepMillis(20);
|
||||
} else return null;
|
||||
|
||||
const endpoint = ipc.createIpcEndpoint() orelse return null;
|
||||
var request = protocol.OpenRequest{ .device_id = device_id };
|
||||
var reply: [@sizeOf(protocol.OpenReply)]u8 = undefined;
|
||||
var request = usb_transfer_protocol.OpenRequest{ .device_id = device_id };
|
||||
var reply: [@sizeOf(usb_transfer_protocol.OpenReply)]u8 = undefined;
|
||||
const result = ipc.callCap(bus, std.mem.asBytes(&request), &reply, endpoint) catch return null;
|
||||
if (result.len < @sizeOf(protocol.OpenReply)) return null;
|
||||
const open_reply = std.mem.bytesToValue(protocol.OpenReply, reply[0..@sizeOf(protocol.OpenReply)]);
|
||||
if (result.len < @sizeOf(usb_transfer_protocol.OpenReply)) return null;
|
||||
const open_reply = std.mem.bytesToValue(usb_transfer_protocol.OpenReply, reply[0..@sizeOf(usb_transfer_protocol.OpenReply)]);
|
||||
if (open_reply.status != 0) return null;
|
||||
|
||||
var device = Device{
|
||||
@@ -139,24 +143,8 @@ pub fn open(device_id: u64) ?Device {
|
||||
.subclass = open_reply.interface_subclass,
|
||||
.protocol_code = open_reply.interface_protocol,
|
||||
.interface_number = open_reply.interface_number,
|
||||
.endpoint_count = @min(open_reply.endpoint_count, protocol.max_reported_endpoints),
|
||||
.endpoint_count = @min(open_reply.endpoint_count, usb_transfer_protocol.max_reported_endpoints),
|
||||
};
|
||||
for (0..device.endpoint_count) |index| device.endpoints[index] = open_reply.endpoints[index];
|
||||
return device;
|
||||
}
|
||||
|
||||
/// Hello the device manager as a class driver (Role.device) so a supervised
|
||||
/// spawn meets its hello deadline. Retries while the manager comes up.
|
||||
pub fn helloManager(device_id: u64) bool {
|
||||
var attempts: usize = 0;
|
||||
const manager = while (attempts < 100) : (attempts += 1) {
|
||||
if (ipc.lookup(.device_manager)) |handle| break handle;
|
||||
system.sleep(20);
|
||||
} else return false;
|
||||
|
||||
const hello = device_manager.Hello{ .role = @intFromEnum(device_manager.Role.device), .device_id = device_id };
|
||||
var reply: [device_manager.message_maximum]u8 = undefined;
|
||||
const length = ipc.call(manager, std.mem.asBytes(&hello), &reply) catch return false;
|
||||
if (length < device_manager.reply_size) return false;
|
||||
return std.mem.bytesToValue(device_manager.HelloReply, reply[0..device_manager.reply_size]).status == 0;
|
||||
}
|
||||
@@ -11,13 +11,14 @@
|
||||
//! shape, unlike the POSIX fd model the old shim emulated.
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const protocol = @import("vfs-protocol");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call");
|
||||
const ipc = @import("ipc");
|
||||
const vfs_protocol = @import("vfs-protocol");
|
||||
|
||||
/// The kind of a filesystem node — re-exported so a caller need not import the
|
||||
/// wire protocol.
|
||||
pub const Kind = protocol.NodeKind;
|
||||
pub const Kind = vfs_protocol.NodeKind;
|
||||
|
||||
/// A node's metadata (the answer to a status request).
|
||||
pub const Attributes = struct {
|
||||
@@ -54,9 +55,9 @@ pub const OpenOptions = struct {
|
||||
|
||||
fn wireFlags(self: OpenOptions) u32 {
|
||||
var f: u32 = 0;
|
||||
if (self.create) f |= protocol.create;
|
||||
if (self.directory) f |= protocol.directory;
|
||||
if (self.truncate) f |= protocol.truncate;
|
||||
if (self.create) f |= vfs_protocol.create;
|
||||
if (self.directory) f |= vfs_protocol.directory;
|
||||
if (self.truncate) f |= vfs_protocol.truncate;
|
||||
return f;
|
||||
}
|
||||
};
|
||||
@@ -76,7 +77,7 @@ const Route = union(enum) {
|
||||
|
||||
fn resolve(path: []const u8, flags: usize) ?Route {
|
||||
var out: [224]u8 = undefined;
|
||||
const route = system.fsResolve(path, flags, &out) orelse return null;
|
||||
const route = fsResolve(path, flags, &out) orelse return null;
|
||||
switch (route) {
|
||||
.kernel => |token| return .{ .kernel = token },
|
||||
.backend => |b| {
|
||||
@@ -87,22 +88,22 @@ fn resolve(path: []const u8, flags: usize) ?Route {
|
||||
}
|
||||
}
|
||||
|
||||
const Result = struct { reply: protocol.Reply, payload: []u8 };
|
||||
const Result = struct { reply: vfs_protocol.Reply, payload: []u8 };
|
||||
|
||||
// One request/reply round trip: [Request header][send payload] -> backend ->
|
||||
// [Reply header][receive payload]. The receive payload lands in `out`.
|
||||
fn transact(h: ipc.Handle, request: protocol.Request, send: []const u8, out: []u8) ?Result {
|
||||
var message: [protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(message[0..protocol.request_size], std.mem.asBytes(&request));
|
||||
const slen = @min(send.len, protocol.maximum_payload);
|
||||
@memcpy(message[protocol.request_size..][0..slen], send[0..slen]);
|
||||
fn transact(h: ipc.Handle, request: vfs_protocol.Request, send: []const u8, out: []u8) ?Result {
|
||||
var message: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(message[0..vfs_protocol.request_size], std.mem.asBytes(&request));
|
||||
const slen = @min(send.len, vfs_protocol.maximum_payload);
|
||||
@memcpy(message[vfs_protocol.request_size..][0..slen], send[0..slen]);
|
||||
|
||||
var rbuf: [protocol.message_maximum]u8 = undefined;
|
||||
const n = ipc.call(h, message[0 .. protocol.request_size + slen], &rbuf) catch return null;
|
||||
if (n < protocol.reply_size) return null;
|
||||
const reply = std.mem.bytesToValue(protocol.Reply, rbuf[0..protocol.reply_size]);
|
||||
const rpl = @min(n - protocol.reply_size, out.len);
|
||||
@memcpy(out[0..rpl], rbuf[protocol.reply_size..][0..rpl]);
|
||||
var rbuf: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
const n = ipc.call(h, message[0 .. vfs_protocol.request_size + slen], &rbuf) catch return null;
|
||||
if (n < vfs_protocol.reply_size) return null;
|
||||
const reply = std.mem.bytesToValue(vfs_protocol.Reply, rbuf[0..vfs_protocol.reply_size]);
|
||||
const rpl = @min(n - vfs_protocol.reply_size, out.len);
|
||||
@memcpy(out[0..rpl], rbuf[vfs_protocol.reply_size..][0..rpl]);
|
||||
return .{ .reply = reply, .payload = out[0..rpl] };
|
||||
}
|
||||
|
||||
@@ -118,12 +119,12 @@ pub const File = struct {
|
||||
/// null on error.
|
||||
pub fn read(self: *File, buffer: []u8) ?usize {
|
||||
const h = self.backend orelse {
|
||||
const n = system.fsNodeRead(self.node, self.offset, buffer) orelse return null;
|
||||
const n = fsNodeRead(self.node, self.offset, buffer) orelse return null;
|
||||
self.offset += n;
|
||||
return n;
|
||||
};
|
||||
const want: u32 = @intCast(@min(buffer.len, protocol.maximum_payload));
|
||||
const request = protocol.Request{ .operation = .read, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
|
||||
const want: u32 = @intCast(@min(buffer.len, vfs_protocol.maximum_payload));
|
||||
const request = vfs_protocol.Request{ .operation = .read, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
|
||||
const r = transact(h, request, &.{}, buffer) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
self.offset += r.reply.len;
|
||||
@@ -136,8 +137,8 @@ pub const File = struct {
|
||||
/// are read-only).
|
||||
pub fn write(self: *File, data: []const u8) ?usize {
|
||||
const h = self.backend orelse return null;
|
||||
const want: u32 = @intCast(@min(data.len, protocol.maximum_payload));
|
||||
const request = protocol.Request{ .operation = .write, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
|
||||
const want: u32 = @intCast(@min(data.len, vfs_protocol.maximum_payload));
|
||||
const request = vfs_protocol.Request{ .operation = .write, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
|
||||
const r = transact(h, request, data[0..want], &.{}) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
self.offset += r.reply.len;
|
||||
@@ -164,14 +165,14 @@ pub const File = struct {
|
||||
/// This file's metadata.
|
||||
pub fn attributes(self: *File) ?Attributes {
|
||||
const h = self.backend orelse {
|
||||
const a = system.fsNodeStatus(self.node) orelse return null;
|
||||
return .{ .size = a.size, .kind = if (a.kind == system.file_kind_directory) .directory else .regular, .mtime = a.mtime };
|
||||
const a = fsNodeStatus(self.node) orelse return null;
|
||||
return .{ .size = a.size, .kind = if (a.kind == file_kind_directory) .directory else .regular, .mtime = a.mtime };
|
||||
};
|
||||
const request = protocol.Request{ .operation = .status, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
var buffer: [@sizeOf(protocol.FileStatus)]u8 = undefined;
|
||||
const request = vfs_protocol.Request{ .operation = .status, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
var buffer: [@sizeOf(vfs_protocol.FileStatus)]u8 = undefined;
|
||||
const r = transact(h, request, &.{}, &buffer) orelse return null;
|
||||
if (r.reply.status != 0 or r.payload.len < @sizeOf(protocol.FileStatus)) return null;
|
||||
const status = std.mem.bytesToValue(protocol.FileStatus, buffer[0..@sizeOf(protocol.FileStatus)]);
|
||||
if (r.reply.status != 0 or r.payload.len < @sizeOf(vfs_protocol.FileStatus)) return null;
|
||||
const status = std.mem.bytesToValue(vfs_protocol.FileStatus, buffer[0..@sizeOf(vfs_protocol.FileStatus)]);
|
||||
return .{ .size = status.size, .kind = kindFromWire(status.kind), .mtime = status.mtime };
|
||||
}
|
||||
|
||||
@@ -179,7 +180,7 @@ pub const File = struct {
|
||||
/// tokens are permanent — nothing to release.
|
||||
pub fn close(self: *File) void {
|
||||
const h = self.backend orelse return;
|
||||
const request = protocol.Request{ .operation = .close, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
const request = vfs_protocol.Request{ .operation = .close, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
_ = transact(h, request, &.{}, &.{});
|
||||
}
|
||||
};
|
||||
@@ -191,7 +192,7 @@ pub fn open(path: []const u8, options: OpenOptions) ?File {
|
||||
.kernel => |token| return .{ .node = token, .backend = null },
|
||||
.backend => |b| {
|
||||
const relative = route.backendPath();
|
||||
const request = protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = options.wireFlags() };
|
||||
const request = vfs_protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = options.wireFlags() };
|
||||
const r = transact(b.handle, request, relative, &.{}) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
return .{ .node = r.reply.node, .backend = b.handle };
|
||||
@@ -234,27 +235,27 @@ pub const Directory = struct {
|
||||
/// on error.
|
||||
pub fn next(self: *Directory, entry: *Entry) bool {
|
||||
const h = self.backend orelse {
|
||||
var buffer: [@sizeOf(system.DirectoryEntryHeader) + 64]u8 = undefined;
|
||||
const n = system.fsNodeReaddir(self.node, self.cursor, &buffer) orelse return false;
|
||||
if (n < @sizeOf(system.DirectoryEntryHeader)) return false; // end
|
||||
const header = std.mem.bytesToValue(system.DirectoryEntryHeader, buffer[0..@sizeOf(system.DirectoryEntryHeader)]);
|
||||
entry.kind = if (header.kind == system.file_kind_directory) .directory else .regular;
|
||||
var buffer: [@sizeOf(DirectoryEntryHeader) + 64]u8 = undefined;
|
||||
const n = fsNodeReaddir(self.node, self.cursor, &buffer) orelse return false;
|
||||
if (n < @sizeOf(DirectoryEntryHeader)) return false; // end
|
||||
const header = std.mem.bytesToValue(DirectoryEntryHeader, buffer[0..@sizeOf(DirectoryEntryHeader)]);
|
||||
entry.kind = if (header.kind == file_kind_directory) .directory else .regular;
|
||||
entry.size = header.size;
|
||||
const nlen = @min(@as(usize, header.name_len), entry.name_buffer.len);
|
||||
@memcpy(entry.name_buffer[0..nlen], buffer[@sizeOf(system.DirectoryEntryHeader)..][0..nlen]);
|
||||
@memcpy(entry.name_buffer[0..nlen], buffer[@sizeOf(DirectoryEntryHeader)..][0..nlen]);
|
||||
entry.name_len = nlen;
|
||||
self.cursor += 1;
|
||||
return true;
|
||||
};
|
||||
const request = protocol.Request{ .operation = .readdir, .node = self.node, .offset = self.cursor, .len = 0, .flags = 0 };
|
||||
var buffer: [protocol.message_maximum]u8 = undefined;
|
||||
const request = vfs_protocol.Request{ .operation = .readdir, .node = self.node, .offset = self.cursor, .len = 0, .flags = 0 };
|
||||
var buffer: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
const r = transact(h, request, &.{}, &buffer) orelse return false;
|
||||
if (r.reply.status != 0 or r.reply.len == 0) return false; // error or EOF
|
||||
if (r.payload.len < protocol.directory_entry_size) return false;
|
||||
const header = std.mem.bytesToValue(protocol.DirectoryEntry, r.payload[0..protocol.directory_entry_size]);
|
||||
if (r.payload.len < vfs_protocol.directory_entry_size) return false;
|
||||
const header = std.mem.bytesToValue(vfs_protocol.DirectoryEntry, r.payload[0..vfs_protocol.directory_entry_size]);
|
||||
entry.kind = kindFromWire(header.kind);
|
||||
entry.size = header.size;
|
||||
const source = r.payload[protocol.directory_entry_size..];
|
||||
const source = r.payload[vfs_protocol.directory_entry_size..];
|
||||
const nlen = @min(@min(@as(usize, header.name_len), source.len), entry.name_buffer.len);
|
||||
@memcpy(entry.name_buffer[0..nlen], source[0..nlen]);
|
||||
entry.name_len = nlen;
|
||||
@@ -278,11 +279,11 @@ pub fn openDirectory(path: []const u8) ?Directory {
|
||||
// A path-based request that returns only a status (mkdir, unlink). Kernel-served
|
||||
// paths (the read-only /system) refuse mutation by construction: the resolve
|
||||
// must land on a backend.
|
||||
fn pathOperation(operation: protocol.Operation, path: []const u8) bool {
|
||||
fn pathOperation(operation: vfs_protocol.Operation, path: []const u8) bool {
|
||||
const route = resolve(path, 0) orelse return false;
|
||||
if (route != .backend) return false;
|
||||
const relative = route.backendPath();
|
||||
const request = protocol.Request{ .operation = operation, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = 0 };
|
||||
const request = vfs_protocol.Request{ .operation = operation, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = 0 };
|
||||
const r = transact(route.backend.handle, request, relative, &.{}) orelse return false;
|
||||
return r.reply.status == 0;
|
||||
}
|
||||
@@ -329,12 +330,12 @@ pub fn rename(old_path: []const u8, new_path: []const u8) bool {
|
||||
const old_relative = old_route.backendPath();
|
||||
const new_relative = new_route.backendPath();
|
||||
const total = old_relative.len + 1 + new_relative.len;
|
||||
if (total > protocol.maximum_payload) return false;
|
||||
var payload: [protocol.maximum_payload]u8 = undefined;
|
||||
if (total > vfs_protocol.maximum_payload) return false;
|
||||
var payload: [vfs_protocol.maximum_payload]u8 = undefined;
|
||||
@memcpy(payload[0..old_relative.len], old_relative);
|
||||
payload[old_relative.len] = 0;
|
||||
@memcpy(payload[old_relative.len + 1 ..][0..new_relative.len], new_relative);
|
||||
const request = protocol.Request{ .operation = .rename, .node = 0, .offset = 0, .len = @intCast(total), .flags = 0 };
|
||||
const request = vfs_protocol.Request{ .operation = .rename, .node = 0, .offset = 0, .len = @intCast(total), .flags = 0 };
|
||||
const r = transact(old_route.backend.handle, request, payload[0..total], &.{}) orelse return false;
|
||||
return r.reply.status == 0;
|
||||
}
|
||||
@@ -343,12 +344,84 @@ pub fn rename(old_path: []const u8, new_path: []const u8) bool {
|
||||
/// the kernel VFS then routes everything under `target` to that backend.
|
||||
/// Possession of the endpoint handle is the capability. Returns true on success.
|
||||
pub fn mount(target: []const u8, backend: ipc.Handle) bool {
|
||||
return system.fsMount(target, backend, "");
|
||||
return fsMount(target, backend, "");
|
||||
}
|
||||
|
||||
/// As `mount`, with a backend-side rewrite prefix: a path under `target` reaches
|
||||
/// the backend as `rewrite` + the mount-relative tail. How one volume serves two
|
||||
/// mounts ("/mnt/usb" from its root, "/var" from its /var subtree).
|
||||
pub fn mountRewritten(target: []const u8, backend: ipc.Handle, rewrite: []const u8) bool {
|
||||
return system.fsMount(target, backend, rewrite);
|
||||
return fsMount(target, backend, rewrite);
|
||||
}
|
||||
|
||||
// --- raw filesystem syscalls, formerly in the system.zig dumping ground ---
|
||||
|
||||
pub const FileAttributes = abi.FileAttributes;
|
||||
pub const DirectoryEntryHeader = abi.DirectoryEntryHeader;
|
||||
pub const file_kind_regular = abi.file_kind_regular;
|
||||
pub const file_kind_directory = abi.file_kind_directory;
|
||||
|
||||
/// Where fs_resolve routed a path: served by the kernel (a permanent node token for
|
||||
/// `fs_node`) or by a user-space filesystem backend (an endpoint handle plus the rewritten
|
||||
/// mount-relative path returned in the caller's buffer).
|
||||
pub const FsRoute = union(enum) {
|
||||
kernel: u64,
|
||||
backend: struct { handle: usize, path_len: usize },
|
||||
};
|
||||
|
||||
/// Route `path` through the kernel VFS. For a backend route the rewritten mount-relative
|
||||
/// path lands in `out` (behind a kernel-written length prefix, already stripped here:
|
||||
/// out[0..path_len] is the path).
|
||||
pub fn fsResolve(path: []const u8, flags: usize, out: []u8) ?FsRoute {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = flags; // in: flags (arg #3); out: node token / backend handle
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "+{rdx}" (rdx),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.fs_resolve)),
|
||||
[a0] "{rdi}" (@intFromPtr(path.ptr)),
|
||||
[a1] "{rsi}" (path.len),
|
||||
[a3] "{r10}" (@intFromPtr(out.ptr)),
|
||||
[a4] "{r8}" (out.len),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (@as(isize, @bitCast(rax)) < 0) return null;
|
||||
if (rax == abi.fs_route_kernel) return .{ .kernel = rdx };
|
||||
if (rax != abi.fs_route_backend) return null;
|
||||
const path_len = @as(usize, out[0]) | (@as(usize, out[1]) << 8);
|
||||
if (path_len + 2 > out.len) return null;
|
||||
std.mem.copyForwards(u8, out[0..path_len], out[2..][0..path_len]);
|
||||
return .{ .backend = .{ .handle = rdx, .path_len = path_len } };
|
||||
}
|
||||
|
||||
/// Read `out.len` bytes of a kernel-served node at `offset` (fs_node read).
|
||||
pub fn fsNodeRead(node_token: u64, offset: u64, out: []u8) ?usize {
|
||||
const r = sc.systemCall5(.fs_node, abi.fs_node_read, node_token, offset, @intFromPtr(out.ptr), out.len);
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return r;
|
||||
}
|
||||
|
||||
/// A kernel-served node's metadata (fs_node status).
|
||||
pub fn fsNodeStatus(node_token: u64) ?abi.FileAttributes {
|
||||
var attrs: abi.FileAttributes = undefined;
|
||||
const r = sc.systemCall5(.fs_node, abi.fs_node_status, node_token, 0, @intFromPtr(&attrs), @sizeOf(abi.FileAttributes));
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return attrs;
|
||||
}
|
||||
|
||||
/// The `cursor`th child of a kernel-served directory (fs_node readdir): fills `out` with
|
||||
/// [DirectoryEntryHeader][name]; returns total bytes (0 = end).
|
||||
pub fn fsNodeReaddir(node_token: u64, cursor: u64, out: []u8) ?usize {
|
||||
const r = sc.systemCall5(.fs_node, abi.fs_node_readdir, node_token, cursor, @intFromPtr(out.ptr), out.len);
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return r;
|
||||
}
|
||||
|
||||
/// Mount a userspace filesystem's endpoint at `prefix`, with an optional backend-side
|
||||
/// `rewrite` prefix ("" = none). Possession of the endpoint handle is the capability.
|
||||
pub fn fsMount(prefix: []const u8, backend: usize, rewrite: []const u8) bool {
|
||||
return sc.systemCall5(.fs_mount, @intFromPtr(prefix.ptr), prefix.len, backend, @intFromPtr(rewrite.ptr), rewrite.len) == 0;
|
||||
}
|
||||
|
||||
pub fn fsUnmount(prefix: []const u8) bool {
|
||||
return sc.systemCall2(.fs_unmount, @intFromPtr(prefix.ptr), prefix.len) == 0;
|
||||
}
|
||||
@@ -4,7 +4,7 @@
|
||||
//! added with the first server binary.
|
||||
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const sc = @import("system-call");
|
||||
|
||||
/// A small-int handle into the calling process's handle table.
|
||||
pub const Handle = usize;
|
||||
@@ -0,0 +1,97 @@
|
||||
//! The per-process logger: std.log wired to the tagged kernel log ring.
|
||||
//!
|
||||
//! A program just calls `std.log.info("mounted {s}", .{path})` (or a scoped
|
||||
//! logger); this backend formats the line into a fixed buffer and emits ONE
|
||||
//! `debug_write` record carrying the level. The kernel stamps the record with
|
||||
//! the sender's pid and task name (its binary path) — the process does NOT put
|
||||
//! its own name in the payload; attribution is the kernel's, structural and
|
||||
//! unforgeable. Serial shows the kernel-rendered `<path>: message` line, and
|
||||
//! the logger service demultiplexes the ring into one file per process.
|
||||
//!
|
||||
//! Installed for every user binary by the root shim (library/runtime/root.zig)
|
||||
//! via `std_options`; a program can override by declaring its own
|
||||
//! `pub const std_options`.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call");
|
||||
|
||||
// --- the tagged log ring: raw wrappers + record types, formerly in the system.zig dump ---
|
||||
|
||||
/// A log record's level and the ring's framing types (re-exported from the shared ABI so
|
||||
/// callers and the logger service don't import `abi` themselves).
|
||||
pub const KlogLevel = abi.KlogLevel;
|
||||
pub const KlogStatus = abi.KlogStatus;
|
||||
pub const KlogRecordHeader = abi.KlogRecordHeader;
|
||||
pub const klog_record_header_size = abi.klog_record_header_size;
|
||||
pub const klog_record_alignment = abi.klog_record_alignment;
|
||||
pub const klog_record_magic = abi.klog_record_magic;
|
||||
pub const klog_flag_truncated = abi.klog_flag_truncated;
|
||||
pub const klog_maximum_message = abi.klog_maximum_message;
|
||||
pub const maximum_process_name = abi.maximum_process_name;
|
||||
|
||||
/// Write raw bytes to the kernel log (bring-up/panic diagnostics; ordinary output goes
|
||||
/// through std.log -> writeRecord). The kernel stamps the record with this process's id
|
||||
/// and name. Returns the byte count, or a wrapped -1.
|
||||
pub fn write(message: []const u8) usize {
|
||||
return writeRecord(.raw, message);
|
||||
}
|
||||
|
||||
/// Emit one leveled record into the tagged kernel log ring. The kernel stamps
|
||||
/// pid/name/sequence/timestamp; the payload should be a single line.
|
||||
pub fn writeRecord(level: KlogLevel, message: []const u8) usize {
|
||||
return sc.systemCall3(.debug_write, @intFromPtr(message.ptr), message.len, @intFromEnum(level));
|
||||
}
|
||||
|
||||
/// Copy framed records out of the tagged kernel log ring starting at stream `offset` into
|
||||
/// `out`. Returns the byte count (0 = caught up), or null when `offset` fell behind the
|
||||
/// ring's tail or lies past its head (re-sync via `klogStatus`).
|
||||
pub fn klogRead(offset: u64, out: []u8) ?usize {
|
||||
const r = sc.systemCall3(.klog_read, offset, @intFromPtr(out.ptr), out.len);
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return r;
|
||||
}
|
||||
|
||||
/// The log ring's live cursors (oldest retained offset, end of stream, next sequence)
|
||||
/// plus the wall-clock time of boot — how a log reader starts, detects loss, and names a
|
||||
/// per-boot log directory.
|
||||
pub fn klogStatus() ?KlogStatus {
|
||||
var status: KlogStatus = undefined;
|
||||
if (@as(isize, @bitCast(sc.systemCall1(.klog_status, @intFromPtr(&status)))) != 0) return null;
|
||||
return status;
|
||||
}
|
||||
|
||||
fn levelOf(comptime level: std.log.Level) KlogLevel {
|
||||
return switch (level) {
|
||||
.err => .err,
|
||||
.warn => .warn,
|
||||
.info => .info,
|
||||
.debug => .debug,
|
||||
};
|
||||
}
|
||||
|
||||
pub fn logFn(
|
||||
comptime level: std.log.Level,
|
||||
comptime scope: @EnumLiteral(),
|
||||
comptime format: []const u8,
|
||||
args: anytype,
|
||||
) void {
|
||||
// One record = one line = at most klog_maximum_message bytes of payload.
|
||||
// On overflow keep what fits and end with "~" so the record is still a
|
||||
// whole line (the kernel would split an embedded rest anyway).
|
||||
var buffer: [256]u8 = undefined;
|
||||
const prefix = if (scope == .default) "" else "(" ++ @tagName(scope) ++ ") ";
|
||||
const line = std.fmt.bufPrint(&buffer, prefix ++ format, args) catch truncated: {
|
||||
buffer[buffer.len - 1] = '~';
|
||||
break :truncated buffer[0..];
|
||||
};
|
||||
_ = writeRecord(levelOf(level), line);
|
||||
}
|
||||
|
||||
/// The std.Options the root shim installs unless the program overrides it.
|
||||
/// Debug level: filtering is the log *reader's* job here — the ring is cheap,
|
||||
/// serial is a dev convenience, and the logger service keeps everything.
|
||||
pub const default_options: std.Options = .{
|
||||
.log_level = .debug,
|
||||
.logFn = logFn,
|
||||
};
|
||||
@@ -5,7 +5,7 @@
|
||||
//! `/lib/mmio` (fill the ring, `wmb()`, ring the doorbell). See docs/driver-model.md.
|
||||
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const sc = @import("system-call");
|
||||
|
||||
/// Allocation flags. `coherent` (uncacheable) is the portable default; the rest are
|
||||
/// opt-in for specific hardware — see `abi`.
|
||||
@@ -19,8 +19,8 @@
|
||||
const std = @import("std");
|
||||
const builtin = @import("builtin");
|
||||
const abi = @import("abi");
|
||||
const system_calls = @import("system.zig");
|
||||
const Mutex = @import("thread.zig").Thread.Mutex;
|
||||
const sc = @import("system-call");
|
||||
const Mutex = @import("thread").Thread.Mutex;
|
||||
|
||||
const page_size = abi.page_size;
|
||||
|
||||
@@ -63,8 +63,8 @@ fn payloadOf(block: *Block) [*]u8 {
|
||||
/// grants usually are adjacent). Returns false if the kernel is out of memory.
|
||||
fn grow(minimum_bytes: usize) bool {
|
||||
const bytes = alignUp(@max(minimum_bytes, chunk), page_size);
|
||||
const ret = system_calls.mmap(bytes, system_calls.PROT_READ | system_calls.PROT_WRITE);
|
||||
if (system_calls.mmapFailed(ret)) return false;
|
||||
const ret = sc.systemCall2(.mmap, bytes, abi.prot_read | abi.prot_write);
|
||||
if (ret > ~@as(usize, 0) - 4095) return false; // a wrapped -errno lands in the top page
|
||||
|
||||
const block: *Block = @ptrFromInt(ret);
|
||||
block.size = bytes;
|
||||
@@ -0,0 +1,48 @@
|
||||
//! library/kernel/memory — the process's memory interface: the heap allocator, DMA-capable
|
||||
//! buffers, shared-memory regions, and the raw `mmap` grant they all sit on. One flat module
|
||||
//! (formerly runtime.heap / runtime.dma / runtime.shared_memory, plus the `mmap` wrappers that
|
||||
//! lived in the system.zig dumping ground). Its private files are heap.zig, dma.zig, and
|
||||
//! shared-memory.zig — imported only here, so the heap's state and C symbols exist once.
|
||||
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call");
|
||||
const heap = @import("heap.zig");
|
||||
const dma = @import("dma.zig");
|
||||
const shared = @import("shared-memory.zig");
|
||||
|
||||
// --- the heap: a std.mem.Allocator over a first-fit free list (C malloc/free are also
|
||||
// exported from heap.zig, compiled once here) ---
|
||||
pub const allocator = heap.allocator;
|
||||
|
||||
// --- the raw grant every allocation sits on ---
|
||||
pub const PROT_READ: usize = abi.prot_read;
|
||||
pub const PROT_WRITE: usize = abi.prot_write;
|
||||
pub const PROT_EXEC: usize = abi.prot_exec;
|
||||
|
||||
/// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable memory and
|
||||
/// return the base virtual address. On failure returns a value in the top page (`mmapFailed`).
|
||||
pub fn mmap(len: usize, prot: usize) usize {
|
||||
return sc.systemCall2(.mmap, len, prot);
|
||||
}
|
||||
/// Release a range previously handed out by `mmap`.
|
||||
pub fn munmap(base: usize, len: usize) usize {
|
||||
return sc.systemCall2(.munmap, base, len);
|
||||
}
|
||||
/// Whether an `mmap` return value is an error (a wrapped -errno lands in the top page).
|
||||
pub inline fn mmapFailed(ret: usize) bool {
|
||||
return ret > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
|
||||
// --- DMA-capable buffers: physically contiguous, pinned, uncacheable, physical address known ---
|
||||
pub const DmaRegion = dma.Region;
|
||||
pub const dma_coherent = dma.coherent;
|
||||
pub const dma_write_combining = dma.write_combining;
|
||||
pub const dma_below_4g = dma.below_4g;
|
||||
pub const dmaAlloc = dma.alloc;
|
||||
pub const dmaFree = dma.free;
|
||||
|
||||
// --- shared-memory regions: a capability handed to another process over an ipc_call send_cap ---
|
||||
pub const SharedRegion = shared.Region;
|
||||
pub const sharedCreate = shared.create;
|
||||
pub const sharedMap = shared.map;
|
||||
pub const sharedPhysical = shared.physical;
|
||||
@@ -6,8 +6,8 @@
|
||||
//! generalization of capability passing from endpoints to memory objects.
|
||||
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const ipc = @import("ipc.zig");
|
||||
const sc = @import("system-call");
|
||||
const ipc = @import("ipc");
|
||||
|
||||
inline fn failed(r: usize) bool {
|
||||
return r > ~@as(usize, 0) - 4095; // a wrapped -errno lands in the top page
|
||||
@@ -7,9 +7,9 @@
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const sc = @import("system-call");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
|
||||
/// Everything a program receives at entry. Passed to
|
||||
/// `pub fn main(init: runtime.process.Init)`; programs that need nothing keep
|
||||
@@ -112,14 +112,14 @@ pub fn sendSignal(id: u32, signal: Signal) bool {
|
||||
/// (arm `system.timerOnce`, keep serving) instead of calling this.
|
||||
pub fn stop(id: u32, deadline_ms: u64, exit_endpoint: usize) void {
|
||||
_ = sendSignal(id, .terminate);
|
||||
_ = system.timerOnce(exit_endpoint, deadline_ms);
|
||||
_ = time.timerOnce(exit_endpoint, deadline_ms);
|
||||
var receive: [8]u8 = undefined;
|
||||
while (true) {
|
||||
const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null);
|
||||
if (got.isChildExit() and got.childProcessId() == id) return;
|
||||
if (got.isTimer()) break; // the deadline passed first — escalate
|
||||
}
|
||||
_ = system.kill(id);
|
||||
_ = kill(id);
|
||||
while (true) {
|
||||
const got = ipc.replyWait(exit_endpoint, &.{}, &receive, null);
|
||||
if (got.isChildExit() and got.childProcessId() == id) return;
|
||||
@@ -135,3 +135,78 @@ pub fn stop(id: u32, deadline_ms: u64, exit_endpoint: usize) void {
|
||||
pub fn subscribeExits(endpoint: usize) bool {
|
||||
return sc.systemCall1(.process_subscribe, endpoint) == 0;
|
||||
}
|
||||
|
||||
// --- raw process syscalls, formerly in the system.zig dumping ground ---
|
||||
|
||||
/// One `processes` entry — re-exported from the shared ABI so a program can declare its
|
||||
/// snapshot buffer without importing `abi` itself.
|
||||
pub const ProcessDescriptor = abi.ProcessDescriptor;
|
||||
|
||||
/// Give up the rest of this quantum.
|
||||
pub fn yield() void {
|
||||
_ = sc.systemCall0(.yield);
|
||||
}
|
||||
|
||||
/// End the process. Never returns.
|
||||
pub fn exit(code: usize) noreturn {
|
||||
_ = sc.systemCall1(.exit, code);
|
||||
unreachable; // the kernel never returns from exit
|
||||
}
|
||||
|
||||
/// Start the binary bundled in the initial-ramdisk under `name` as a new ring-3 process,
|
||||
/// returning the child's process id (or null). argv[0] is `name`, and the caller becomes
|
||||
/// its **supervisor** — the only process allowed to `kill` it.
|
||||
pub fn spawn(name: []const u8) ?u32 {
|
||||
return spawnSupervised(name, &.{}, null);
|
||||
}
|
||||
|
||||
/// Like `spawn`, but hands the child argv[1..] (argv[0] is still `name`).
|
||||
pub fn spawnWithArguments(name: []const u8, arguments: []const []const u8) ?u32 {
|
||||
return spawnSupervised(name, arguments, null);
|
||||
}
|
||||
|
||||
/// The full spawn: argv[1..] for the child, and an optional endpoint the kernel notifies
|
||||
/// when the child ends (any way — clean exit, fault, or `kill`), delivered via
|
||||
/// `ipc.replyWait` as a child-exit badge (`ipc.Received.isChildExit`/`childProcessId`), so
|
||||
/// one endpoint can supervise many children. Returns the child's process id, or null.
|
||||
pub fn spawnSupervised(name: []const u8, arguments: []const []const u8, exit_endpoint: ?usize) ?u32 {
|
||||
var blob: [256]u8 = undefined;
|
||||
var len: usize = 0;
|
||||
for (arguments, 0..) |argument, i| {
|
||||
if (i != 0) {
|
||||
if (len >= blob.len) return null;
|
||||
blob[len] = 0;
|
||||
len += 1;
|
||||
}
|
||||
if (len + argument.len > blob.len) return null;
|
||||
@memcpy(blob[len..][0..argument.len], argument);
|
||||
len += argument.len;
|
||||
}
|
||||
const r = sc.systemCall5(.system_spawn, @intFromPtr(name.ptr), name.len, if (len == 0) 0 else @intFromPtr(&blob), len, exit_endpoint orelse abi.no_cap);
|
||||
if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno
|
||||
return @intCast(r);
|
||||
}
|
||||
|
||||
/// Snapshot the process table into `out` and return the total number of live processes
|
||||
/// (which may exceed `out.len`; call again with a larger buffer). Kernel tasks are
|
||||
/// included, with an empty name. The primitive `ps` is built on.
|
||||
pub fn processes(out: []ProcessDescriptor) usize {
|
||||
return sc.systemCall2(.process_enumerate, @intFromPtr(out.ptr), out.len);
|
||||
}
|
||||
|
||||
/// Whether a process spawned under `name` (its argv[0]) is currently alive.
|
||||
pub fn isProcessRunning(name: []const u8) bool {
|
||||
var table: [32]ProcessDescriptor = undefined;
|
||||
const total = processes(&table);
|
||||
for (table[0..@min(total, table.len)]) |descriptor| {
|
||||
if (std.mem.eql(u8, descriptor.name[0..descriptor.name_length], name)) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// End process `id`. Only its supervisor — the process that spawned it — may; anyone else
|
||||
/// gets false, as does a stale or unknown id. Delivery is prompt but asynchronous, like a
|
||||
/// signal. True means the kill is accepted and irrevocable.
|
||||
pub fn kill(id: u32) bool {
|
||||
return sc.systemCall1(.process_kill, id) == 0;
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
//! The root module every user binary is compiled through (build.zig,
|
||||
//! `addUserBinary`). The program's own file is imported as `program`, and this
|
||||
//! shim contributes the declarations Zig resolves from the compilation root —
|
||||
//! `main` (dispatched by start's comptime dispatch) and the panic handler — and
|
||||
//! pulls in the `_start` entry shim. A program therefore only defines
|
||||
//! `pub fn main`; nothing else is required in its source file.
|
||||
|
||||
const start = @import("start");
|
||||
const logging = @import("logging");
|
||||
const program = @import("program");
|
||||
|
||||
/// Resolved as `@import("root").main` by start's comptime dispatch.
|
||||
pub const main = program.main;
|
||||
|
||||
/// The panic handler for every safety check in the image (start.panic).
|
||||
pub const panic = start.panic;
|
||||
|
||||
/// std.log for every user binary goes to the tagged kernel log ring (the kernel
|
||||
/// stamps the sender; see the logging module). A program overrides by declaring
|
||||
/// its own `pub const std_options`.
|
||||
pub const std_options: @import("std").Options =
|
||||
if (@hasDecl(program, "std_options")) program.std_options else logging.default_options;
|
||||
|
||||
comptime {
|
||||
_ = &start._start; // pull the entry shim into the image
|
||||
}
|
||||
@@ -13,8 +13,8 @@
|
||||
//! is the diagnosis (see docs/ipc.md).
|
||||
|
||||
const abi = @import("abi");
|
||||
const ipc = @import("ipc.zig");
|
||||
const process = @import("process.zig");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
|
||||
pub const Callbacks = struct {
|
||||
/// Called once with the service's endpoint before the loop starts — the
|
||||
@@ -4,8 +4,8 @@
|
||||
//! the whole runtime is linked in.
|
||||
|
||||
const std = @import("std");
|
||||
const system = @import("system.zig");
|
||||
const process = @import("process.zig");
|
||||
const logging = @import("logging");
|
||||
const process = @import("process");
|
||||
|
||||
/// The kernel enters at `_start` with rsp 16-aligned, pointing at the System V
|
||||
/// process-entry block it built: argc, argv pointers, NULL, envp terminator, the
|
||||
@@ -31,7 +31,7 @@ export fn rt_start(stack: [*]const u64) callconv(.c) noreturn {
|
||||
.count = stack[0],
|
||||
.vector = @ptrCast(stack + 1),
|
||||
} };
|
||||
system.exit(callMain(init));
|
||||
process.exit(callMain(init));
|
||||
}
|
||||
|
||||
/// Comptime-dispatch on root.main's signature, in the spirit of std's start.zig:
|
||||
@@ -68,7 +68,7 @@ fn callMain(init: process.Init) u8 {
|
||||
const payload = @call(.auto, root.main, call_arguments) catch |err| {
|
||||
var buffer: [128]u8 = undefined;
|
||||
const line = std.fmt.bufPrint(&buffer, "main returned error: {s}\n", .{@errorName(err)}) catch "main returned an error\n";
|
||||
_ = system.write(line);
|
||||
_ = logging.write(line);
|
||||
return 1; // distinct from panic's 127
|
||||
};
|
||||
if (@TypeOf(payload) == void) return 0;
|
||||
@@ -82,6 +82,6 @@ fn callMain(init: process.Init) u8 {
|
||||
/// No runtime to unwind into — report a panic as a nonzero exit code.
|
||||
pub const panic = std.debug.FullPanic(struct {
|
||||
fn panic(_: []const u8, _: ?usize) noreturn {
|
||||
system.exit(127);
|
||||
process.exit(127);
|
||||
}
|
||||
}.panic);
|
||||
@@ -13,8 +13,19 @@
|
||||
const std = @import("std");
|
||||
const builtin = @import("builtin");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const system = @import("system.zig");
|
||||
const sc = @import("system-call");
|
||||
|
||||
// A thread allocates its own stack straight from the mmap syscall (not through the
|
||||
// `memory` module) so `memory`'s heap can depend on this module's Mutex without a cycle.
|
||||
inline fn mmapStack(len: usize) usize {
|
||||
return sc.systemCall2(.mmap, len, abi.prot_read | abi.prot_write);
|
||||
}
|
||||
inline fn mmapFailed(ret: usize) bool {
|
||||
return ret > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
inline fn munmapStack(base: usize, len: usize) void {
|
||||
_ = sc.systemCall2(.munmap, base, len);
|
||||
}
|
||||
|
||||
/// True in a real danos binary; false when this module is compiled for host unit tests.
|
||||
/// The `Futex` seam and the test blocks below branch on it so the lock/condvar state
|
||||
@@ -66,8 +77,8 @@ pub const Thread = struct {
|
||||
}
|
||||
};
|
||||
|
||||
const base = system.mmap(config.stack_size, system.PROT_READ | system.PROT_WRITE);
|
||||
if (system.mmapFailed(base)) return error.SystemResources;
|
||||
const base = mmapStack(config.stack_size);
|
||||
if (mmapFailed(base)) return error.SystemResources;
|
||||
|
||||
// Top of the thread's own stack, downward: the closure, then a small per-thread TLS
|
||||
// block (the thread pointer points here; slot 0 is the variant-II self-pointer, the rest is
|
||||
@@ -88,7 +99,7 @@ pub const Thread = struct {
|
||||
|
||||
const tid = threadSpawn(@intFromPtr(&Closure.entry), stack_top, closure_addr);
|
||||
if (threadSpawnFailed(tid)) {
|
||||
_ = system.munmap(base, config.stack_size);
|
||||
munmapStack(base, config.stack_size);
|
||||
return error.SystemResources;
|
||||
}
|
||||
return .{ .tid = @intCast(tid), .stack_base = base, .stack_size = config.stack_size };
|
||||
@@ -99,7 +110,7 @@ pub const Thread = struct {
|
||||
/// child-exit notification on it is this thread's.
|
||||
pub fn join(self: Thread) void {
|
||||
_ = sc.systemCall1(.thread_join, self.tid); // block until the thread has exited
|
||||
_ = system.munmap(self.stack_base, self.stack_size); // reclaim its (now-vacated) stack
|
||||
munmapStack(self.stack_base, self.stack_size); // reclaim its (now-vacated) stack
|
||||
}
|
||||
|
||||
/// Relinquish the right to join: never wait for or reclaim this thread. Its stack is
|
||||
@@ -11,7 +11,33 @@
|
||||
//! CLOCK_REALTIME) layered on top later.
|
||||
|
||||
const std = @import("std");
|
||||
const system = @import("system.zig");
|
||||
const sc = @import("system-call");
|
||||
|
||||
// --- raw syscall wrappers, formerly in the system.zig dumping ground ---
|
||||
|
||||
/// Monotonic nanoseconds since boot — the raw reading; `now()` wraps it in an `Instant`.
|
||||
/// Never runs backward. Not wall-clock time (see `wallClock`).
|
||||
pub fn clock() u64 {
|
||||
return @intCast(sc.systemCall0(.clock));
|
||||
}
|
||||
|
||||
/// Wall-clock time in Unix epoch seconds (UTC), from the RTC — the real date/time, what a
|
||||
/// filesystem stamps as an mtime. Unlike `clock` (monotonic since boot), this is calendar time.
|
||||
pub fn wallClock() u64 {
|
||||
return @intCast(sc.systemCall0(.wall_clock));
|
||||
}
|
||||
|
||||
/// Block the caller for `ms` milliseconds — the raw, coarse, allocation-free form.
|
||||
pub fn sleepMillis(ms: u64) void {
|
||||
_ = sc.systemCall1(.sleep, ms);
|
||||
}
|
||||
|
||||
/// Arm a one-shot timer: after `ms` the kernel posts a timer notification
|
||||
/// (`ipc.Received.isTimer`) to `endpoint` (a handle from `ipc.createIpcEndpoint`). Unlike
|
||||
/// `sleep`, does not block — a service keeps serving IPC while the deadline is pending.
|
||||
pub fn timerOnce(endpoint: usize, ms: u64) bool {
|
||||
return sc.systemCall2(.timer_bind, endpoint, ms) == 0;
|
||||
}
|
||||
|
||||
const nanos_per_micro: u64 = 1_000;
|
||||
const nanos_per_milli: u64 = 1_000_000;
|
||||
@@ -90,32 +116,27 @@ pub const Instant = struct {
|
||||
|
||||
/// The current monotonic time.
|
||||
pub fn now() Instant {
|
||||
return .{ .ns = system.clock() };
|
||||
return .{ .ns = clock() };
|
||||
}
|
||||
|
||||
/// Monotonic nanoseconds since boot — the raw `clock()` reading, for callers that
|
||||
/// want a plain integer instead of an `Instant`.
|
||||
pub fn monotonicNanos() u64 {
|
||||
return system.clock();
|
||||
return clock();
|
||||
}
|
||||
|
||||
/// Whether the monotonic clock is usable. The kernel returns 0 until the TSC is
|
||||
/// calibrated (`tsc_hz == 0`); a caller that needs real time can treat that as
|
||||
/// "unavailable" instead of assuming the clock advances.
|
||||
pub fn available() bool {
|
||||
return system.clock() != 0;
|
||||
return clock() != 0;
|
||||
}
|
||||
|
||||
/// Block the caller for at least `d`, rounded up to the kernel's millisecond
|
||||
/// granularity. For sub-millisecond precision the scheduler cannot express, use
|
||||
/// `spin`.
|
||||
/// `spin`. (The raw millisecond form is `sleepMillis`.)
|
||||
pub fn sleep(d: Duration) void {
|
||||
system.sleep(d.ceilMillis());
|
||||
}
|
||||
|
||||
/// Block the caller for `ms` milliseconds — the coarse, allocation-free form.
|
||||
pub fn sleepMillis(ms: u64) void {
|
||||
system.sleep(ms);
|
||||
sleepMillis(d.ceilMillis());
|
||||
}
|
||||
|
||||
/// Busy-wait until `d` has elapsed, polling the monotonic clock. This burns the CPU
|
||||
@@ -126,13 +147,10 @@ pub fn spin(d: Duration) void {
|
||||
while (!deadline.reached()) {}
|
||||
}
|
||||
|
||||
/// Arm a one-shot timer against `endpoint` (a handle from `ipc.createIpcEndpoint`):
|
||||
/// after `d` the kernel posts a timer notification (`ipc.Received.isTimer`) there.
|
||||
/// Unlike `sleep`, this does not block — a service can keep serving IPC on the same
|
||||
/// endpoint while the deadline is pending. Rounds `d` up to milliseconds; returns
|
||||
/// false if the timer could not be armed. See `system.timerOnce`.
|
||||
/// The ergonomic `Duration` form of `timerOnce`: arm a one-shot timer against `endpoint`
|
||||
/// for `d` (rounded up to milliseconds). Returns false if the timer could not be armed.
|
||||
pub fn after(endpoint: usize, d: Duration) bool {
|
||||
return system.timerOnce(endpoint, d.ceilMillis());
|
||||
return timerOnce(endpoint, d.ceilMillis());
|
||||
}
|
||||
|
||||
test "Duration unit conversions round toward zero" {
|
||||
@@ -8,7 +8,10 @@
|
||||
//! side is `runtime.fs` (library/runtime/fs.zig), which programs use directly.
|
||||
//!
|
||||
//! This is user-space only — the kernel knows nothing of files or paths; it only moves the bytes.
|
||||
//! Shared by library/runtime/fs.zig (client) and system/services/vfs/vfs.zig (server).
|
||||
//! Shared by library/runtime/fs.zig (the client) and the mount backends that serve it (today
|
||||
//! the fat server, system/services/fat/). The standalone user-space VFS server it was first
|
||||
//! written against has retired — path routing moved into the kernel (system/kernel/vfs.zig,
|
||||
//! fs_resolve) — but the protocol module outlived it.
|
||||
|
||||
pub const Operation = enum(u32) {
|
||||
open, // open(path) -> node id
|
||||
@@ -1,51 +0,0 @@
|
||||
//! The per-process logger: std.log wired to the tagged kernel log ring.
|
||||
//!
|
||||
//! A program just calls `std.log.info("mounted {s}", .{path})` (or a scoped
|
||||
//! logger); this backend formats the line into a fixed buffer and emits ONE
|
||||
//! `debug_write` record carrying the level. The kernel stamps the record with
|
||||
//! the sender's pid and task name (its binary path) — the process does NOT put
|
||||
//! its own name in the payload; attribution is the kernel's, structural and
|
||||
//! unforgeable. Serial shows the kernel-rendered `<path>: message` line, and
|
||||
//! the logger service demultiplexes the ring into one file per process.
|
||||
//!
|
||||
//! Installed for every user binary by the root shim (library/runtime/root.zig)
|
||||
//! via `std_options`; a program can override by declaring its own
|
||||
//! `pub const std_options`.
|
||||
|
||||
const std = @import("std");
|
||||
const system = @import("system.zig");
|
||||
|
||||
fn levelOf(comptime level: std.log.Level) system.KlogLevel {
|
||||
return switch (level) {
|
||||
.err => .err,
|
||||
.warn => .warn,
|
||||
.info => .info,
|
||||
.debug => .debug,
|
||||
};
|
||||
}
|
||||
|
||||
pub fn logFn(
|
||||
comptime level: std.log.Level,
|
||||
comptime scope: @EnumLiteral(),
|
||||
comptime format: []const u8,
|
||||
args: anytype,
|
||||
) void {
|
||||
// One record = one line = at most klog_maximum_message bytes of payload.
|
||||
// On overflow keep what fits and end with "~" so the record is still a
|
||||
// whole line (the kernel would split an embedded rest anyway).
|
||||
var buffer: [256]u8 = undefined;
|
||||
const prefix = if (scope == .default) "" else "(" ++ @tagName(scope) ++ ") ";
|
||||
const line = std.fmt.bufPrint(&buffer, prefix ++ format, args) catch truncated: {
|
||||
buffer[buffer.len - 1] = '~';
|
||||
break :truncated buffer[0..];
|
||||
};
|
||||
_ = system.writeRecord(levelOf(level), line);
|
||||
}
|
||||
|
||||
/// The std.Options the root shim installs unless the program overrides it.
|
||||
/// Debug level: filtering is the log *reader's* job here — the ring is cheap,
|
||||
/// serial is a dev convenience, and the logger service keeps everything.
|
||||
pub const default_options: std.Options = .{
|
||||
.log_level = .debug,
|
||||
.logFn = logFn,
|
||||
};
|
||||
@@ -1,25 +0,0 @@
|
||||
//! The root module every user binary is compiled through (build.zig,
|
||||
//! `addUserBinary`). The program's own file is imported as `program`, and this
|
||||
//! shim contributes the declarations Zig resolves from the compilation root —
|
||||
//! `main` (dispatched by runtime.start) and the panic handler — and pulls in the
|
||||
//! `_start` entry shim. A program therefore only defines `pub fn main`; nothing
|
||||
//! else is required in its source file.
|
||||
|
||||
const runtime = @import("runtime");
|
||||
const program = @import("program");
|
||||
|
||||
/// Resolved as `@import("root").main` by runtime.start's comptime dispatch.
|
||||
pub const main = program.main;
|
||||
|
||||
/// The panic handler for every safety check in the image (runtime.start.panic).
|
||||
pub const panic = runtime.panic;
|
||||
|
||||
/// std.log for every user binary goes to the tagged kernel log ring (the kernel
|
||||
/// stamps the sender; see runtime.log). A program overrides by declaring its
|
||||
/// own `pub const std_options`.
|
||||
pub const std_options: @import("std").Options =
|
||||
if (@hasDecl(program, "std_options")) program.std_options else runtime.log.default_options;
|
||||
|
||||
comptime {
|
||||
_ = &runtime.start._start; // pull the runtime entry shim into the image
|
||||
}
|
||||
@@ -1,83 +0,0 @@
|
||||
//! danos user-space runtime library — a nascent libc. Every user binary (init,
|
||||
//! and later the VFS server + device drivers) imports this as `@import("runtime")`:
|
||||
//! system_call wrappers, the C-convention heap, IPC helpers, and the process start
|
||||
//! shim. It is compiled into each binary (inheriting its `.large` code model and
|
||||
//! freestanding target), so all user programs share one implementation.
|
||||
//!
|
||||
//! A user binary only defines a `pub fn main() void` or
|
||||
//! `pub fn main(init: runtime.process.Init) void` (arguments arrive via `init`).
|
||||
//! The panic handler and the `_start` entry pull live in the shared compilation
|
||||
//! root, library/runtime/root.zig, which build.zig wires around every program —
|
||||
//! nothing to declare per source file.
|
||||
|
||||
pub const system = @import("system.zig");
|
||||
pub const log = @import("log.zig");
|
||||
/// Monotonic time, delays, and deadlines over the kernel clock/sleep/timer syscalls
|
||||
/// — an `Instant`/`Duration` front door, no time service (docs/timers.md).
|
||||
pub const time = @import("time.zig");
|
||||
pub const heap = @import("heap.zig");
|
||||
pub const ipc = @import("ipc.zig");
|
||||
pub const start = @import("start.zig");
|
||||
/// The VFS wire protocol (shared with the VFS server).
|
||||
pub const vfs_protocol = @import("vfs-protocol");
|
||||
|
||||
/// The device-manager protocol: hello + tree reports (docs/device-manager.md).
|
||||
pub const device_manager_protocol = @import("device-manager-protocol");
|
||||
|
||||
/// The power protocol: events (button, lid, battery) + shutdown (docs/power.md).
|
||||
pub const power_protocol = @import("power-protocol");
|
||||
/// Keyboard-event listening (subscribe/next) and broadcasting (publish), over the input
|
||||
/// service. See library/runtime/input.zig and system/services/input/.
|
||||
pub const input = @import("input.zig");
|
||||
/// The input wire protocol (shared with the input service and its clients).
|
||||
pub const input_protocol = @import("input-protocol");
|
||||
/// POSIX-style file API: open/read/write/lseek/stat/close.
|
||||
/// C stdio: fopen/fread/fwrite/fseek/ftell/fclose over unistd.
|
||||
/// Device access for drivers: enumerate/claim/mmioMap.
|
||||
pub const device = @import("device.zig");
|
||||
/// DMA-capable memory for drivers: contiguous, pinned, uncacheable buffers.
|
||||
pub const dma = @import("dma.zig");
|
||||
|
||||
/// Shared cacheable memory: create a region + capability, pass the capability to another
|
||||
/// process (an `ipc_call` send_cap), map the same pages there. See library/runtime/shared-memory.zig
|
||||
/// and docs/display-v2.md.
|
||||
pub const shared_memory = @import("shared-memory.zig");
|
||||
|
||||
/// USB class-driver client: open a device on the xHCI bus and drive it
|
||||
/// (control / interrupt / bulk transfers). See library/runtime/usb.zig.
|
||||
pub const usb = @import("usb.zig");
|
||||
|
||||
/// Block-device client: read/write a block device (a USB stick, via
|
||||
/// usb-storage). See library/runtime/block.zig.
|
||||
pub const block = @import("block.zig");
|
||||
|
||||
/// Display-service client: query the mode, and (from D3) create layers, draw, and
|
||||
/// present frames. See library/runtime/display.zig and system/services/display/.
|
||||
pub const display = @import("display.zig");
|
||||
/// The display wire protocol (shared with the display service and its clients).
|
||||
pub const display_protocol = @import("display-protocol");
|
||||
/// The scanout wire protocol: the compositor's present channel to a native scanout driver
|
||||
/// (virtio-gpu). See system/services/display/scanout-protocol.zig and docs/display-v2.md.
|
||||
pub const scanout_protocol = @import("scanout-protocol");
|
||||
|
||||
/// The danos-native file API (open/read/write/list over the user-space VFS) — the
|
||||
/// layer danos programs use directly, and where the operations that later become
|
||||
/// `std.os.danos` are staged. See docs/zig-self-hosting.md.
|
||||
pub const fs = @import("fs.zig");
|
||||
|
||||
/// Re-exported so the root shim (root.zig) can install it as the panic handler.
|
||||
pub const panic = start.panic;
|
||||
|
||||
/// Process entry types: the `Init` handed to `main`, and its `Arguments`.
|
||||
pub const process = @import("process.zig");
|
||||
|
||||
/// Threads: `runtime.Thread`, std.Thread-shaped, over the private thread ABI
|
||||
/// (docs/threading.md). A binary must be built multi-threaded to spawn.
|
||||
pub const Thread = @import("thread.zig").Thread;
|
||||
|
||||
/// The service harness: one replyWait loop folding requests, signals, and
|
||||
/// notifications into callbacks (docs/process-lifecycle.md).
|
||||
pub const service = @import("service.zig");
|
||||
|
||||
/// The heap as a `std.mem.Allocator`, for Zig `std` containers in user code.
|
||||
pub const allocator = heap.allocator;
|
||||
@@ -1,266 +0,0 @@
|
||||
//! Typed system_call surface for user space — thin wrappers over the raw `system_call`
|
||||
//! stubs, one per kernel call. Numbers come from `abi.SystemCall`, the single
|
||||
//! source of truth shared with the kernel dispatcher.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call.zig");
|
||||
|
||||
/// `mmap` protection flags (matching the usual C bit values). Grants are always
|
||||
/// readable+writable today; the kernel does not yet honour finer prot.
|
||||
pub const PROT_READ: usize = abi.prot_read;
|
||||
pub const PROT_WRITE: usize = abi.prot_write;
|
||||
pub const PROT_EXEC: usize = abi.prot_exec;
|
||||
|
||||
/// One `processes` entry — re-exported from the shared ABI so a user program can
|
||||
/// declare its snapshot buffer without importing `abi` itself.
|
||||
pub const ProcessDescriptor = abi.ProcessDescriptor;
|
||||
|
||||
/// Give up the rest of this quantum.
|
||||
pub fn yield() void {
|
||||
_ = sc.systemCall0(.yield);
|
||||
}
|
||||
|
||||
/// The tagged-log level of a record — re-exported so runtime.log and the logger
|
||||
/// service don't import `abi` themselves.
|
||||
pub const KlogLevel = abi.KlogLevel;
|
||||
pub const KlogStatus = abi.KlogStatus;
|
||||
pub const KlogRecordHeader = abi.KlogRecordHeader;
|
||||
pub const klog_record_header_size = abi.klog_record_header_size;
|
||||
pub const klog_record_alignment = abi.klog_record_alignment;
|
||||
pub const klog_record_magic = abi.klog_record_magic;
|
||||
pub const klog_flag_truncated = abi.klog_flag_truncated;
|
||||
pub const klog_maximum_message = abi.klog_maximum_message;
|
||||
pub const maximum_process_name = abi.maximum_process_name;
|
||||
pub const FileAttributes = abi.FileAttributes;
|
||||
pub const DirectoryEntryHeader = abi.DirectoryEntryHeader;
|
||||
pub const file_kind_regular = abi.file_kind_regular;
|
||||
pub const file_kind_directory = abi.file_kind_directory;
|
||||
|
||||
/// Write raw bytes to the kernel log (bring-up/panic diagnostics; ordinary
|
||||
/// output goes through std.log -> writeRecord). The kernel stamps the record
|
||||
/// with this process's id and name. Returns the byte count, or a wrapped -1.
|
||||
pub fn write(message: []const u8) usize {
|
||||
return writeRecord(.raw, message);
|
||||
}
|
||||
|
||||
/// Emit one leveled record into the tagged kernel log ring. The kernel stamps
|
||||
/// pid/name/sequence/timestamp; the payload should be a single line (embedded
|
||||
/// newlines split into further records).
|
||||
pub fn writeRecord(level: KlogLevel, message: []const u8) usize {
|
||||
return sc.systemCall3(.debug_write, @intFromPtr(message.ptr), message.len, @intFromEnum(level));
|
||||
}
|
||||
|
||||
/// Block the caller for `ms` milliseconds.
|
||||
pub fn sleep(ms: usize) void {
|
||||
_ = sc.systemCall1(.sleep, ms);
|
||||
}
|
||||
|
||||
/// Arm a one-shot timer: after `ms` milliseconds the kernel posts a timer
|
||||
/// notification (`ipc.Received.isTimer`) to `endpoint`. The timed wait of
|
||||
/// docs/process-lifecycle.md — a service arms a deadline and keeps serving,
|
||||
/// instead of blocking in sleep; what stop-sequence escalation, hello deadlines,
|
||||
/// and restart backoff are built from.
|
||||
pub fn timerOnce(endpoint: usize, ms: u64) bool {
|
||||
return sc.systemCall2(.timer_bind, endpoint, ms) == 0;
|
||||
}
|
||||
|
||||
/// Monotonic nanoseconds since boot — a time source for timeouts and short delays. It
|
||||
/// only ever moves forward. This is *not* wall-clock time (no date, no timezone — that
|
||||
/// is a user-space service layered on top). Deadline pattern for a bounded poll loop:
|
||||
///
|
||||
/// const deadline = clock() + timeout_ns;
|
||||
/// while (clock() < deadline) { ... }
|
||||
pub fn clock() u64 {
|
||||
return @intCast(sc.systemCall0(.clock));
|
||||
}
|
||||
|
||||
/// Wall-clock time in Unix epoch seconds (UTC) — the real date/time, from the RTC.
|
||||
/// Unlike `clock` (monotonic since boot), this tracks calendar time, so it is what a
|
||||
/// filesystem stamps as a file's modification time. Formatting it into a calendar
|
||||
/// date/timezone is user-space policy layered on top.
|
||||
pub fn wallClock() u64 {
|
||||
return @intCast(sc.systemCall0(.wall_clock));
|
||||
}
|
||||
|
||||
/// Copy bytes out of the tagged kernel log ring — framed records of everything
|
||||
/// every process (and the kernel) has emitted — starting at stream offset
|
||||
/// `offset`, into `out`. Returns the byte count (0 = caught up), or null when
|
||||
/// `offset` fell behind the ring's tail (those records were overwritten) or
|
||||
/// lies past its head; re-sync via `klogStatus`. A reader parses
|
||||
/// [KlogRecordHeader][name][message] frames (8-byte aligned) from the bytes.
|
||||
pub fn klogRead(offset: u64, out: []u8) ?usize {
|
||||
const r = sc.systemCall3(.klog_read, offset, @intFromPtr(out.ptr), out.len);
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return r;
|
||||
}
|
||||
|
||||
/// The log ring's live cursors (oldest retained offset, end of stream, next
|
||||
/// sequence number) plus the wall-clock time of boot — how a log reader starts,
|
||||
/// detects loss, and names a per-boot log directory.
|
||||
pub fn klogStatus() ?KlogStatus {
|
||||
var status: KlogStatus = undefined;
|
||||
if (@as(isize, @bitCast(sc.systemCall1(.klog_status, @intFromPtr(&status)))) != 0) return null;
|
||||
return status;
|
||||
}
|
||||
|
||||
/// Where fs_resolve routed a path: served by the kernel (a permanent node
|
||||
/// token for fs_node) or by a userspace filesystem backend (an endpoint handle
|
||||
/// plus the rewritten mount-relative path, returned in the caller's buffer).
|
||||
pub const FsRoute = union(enum) {
|
||||
kernel: u64,
|
||||
backend: struct { handle: usize, path_len: usize },
|
||||
};
|
||||
|
||||
/// Route `path` through the kernel VFS. For a backend route the rewritten
|
||||
/// mount-relative path lands in `out` (behind a kernel-written length prefix,
|
||||
/// already stripped here: out[0..path_len] is the path).
|
||||
pub fn fsResolve(path: []const u8, flags: usize, out: []u8) ?FsRoute {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = flags; // in: flags (arg #3); out: node token / backend handle
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "+{rdx}" (rdx),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.fs_resolve)),
|
||||
[a0] "{rdi}" (@intFromPtr(path.ptr)),
|
||||
[a1] "{rsi}" (path.len),
|
||||
[a3] "{r10}" (@intFromPtr(out.ptr)),
|
||||
[a4] "{r8}" (out.len),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (@as(isize, @bitCast(rax)) < 0) return null;
|
||||
if (rax == abi.fs_route_kernel) return .{ .kernel = rdx };
|
||||
if (rax != abi.fs_route_backend) return null;
|
||||
const path_len = @as(usize, out[0]) | (@as(usize, out[1]) << 8);
|
||||
if (path_len + 2 > out.len) return null;
|
||||
std.mem.copyForwards(u8, out[0..path_len], out[2..][0..path_len]);
|
||||
return .{ .backend = .{ .handle = rdx, .path_len = path_len } };
|
||||
}
|
||||
|
||||
/// Read `out.len` bytes of a kernel-served node at `offset` (fs_node read).
|
||||
pub fn fsNodeRead(node_token: u64, offset: u64, out: []u8) ?usize {
|
||||
const r = sc.systemCall5(.fs_node, abi.fs_node_read, node_token, offset, @intFromPtr(out.ptr), out.len);
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return r;
|
||||
}
|
||||
|
||||
/// A kernel-served node's metadata (fs_node status).
|
||||
pub fn fsNodeStatus(node_token: u64) ?abi.FileAttributes {
|
||||
var attributes: abi.FileAttributes = undefined;
|
||||
const r = sc.systemCall5(.fs_node, abi.fs_node_status, node_token, 0, @intFromPtr(&attributes), @sizeOf(abi.FileAttributes));
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return attributes;
|
||||
}
|
||||
|
||||
/// The `cursor`th child of a kernel-served directory (fs_node readdir): fills
|
||||
/// `out` with [DirectoryEntryHeader][name]; returns total bytes (0 = end).
|
||||
pub fn fsNodeReaddir(node_token: u64, cursor: u64, out: []u8) ?usize {
|
||||
const r = sc.systemCall5(.fs_node, abi.fs_node_readdir, node_token, cursor, @intFromPtr(out.ptr), out.len);
|
||||
if (@as(isize, @bitCast(r)) < 0) return null;
|
||||
return r;
|
||||
}
|
||||
|
||||
/// Mount a userspace filesystem's endpoint at `prefix`, with an optional
|
||||
/// backend-side `rewrite` prefix ("" = none). Possession of the endpoint
|
||||
/// handle is the capability.
|
||||
pub fn fsMount(prefix: []const u8, backend: usize, rewrite: []const u8) bool {
|
||||
return sc.systemCall5(.fs_mount, @intFromPtr(prefix.ptr), prefix.len, backend, @intFromPtr(rewrite.ptr), rewrite.len) == 0;
|
||||
}
|
||||
|
||||
pub fn fsUnmount(prefix: []const u8) bool {
|
||||
return sc.systemCall2(.fs_unmount, @intFromPtr(prefix.ptr), prefix.len) == 0;
|
||||
}
|
||||
|
||||
/// End the process. Never returns.
|
||||
pub fn exit(code: usize) noreturn {
|
||||
_ = sc.systemCall1(.exit, code);
|
||||
unreachable; // the kernel never returns from exit
|
||||
}
|
||||
|
||||
/// Start the binary bundled in the initial-ramdisk under `name` as a new ring-3
|
||||
/// process, returning the child's process id (or null on failure). The child's
|
||||
/// argv[0] is `name`, and the caller becomes its **supervisor** — the only process
|
||||
/// allowed to `kill` it. This is how a supervisor (the device manager) launches a
|
||||
/// driver it matched — danos-native, not POSIX (a spawn/exec family comes with the
|
||||
/// POSIX layer later).
|
||||
pub fn spawn(name: []const u8) ?u32 {
|
||||
return spawnSupervised(name, &.{}, null);
|
||||
}
|
||||
|
||||
/// Like `spawn`, but hands the child command-line arguments: they arrive as
|
||||
/// argv[1..] on its System V entry stack (argv[0] is still `name`).
|
||||
pub fn spawnWithArguments(name: []const u8, arguments: []const []const u8) ?u32 {
|
||||
return spawnSupervised(name, arguments, null);
|
||||
}
|
||||
|
||||
/// The full spawn: command-line arguments for the child, and an optional endpoint
|
||||
/// (a handle from `ipc.createIpcEndpoint`) the kernel notifies when the child ends
|
||||
/// — any way it ends: clean exit, fault, or `kill`. The notification arrives via
|
||||
/// `ipc.replyWait` as a badge with the child-exit bit set and the child's id in
|
||||
/// the low bits (`ipc.Received.isChildExit`/`childProcessId`), so one endpoint can
|
||||
/// supervise many children. Arguments are marshalled to the kernel as one
|
||||
/// NUL-separated blob; the combined arguments must fit `blob` (the kernel caps the
|
||||
/// blob at 256 bytes and argc at 8 anyway). Returns the child's process id, or
|
||||
/// null on failure.
|
||||
pub fn spawnSupervised(name: []const u8, arguments: []const []const u8, exit_endpoint: ?usize) ?u32 {
|
||||
var blob: [256]u8 = undefined;
|
||||
var len: usize = 0;
|
||||
for (arguments, 0..) |argument, i| {
|
||||
if (i != 0) {
|
||||
if (len >= blob.len) return null;
|
||||
blob[len] = 0;
|
||||
len += 1;
|
||||
}
|
||||
if (len + argument.len > blob.len) return null;
|
||||
@memcpy(blob[len..][0..argument.len], argument);
|
||||
len += argument.len;
|
||||
}
|
||||
const r = sc.systemCall5(.system_spawn, @intFromPtr(name.ptr), name.len, if (len == 0) 0 else @intFromPtr(&blob), len, exit_endpoint orelse abi.no_cap);
|
||||
if (r > ~@as(usize, 0) - 4095) return null; // a wrapped -errno
|
||||
return @intCast(r);
|
||||
}
|
||||
|
||||
/// Snapshot the process table into `out` (up to its length) and return the total
|
||||
/// number of live processes — which may exceed `out.len`; call again with a larger
|
||||
/// buffer for the full listing. Kernel tasks are included, with an empty name.
|
||||
/// The primitive `ps` is built on.
|
||||
pub fn processes(out: []abi.ProcessDescriptor) usize {
|
||||
return sc.systemCall2(.process_enumerate, @intFromPtr(out.ptr), out.len);
|
||||
}
|
||||
|
||||
/// Whether a process spawned under `name` (its argv[0]) is currently alive.
|
||||
pub fn isProcessRunning(name: []const u8) bool {
|
||||
var table: [32]ProcessDescriptor = undefined;
|
||||
const total = processes(&table);
|
||||
for (table[0..@min(total, table.len)]) |descriptor| {
|
||||
if (std.mem.eql(u8, descriptor.name[0..descriptor.name_length], name)) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// End process `id`. Only its supervisor — the process that spawned it — may;
|
||||
/// anyone else gets false, as does a stale or unknown id (ids are never reused).
|
||||
/// Delivery is prompt but asynchronous, like a signal: a target caught running on
|
||||
/// another core dies at its next system call or timer tick. True means the kill
|
||||
/// is accepted and irrevocable; the exit notification (if an endpoint was given
|
||||
/// at spawn) confirms completion.
|
||||
pub fn kill(id: u32) bool {
|
||||
return sc.systemCall1(.process_kill, id) == 0;
|
||||
}
|
||||
|
||||
/// Grant `len` bytes (rounded up to whole pages) of fresh, zeroed, writable
|
||||
/// memory and return the base virtual address. On failure returns a value in the
|
||||
/// top page (see `mmapFailed`). The user heap grows through this call.
|
||||
pub fn mmap(len: usize, prot: usize) usize {
|
||||
return sc.systemCall2(.mmap, len, prot);
|
||||
}
|
||||
|
||||
/// Release a range previously handed out by `mmap`.
|
||||
pub fn munmap(base: usize, len: usize) usize {
|
||||
return sc.systemCall2(.munmap, base, len);
|
||||
}
|
||||
|
||||
/// Whether an `mmap` return value is an error (the kernel returns a wrapped
|
||||
/// -errno, which lands in the top page — no real grant base is ever that high).
|
||||
pub inline fn mmapFailed(ret: usize) bool {
|
||||
return ret > ~@as(usize, 0) - 4095;
|
||||
}
|
||||
@@ -1,6 +1,6 @@
|
||||
# xkeyboard-config — X11 keyboard layouts, compiled to Zig
|
||||
|
||||
This module turns a physical key (a **USB HID usage**, as the [input module](../../docs/input.md)
|
||||
This module turns a physical key (a **USB HID usage**, as the [input module](../../docs/device-driver-development-guide/input.md)
|
||||
delivers in `KeyEvent.keycode`) plus a modifier state into a **keysym** and, when the key
|
||||
produces one, a **character** (a Unicode scalar). It is what lets a `keycode` become a
|
||||
`character` — a keymap — without danos shipping an X11 runtime.
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user