Compare commits
84
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
1b1c587c14 | ||
|
|
2719b93530 | ||
|
|
d2dfbcabf8 | ||
|
|
b004b9c3eb | ||
|
|
3e6e21bf0a | ||
|
|
0fbd2c8f12 | ||
|
|
73aea4582b | ||
|
|
1379b699f3 | ||
|
|
1ff0991452 | ||
|
|
ac01f627d1 | ||
|
|
f3bc23cb81 | ||
|
|
8d4a7cf240 | ||
|
|
c4f16a5448 | ||
|
|
e4da4e0610 | ||
|
|
9a3238025d | ||
|
|
c96ef87714 | ||
|
|
de9870175f | ||
|
|
be04ebe954 | ||
|
|
62d6a7a150 | ||
|
|
fa8203cdba | ||
|
|
e53d6ebafb | ||
|
|
4476208361 | ||
|
|
c621b649f6 | ||
|
|
d27670ec39 | ||
|
|
ab7594df6e | ||
|
|
d03942b543 | ||
|
|
3f9b6813f7 | ||
|
|
bc2eb67581 | ||
|
|
cb98a9844e | ||
|
|
4701fbd123 | ||
|
|
3b23b11b0e | ||
|
|
902e4a0a9e | ||
|
|
15575960bd | ||
|
|
6f4fdc2789 | ||
|
|
721288c516 | ||
|
|
4194bb6e32 | ||
|
|
fc0b934b7f | ||
|
|
6a687fbc2b | ||
|
|
4e7cbc9792 | ||
|
|
e94adcfc02 | ||
|
|
f477ef7d9f | ||
|
|
9e649178bf | ||
|
|
e376c9e908 | ||
|
|
48b9ed4001 | ||
|
|
081ba1d74e | ||
|
|
203528c8a7 | ||
|
|
bf0c3fd3e0 | ||
|
|
3af0110483 | ||
|
|
757c6f14c3 | ||
|
|
52d6e372fd | ||
|
|
f023f1cfd6 | ||
|
|
b9cec7d1be | ||
|
|
6271278d4d | ||
|
|
37326c7664 | ||
|
|
23bcd77c58 | ||
|
|
dded46726b | ||
|
|
60f32ee9ff | ||
|
|
3e69712b97 | ||
|
|
11f567ee20 | ||
|
|
3a155cdc7d | ||
|
|
5b874fc756 | ||
|
|
7d540c4b2f | ||
|
|
ac2d102878 | ||
|
|
3c9475e33a | ||
|
|
44122bd44d | ||
|
|
9ef22d6e55 | ||
|
|
07c901c18c | ||
|
|
25cc4d610e | ||
|
|
f90dc6c121 | ||
|
|
1d1234c963 | ||
|
|
a7f0c1a450 | ||
|
|
f86f2987d5 | ||
|
|
794a8b5782 | ||
|
|
ea470afe84 | ||
|
|
d27377ab1e | ||
|
|
cacbacd76b | ||
|
|
e63d6ef5ef | ||
|
|
c8191570e1 | ||
|
|
2bc2a0d70d | ||
|
|
1882161cb4 | ||
|
|
08e139ebba | ||
|
|
b09a62bc36 | ||
|
|
daca0d9216 | ||
|
|
e78810195d |
@@ -52,7 +52,8 @@ zig build
|
||||
Produces a FHS-shaped `zig-out/` that *is* the danos filesystem and the boot volume:
|
||||
the UEFI bootloader at `zig-out/EFI/BOOT/BOOTX64.efi`, the kernel at
|
||||
`zig-out/system/kernel`, init at `zig-out/system/services/init`, drivers under
|
||||
`zig-out/system/drivers/`, and the initial-ramdisk at `zig-out/boot/`.
|
||||
`zig-out/system/drivers/`, the test fixtures under `zig-out/test/system/services/`,
|
||||
and the initial-ramdisk at `zig-out/boot/`.
|
||||
|
||||
## Release media
|
||||
|
||||
@@ -62,7 +63,7 @@ zig build release-x86-64
|
||||
|
||||
Produces `zig-out/danos-x86-64.iso`, a hybrid ISO that boots flashed raw to a
|
||||
USB stick (balenaEtcher, dd) or burned to optical media — see
|
||||
[docs/release-iso.md](docs/release-iso.md). `zig build check-iso-image`
|
||||
[docs/release-iso.md](docs/os-development/release-iso.md). `zig build check-iso-image`
|
||||
validates it without booting.
|
||||
|
||||
## Run
|
||||
|
||||
+22
-14
@@ -16,11 +16,12 @@ const MemoryMapSlice = uefi.tables.MemoryMapSlice;
|
||||
/// The kernel image: /system/kernel.
|
||||
const kernel_file_name = std.unicode.utf8ToUtf16LeStringLiteral("system\\kernel");
|
||||
|
||||
/// The user binaries: everything under /system except the kernel itself. The
|
||||
/// loader walks this tree and packs it into the in-RAM initial_ramdisk image —
|
||||
/// the volume's file structure is the single source of truth (no packed image
|
||||
/// artifact on disk).
|
||||
/// The user binaries: everything under /system except the kernel itself, plus
|
||||
/// the test fixtures under /test. The loader walks both trees and packs them
|
||||
/// into the in-RAM initial_ramdisk image — the volume's file structure is the
|
||||
/// single source of truth (no packed image artifact on disk).
|
||||
const system_directory_name = std.unicode.utf8ToUtf16LeStringLiteral("system");
|
||||
const test_directory_name = std.unicode.utf8ToUtf16LeStringLiteral("test");
|
||||
|
||||
/// Physical page size, and the sentinel UEFI uses to seek to end-of-file.
|
||||
const page_size = 4096;
|
||||
@@ -364,16 +365,17 @@ fn handoff(cr3: u64, entry: usize, boot_information: *const BootInformation) nor
|
||||
unreachable;
|
||||
}
|
||||
|
||||
// --- the /system tree -> initial_ramdisk ------------------------------------
|
||||
// --- the /system and /test trees -> initial_ramdisk --------------------------
|
||||
|
||||
/// Cap on bundled binaries. Generous: the tree carries ~30 today.
|
||||
const maximum_bundled = 64;
|
||||
|
||||
/// How deep the walk goes below /system ("/system/services/x" is depth 1).
|
||||
/// How deep the walk goes below a tree root ("/system/services/x" is depth 1,
|
||||
/// "/test/system/services/x" is depth 2).
|
||||
const maximum_tree_depth = 3;
|
||||
|
||||
/// One binary discovered under /system: its FHS path (UTF-8, '/'-separated,
|
||||
/// NUL-free) and its contents in a transient pool buffer.
|
||||
/// One binary discovered under a walked tree: its FHS path (UTF-8,
|
||||
/// '/'-separated, NUL-free) and its contents in a transient pool buffer.
|
||||
const Bundled = struct {
|
||||
path: [initial_ramdisk.maximum_name]u8,
|
||||
path_len: usize,
|
||||
@@ -389,10 +391,10 @@ const Bundled = struct {
|
||||
/// 1. /system/manifest (written by the build): each listed path is opened BY
|
||||
/// NAME — the case-insensitive lookup every firmware FAT driver gets
|
||||
/// right, and the only file access the pre-tree loader ever used.
|
||||
/// 2. No manifest: ENUMERATE the /system tree. Portable in principle, but
|
||||
/// firmware differs in what names enumeration returns (bare 8.3 entries
|
||||
/// come back uppercase on some drivers), so this is the fallback for
|
||||
/// hand-assembled sticks, not the primary path.
|
||||
/// 2. No manifest: ENUMERATE the /system and /test trees. Portable in
|
||||
/// principle, but firmware differs in what names enumeration returns
|
||||
/// (bare 8.3 entries come back uppercase on some drivers), so this is
|
||||
/// the fallback for hand-assembled sticks, not the primary path.
|
||||
fn loadSystemTree(bs: *uefi.tables.BootServices, boot_information: *BootInformation) !void {
|
||||
const loaded = (try bs.handleProtocol(uefi.protocol.LoadedImage, uefi.handle)) orelse
|
||||
return error.NoLoadedImage;
|
||||
@@ -426,6 +428,11 @@ fn loadSystemTree(bs: *uefi.tables.BootServices, boot_information: *BootInformat
|
||||
const system_directory = try root.open(system_directory_name, .read, .{});
|
||||
defer _ = system_directory.close() catch {};
|
||||
try walkDirectory(bs, system_directory, "/system", 0, &list, &count);
|
||||
// The /test tree is optional: a stick without fixtures still boots.
|
||||
if (root.open(test_directory_name, .read, .{})) |test_directory| {
|
||||
defer _ = test_directory.close() catch {};
|
||||
try walkDirectory(bs, test_directory, "/test", 0, &list, &count);
|
||||
} else |_| {}
|
||||
}
|
||||
if (count == 0) return error.NoBinaries;
|
||||
|
||||
@@ -452,7 +459,7 @@ fn loadSystemTree(bs: *uefi.tables.BootServices, boot_information: *BootInformat
|
||||
|
||||
boot_information.initial_ramdisk_base = @intFromPtr(image.ptr);
|
||||
boot_information.initial_ramdisk_len = total;
|
||||
log("EFI: /system tree loaded, starting the kernel\r\n");
|
||||
log("EFI: boot tree loaded, starting the kernel\r\n");
|
||||
}
|
||||
|
||||
/// The boot capsule: the bundled binaries as one v2 initial_ramdisk image.
|
||||
@@ -522,7 +529,8 @@ fn loadByManifest(bs: *uefi.tables.BootServices, root: *uefi.protocol.File, list
|
||||
|
||||
/// Recursively collect the regular files below `directory` into `list`. Top-level
|
||||
/// files (depth 0) are skipped: the only one is /system/kernel, which loadKernel
|
||||
/// has already consumed and which is not a spawnable user binary.
|
||||
/// has already consumed and which is not a spawnable user binary (/test has no
|
||||
/// top-level files, so the skip is a no-op there).
|
||||
fn walkDirectory(
|
||||
bs: *uefi.tables.BootServices,
|
||||
directory: *uefi.protocol.File,
|
||||
|
||||
@@ -0,0 +1,131 @@
|
||||
//! The danos build API (docs/build-packages-plan.md): the one shared recipe
|
||||
//! for building a user-space binary. A binary package's build.zig names its
|
||||
//! binary and EXACTLY the modules its source imports — the moral equivalent
|
||||
//! of a C file's include list — and `userBinary` resolves each name from the
|
||||
//! library domain package that exports it. Nothing is pre-wired: an @import
|
||||
//! the package did not declare is a compile error, and a domain none of the
|
||||
//! imports come from never appears in the package's manifest. The only
|
||||
//! implicit dependency is the kernel package, because the shared root shim
|
||||
//! (root.zig, user.ld) lives there and itself reaches start + logging.
|
||||
//!
|
||||
//! Consumers declare this package in their build.zig.zon (as "build-support")
|
||||
//! and @import its build.zig from their own build.zig; nothing is compiled
|
||||
//! from this package itself — it exports build-time functions only.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
_ = b; // nothing to build: this package exports build-time functions only
|
||||
}
|
||||
|
||||
/// The freestanding x86-64 target every danos binary (kernel and user) is
|
||||
/// built for. SSE2 is part of the x86_64 baseline and UEFI leaves it enabled
|
||||
/// at handoff, so we keep it: disabling it forces soft-float and makes the
|
||||
/// compiler unable to encode the vector ops that std's formatting/runtime
|
||||
/// still emit.
|
||||
pub fn freestandingTarget(b: *std.Build) std.Build.ResolvedTarget {
|
||||
return b.resolveTargetQuery(.{
|
||||
.cpu_arch = .x86_64,
|
||||
.os_tag = .freestanding,
|
||||
.abi = .none,
|
||||
});
|
||||
}
|
||||
|
||||
/// Resolve one imported module by searching the packages this binary DECLARED
|
||||
/// in its own build.zig.zon — the C include path made literal: an import can
|
||||
/// only be satisfied by a domain the binary claims, and each domain's own
|
||||
/// build.zig (its addModule exports) is the single statement of who owns
|
||||
/// what. There is no name table here to drift.
|
||||
fn moduleFromDeclaredDependencies(b: *std.Build, name: []const u8) *std.Build.Module {
|
||||
for (b.available_deps) |declared| {
|
||||
const dependency = b.dependency(declared[0], .{});
|
||||
if (dependency.builder.modules.get(name)) |module| return module;
|
||||
}
|
||||
@panic(b.fmt(
|
||||
"no declared dependency exports a module named '{s}' — declare the domain that owns it in this package's build.zig.zon",
|
||||
.{name},
|
||||
));
|
||||
}
|
||||
|
||||
/// What `userBinary` needs to know about one user binary.
|
||||
pub const UserBinaryOptions = struct {
|
||||
name: []const u8,
|
||||
/// The program's own source file — it becomes the `program` module the
|
||||
/// root shim imports; a program only defines `pub fn main`.
|
||||
root_source_file: std.Build.LazyPath,
|
||||
/// Exactly the modules the program's source @imports (directly or through
|
||||
/// its same-directory files) — no more, no less. Order is free; sorted
|
||||
/// reads best. An undeclared @import fails the compile; a name no
|
||||
/// declared domain exports fails the build graph, naming the miss.
|
||||
imports: []const []const u8,
|
||||
/// Built multi-threaded (`single_threaded = false`) so real atomics/TLS
|
||||
/// work — required before a binary may call `Thread.spawn`
|
||||
/// (docs/threading.md). Threads are a deliberate per-binary opt-in.
|
||||
threaded: bool = false,
|
||||
};
|
||||
|
||||
/// Build one user-space binary the same way for every program (init, the
|
||||
/// services, the drivers): freestanding, ReleaseSmall, `.large` code model
|
||||
/// (the image base is above 4 GiB — smaller models emit 32-bit relocations
|
||||
/// that can't reach), linked with the shared user link script. Pinned to
|
||||
/// LLVM + LLD so the script's PHDRS (segment permissions) are authoritative —
|
||||
/// the kernel's W^X user-ELF loader requires exact perms.
|
||||
///
|
||||
/// The compilation root is not the program's own file but the shared shim
|
||||
/// (the kernel package's root.zig), which supplies the root declarations
|
||||
/// (`main` re-export, panic handler, `_start` pull) so a program only defines
|
||||
/// `pub fn main`. The program's file becomes the `program` module the shim
|
||||
/// imports; reach it through `programModule` to add per-binary non-library
|
||||
/// modules (compile-time options).
|
||||
pub fn userBinary(b: *std.Build, options: UserBinaryOptions) *std.Build.Step.Compile {
|
||||
const kernel = b.dependency("kernel", .{});
|
||||
var imports: std.ArrayListUnmanaged(std.Build.Module.Import) = .empty;
|
||||
for (options.imports) |name| {
|
||||
imports.append(b.allocator, .{
|
||||
.name = name,
|
||||
.module = moduleFromDeclaredDependencies(b, name),
|
||||
}) catch @panic("OOM");
|
||||
}
|
||||
// Settings (target, optimize, code model, ...) live on the root module
|
||||
// only; the program module inherits them.
|
||||
const program_module = b.createModule(.{
|
||||
.root_source_file = options.root_source_file,
|
||||
.imports = imports.items,
|
||||
});
|
||||
const exe = b.addExecutable(.{
|
||||
.name = options.name,
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = kernel.path("root.zig"),
|
||||
.target = freestandingTarget(b),
|
||||
.optimize = .ReleaseSmall,
|
||||
.code_model = .large,
|
||||
.single_threaded = !options.threaded, // a threaded binary needs real atomics/TLS
|
||||
.sanitize_c = .off,
|
||||
.stack_check = false,
|
||||
.stack_protector = false,
|
||||
// The root shim itself imports only start (_start + panic) and
|
||||
// logging (std_options) — straight from the kernel package, so a
|
||||
// program's own import list stays exactly its own.
|
||||
.imports = &.{
|
||||
.{ .name = "start", .module = kernel.module("start") },
|
||||
.{ .name = "logging", .module = kernel.module("logging") },
|
||||
.{ .name = "program", .module = program_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
exe.setLinkerScript(kernel.path("user.ld"));
|
||||
exe.entry = .{ .symbol_name = "_start" };
|
||||
exe.image_base = 0x7000_0000_0000;
|
||||
exe.use_llvm = true;
|
||||
exe.use_lld = true;
|
||||
return exe;
|
||||
}
|
||||
|
||||
/// The `program` module of a binary built by `userBinary` — the module rooted
|
||||
/// at the program's own source file. Per-binary non-library modules (an
|
||||
/// addOptions build_options) go here, not on the root shim: module imports
|
||||
/// are not transitive, so an import added to the root would be invisible to
|
||||
/// the program's code.
|
||||
pub fn programModule(exe: *std.Build.Step.Compile) *std.Build.Module {
|
||||
return exe.root_module.import_table.get("program").?;
|
||||
}
|
||||
@@ -0,0 +1,8 @@
|
||||
.{
|
||||
.name = .build_support,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0xad91962994f4be41, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{},
|
||||
.paths = .{""},
|
||||
}
|
||||
+48
-8
@@ -32,6 +32,53 @@
|
||||
// Once all dependencies are fetched, `zig build` no longer requires
|
||||
// internet connectivity.
|
||||
.dependencies = .{
|
||||
// The danos build API — the shared user-binary recipe every build file
|
||||
// (root and per-binary packages) consumes (docs/build-packages-plan.md).
|
||||
.@"build-support" = .{ .path = "build-support" },
|
||||
// The library domains, each a package exporting its modules.
|
||||
.kernel = .{ .path = "library/kernel" },
|
||||
.device = .{ .path = "library/device" },
|
||||
.client = .{ .path = "library/client" },
|
||||
.protocol = .{ .path = "library/protocol" },
|
||||
.csv = .{ .path = "library/csv" },
|
||||
.@"xkeyboard-config" = .{ .path = "library/xkeyboard-config" },
|
||||
// Binary packages (phase 2), consumed as artifacts for the boot image.
|
||||
.@"pci-bus" = .{ .path = "system/drivers/pci-bus" },
|
||||
.init = .{ .path = "system/services/init" },
|
||||
.fat = .{ .path = "system/services/fat" },
|
||||
.display = .{ .path = "system/services/display" },
|
||||
.@"display-demo" = .{ .path = "system/services/display-demo" },
|
||||
.@"device-manager" = .{ .path = "system/services/device-manager" },
|
||||
.input = .{ .path = "system/services/input" },
|
||||
.logger = .{ .path = "system/services/logger" },
|
||||
// The discovery pair and the /test fixtures are lazy: only what a
|
||||
// given build actually ships gets its build file loaded and compiled
|
||||
// (-Ddiscovery picks one of the pair; -Dtest-case pulls the fixtures).
|
||||
.acpi = .{ .path = "system/services/acpi", .lazy = true },
|
||||
.fdt = .{ .path = "system/services/fdt", .lazy = true },
|
||||
.@"ps2-bus" = .{ .path = "system/drivers/ps2-bus" },
|
||||
.@"usb-xhci-bus" = .{ .path = "system/drivers/usb-xhci-bus" },
|
||||
.@"usb-hid" = .{ .path = "system/drivers/usb-hid" },
|
||||
.@"usb-storage" = .{ .path = "system/drivers/usb-storage" },
|
||||
.@"virtio-gpu" = .{ .path = "system/drivers/virtio-gpu" },
|
||||
.@"vfs-test" = .{ .path = "test/system/services/vfs-test", .lazy = true },
|
||||
.@"fat-test" = .{ .path = "test/system/services/fat-test", .lazy = true },
|
||||
.@"badge-scope-test" = .{ .path = "test/system/services/badge-scope-test", .lazy = true },
|
||||
.@"shared-memory-server" = .{ .path = "test/system/services/shared-memory-server", .lazy = true },
|
||||
.@"shared-memory-client" = .{ .path = "test/system/services/shared-memory-client", .lazy = true },
|
||||
.@"crash-test" = .{ .path = "test/system/services/crash-test", .lazy = true },
|
||||
.@"device-list" = .{ .path = "test/system/services/device-list", .lazy = true },
|
||||
.@"pci-cap-test" = .{ .path = "test/system/services/pci-cap-test", .lazy = true },
|
||||
.@"iommu-fault-test" = .{ .path = "test/system/services/iommu-fault-test", .lazy = true },
|
||||
.@"input-source" = .{ .path = "test/system/services/input-source", .lazy = true },
|
||||
.@"input-test" = .{ .path = "test/system/services/input-test", .lazy = true },
|
||||
.@"args-echo" = .{ .path = "test/system/services/args-echo", .lazy = true },
|
||||
.@"process-test" = .{ .path = "test/system/services/process-test", .lazy = true },
|
||||
.@"thread-test" = .{ .path = "test/system/services/thread-test", .lazy = true },
|
||||
.@"user-memory-test" = .{ .path = "test/system/services/user-memory-test", .lazy = true },
|
||||
.@"protocol-registry-test" = .{ .path = "test/system/services/protocol-registry-test", .lazy = true },
|
||||
.@"protocol-denied-test" = .{ .path = "test/system/services/protocol-denied-test", .lazy = true },
|
||||
.@"protocol-conformance-test" = .{ .path = "test/system/services/protocol-conformance-test", .lazy = true },
|
||||
// See `zig fetch --save <url>` for a command-line interface for adding dependencies.
|
||||
//.example = .{
|
||||
// // When updating this field to a new URL, be sure to delete the corresponding
|
||||
@@ -70,12 +117,5 @@
|
||||
// Paths are relative to the build root. Use the empty string (`""`) to refer to
|
||||
// the build root itself.
|
||||
// A directory listed here means that all files within, recursively, are included.
|
||||
.paths = .{
|
||||
"build.zig",
|
||||
"build.zig.zon",
|
||||
"src",
|
||||
// For example...
|
||||
//"LICENSE",
|
||||
//"README.md",
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
|
||||
@@ -0,0 +1,166 @@
|
||||
//! Boot-image assembly (docs/build-packages-plan.md, phase 3): everything
|
||||
//! between "here are the built binaries" and "here is a bootable volume".
|
||||
//! The FHS-shaped zig-out install tree, the boot manifest, the boot capsule,
|
||||
//! the FAT32 USB image (+ its serial-enabled twin for the QEMU run steps),
|
||||
//! and the release ISO — with their check steps. The root build.zig decides
|
||||
//! WHAT ships (the bundled list); this file owns HOW it becomes an image.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// One user binary and its FHS home on the boot volume (and in zig-out).
|
||||
pub const BundledBinary = struct { path: []const u8, binary: std.Build.LazyPath };
|
||||
|
||||
pub const Options = struct {
|
||||
/// The installed/flashable kernel (serial follows the root -Dserial).
|
||||
kernel: *std.Build.Step.Compile,
|
||||
/// The serial-enabled kernel variant the `run-x86-64` image boots.
|
||||
kernel_serial: *std.Build.Step.Compile,
|
||||
/// The UEFI loader (BOOTX64).
|
||||
efi: *std.Build.Step.Compile,
|
||||
/// Every user binary and data file at its FHS path.
|
||||
bundled: []const BundledBinary,
|
||||
};
|
||||
|
||||
/// Wire up the install tree, both FAT32 boot images, the release ISO, and the
|
||||
/// check steps. Returns the serial-enabled FAT image for the QEMU run steps.
|
||||
pub fn addImageSteps(b: *std.Build, options: Options) std.Build.LazyPath {
|
||||
// Everything installs into a FHS-shaped zig-out: it IS the danos filesystem *and*
|
||||
// the boot volume. Each binary lands at its addressed, leaf-collapsed path — the
|
||||
// kernel at zig-out/system/kernel (from system/kernel/kernel.zig), init at
|
||||
// zig-out/system/services/init, and so on (see docs/README.md). The bootloader
|
||||
// then loads these FHS paths off the volume.
|
||||
const kernel_install = b.addInstallArtifact(options.kernel, .{ .dest_dir = .{ .override = .{ .custom = "system" } } });
|
||||
b.getInstallStep().dependOn(&kernel_install.step);
|
||||
|
||||
// UEFI firmware requires the removable-media loader at exactly \EFI\BOOT\BOOTX64.efi,
|
||||
// so that path is fixed by the firmware (it is /boot's EFI stub, conceptually).
|
||||
const efi_install = b.addInstallArtifact(options.efi, .{ .dest_dir = .{ .override = .{ .custom = "EFI/BOOT" } } });
|
||||
b.getInstallStep().dependOn(&efi_install.step);
|
||||
|
||||
// The boot manifest: the FHS path of every bundled binary, one per line. The
|
||||
// EFI loader reads THIS by name and opens each listed path by name — FAT
|
||||
// name lookup is case-insensitive and firmware-portable, unlike directory
|
||||
// ENUMERATION, whose returned names vary by firmware (bare 8.3 entries come
|
||||
// back uppercase on some FAT drivers). The tree walk remains only as the
|
||||
// loader's fallback for hand-assembled sticks without a manifest.
|
||||
var manifest_text: std.ArrayListUnmanaged(u8) = .empty;
|
||||
for (options.bundled) |item| {
|
||||
manifest_text.append(b.allocator, '/') catch @panic("OOM");
|
||||
manifest_text.appendSlice(b.allocator, item.path) catch @panic("OOM");
|
||||
manifest_text.append(b.allocator, '\n') catch @panic("OOM");
|
||||
}
|
||||
const manifest_files = b.addWriteFiles();
|
||||
const manifest_file = manifest_files.add("manifest", manifest_text.items);
|
||||
const manifest_install = b.addInstallFileWithDir(manifest_file, .prefix, "system/manifest");
|
||||
b.getInstallStep().dependOn(&manifest_install.step);
|
||||
|
||||
// The boot capsule: the same bundled list packed into ONE file (v2
|
||||
// initial_ramdisk format), because a single open + sequential read is the
|
||||
// only firmware file I/O shape that is fast everywhere — a per-file tree
|
||||
// walk measured MINUTES on real firmware. The loader tries this first,
|
||||
// then the manifest, then the walk; the running system cannot tell the
|
||||
// difference (it always receives the same in-RAM table). Derived from the
|
||||
// tree in the same build graph, so the two cannot drift.
|
||||
const mk_capsule = b.addSystemCommand(&.{"python3"});
|
||||
mk_capsule.addFileArg(b.path("tools/pack-system-image.py"));
|
||||
const capsule_img = mk_capsule.addOutputFileArg("system.img");
|
||||
for (options.bundled) |item| {
|
||||
mk_capsule.addArg(item.path);
|
||||
mk_capsule.addFileArg(item.binary);
|
||||
}
|
||||
const capsule_install = b.addInstallFile(capsule_img, "boot/system.img");
|
||||
b.getInstallStep().dependOn(&capsule_install.step);
|
||||
|
||||
// Install every bundled binary to its FHS home, so zig-out is a true image of
|
||||
// the filesystem — the same tree make-fat-image.py lays out on the boot volume.
|
||||
for (options.bundled) |item| {
|
||||
const install = b.addInstallFileWithDir(item.binary, .prefix, item.path);
|
||||
b.getInstallStep().dependOn(&install.step);
|
||||
}
|
||||
|
||||
// --- danos-usb.img: the bootable FAT32 USB image ---
|
||||
// Format a real FAT32 image (the in-repo Python builder, no external tools)
|
||||
// holding the EFI stub, the kernel, and the whole /system tree of user
|
||||
// binaries at their FHS paths. QEMU presents this image as a USB mass-storage
|
||||
// device the guest boots from (see run-x86-64 and the test harness), and the
|
||||
// danos fat driver mounts the same image at /volumes/usb.
|
||||
const fat_image = addBootImage(b, options.kernel.getEmittedBin(), options.efi.getEmittedBin(), manifest_file, capsule_img, options.bundled);
|
||||
const fat_image_install = b.addInstallFile(fat_image, "danos-usb.img");
|
||||
b.getInstallStep().dependOn(&fat_image_install.step);
|
||||
|
||||
// The image `run-x86-64` boots: identical to the flashable one but with the
|
||||
// serial log sink compiled in, so a developer always gets the machine-readable
|
||||
// log captured to serial0 — without baking serial into the image users flash.
|
||||
// Built lazily (only when `run-x86-64` is requested), and never installed.
|
||||
const fat_image_serial = addBootImage(b, options.kernel_serial.getEmittedBin(), options.efi.getEmittedBin(), manifest_file, capsule_img, options.bundled);
|
||||
|
||||
// `zig build check-fat-image` — validate the produced image is a real FAT32
|
||||
// with the EFI stub present (the builder's own --verify, no external tools).
|
||||
const check_fat = b.addSystemCommand(&.{"python3"});
|
||||
check_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||
check_fat.addArg("--verify");
|
||||
check_fat.addFileArg(fat_image);
|
||||
const check_fat_step = b.step("check-fat-image", "Verify the FAT32 USB image is valid and bootable");
|
||||
check_fat_step.dependOn(&check_fat.step);
|
||||
|
||||
// --- release-x86-64: danos-x86-64.iso, the flashable release image ---
|
||||
// Wrap the FAT32 boot volume in a hybrid ISO (the in-repo Python builder
|
||||
// again, no xorriso/isohybrid): an ISO9660 whose El Torito EFI boot entry
|
||||
// and MBR ESP partition entry both point at the embedded FAT image. One
|
||||
// file then boots every way release media is consumed — flashed raw to a
|
||||
// USB stick with Etcher or dd, or burned to optical media — while
|
||||
// danos-usb.img stays the raw superfloppy QEMU and the test harness boot.
|
||||
const mk_iso = b.addSystemCommand(&.{"python3"});
|
||||
mk_iso.addFileArg(b.path("tools/make-iso-image.py"));
|
||||
const iso_image = mk_iso.addOutputFileArg("danos-x86-64.iso");
|
||||
mk_iso.addFileArg(fat_image);
|
||||
const iso_install = b.addInstallFile(iso_image, "danos-x86-64.iso");
|
||||
const release_step = b.step("release-x86-64", "Build the flashable x86-64 release ISO (zig-out/danos-x86-64.iso; flash with Etcher or dd)");
|
||||
release_step.dependOn(&iso_install.step);
|
||||
|
||||
// `zig build check-iso-image` — the ISO builder's own --verify (mirroring
|
||||
// check-fat-image): the MBR partition, the El Torito catalog, and the
|
||||
// embedded FAT32 image must all agree.
|
||||
const check_iso = b.addSystemCommand(&.{"python3"});
|
||||
check_iso.addFileArg(b.path("tools/make-iso-image.py"));
|
||||
check_iso.addArg("--verify");
|
||||
check_iso.addFileArg(iso_image);
|
||||
const check_iso_step = b.step("check-iso-image", "Verify the release ISO is a valid hybrid (MBR ESP partition + El Torito EFI entry)");
|
||||
check_iso_step.dependOn(&check_iso.step);
|
||||
|
||||
return fat_image_serial;
|
||||
}
|
||||
|
||||
/// Assemble the bootable FAT32 image (the in-repo Python builder) holding the
|
||||
/// EFI stub, the kernel, and every user binary at its FHS path — the volume's
|
||||
/// /system tree IS the system image; the EFI loader walks it at boot and builds
|
||||
/// the in-RAM initial_ramdisk from it. Factored so the serial-enabled
|
||||
/// `run-x86-64` variant can bundle its own serial kernel while sharing the
|
||||
/// loader and user tree (the loader's boot breadcrumbs and init's heartbeat both
|
||||
/// follow the top-level -Dserial). Returns the image's LazyPath.
|
||||
fn addBootImage(
|
||||
b: *std.Build,
|
||||
kernel_bin: std.Build.LazyPath,
|
||||
efi_bin: std.Build.LazyPath,
|
||||
manifest: std.Build.LazyPath,
|
||||
capsule: std.Build.LazyPath,
|
||||
bundled: []const BundledBinary,
|
||||
) std.Build.LazyPath {
|
||||
const mk_fat = b.addSystemCommand(&.{"python3"});
|
||||
mk_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||
const fat_image = mk_fat.addOutputFileArg("danos-usb.img");
|
||||
mk_fat.addArg("64"); // MiB
|
||||
mk_fat.addArg("EFI/BOOT/BOOTX64.efi");
|
||||
mk_fat.addFileArg(efi_bin);
|
||||
mk_fat.addArg("system/kernel");
|
||||
mk_fat.addFileArg(kernel_bin);
|
||||
mk_fat.addArg("system/manifest");
|
||||
mk_fat.addFileArg(manifest);
|
||||
mk_fat.addArg("boot/system.img");
|
||||
mk_fat.addFileArg(capsule);
|
||||
for (bundled) |item| {
|
||||
mk_fat.addArg(item.path);
|
||||
mk_fat.addFileArg(item.binary);
|
||||
}
|
||||
return fat_image;
|
||||
}
|
||||
+176
@@ -0,0 +1,176 @@
|
||||
//! The QEMU run steps (docs/build-packages-plan.md, phase 3): `run-x86-64`
|
||||
//! boots the serial-enabled FAT image via UEFI/OVMF; `run-x86-64-gpu` adds a
|
||||
//! virtio-gpu adapter for the native-present display path. OVMF firmware is
|
||||
//! probed across distro/OS layouts (-Dovmf-code / -Dovmf-vars override).
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// Wire up the `run-x86-64` and `run-x86-64-gpu` steps around the given
|
||||
/// serial-enabled boot image (the guest boots that self-contained image
|
||||
/// attached as USB storage, not the installed FHS zig-out).
|
||||
pub fn addRunSteps(b: *std.Build, fat_image_serial: std.Build.LazyPath) void {
|
||||
// Firmware lives in different places per OS/distro, so probe the known
|
||||
// layouts (Architecture, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
|
||||
// that exists. Override with -Dovmf-code / -Dovmf-vars if yours is elsewhere.
|
||||
const ovmf_code = b.option(
|
||||
[]const u8,
|
||||
"ovmf-code",
|
||||
"Path to the OVMF_CODE firmware image",
|
||||
) orelse firstExisting(b.graph.io, &.{
|
||||
"/usr/share/edk2/x64/OVMF_CODE.4m.fd", // Architecture
|
||||
"/usr/share/OVMF/OVMF_CODE_4M.fd", // Debian/Ubuntu
|
||||
"/usr/share/OVMF/OVMF_CODE.fd", // older Debian/Ubuntu
|
||||
"/usr/share/edk2-ovmf/x64/OVMF_CODE.fd", // Fedora
|
||||
"/opt/homebrew/share/qemu/edk2-x86_64-code.fd", // macOS Homebrew (Apple Silicon)
|
||||
"/usr/local/share/qemu/edk2-x86_64-code.fd", // macOS Homebrew (Intel)
|
||||
});
|
||||
const ovmf_vars = b.option(
|
||||
[]const u8,
|
||||
"ovmf-vars",
|
||||
"Path to the OVMF_VARS firmware image (a writable copy is made)",
|
||||
) orelse firstExisting(b.graph.io, &.{
|
||||
"/usr/share/edk2/x64/OVMF_VARS.4m.fd", // Architecture
|
||||
"/usr/share/OVMF/OVMF_VARS_4M.fd", // Debian/Ubuntu
|
||||
"/usr/share/OVMF/OVMF_VARS.fd", // older Debian/Ubuntu
|
||||
"/usr/share/edk2-ovmf/x64/OVMF_VARS.fd", // Fedora
|
||||
"/opt/homebrew/share/qemu/edk2-i386-vars.fd", // macOS Homebrew (Apple Silicon)
|
||||
"/usr/local/share/qemu/edk2-i386-vars.fd", // macOS Homebrew (Intel)
|
||||
});
|
||||
|
||||
// The firmware needs to write NVRAM, so give it a writable copy of the vars.
|
||||
const vars_copy = b.addSystemCommand(&.{ "cp", "-f", ovmf_vars });
|
||||
const vars_out = vars_copy.addOutputFileArg("OVMF_VARS.4m.fd");
|
||||
|
||||
// Capture the guest's serial0 (danos's machine-readable log) to the qemu-test
|
||||
// scratch area — a dev/host artifact, kept out of the boot volume we mount.
|
||||
// (/system/logs on the volume belongs to the guest's own logger.) One
|
||||
// timestamped file per run.
|
||||
const log_dir = b.fmt("{s}/qemu-test", .{b.install_path});
|
||||
const make_log_dir = b.addSystemCommand(&.{ "mkdir", "-p", log_dir });
|
||||
|
||||
// --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF ---
|
||||
const run_efi = b.addSystemCommand(&.{
|
||||
"qemu-system-x86_64",
|
||||
"-device",
|
||||
"qemu-xhci,id=xhci",
|
||||
"-device",
|
||||
"usb-mouse,bus=xhci.0",
|
||||
"-device",
|
||||
"usb-kbd,bus=xhci.0",
|
||||
"-machine",
|
||||
"q35",
|
||||
"-m",
|
||||
"128M",
|
||||
"-drive",
|
||||
b.fmt("if=pflash,format=raw,readonly=on,file={s}", .{ovmf_code}),
|
||||
});
|
||||
run_efi.addArg("-drive");
|
||||
run_efi.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
||||
// Boot off the FAT32 USB image: a mass-storage device on the same xHCI bus as
|
||||
// the keyboard and mouse. OVMF finds \EFI\BOOT\BOOTX64.efi on it and boots.
|
||||
// The serial-enabled variant, so serial0 carries the log for this dev boot.
|
||||
run_efi.addArg("-drive");
|
||||
run_efi.addPrefixedFileArg("if=none,id=bootusb,format=raw,file=", fat_image_serial);
|
||||
run_efi.addArgs(&.{
|
||||
"-device",
|
||||
"usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
||||
"-net",
|
||||
"none",
|
||||
// Emulated display advertising 1280x720 as its native (EDID preferred)
|
||||
// resolution, so the kernel's native-resolution switch has something to
|
||||
// find. `-vga none` avoids a second, default adapter.
|
||||
"-vga",
|
||||
"none",
|
||||
"-device",
|
||||
"VGA,edid=on,xres=1280,yres=720",
|
||||
});
|
||||
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
||||
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
|
||||
// We boot the self-contained `fat_image_serial` (added as a file arg above, so
|
||||
// it's already a dependency) — not the installed FHS zig-out — so `run-x86-64`
|
||||
// builds only the serial kernel, never the flashable one. Just make the serial
|
||||
// scratch dir first.
|
||||
run_efi.step.dependOn(&make_log_dir.step);
|
||||
|
||||
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/qemu-test/run-x86-64-serial0-<timestamp>.log");
|
||||
run_efi_step.dependOn(&run_efi.step);
|
||||
|
||||
// --- run-x86-64-gpu: the same boot plus a virtio-gpu adapter ---
|
||||
// The VGA device still supplies the boot (GOP) framebuffer the compositor starts
|
||||
// on; the virtio-gpu function is discovered by the device-manager stack, its
|
||||
// driver announces a shared scanout, and the compositor upgrades off the GOP
|
||||
// floor to fenced, tear-free native presents (docs/display-v2.md).
|
||||
// This is the interactive twin of the `display-native` test case, and 512M
|
||||
// matches it (the whole driver stack + the compositor's surfaces at once).
|
||||
// QEMU shows one head per adapter: pick the virtio-gpu head in the View menu
|
||||
// to watch the native output.
|
||||
const run_gpu = b.addSystemCommand(&.{
|
||||
"qemu-system-x86_64",
|
||||
"-device",
|
||||
"qemu-xhci,id=xhci",
|
||||
"-device",
|
||||
"usb-mouse,bus=xhci.0",
|
||||
"-device",
|
||||
"usb-kbd,bus=xhci.0",
|
||||
"-machine",
|
||||
"q35",
|
||||
"-m",
|
||||
"512M",
|
||||
"-drive",
|
||||
b.fmt("if=pflash,format=raw,readonly=on,file={s}", .{ovmf_code}),
|
||||
});
|
||||
run_gpu.addArg("-drive");
|
||||
run_gpu.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
||||
run_gpu.addArg("-drive");
|
||||
run_gpu.addPrefixedFileArg("if=none,id=bootusb,format=raw,file=", fat_image_serial);
|
||||
run_gpu.addArgs(&.{
|
||||
"-device",
|
||||
"usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
||||
"-net",
|
||||
"none",
|
||||
"-vga",
|
||||
"none",
|
||||
"-device",
|
||||
"VGA,edid=on,xres=1280,yres=720",
|
||||
"-device",
|
||||
"virtio-gpu-pci",
|
||||
});
|
||||
const gpu_serial_log = b.fmt("{s}/run-x86-64-gpu-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
||||
run_gpu.addArgs(&.{ "-serial", b.fmt("file:{s}", .{gpu_serial_log}) });
|
||||
run_gpu.step.dependOn(&make_log_dir.step);
|
||||
|
||||
const run_gpu_step = b.step("run-x86-64-gpu", "Boot in QEMU with a virtio-gpu adapter: the compositor upgrades to fenced (tear-free) native presents; watch the virtio-gpu head in QEMU's View menu");
|
||||
run_gpu_step.dependOn(&run_gpu.step);
|
||||
}
|
||||
|
||||
/// Return the first path in `candidates` that exists on the build host, else the
|
||||
/// first candidate as a fallback so a missing-firmware error still names a
|
||||
/// concrete (and, by convention, the primary) path. Used to locate OVMF firmware
|
||||
/// across distro/OS layouts without configuration.
|
||||
fn firstExisting(io: std.Io, candidates: []const []const u8) []const u8 {
|
||||
for (candidates) |path| {
|
||||
std.Io.Dir.accessAbsolute(io, path, .{}) catch continue;
|
||||
return path;
|
||||
}
|
||||
return candidates[0];
|
||||
}
|
||||
|
||||
/// A UTC timestamp like "20260708-153045", for naming a per-run artifact so
|
||||
/// repeated runs don't clobber each other's logs. Resolved when `zig build`
|
||||
/// runs, which is moments before QEMU launches.
|
||||
fn timestamp(b: *std.Build) []const u8 {
|
||||
const ns = std.Io.Clock.now(.real, b.graph.io).nanoseconds;
|
||||
const secs: u64 = @intCast(@divFloor(ns, std.time.ns_per_s));
|
||||
const es = std.time.epoch.EpochSeconds{ .secs = secs };
|
||||
const yd = es.getEpochDay().calculateYearDay();
|
||||
const md = yd.calculateMonthDay();
|
||||
const ds = es.getDaySeconds();
|
||||
return b.fmt("{d:0>4}{d:0>2}{d:0>2}-{d:0>2}{d:0>2}{d:0>2}", .{
|
||||
yd.year,
|
||||
md.month.numeric(),
|
||||
@as(u32, md.day_index) + 1,
|
||||
ds.getHoursIntoDay(),
|
||||
ds.getMinutesIntoHour(),
|
||||
ds.getSecondsIntoMinute(),
|
||||
});
|
||||
}
|
||||
+143
-93
@@ -3,94 +3,102 @@
|
||||
Notes on how danos boots and draws, written to explain the *why* behind the code
|
||||
rather than restate it. Roughly in the order things happen at runtime:
|
||||
|
||||
1. **[efi.md](efi.md) — EFI / the boot process.** How UEFI firmware finds and
|
||||
1. **[efi.md](os-development/efi.md) — EFI / the boot process.** How UEFI firmware finds and
|
||||
runs the bootloader, what the loader gathers before `ExitBootServices`, how it
|
||||
loads the kernel ELF, and the ABI contract for the jump into the kernel. Start
|
||||
here.
|
||||
2. **[gop.md](gop.md) — the Graphics Output Protocol.** How UEFI exposes graphics
|
||||
2. **[system-image.md](os-development/system-image.md) — system.img, the boot capsule.** The
|
||||
bundled user binaries packed into one file in the initial-ramdisk wire
|
||||
format, because one open + one sequential read is the only file I/O shape
|
||||
firmware is fast at. The trivial container format, the three artifacts one
|
||||
build list derives (tree, manifest, capsule), the loader's three-strategy
|
||||
fallback chain, and the capsule's kernel-side life as both the spawn table
|
||||
and the read-only `/system` mount.
|
||||
3. **[gop.md](os-development/gop.md) — the Graphics Output Protocol.** How UEFI exposes graphics
|
||||
modes (unlike fixed VGA modes), how we detect the monitor's native resolution
|
||||
from EDID and switch to it, and the pixel formats we accept or reject.
|
||||
3. **[framebuffer.md](framebuffer.md) — the framebuffer.** What the linear
|
||||
4. **[framebuffer.md](os-development/framebuffer.md) — the framebuffer.** What the linear
|
||||
framebuffer the loader hands over actually is, and what **pitch** (stride)
|
||||
means versus width — the detail you have to get right to avoid a skewed image.
|
||||
4. **[memory-map.md](memory-map.md) — the memory map.** How the loader learns what
|
||||
5. **[memory-map.md](os-development/memory-map.md) — the memory map.** How the loader learns what
|
||||
physical RAM exists and hands it to the kernel in danos's own neutral format,
|
||||
rather than leaking UEFI's memory descriptors across the boundary.
|
||||
5. **[frame-allocator.md](frame-allocator.md) — the physical frame allocator.** The
|
||||
6. **[frame-allocator.md](os-development/frame-allocator.md) — the physical frame allocator.** The
|
||||
bitmap allocator that hands out and reclaims 4 KiB physical frames from that
|
||||
map — the primitive page tables and the heap are built on.
|
||||
6. **[interrupts.md](interrupts.md) — interrupts and exceptions.** The GDT, IDT and
|
||||
7. **[interrupts.md](os-development/interrupts.md) — interrupts and exceptions.** The GDT, IDT and
|
||||
TSS, the exception stubs, and the handler that reports a CPU fault in red instead
|
||||
of letting it triple-fault into a silent reset.
|
||||
7. **[paging.md](paging.md) — the kernel's page tables.** Building our own 4-level
|
||||
8. **[paging.md](os-development/paging.md) — the kernel's page tables.** Building our own 4-level
|
||||
page tables, identity-mapping the low 4 GiB, and switching CR3 off the firmware's
|
||||
tables onto ours.
|
||||
8. **[device-interrupts.md](device-interrupts.md) — device interrupts.** The Local
|
||||
9. **[device-interrupts.md](device-driver-development/device-interrupts.md) — device interrupts.** The Local
|
||||
APIC and its timer — the kernel's first interrupt that is *handled and returned
|
||||
from*, giving it a heartbeat.
|
||||
9. **[heap.md](heap.md) — the kernel heap.** A growable free-list allocator built on
|
||||
10. **[heap.md](os-development/heap.md) — the kernel heap.** A growable free-list allocator built on
|
||||
the VMM, exposed as a `std.mem.Allocator` so std containers work — dynamic
|
||||
allocation for the kernel.
|
||||
10. **[scheduling.md](scheduling.md) — the scheduler.** Fixed-priority preemptive
|
||||
11. **[scheduling.md](os-development/scheduling.md) — the scheduler.** Fixed-priority preemptive
|
||||
multitasking: kernel threads, the context switch, O(1) priority selection, and
|
||||
blocking (sleep, wait queues) — the leap to a running system.
|
||||
11. **[ipc.md](ipc.md) — inter-process communication.** Bounded blocking
|
||||
12. **[ipc.md](device-driver-development/ipc.md) — inter-process communication.** Bounded blocking
|
||||
message-passing channels, then synchronous call/reply between *processes* over
|
||||
endpoints — the backbone the microkernel's isolated servers talk over.
|
||||
12. **[syscall.md](syscall.md) — system calls.** How ring 3 asks the kernel for
|
||||
13. **[syscall.md](os-development/syscall.md) — system calls.** How ring 3 asks the kernel for
|
||||
something: the `syscall`/`sysret` fast path, the trap frame, and why the table is
|
||||
deliberately tiny. The numbers are a **private** ABI — [vdso.md](vdso.md) designs
|
||||
deliberately tiny. The numbers are a **private** ABI — [vdso.md](os-development/vdso.md) designs
|
||||
the public boundary that will hide them.
|
||||
13. **[vfs-protocol.md](vfs-protocol.md) — the VFS wire protocol.** The language-neutral
|
||||
14. **[vfs-protocol.md](file-system-development/vfs-protocol.md) — the VFS wire protocol.** The language-neutral
|
||||
byte-level spec of the file protocol spoken over IPC: request/reply headers,
|
||||
the operation table, mount routing, and the append-only evolution rules — the
|
||||
first IPC protocol documented as public ABI.
|
||||
14. **[drivers.md](drivers.md) — writing a driver.** The payoff: a driver is an
|
||||
15. **[drivers.md](device-driver-development/drivers.md) — writing a driver.** The payoff: a driver is an
|
||||
ordinary ring-3 process that claims a device, maps its registers, and **sleeps
|
||||
until its hardware interrupts it**. The claim is the capability; `irq_ack` is the
|
||||
unmask.
|
||||
15. **[driver-model.md](driver-model.md) — buses, classes and host controllers.** How
|
||||
16. **[driver-model.md](device-driver-development/driver-model.md) — buses, classes and host controllers.** How
|
||||
real driver stacks factor into three shapes and how families share code. The
|
||||
three primitives it proposed are long since built (M13 capability passing,
|
||||
M14 DMA + barriers, M15 MSI), and the driver *contract* on top of them —
|
||||
hello, supervision, restart — is built too (device-manager.md, M18).
|
||||
16. **[usb-hub.md](usb-hub.md) — USB hubs.** Built (M22): why hub topology is handled
|
||||
17. **[usb-hub.md](device-driver-development/usb-hub.md) — USB hubs.** Built (M22): why hub topology is handled
|
||||
*inside* the `usb-xhci-bus` driver rather than a separate hub class driver — a
|
||||
device behind a hub is reached by the **controller**, programmed with a route
|
||||
string in its slot context — plus the compound-hub reality (a USB 3.0 hub is
|
||||
physically two hubs) and detection via the hub's status-change interrupt endpoint.
|
||||
17. **[process-management.md](process-management.md) — process management.** The
|
||||
18. **[process-management.md](os-development/process-management.md) — process management.** The
|
||||
microkernel's `ps`/`kill`/SIGCHLD: enumerate as a table snapshot, the
|
||||
supervision link as the kill authority, and child-exit notifications over the
|
||||
same endpoints IRQs arrive on.
|
||||
18. **[process-lifecycle.md](process-lifecycle.md) — the process lifecycle.** Built
|
||||
19. **[process-lifecycle.md](os-development/process-lifecycle.md) — the process lifecycle.** Built
|
||||
(M17): signals over IPC as the one lifecycle vocabulary every process speaks — the
|
||||
POSIX.1-1990 words with message delivery instead of stack hijack, the stable
|
||||
`runtime.process` interface, exit reasons, published exit events any stateful
|
||||
`process` module interface, exit reasons, published exit events any stateful
|
||||
service can subscribe to (the VFS releasing dead clients' handles), and the two
|
||||
iron rules (cleanup is the kernel's job; kill is not a signal).
|
||||
19. **[device-manager.md](device-manager.md) — the device manager.** Built (M18,
|
||||
20. **[device-manager.md](device-driver-development/device-manager.md) — the device manager.** Built (M18,
|
||||
through the app surface): the
|
||||
tree, the matcher, and the supervisor. Tree structure lives in the manager,
|
||||
authority stays in the kernel; bus drivers report what they see; drivers are
|
||||
restarted through the lifecycle vocabulary — the plan that turns
|
||||
[resilience.md](resilience.md)'s restart goal into increments.
|
||||
20. **[input.md](input.md) — the input module.** Broadcasting input events (keyboard,
|
||||
[resilience.md](os-development/resilience.md)'s restart goal into increments.
|
||||
21. **[input.md](device-driver-development/input.md) — the input module.** Broadcasting input events (keyboard,
|
||||
mouse, joystick): why a synchronous rendezvous can't fan out to many listeners, the
|
||||
asynchronous `ipc_send` primitive built to fix it, and the per-device subscribe/publish
|
||||
service layered on top.
|
||||
21. **[display.md](display.md) — the display service.** The display half of the GUI
|
||||
22. **[display.md](device-driver-development/display.md) — the display service.** The display half of the GUI
|
||||
track: a user-space compositor that owns the framebuffer, composes a layer stack into
|
||||
a double buffer, and presents it. Why GOP and the PCI display device are two views of
|
||||
one controller, the device-node + write-combining handoff, and what flicker-free buys
|
||||
that tear-free doesn't. Plan: [display-plan.md](display-plan.md). **v2** (complete) makes
|
||||
that tear-free doesn't. Plan: [display-plan.md](device-driver-development/display-plan.md). **v2** (complete) makes
|
||||
scanout a pluggable backend — GOP floor + a native virtio-gpu driver, hot-attached, with
|
||||
runtime mode-set, EDID, fenced vsync presents, and restart re-attach:
|
||||
[display-v2.md](display-v2.md), plan [display-v2-plan.md](display-v2-plan.md). Looking
|
||||
further out, two research snapshots survey what a *native* driver for real GPU silicon
|
||||
would take as another `.scanout` backend: [nvidia-gpus.md](nvidia-gpus.md) (RTX 3060 /
|
||||
Ampere) and [intel-igpu.md](intel-igpu.md) (Intel iGPU).
|
||||
22. **[halting.md](halting.md) — halting.** Why a kernel can't just "exit", and
|
||||
[display-v2.md](device-driver-development/display-v2.md), plan [display-v2-plan.md](device-driver-development/display-v2-plan.md). Looking
|
||||
further out, three research snapshots survey what a *native* driver for real GPU silicon
|
||||
would take as another `.scanout` backend: [nvidia-gpus.md](device-driver-development/nvidia-gpus.md) (RTX 3060 /
|
||||
Ampere), [amd-gpus.md](device-driver-development/amd-gpus.md) (RX 6600 / RDNA2), and [intel-igpu.md](device-driver-development/intel-igpu.md)
|
||||
(Intel iGPU).
|
||||
23. **[halting.md](os-development/halting.md) — halting.** Why a kernel can't just "exit", and
|
||||
how `while (true) hlt` parks the CPU safely once there's nothing left to do.
|
||||
|
||||
Start with the north star:
|
||||
@@ -100,28 +108,28 @@ Start with the north star:
|
||||
**resilience** (restartable components). Win condition: runs on the author's PC and
|
||||
both Raspberry Pis, ideally with a GUI. Real-time is an option to explore, not a
|
||||
requirement. The *why* that shapes everything below.
|
||||
- **[resilience.md](resilience.md) — resilience.** A design note (not built yet) on
|
||||
- **[resilience.md](os-development/resilience.md) — resilience.** A design note (not built yet) on
|
||||
fault isolation + live restart — the reincarnation-server + capability model that
|
||||
makes "if I break it, I can restart it" real. danos's core motivation.
|
||||
- **[zig-self-hosting.md](zig-self-hosting.md) — running Zig on danos.** A design note
|
||||
(not built yet) on making danos a real Zig target (`-target x86_64-danos`) and
|
||||
eventually running the compiler on it. The key realisation: Zig 0.16 reduces an OS
|
||||
port to **one seam** (`std.os.danos`), so we build `runtime.os` (→ that seam) plus a
|
||||
thin `runtime.fs`, retire the `posix` shim, and follow a phased path to
|
||||
port to **one seam** (`std.os.danos`), so we build an `os` seam module (→ that seam) plus
|
||||
the thin `file-system` module, retire the `posix` shim, and follow a phased path to
|
||||
`zig build-exe hello.zig` running on danos — **not** Linux-ABI emulation.
|
||||
- **[threading.md](threading.md) — threads, the std-shaped way.** **Built** (M1–M6):
|
||||
`runtime.Thread` mirrors `std.Thread`'s API (spawn/join/detach, Mutex/Condition/
|
||||
- **[threading.md](os-development/threading.md) — threads, the std-shaped way.** **Built** (M1–M6):
|
||||
the `thread` module's `Thread` mirrors `std.Thread`'s API (spawn/join/detach, Mutex/Condition/
|
||||
Semaphore) over a **private** thread ABI — several tasks sharing one address space via
|
||||
a `thread_spawn` syscall, futex-backed blocking, address-space refcounting. Why it's the
|
||||
native type and not literal `std.Thread` (the [private ABI](syscall.md)), and why
|
||||
threads stay a narrow opt-in against the [resilience](resilience.md) default. Build
|
||||
plan + gates: [threading-plan.md](threading-plan.md).
|
||||
- **[vdso.md](vdso.md) — the vDSO, the public system-call boundary.** A design note
|
||||
native type and not literal `std.Thread` (the [private ABI](os-development/syscall.md)), and why
|
||||
threads stay a narrow opt-in against the [resilience](os-development/resilience.md) default. Build
|
||||
plan + gates: [threading-plan.md](os-development/threading-plan.md).
|
||||
- **[vdso.md](os-development/vdso.md) — the vDSO, the public system-call boundary.** A design note
|
||||
(not built yet) on keeping `abi.zig` genuinely private: a kernel-supplied, C-ABI
|
||||
entry blob mapped into every process as the *only* way into the kernel — so the
|
||||
syscall numbers can be renumbered or randomised at will, and Rust/C binaries get a
|
||||
stable boundary without danos growing a dynamic linker. danos's public ABI = the
|
||||
vDSO + the documented IPC wire protocols ([vfs-protocol.md](vfs-protocol.md) first).
|
||||
vDSO + the documented IPC wire protocols ([vfs-protocol.md](file-system-development/vfs-protocol.md) first).
|
||||
|
||||
Cutting across all of these:
|
||||
|
||||
@@ -129,69 +137,69 @@ Cutting across all of these:
|
||||
hardware needed to run danos: minimum specs (UEFI x86-64, ACPI, PCIe ECAM,
|
||||
xHCI, ~128 MiB RAM) grounded in what the boot path actually assumes, plus a
|
||||
plain-language guide matching Intel/AMD CPU generations by name.
|
||||
- **[release-iso.md](release-iso.md) — the release ISO.** The flashable boot
|
||||
- **[release-iso.md](os-development/release-iso.md) — the release ISO.** The flashable boot
|
||||
media: `zig build release-x86-64` wraps the FAT32 boot volume in a hybrid ISO
|
||||
(MBR ESP partition + El Torito EFI entry, one embedded image) that Etcher/dd
|
||||
flash to USB or a burner writes to disc — built by an in-repo pure-Python
|
||||
tool, like the FAT image itself.
|
||||
- **[architecture.md](architecture.md) — the architecture split.** How CPU-specific code is kept
|
||||
- **[architecture.md](os-development/architecture.md) — the architecture split.** How CPU-specific code is kept
|
||||
behind a build-time `arch` module so the generic kernel never names x86_64,
|
||||
leaving room for other systems (e.g. an AArch64 Raspberry Pi) later.
|
||||
- **[arm.md](arm.md) — ARM targets.** The Raspberry Pi landscape the arch split is
|
||||
- **[arm.md](os-development/arm.md) — ARM targets.** The Raspberry Pi landscape the arch split is
|
||||
aiming at: `arm` (32-bit, Pi Zero W) vs `aarch64` (64-bit, Pi 3-5), UEFI vs
|
||||
device-tree boot, and what each layer needs.
|
||||
- **[discovery.md](discovery.md) — device discovery.** A design note on learning what
|
||||
- **[discovery.md](os-development/discovery.md) — device discovery.** A design note on learning what
|
||||
hardware exists via ACPI (x86) or device tree (ARM) behind one neutral device model —
|
||||
when to build it, and how to keep it architecture-agnostic.
|
||||
- **[acpi.md](acpi.md) — finding the ACPI tables.** The concrete x86 locator chain:
|
||||
- **[acpi.md](os-development/acpi.md) — finding the ACPI tables.** The concrete x86 locator chain:
|
||||
how the loader captures the **RSDP**, hands its physical address across in `BootInformation`,
|
||||
and how the platform derives the **RSDT/XSDT** from it and walks the SDTs — plus the
|
||||
live event side (the SCI, the power button, GPE/Notify) the ring-3 acpi service runs.
|
||||
- **[power.md](power.md) — the power service.** System power as a domain-named
|
||||
- **[power.md](os-development/power.md) — the power service.** System power as a domain-named
|
||||
service: button/lid/battery events published to subscribers, and init's orderly
|
||||
shutdown composing the [lifecycle](process-lifecycle.md) stop sequence with an ACPI
|
||||
shutdown composing the [lifecycle](os-development/process-lifecycle.md) stop sequence with an ACPI
|
||||
S5 write. Firmware-neutral — a PSCI backend drops in on ARM.
|
||||
- **[timers.md](timers.md) — timers and time.** The ring-3 surface for reading the
|
||||
- **[timers.md](os-development/timers.md) — timers and time.** The ring-3 surface for reading the
|
||||
clock and waiting: why `now()` is a syscall rather than a service, and the one-shot
|
||||
timer notification (`timer_bind`) that gives supervisors a timed wait — built on the
|
||||
LAPIC heartbeat and calibrated TSC of [device-interrupts.md](device-interrupts.md).
|
||||
- **[smp.md](smp.md) — multiple cores.** A design/research note on how microkernels
|
||||
LAPIC heartbeat and calibrated TSC of [device-interrupts.md](device-driver-development/device-interrupts.md).
|
||||
- **[smp.md](os-development/smp.md) — multiple cores.** A design/research note on how microkernels
|
||||
(L4, seL4) handle SMP — big kernel lock vs per-CPU vs multikernel — and how the
|
||||
right choice depends on whether danos is chasing real-time or resilience.
|
||||
- **[coding-standards.md](coding-standards.md) — coding standards.** The naming rule the
|
||||
tree follows: non-acronyms are spelled out in full (`message`, not `msg`), files are
|
||||
`kebab-case`, code follows Zig's case conventions, and the handful of exceptions
|
||||
(POSIX/C ABI names, `init`/`len`/`ptr`, acronyms).
|
||||
- **[sysv.md](sysv.md) — the calling convention.** What "the kernel is SysV" means,
|
||||
- **[sysv.md](os-development/sysv.md) — the calling convention.** What "the kernel is SysV" means,
|
||||
and why the loader→kernel boundary has to pin it (the RDI-vs-RCX handoff).
|
||||
- **[testing.md](testing.md) — testing.** How the kernel is tested by booting it in
|
||||
QEMU and asserting on its serial output — reproducibly, and structured so the
|
||||
same tests run across architectures.
|
||||
- **[logging.md](logging.md) — logging.** The multi-sink diagnostic log (serial,
|
||||
- **[logging.md](os-development/logging.md) — logging.** The multi-sink diagnostic log (serial,
|
||||
0xE9 debugcon, file later) kept separate from the framebuffer display, plus the
|
||||
robustness path: optional framebuffer, POST-code checkpoints, and a persistent
|
||||
panic breadcrumb so the kernel survives — and can be diagnosed — with no output.
|
||||
|
||||
## How the pieces relate
|
||||
|
||||
The boot flow ties them together: UEFI runs the loader ([efi.md](efi.md)), which
|
||||
queries the **GOP** to pick a graphics mode ([gop.md](gop.md)), hands the kernel a
|
||||
**framebuffer** to draw into ([framebuffer.md](framebuffer.md)) and a **memory
|
||||
map** of physical RAM ([memory-map.md](memory-map.md)); the kernel turns that map
|
||||
into a **frame allocator** ([frame-allocator.md](frame-allocator.md)), installs
|
||||
its **descriptor tables** so CPU faults are caught ([interrupts.md](interrupts.md)),
|
||||
builds its own **page tables** and switches onto them ([paging.md](paging.md)),
|
||||
brings up the **heap** for dynamic allocation ([heap.md](heap.md)), starts the
|
||||
**scheduler** ([scheduling.md](scheduling.md)) and the **timer** that preempts it
|
||||
([device-interrupts.md](device-interrupts.md)) — with tasks blocking, sleeping and
|
||||
passing messages over **[IPC](ipc.md)** channels — runs, its CPU-specific bits
|
||||
behind the [architecture](architecture.md) boundary, and when idle, or on a panic, it **halts**
|
||||
([halting.md](halting.md)).
|
||||
The boot flow ties them together: UEFI runs the loader ([efi.md](os-development/efi.md)), which
|
||||
queries the **GOP** to pick a graphics mode ([gop.md](os-development/gop.md)), hands the kernel a
|
||||
**framebuffer** to draw into ([framebuffer.md](os-development/framebuffer.md)) and a **memory
|
||||
map** of physical RAM ([memory-map.md](os-development/memory-map.md)); the kernel turns that map
|
||||
into a **frame allocator** ([frame-allocator.md](os-development/frame-allocator.md)), installs
|
||||
its **descriptor tables** so CPU faults are caught ([interrupts.md](os-development/interrupts.md)),
|
||||
builds its own **page tables** and switches onto them ([paging.md](os-development/paging.md)),
|
||||
brings up the **heap** for dynamic allocation ([heap.md](os-development/heap.md)), starts the
|
||||
**scheduler** ([scheduling.md](os-development/scheduling.md)) and the **timer** that preempts it
|
||||
([device-interrupts.md](device-driver-development/device-interrupts.md)) — with tasks blocking, sleeping and
|
||||
passing messages over **[IPC](device-driver-development/ipc.md)** channels — runs, its CPU-specific bits
|
||||
behind the [architecture](os-development/architecture.md) boundary, and when idle, or on a panic, it **halts**
|
||||
([halting.md](os-development/halting.md)).
|
||||
|
||||
Above that line the microkernel proper begins: **discovery** ([discovery.md](discovery.md),
|
||||
[acpi.md](acpi.md)) learns what hardware exists, ring-3 processes ask the kernel for
|
||||
things through the small **[syscall](syscall.md)** table, isolated servers reach each
|
||||
other over IPC **endpoints** ([ipc.md](ipc.md)), and a **[driver](drivers.md)** claims
|
||||
Above that line the microkernel proper begins: **discovery** ([discovery.md](os-development/discovery.md),
|
||||
[acpi.md](os-development/acpi.md)) learns what hardware exists, ring-3 processes ask the kernel for
|
||||
things through the small **[syscall](os-development/syscall.md)** table, isolated servers reach each
|
||||
other over IPC **endpoints** ([ipc.md](device-driver-development/ipc.md)), and a **[driver](device-driver-development/drivers.md)** claims
|
||||
a device, maps its registers, and sleeps until the hardware interrupts it — which is
|
||||
the whole reason for the arrangement ([vision.md](vision.md)).
|
||||
|
||||
@@ -200,7 +208,7 @@ the whole reason for the arrangement ([vision.md](vision.md)).
|
||||
danos is a **monorepo of sub-projects**. Each service or driver is a directory that is
|
||||
its own Zig module — it can hold as many files as it needs, and other sub-projects
|
||||
reach it *by module name*, never by a path into its files. The source tree deliberately
|
||||
**mirrors the runtime FHS** ([danos-file-system-hierarchy-FSH.md](danos-file-system-hierarchy-FSH.md)):
|
||||
**mirrors the runtime file-system hierarchy** ([file-system-hierarchy.md](file-system-development/file-system-hierarchy.md)):
|
||||
what you see under `system/` in the source is what a running danos represents under
|
||||
`/system`.
|
||||
|
||||
@@ -208,11 +216,12 @@ what you see under `system/` in the source is what a running danos represents un
|
||||
name.** `system/services/init/` contains `init.zig` (its root), and produces a binary
|
||||
addressed as **`system/services/init`** — the repeated leaf resolves away:
|
||||
|
||||
| Source (root file) | Addressed as (module / binary / FHS path) |
|
||||
| Source (root file) | Addressed as (module / binary / hierarchy path) |
|
||||
|----------------------------------------|--------------------------------------------|
|
||||
| `system/services/init/init.zig` | `system/services/init` → `/system/services/init` |
|
||||
| `system/drivers/ps2-bus/ps2-bus.zig` | `system/drivers/ps2-bus` → `/system/drivers/ps2-bus` |
|
||||
| `library/runtime/runtime.zig` | `library/runtime` (the `runtime` module) |
|
||||
| `test/system/services/vfs-test/vfs-test.zig` | `test/system/services/vfs-test` → `/test/system/services/vfs-test` |
|
||||
| `library/device/pci/pci.zig` | `library/device/pci` (the `pci` module) |
|
||||
|
||||
In **source**, a sub-project is a directory so it can hold many files — the entry is
|
||||
`fat/fat.zig`, beside it `fat/engine.zig`, `fat/on-disk.zig`, and so on. When
|
||||
@@ -226,32 +235,69 @@ A sub-project's extra files are reached through the module, never as separate pa
|
||||
```
|
||||
system/ → /system danos's own internals (the self-representation)
|
||||
boot-handoff.zig the loader↔kernel contract (the `boot-handoff` module)
|
||||
abi.zig the private kernel↔runtime syscall ABI (the `abi` module)
|
||||
parameters.zig initial-ramdisk.zig vfs-protocol.zig shared contracts
|
||||
abi.zig the private kernel↔userspace syscall ABI (the `abi` module)
|
||||
parameters.zig initial-ramdisk.zig shared contracts
|
||||
kernel/ IPC, memory, scheduling, the VFS root, the private syscall dispatch
|
||||
architecture/x86_64/ the `architecture` module (never named by generic code)
|
||||
devices/ the device model /system/devices reflects (+ aml/)
|
||||
device-abi.zig the device wire types (the `device-abi` module)
|
||||
devices-broker.zig the syscall-facing device table
|
||||
platform.zig acpi.zig fdt.zig device-model.zig firmware discovery + the kernel's
|
||||
device model — the implementation of what /system/devices reflects
|
||||
drivers/ pci-bus/ ps2-bus/ usb-xhci-bus/ one sub-project per driver → /system/drivers
|
||||
services/ init/ fat/ device-manager/ system servers → /system/services (fat/ holds
|
||||
fat.zig, engine.zig, on-disk.zig)
|
||||
library/ → /lib libraries, one sub-directory each
|
||||
runtime/ the danos-native runtime + file API (fs) — the stable application ABI
|
||||
kernel/ the danos-native system library (kernel32-style): the syscall
|
||||
surface split by concern — ipc, memory (heap/dma/shared-memory),
|
||||
process, time, logging, file-system, thread, service, plus the
|
||||
system-call stubs and the start/root entry shim
|
||||
device/ device code by domain — mmio/ model/ pci/ usb/ acpi/ driver/
|
||||
block/ — each a shareable data module (device-abi, pci-class,
|
||||
usb-abi/ids, acpi-ids) plus a logic module (mmio, pci, usb, aml,
|
||||
driver — the device-access + device-manager-hello client)
|
||||
client/ userspace service clients (display, input) — a program's view of
|
||||
a service, layered over that service's protocol
|
||||
protocol/ driver↔service wire contracts (vfs block display scanout input
|
||||
power device-manager usb-transfer), one module per directory
|
||||
boot/ → /boot the loaders
|
||||
tools/ test/ host-side build + QEMU test harness
|
||||
test/ → /test the test tree: the QEMU harness (qemu_test.py, host-side)
|
||||
system/services/ beside the on-image test fixtures — vfs-test/ thread-test/
|
||||
crash-test/ … — whose repo path IS their boot-volume path
|
||||
(/test/system/services/<name>)
|
||||
build-support/ the danos build API (build-time only, nothing on the image):
|
||||
the shared user-binary recipe + default-import wiring every
|
||||
build file consumes (docs/build-packages-plan.md)
|
||||
build/ root-build helpers: image assembly (images.zig) + the QEMU
|
||||
run steps (qemu.zig)
|
||||
tools/ host-side build scripts
|
||||
```
|
||||
|
||||
A sub-project exposes its **public interface as a module**: the `usb-xhci-bus` driver
|
||||
owns the USB transfer protocol (`usb-transfer-protocol.zig`, the `usb-transfer-protocol`
|
||||
module), which `runtime.usb` imports by name — the USB class drivers reach the
|
||||
protocol through that wrapper; `block` exposes its protocol (`block-protocol`) the
|
||||
same way. The VFS wire protocol is the one that outgrew its
|
||||
sub-project: the VFS root moved into the kernel (`system/kernel/vfs.zig`), so the
|
||||
protocol lives as a shared contract at `system/vfs-protocol.zig` (the `vfs-protocol`
|
||||
module), which the runtime's file API (`runtime.fs`) imports by name.
|
||||
**Builds are packages** (docs/build-packages-plan.md): each `library/` domain owns a
|
||||
`build.zig`/`build.zig.zon` exporting its modules (with a standalone `zig build test`),
|
||||
every binary directory is a ~15-line package build, and the root `build.zig`
|
||||
orchestrates — the kernel + loader, what ships, and the aggregate test step — with
|
||||
image assembly in `build/images.zig` and the QEMU run steps in `build/qemu.zig`.
|
||||
|
||||
**Wire protocols live in `library/protocol/`**, one module per directory
|
||||
(`library/protocol/vfs/vfs-protocol.zig` is the `vfs-protocol` module), imported by module
|
||||
name. A protocol is the seam between a low-level driver and the higher-level service it
|
||||
serves — block ↔ the filesystem, a scanout driver ↔ the compositor — so both sides depend
|
||||
on the contract, not on each other, and the contract belongs to neither sub-project. A
|
||||
client module may *wrap* one for application convenience (the `file-system` module over
|
||||
`vfs-protocol`, and the `block`, `display`, `input` clients over theirs), but the protocol
|
||||
module is the boundary — a client re-exports no protocol, it imports it by name. A driver's
|
||||
private wire to its *hardware* (virtio-gpu's command set) is not a service seam and stays a
|
||||
driver-private file, beside the transport that reaches the same device.
|
||||
|
||||
**Device code lives in `library/device/<domain>/`**, grouped by what it is about (pci, usb,
|
||||
acpi, and the cross-cutting device model) and split by dependency weight: a data module of
|
||||
enums and wire types that is `std`-only and cheap for anyone to import, and a logic module
|
||||
that needs `mmio` or IPC. This is what keeps the microkernel out of device business — it
|
||||
imports exactly one `library/` module, `device-abi` (the descriptor types its broker
|
||||
marshals across the syscall boundary), and nothing with logic or a taxonomy in it. That
|
||||
lone pure-data import is the only edge from `system/kernel/` into `library/`.
|
||||
|
||||
There is **no POSIX/C compatibility layer today**: danos programs do file I/O through the
|
||||
danos-native `runtime.fs` (open/read/write/list over the VFS). A hand-rolled POSIX shim
|
||||
danos-native `file-system` module (open/read/write/list over the VFS). A hand-rolled POSIX shim
|
||||
(`library/posix/`) was retired as premature — the real POSIX/C surface will come later
|
||||
from the `std.os.danos` seam (and, eventually, musl) when danos becomes a Zig target (see
|
||||
[zig-self-hosting.md](zig-self-hosting.md)). When it does, the foreign-ABI naming
|
||||
@@ -264,8 +310,8 @@ exception in [coding-standards.md](coding-standards.md) applies to that seam.
|
||||
| Boot methods (one per way of booting the kernel) | `boot/` — `efi.zig` (UEFI) → `BOOTX64.efi` |
|
||||
| Kernel entry, panic, bring-up | `system/kernel/kernel.zig` |
|
||||
| Loader↔kernel handoff (`BootInformation`, `Framebuffer`, `MemoryMap`, VM layout) | `system/boot-handoff.zig` |
|
||||
| Private kernel↔runtime syscall ABI (`SystemCall`, mmap prot flags, `page_size`) — the runtime speaks it, not apps | `system/abi.zig` |
|
||||
| Device wire types (`DeviceDescriptor`, `DeviceClass`, …) | `system/devices/device-abi.zig` |
|
||||
| Private kernel↔userspace syscall ABI (`SystemCall`, mmap prot flags, `page_size`) — the system library speaks it, not apps | `system/abi.zig` |
|
||||
| Device wire types (`DeviceDescriptor`, `DeviceClass`, …) | `library/device/model/device-abi.zig` |
|
||||
| Physical frame allocator | `system/kernel/pmm.zig` |
|
||||
| Kernel heap (`std.mem.Allocator`) | `system/kernel/heap.zig` |
|
||||
| Scheduler (fixed-priority preemptive; blocking, wait queues) | `system/kernel/scheduler.zig` |
|
||||
@@ -273,15 +319,19 @@ exception in [coding-standards.md](coding-standards.md) applies to that seam.
|
||||
| IPC channels between kernel threads (message passing) | `system/kernel/ipc.zig` |
|
||||
| IPC endpoints: cross-address-space call/reply, handles, notifications | `system/kernel/ipc-synchronous.zig` |
|
||||
| User processes: ELF loading, address spaces, the syscall table | `system/kernel/process.zig` |
|
||||
| VFS root: mount table + kernel-served nodes (`fs_resolve`/`fs_node`); wire protocol in `system/vfs-protocol.zig` | `system/kernel/vfs.zig` |
|
||||
| VFS root: mount table + kernel-served nodes (`fs_resolve`/`fs_node`); wire protocol in `library/protocol/vfs/vfs-protocol.zig` | `system/kernel/vfs.zig` |
|
||||
| Device tree + claim capability + `device_register` containment | `system/kernel/devices-broker.zig` |
|
||||
| IRQ-as-IPC: routing a device interrupt to a driver's endpoint | `system/kernel/irq.zig` |
|
||||
| Hardware discovery (ACPI/device tree) behind one neutral device model | `system/devices/` |
|
||||
| Hardware discovery (ACPI/device tree) behind one neutral device model | `system/kernel/platform.zig` |
|
||||
| Framebuffer text console (mirrors to serial) | `system/kernel/console.zig` |
|
||||
| In-kernel test cases | `system/kernel/tests.zig` |
|
||||
| Arch-specific kernel code (`halt`, GDT/IDT/TSS, exception + interrupt stubs, page tables, APIC/IO-APIC/timer, serial, linker script) | `system/kernel/architecture/x86_64/` |
|
||||
| danos-native runtime (`runtime`): syscall wrappers, heap, IPC, device access, the file API (`fs`) — the stable application ABI | `library/runtime/` |
|
||||
| danos-native system library (kernel32-style): the syscall surface by concern — `ipc`, `memory`, `process`, `time`, `logging`, `file-system`, `thread`, `service` — the stable application ABI | `library/kernel/` |
|
||||
| Service clients (a program's view of a service) and device clients | `library/client/` (display, input), `library/device/driver` |
|
||||
| System services (init, the `fat` filesystem, the device-manager) | `system/services/` |
|
||||
| Device drivers, one sub-project each (`pci-bus`, `ps2-bus`, `usb-xhci-bus` bus drivers) | `system/drivers/` |
|
||||
| Build + `run-x86-64` (QEMU/OVMF) + `release-x86-64` (the flashable ISO) | `build.zig` |
|
||||
| On-image test fixtures for the QEMU cases (`vfs-test`, `crash-test`, `thread-test`, …) → `/test/system/services` | `test/system/services/` |
|
||||
| Build orchestration (kernel + loader, what ships, the aggregate test step) | `build.zig` (root; the shared user-binary recipe is `build-support/`, and each `library/` domain + binary package carries its own `build.zig`) |
|
||||
| Image assembly + `release-x86-64` (the flashable ISO) | `build/images.zig` |
|
||||
| `run-x86-64` / `run-x86-64-gpu` (QEMU/OVMF) | `build/qemu.zig` |
|
||||
| QEMU integration test harness | `test/qemu_test.py` |
|
||||
|
||||
@@ -0,0 +1,179 @@
|
||||
# Plan: packages — hierarchical builds for libraries and binaries
|
||||
|
||||
**Status: complete** (branch `claude/build-packages-plan-174144`). Phase 0
|
||||
(`build-support`), phase 1 (all six library domains), phase 2 (every binary —
|
||||
the pci-bus pilot first, then services, drivers, and test fixtures in waves;
|
||||
multi-binary directories like ps2-bus and usb-hid are one package exporting
|
||||
several artifacts, and the acpi/fdt discovery pair each export an artifact
|
||||
named "discovery" that the root's -Ddiscovery picks between), and phase 3 (the
|
||||
root split into `build/images.zig` + `build/qemu.zig`; the root `build.zig` is
|
||||
~460 lines of orchestration, down from ~1,250). Every phase landed green: unit
|
||||
tests, the QEMU suite at parity with main, boot-image file list unchanged.
|
||||
The `lazyDependency` payoff (What-this-buys #4) is in too: the /test fixtures
|
||||
and the unselected discovery package are lazy — a build loads and compiles
|
||||
only what it ships. And imports are exact: the pre-wired default set is gone;
|
||||
every binary names precisely the modules its source imports and carries only
|
||||
those domains in its manifest (rule 1 below).
|
||||
|
||||
## Why
|
||||
|
||||
`build.zig` was ~1,250 lines, growing by three hand-written stanzas per binary;
|
||||
at a driver per device family that does not scale. More fundamentally: in one
|
||||
monolithic build every binary compiles against library *source*, so a library
|
||||
interface break is silently absorbed by whoever edits everything in one commit —
|
||||
the interface never has to be honest. danos is about isolation; the build should
|
||||
mirror it.
|
||||
|
||||
A **package** here is a build-time unit only — a directory owning a `build.zig`
|
||||
(recipe: what it exports, how to test it) and a `build.zig.zon` (manifest: name
|
||||
+ dependencies). Binaries remain fully static freestanding ELFs; packages change
|
||||
who declares what, not what links to what. Source code is untouched: `@import`
|
||||
uses module names (`"pci"`, `"service"`) exactly as today — only build files
|
||||
know where anything lives.
|
||||
|
||||
## Target shape
|
||||
|
||||
```
|
||||
build-support/ package: the danos build API (userBinary(), defaultImports(), targets)
|
||||
library/kernel/ package "kernel": modules abi, ipc, service, memory, process, logging, time, ... (depends on protocol)
|
||||
library/device/ package "device": modules driver, pci, usb-abi, model, ... (depends on kernel, protocol, csv)
|
||||
library/protocol/ package "protocol": the wire protocols
|
||||
library/client/ package "client" (depends on kernel, protocol)
|
||||
library/csv/ package "csv"
|
||||
library/xkeyboard-config/ package "xkeyboard-config"
|
||||
system/services/<name>/ one package per binary: ~15-line build.zig + zon
|
||||
system/drivers/<name>/ one package per binary
|
||||
build.zig (root) orchestrator: dependency() per binary, image assembly, QEMU, test steps
|
||||
```
|
||||
|
||||
The three shared contracts: `boot-handoff` stays a root module (only the
|
||||
loader↔kernel pair speaks it); `abi` is exported by the kernel package from
|
||||
`../../system/abi.zig` (the source stays with the kernel; userspace's one view
|
||||
of it lives in the package, so every consumer names the same module instance);
|
||||
`device-abi` is exported by device. Reaching outside the package root means the
|
||||
kernel package is valid only as an in-repo path dependency — it could never be
|
||||
fetched by hash — which is fine: path dependencies are the only way any of
|
||||
these packages is consumed.
|
||||
|
||||
Rules:
|
||||
|
||||
- **Imports are exact and per binary.** A binary's build.zig names precisely
|
||||
the modules its source `@import`s — the moral equivalent of a C file's
|
||||
include list — and its zon names only the domains those modules come from
|
||||
(plus `build-support` and `kernel`, which is implicit in every binary: the
|
||||
root shim and user link script live there). Nothing is pre-wired: an
|
||||
undeclared `@import` is a compile error, and build-support resolves each
|
||||
name by searching the packages the zon declares — the domains' own
|
||||
addModule exports are the single statement of who owns what, with no name
|
||||
table anywhere to drift. Availability
|
||||
never meant bloat — Zig only compiles what a program actually imports — but
|
||||
exactness makes the declared interface honest and machine-checked.
|
||||
- **Modules export source, not artifacts** — each consumer compiles libraries
|
||||
with its own flags, so per-binary optimization choices keep working; Zig's
|
||||
cache deduplicates.
|
||||
- **Zon paths are relative and that is accepted.** Binaries sit exactly three
|
||||
levels deep, so the `../../../` prefix is a constant idiom; a library-domain
|
||||
move is a rare, already-breaking event fixed by one sed across manifests, and
|
||||
a stale path fails loudly before anything compiles.
|
||||
- **Cross-cutting build changes live in `build-support` only** — that is the
|
||||
contract that keeps per-binary build files declarative.
|
||||
|
||||
## What this buys
|
||||
|
||||
1. Library interfaces become machine-checked: a consumer can only import what
|
||||
it declared — per binary, down to the single module — and each domain's zon
|
||||
declares what it needs (claim-before-touch, applied to source). A keyboard
|
||||
driver carries `xkeyboard-config` in its manifest; nothing else does.
|
||||
2. Each library domain gets a standalone `zig build test` — runtime-library
|
||||
stability testing in isolation.
|
||||
3. Adding a binary = adding a directory (source + two small files), not editing
|
||||
three places in a 1,250-line file.
|
||||
4. `lazyDependency` lets an image target build only what it ships: the /test
|
||||
fixtures resolve only under -Dtest-case, and only the -Ddiscovery-selected
|
||||
discovery package ever loads.
|
||||
|
||||
## Phases
|
||||
|
||||
Each phase ends green: `zig build test` passes (88/88 QEMU) and the boot
|
||||
image's file list is unchanged. Byte-identical binaries are expected but not
|
||||
required (module reorganization can perturb symbol order); file list is the
|
||||
hard gate.
|
||||
|
||||
**Phase 0 — `build-support`.** Extract `addUserBinary`/`addThreadedUserBinary`,
|
||||
the freestanding target setup, and the default-import wiring into the
|
||||
`build-support` package. Root build consumes it; nothing else moves. This is
|
||||
the cross-cutting-change home, so it lands first.
|
||||
|
||||
**Phase 1 — library domains become packages.** In dependency order: `protocol`
|
||||
and `csv` (the roots) → `kernel` (depends on protocol: file-system speaks
|
||||
vfs-protocol) → `device`, `client`; `xkeyboard-config` stands alone. Each gets
|
||||
build.zig + zon + a standalone test step (client's is empty until its modules
|
||||
grow host tests — kept for uniformity, since the root aggregate depends on
|
||||
every domain's test step). The root build swaps its `createModule` calls for
|
||||
`b.dependency("<domain>").module("<name>")`. **No binary moves in this phase**
|
||||
— the root build is the pilot consumer, which proves the packages without
|
||||
touching 30 binaries.
|
||||
|
||||
**Phase 2 — binaries become packages, in waves.** The template was shaken out
|
||||
by the pci-bus pilot (see Status). Wave A: services (done). Wave B: the
|
||||
remaining drivers (done). Wave C: test fixtures (done). Root build shrank to
|
||||
orchestration per wave. init's `-Dserial` heartbeat flag rides a dependency
|
||||
option; a directory with several binaries (ps2-bus, usb-hid) is one package
|
||||
exporting several artifacts.
|
||||
|
||||
**Phase 3 — root cleanup (done).** What remained of the root build split into
|
||||
`build/images.zig` (the FHS install tree, boot manifest + capsule, FAT32
|
||||
images, release ISO, check steps) and `build/qemu.zig` (the run steps + OVMF
|
||||
probing), imported by a short root `build.zig`.
|
||||
|
||||
**Afterwards** (outside this plan): the intel-uhd-graphics-750 driver is
|
||||
(re)created as a greenfield package. The new-driver checklist's build step
|
||||
(docs/device-driver-development/new-driver-checklist.md, step 2) is already
|
||||
rewritten against the package template.
|
||||
|
||||
## Execution notes (the finished shape)
|
||||
|
||||
- The shared recipe lives in `build-support/build.zig`: `userBinary` (what
|
||||
every binary package calls; each named import resolves by searching the
|
||||
packages the binary's zon declares) and `programModule` (for per-binary
|
||||
addOptions modules). The `start` root shim and `user.ld` are named through the kernel
|
||||
package (Dependency.path).
|
||||
- Adding a binary = adding a directory with source + a ~15-line build.zig +
|
||||
zon (copy any existing binary package, e.g.
|
||||
`system/drivers/pci-bus/build.zig`) listing exactly the modules the source
|
||||
imports and the domains they come from, then one dependency + one bundled
|
||||
entry in the root build.zig and one zon line.
|
||||
- The boot-tree array in the root (search `"etc/init.csv"` or
|
||||
`.getEmittedBin()`) is the image file list — the authoritative comparison
|
||||
target for any future build change.
|
||||
- Package unit tests live in each package's own `test` step; the root
|
||||
aggregate depends on every test-bearing package's step, so `zig build test`
|
||||
at the root still runs everything.
|
||||
|
||||
Verification per phase:
|
||||
|
||||
- Unit tests: `zig build test`.
|
||||
- QEMU integration suite: `python3 test/qemu_test.py` (docs/testing.md; the
|
||||
full suite, all cases must pass).
|
||||
- Image file list: the boot-tree array is the source of truth — snapshot it
|
||||
(paths only) before phase 0 and diff after each phase; `zig build
|
||||
check-fat-image` must also stay green.
|
||||
|
||||
Context a fresh session should read first: this doc, docs/testing.md,
|
||||
docs/coding-standards.md (kebab-case names, no abbreviations), and the
|
||||
`userBinary`/`userBinaryFromImports` bodies in build-support/build.zig. Commit
|
||||
style: no Co-Authored-By trailers.
|
||||
|
||||
## Risks / notes
|
||||
|
||||
- Zig version churn: the package API (`b.dependency`, zon schema) has moved
|
||||
between releases; the work pins against the repo's current Zig and any
|
||||
upgrade lands separately, never mid-phase.
|
||||
- The QEMU size-check tests hardcode source paths (e.g. virtio-gpu protocol
|
||||
struct sizes) — they moved into their binaries' packages with their waves,
|
||||
discharging the carry-along obligation.
|
||||
- Doc updates ride each phase: docs/README.md (repo layout + source map),
|
||||
docs/device-driver-development/new-driver-checklist.md (step 2) and
|
||||
devices-csv.md ("Adding a driver"), and the docs that cite the build recipe
|
||||
(driver-model.md, threading.md, system-requirements.md) reference build
|
||||
shapes that keep changing.
|
||||
@@ -0,0 +1,205 @@
|
||||
# The C library compatibility layer
|
||||
|
||||
A design note and milestone plan for **libdanos-c** — the mini C library that lets
|
||||
`zig cc` cross-compile C programs for danos. It is milestone **P0** of
|
||||
[python-on-danos-milestones.md](python-on-danos-milestones.md), expanded here the
|
||||
way [character-devices-and-tty.md](character-devices-and-tty.md) expands P1.
|
||||
CPython is the driving consumer, but the layer is general: any portable C program
|
||||
within its surface should build.
|
||||
|
||||
## What it is — and the three things it is not
|
||||
|
||||
The deliverable is a **sysroot**: a set of C headers plus a static `libdanos-c.a`,
|
||||
handed to `zig cc -target x86_64-freestanding-none` via `-isystem` and linked into
|
||||
every C binary. Three explicit non-goals keep it small:
|
||||
|
||||
- **Not a musl port.** Whole-musl assumes Linux syscall semantics at its bottom
|
||||
(the door the Zig roadmap deferred, twice now). We *lift* musl's pure-computation
|
||||
source files and *write* a danos-native bottom — see the layer split below.
|
||||
- **Not full POSIX — *yet*.** Stage 1's surface is "what CPython's minimal
|
||||
configuration and ordinary portable C need" — roughly 100–150 functions — and
|
||||
at that stage absence is a *feature*: configure scripts probe and adapt, and a
|
||||
linker error is honest. But the end state is a **full C compatibility layer**
|
||||
(see "The road to full coverage" below); the absence table is a schedule of
|
||||
arrivals, not a wall.
|
||||
- **Not a second runtime.** The library is a thin C-ABI re-spelling of the same
|
||||
danos-native surface `runtime` already provides. It contains no policy of its
|
||||
own; when the Zig track's `runtime.os` seam is authored, the libc bottom
|
||||
re-targets it near-mechanically — the fourth appearance of the roadmap's "same
|
||||
surface" symmetry.
|
||||
|
||||
One scoping rule sits above all three — the **size doctrine**: this layer serves
|
||||
**applications only**. The kernel and the system services never link libdanos-c;
|
||||
they stay danos-native Zig over `runtime`, small and static, because leanness is
|
||||
an operating-system property. Applications have their own budget and may be as
|
||||
big as they need to be. The libc is how big software *lands on* danos, never how
|
||||
danos itself is built.
|
||||
|
||||
## The layer split: lift the mathematics, write the plumbing
|
||||
|
||||
The realization that makes 100–150 functions tractable: a libc is two very
|
||||
different kinds of code, and the hard kind is portable.
|
||||
|
||||
| Layer | Contents | Source |
|
||||
|-------|----------|--------|
|
||||
| **Pure computation** | `string.h`/`memcpy` family, all of libm, `strtod`/`dtoa`, `strtol`, `qsort`, `ctype` tables, `gmtime` calendar math, the `printf`/`scanf` engines, `setjmp` (a dozen instructions of x86-64 asm) | **Lift from musl**, vendored under `library/c/third-party/musl/` (MIT; files compile standalone) |
|
||||
| **OS plumbing** | fds (`open`/`read`/`write`/`close`/`lseek`/`stat`/`getcwd`/`chdir`/`isatty`), `mmap`/`munmap`, clocks, `exit`, `getenv`, `getentropy` | **Write in Zig**, exporting C ABI over the `runtime` syscall + VFS client surface |
|
||||
| **The middle** | `malloc` over danos `mmap` (simple free-list; CPython's arenas sit above), `FILE*` buffering, `errno` | **Write in Zig** (small, danos-shaped) |
|
||||
| **Entry** | `crt0`: the existing danos entry shim ([sysv.md](os-development/sysv.md)) bridged to C `main(argc, argv, envp)`, `environ` initialised, `exit` flushing stdio | **Write** |
|
||||
|
||||
Two liftings deserve their own line because getting them wrong is silent
|
||||
corruption rather than a linker error:
|
||||
|
||||
- **`strtod`/float formatting.** Python's float `repr` guarantees shortest
|
||||
round-trip; that property lives entirely in these routines. musl's are correct;
|
||||
an improvised one would be subtly wrong for years. Lift, never write.
|
||||
- **The stdio engines.** musl's `vfprintf`/`vfscanf` are self-contained around
|
||||
its `FILE` abstraction (function-pointer read/write slots), so the whole
|
||||
formatted-I/O engine lifts too — we implement only the fd-backed slots
|
||||
(`__stdio_write`-shaped) and the buffering glue.
|
||||
|
||||
## Header policy
|
||||
|
||||
Hand-write the headers as danos's own minimal set rather than importing musl's
|
||||
(musl's are entangled with Linux ABI details), borrowing declarations freely.
|
||||
Freestanding compiler headers (`stdint.h`, `stddef.h`, `stdarg.h`, `stdbool.h`,
|
||||
`float.h`, `limits.h`) come from clang via `zig cc` — do not duplicate them.
|
||||
`errno.h` values are the danos errno enum re-spelled with POSIX names; there is no
|
||||
Linux numbering to be compatible with, so the enum is the truth.
|
||||
|
||||
Deliberate absences, and their planned arrivals — this table is the
|
||||
compatibility matrix, and "the road to full coverage" below is the schedule
|
||||
that empties it:
|
||||
|
||||
| Absent | Arrives with |
|
||||
|--------|--------------|
|
||||
| `pthread.h` | the post-P5 pthread subset over `thread_spawn`/futex — but see the risk below |
|
||||
| real `signal.h` (beyond no-op `signal()`/`raise` stubs) | M17 signals-over-IPC in the libc |
|
||||
| `dlfcn.h` | [dynamic-libraries.md](dynamic-libraries.md) D1 |
|
||||
| `fork`/`exec*`/`wait*` | P5 exposes danos spawn as `posix_spawn`; `fork` itself never (see below) |
|
||||
| `socket.h` | a future networking track |
|
||||
| locale beyond `"C"` | stage 3 evaluation (CPython is UTF-8-mode happy without it) |
|
||||
| pipes (`pipe()`) | P5 process-control cluster |
|
||||
|
||||
## The road to full coverage
|
||||
|
||||
The layer grows in three stages; only stage 1 is a current milestone (P0), but
|
||||
the stages exist so stage-1 decisions never have to be unmade:
|
||||
|
||||
- **Stage 1 — CPython-minimal** (P0, the slicing below): ~100–150 functions,
|
||||
static-only, absences honest.
|
||||
- **Stage 2 — the danos-complete layer**: the full hosted C11 standard library,
|
||||
plus every POSIX facility danos semantics support, landing as its enabling
|
||||
milestone lands — pipes and `posix_spawn` at P5, real signals at M17, the
|
||||
pthread subset after P5, `dlfcn.h` at
|
||||
[dynamic-libraries](dynamic-libraries.md) D1, sockets with networking. Stage 2
|
||||
is not one milestone but the standing rule that **every system capability
|
||||
gets its C spelling when it ships**, so the matrix above drains as the OS
|
||||
grows.
|
||||
- **Stage 3 — ecosystem grade**: the point where "portable C program" generally
|
||||
means "builds on danos" (autotools-style probing included). Reaching it is
|
||||
mostly stage 2 compounding, plus the long tail (locale, wide-char,
|
||||
`fnmatch`/`glob`/`regex` — the last three lift from musl like the rest). At
|
||||
this stage, re-evaluate hand-grown-vs-musl-port once with real data; the
|
||||
standing recommendation remains danos-native — musl's bottom assumes Linux
|
||||
syscall semantics, and by stage 3 the danos bottom exists and is tested —
|
||||
with musl continuing as the quarry for computation code.
|
||||
|
||||
Two boundaries are permanent and worth stating at every stage: **`fork` never
|
||||
comes** — danos is a spawn-shaped OS, and `fork`'s address-space-duplication
|
||||
semantics are hostile to everything from capabilities to threads; software that
|
||||
hard-requires `fork` (not `posix_spawn`) stays off the platform. And the
|
||||
**public ABI stays the vDSO + IPC protocols** — a full libc is a compatibility
|
||||
*layer*, not a second stable system ABI.
|
||||
|
||||
## Milestone slicing
|
||||
|
||||
1. **sysroot-skeleton** — layout under `library/c/` (a build package:
|
||||
`include/`, Zig sources, vendored musl subtree); `crt0`; string/mem +
|
||||
`ctype` lifted; a `build.zig` step making C binaries first-class targets.
|
||||
*Test:* a C program using only computation links and runs in QEMU
|
||||
(`c-hello` printing via a raw `write` extern to `debug_write`).
|
||||
2. **fd-plumbing** — `errno`; open/read/write/close/lseek/stat/unlink/mkdir/
|
||||
rename over the `runtime` VFS client; `getcwd`/`chdir`/`getenv`/
|
||||
`getentropy` arriving as P1 lands them (stubbed truthfully until then:
|
||||
`getenv` empty, `getentropy` `ENOSYS`). *Test:* QEMU `c-file-io` — create,
|
||||
write, reopen, read back, stat size + mtime through FAT.
|
||||
3. **malloc** — free-list allocator over danos `mmap`; `calloc`/`realloc`/
|
||||
`free`; alignment guarantees documented. *Test:* host + QEMU allocator
|
||||
torture (interleaved sizes, realloc growth, alignment asserts).
|
||||
4. **stdio** — `FILE*`, buffering modes, the lifted printf/scanf engines wired
|
||||
to the fd slots; `snprintf` family; stdin/stdout/stderr over fd 0/1/2.
|
||||
*Test:* host round-trip suite for format engines (especially `%.17g`
|
||||
float round-trip); QEMU `c-stdio` cooked-line echo once P1's console exists.
|
||||
5. **mathematics-and-time** — libm lifted wholesale; `strtod`/`strtol`;
|
||||
`clock_gettime` (monotonic + realtime over `clock`/`wall_clock`);
|
||||
`gmtime`/`mktime`/`strftime` (UTC only — no timezone database);
|
||||
`setjmp`/`longjmp`; `qsort`/`bsearch`; `abort`/`assert`. *Test:* host
|
||||
`strtod`/`dtoa` vectors against known-hard cases; QEMU `c-time` sanity
|
||||
against the wall clock.
|
||||
|
||||
Slices 1, 3, 4-host, and 5-host have **no dependency on P1** and can start
|
||||
immediately; slice 2 and the QEMU halves interleave with P1 as it lands.
|
||||
|
||||
**Exit for the layer as a whole** (= P0's exit): `c-hello` and `c-file-io` green
|
||||
in the QEMU suite, and the host-side computation tests green — at which point P2
|
||||
(CPython configure) becomes the layer's real integration test.
|
||||
|
||||
## Testing strategy: two targets, on purpose
|
||||
|
||||
The computation layer is target-independent, so it is unit-tested **on the host**
|
||||
(built for the host triple, compared against the host libc's answers —
|
||||
thousands of cheap oracle checks for `strtod`, `printf`, libm edge cases). The
|
||||
plumbing layer only means anything **on danos**, so it is tested in the QEMU
|
||||
suite like every other subsystem. Keeping the split explicit stops the slow-QEMU
|
||||
suite from absorbing tests that a host `zig test` runs in milliseconds.
|
||||
|
||||
## Risks and gotchas
|
||||
|
||||
- **CPython's configure may insist on pthreads.** WASI-class targets build
|
||||
threadless, but verify this *first* in P2 bring-up; the fallback is a
|
||||
truthfully-single-threaded `pthread.h` stub set (create returns `EAGAIN`,
|
||||
mutexes are no-ops — valid when only one thread can exist). Decide from
|
||||
evidence, not assumption.
|
||||
- **`long double` is x87 80-bit on x86-64.** musl's libm handles it, but keep
|
||||
CPython away from it (`configure` uses `double` throughout by default);
|
||||
don't hand-write anything touching x87.
|
||||
- **errno is a contract, not a convention.** The Zig plumbing must map every
|
||||
`runtime` error to a POSIX name consistently — CPython turns errno into
|
||||
exception types (`FileNotFoundError` is `ENOENT`). One table, tested.
|
||||
- **`malloc` alignment**: 16-byte minimum on x86-64 (SSE spills in
|
||||
compiled C). The free-list must guarantee it from day one; retrofitting
|
||||
alignment bugs out of an allocator is misery.
|
||||
- **Vendoring discipline.** The musl subtree is lift-only — never edited in
|
||||
place (patches live beside it if ever needed), pinned to one musl release,
|
||||
with the file list documented so a version bump is a re-copy, not an
|
||||
archaeology dig.
|
||||
- **stdio buffering vs. crashes.** Buffered stdout + a crashing program eats
|
||||
output — the classic debugging trap. `stderr` stays unbuffered (per C
|
||||
standard) and `exit`/`abort` flush; document that `_exit` does not.
|
||||
|
||||
## Decisions needing sign-off
|
||||
|
||||
- **Lift-from-musl for all pure computation** (vendored, pinned, unedited) rather
|
||||
than writing or porting whole-musl.
|
||||
- **Hand-written danos-native headers**; danos errno values are the numbering.
|
||||
- **`library/c/` as a build package** producing both the sysroot and the
|
||||
first-class C-binary build step.
|
||||
- The **deliberate-absence table** as the living compatibility matrix, drained
|
||||
by the three-stage road above — with exactly one permanent "never": `fork`.
|
||||
- **Full coverage as the end state** (stage 3), reached by the standing rule
|
||||
that every system capability ships with its C spelling — not by a musl port.
|
||||
|
||||
## Related
|
||||
|
||||
- [python-on-danos-milestones.md](python-on-danos-milestones.md) — this is P0.
|
||||
- [dynamic-libraries.md](dynamic-libraries.md) — ships in this sysroot
|
||||
(`dlfcn.h` + the loader) once its D1 lands.
|
||||
- [python-on-danos.md](python-on-danos.md) — the design note that scoped the
|
||||
layer.
|
||||
- [character-devices-and-tty.md](character-devices-and-tty.md) — P1; supplies
|
||||
the console that makes stdio interactive.
|
||||
- [zig-self-hosting.md](zig-self-hosting.md) — the `runtime.os` seam the
|
||||
plumbing layer will re-target when it exists.
|
||||
- [os-development/sysv.md](os-development/sysv.md) — the entry stack `crt0`
|
||||
bridges.
|
||||
@@ -0,0 +1,174 @@
|
||||
# Character devices, the console, and the tty question
|
||||
|
||||
A design note for the **stream** half of the device world. danos has block devices
|
||||
(the USB storage service behind the FAT mount) but no character devices — and three
|
||||
tracks now need them at once: the terminal application, Zig self-hosting Phase 1
|
||||
("wire fd 0/1/2 to a console byte stream"), and [Python on danos](python-on-danos.md)
|
||||
Phase 1. This note settles what a character device *is* on danos before any of those
|
||||
tracks build one.
|
||||
|
||||
## The Unix picture, briefly
|
||||
|
||||
Unix splits devices in two: **block devices** are seekable arrays of fixed-size
|
||||
sectors (disks); **character devices** are unseekable byte streams (keyboards,
|
||||
serial ports, terminals, `/dev/null`, entropy). A **tty** is the canonical
|
||||
character device — a byte stream plus a *line discipline* (echo, line buffering,
|
||||
erase handling, Ctrl-C-to-signal) that lives in the kernel. A **pty** is a pair of
|
||||
character devices (master/slave) that exists so a *userspace* program — a terminal
|
||||
emulator — can impersonate terminal hardware to the kernel's in-kernel line
|
||||
discipline.
|
||||
|
||||
The identification asked for and confirmed: yes, tty and pty are character
|
||||
devices in this taxonomy.
|
||||
|
||||
## The realization that shapes everything: danos already has the mechanism
|
||||
|
||||
A Unix character device is an in-kernel dispatch table: major/minor numbers route
|
||||
`read()`/`write()` to a driver. danos already has exactly that dispatch — the VFS:
|
||||
`fs_resolve` routes a path to a mounted backend service, and `Operation.mount`
|
||||
attaches a backend *endpoint* at a prefix. What is missing is not a device model;
|
||||
it is **one node kind with stream semantics**. And the protocol already reserved
|
||||
it: `NodeKind.character_device = 2` sits unimplemented in
|
||||
[vfs-protocol.zig](../library/protocol/vfs/vfs-protocol.zig), exactly like
|
||||
`symbolic_link`.
|
||||
|
||||
So the design is small:
|
||||
|
||||
**A character device on danos is a VFS node, served by an ordinary service over
|
||||
the existing VFS wire protocol, whose read/write have stream semantics.**
|
||||
|
||||
No device numbers, no `/dev` special casing, no new syscalls, no new protocol —
|
||||
a service is reachable at a path, clients open it with `runtime.fs` like any
|
||||
file, and the node kind says what it is. (Since the protocol namespace landed
|
||||
in design, that path is `/protocol/console` — a protocol node, see
|
||||
[os-development/protocol-namespace.md](os-development/protocol-namespace.md) —
|
||||
rather than a mounted device file; the stream semantics below are unchanged.)
|
||||
|
||||
### Stream semantics (the actual contract change)
|
||||
|
||||
For a node whose kind is `character_device`:
|
||||
|
||||
- **`offset` is ignored** on read and write; there is no seek position. (`lseek`,
|
||||
when the C layer exists, returns `ESPIPE`.)
|
||||
- **Reads block** until at least one byte is available, then return what is there —
|
||||
**short reads are normal**, not EOF. A zero-length read reply means the stream
|
||||
is closed (hangup), not end-of-file-at-size.
|
||||
- **`FileStatus.size` is 0** and means nothing; `mtime` may be 0.
|
||||
- Writes may be short if the service's buffer is full; the client loops as it
|
||||
already must for the 256-byte message cap.
|
||||
|
||||
This is a semantics note on existing operations, not a wire change — the `Request`
|
||||
and `Reply` structs are untouched. The one true protocol addition is a **`control`
|
||||
operation** (appended to `Operation`, values stable): a typed request the stream's
|
||||
service interprets. Deliberately *not* an `ioctl` grab-bag — the control payloads
|
||||
are enumerated per protocol, starting with the terminal set below.
|
||||
|
||||
## The first character device is a pseudo-device
|
||||
|
||||
The first device is deliberately **not hardware**: an in-memory **loopback** — a
|
||||
byte queue served over the stream contract, where bytes written to one end are
|
||||
read from the other. It is the reference implementation of the semantics above
|
||||
(blocking reads, short reads, hangup on close, the `control` round-trip), it
|
||||
tests deterministically with no QEMU serial scripting, and it keeps hardware off
|
||||
the critical path entirely. `null` and `zero` come along nearly for free as
|
||||
degenerate cases. This is a decision, not a convenience: the dead-COM1 boot bug
|
||||
on real hardware already proved serial cannot be assumed present or alive, so
|
||||
**nothing in this milestone writes to COM1**. (A serial-backed stream node can
|
||||
exist *later* as one more optional backend for headless debugging; it is on
|
||||
nobody's critical path.)
|
||||
|
||||
The loopback is also not throwaway — it is the seed of P5's `pipe()`, which is
|
||||
the same object with two fds.
|
||||
|
||||
## The console service
|
||||
|
||||
A `console` service owns the line discipline — **in userspace**, where a
|
||||
microkernel wants it, not in the kernel as Unix has it:
|
||||
|
||||
- **The discipline is a pure library first**: bytes and key events in, bytes
|
||||
out, no I/O of its own — developed and host-tested against in-memory buffers,
|
||||
then shared verbatim between the console and the future terminal application.
|
||||
- **Input**: subscribes to keyboard `InputEvent` IPC (the structured events that
|
||||
exist today) and cooks them into bytes. Cooked mode is the default: echo, line
|
||||
buffering, backspace/erase, so a line is delivered on Enter. Raw mode delivers
|
||||
bytes as they come (the REPL's line editor and any full-screen program need it).
|
||||
- **Output is a pluggable sink**, and the stream contract is independent of it:
|
||||
the bring-up sink is in-memory (readable back by tests, mirrored to the boot
|
||||
log), and the real one is the framebuffer text renderer when the display
|
||||
track's font work lands.
|
||||
- **Control set** (the `control` payloads): mode raw/cooked, echo on/off, and
|
||||
window-size query — the minimal termios. Ctrl-C-to-signal joins when M17
|
||||
signals-over-IPC lands; until then Ctrl-C is just a byte.
|
||||
- Mounts itself at `/device/console` as a `character_device` node.
|
||||
|
||||
**fd 0/1/2** then stop being special: spawn hands the child three open handles
|
||||
(console by default; anything else if the parent chooses), and `runtime`'s fd
|
||||
table maps 0/1/2 to them. `isatty` is simply "does `status` say
|
||||
`character_device`" — no side channel needed.
|
||||
|
||||
## The pty answer: there is no pty
|
||||
|
||||
The pty exists in Unix *because the line discipline is in the kernel* — userspace
|
||||
terminal emulators need a kernel gadget to impersonate hardware. On danos the
|
||||
terminal emulator is already a userspace server, so the pair collapses:
|
||||
|
||||
**The graphical terminal application serves the VFS stream protocol itself and
|
||||
hands its own endpoints to the children it spawns as their fd 0/1/2.**
|
||||
|
||||
The terminal *is* the console service for its children — same protocol, same
|
||||
control set, same line discipline code (shared as a library with the boot
|
||||
console). No master/slave device pair, no `/dev/pts`, no new kernel object. When
|
||||
CPython arrives, the libc's `isatty`/read/write see a character device and are
|
||||
none the wiser; when xonsh eventually wants job control, that lands as control
|
||||
messages + M17 signals, still with no pty object.
|
||||
|
||||
What this costs: programs that *specifically* manipulate Unix ptys
|
||||
(`os.openpty()`, `pexpect`-style tools) have no direct equivalent — the danos
|
||||
answer is "spawn the child yourself with your own stream endpoints," which is the
|
||||
same capability with less machinery. Accepted.
|
||||
|
||||
## Milestone slicing
|
||||
|
||||
1. **pseudo-devices** — VFS honors `character_device` semantics end to end;
|
||||
`Operation.control` added; the in-memory **loopback** (plus `null`/`zero`)
|
||||
as the first device. QEMU test: one client writes, another reads — open,
|
||||
offsetless read/write, blocking read, short read, hangup on close, control
|
||||
round-trip. No hardware anywhere.
|
||||
2. **console-service** — the line-discipline library (host-tested, pure) plus
|
||||
the console composing keyboard `InputEvent`s with an in-memory output sink;
|
||||
mounted at `/device/console`. QEMU test injects key events and reads cooked
|
||||
lines and raw bytes back through the sink.
|
||||
3. **fd-inheritance** — spawn passes 0/1/2 handles; `runtime` fd table; `isatty`
|
||||
via `status`; existing binaries' stdout migrates from `debug_write` to fd 1
|
||||
(the logger keeps its own path).
|
||||
4. **terminal-as-server** — deferred to the terminal application milestone
|
||||
(Python track P3): the terminal reuses the discipline library and serves its
|
||||
children directly.
|
||||
|
||||
Steps 1–3 are exactly the shared seam that Zig self-hosting Phase 1 and Python
|
||||
Phase 1 both list; neither track repeats them.
|
||||
|
||||
## Decisions needing sign-off
|
||||
|
||||
- **No pty object; the terminal serves its children directly** (the section
|
||||
above) — the load-bearing simplification.
|
||||
- **`control` as an enumerated, typed operation** rather than an ioctl-style
|
||||
opaque pass-through.
|
||||
- **Line discipline in userspace services** (console + terminal, shared library),
|
||||
never in the kernel.
|
||||
|
||||
## Related
|
||||
|
||||
- [python-on-danos.md](python-on-danos.md) — consumes this as its Phase 1.
|
||||
- [zig-self-hosting.md](zig-self-hosting.md) — ditto ("stdio as fds").
|
||||
- [file-system-development/vfs-protocol.md](file-system-development/vfs-protocol.md) —
|
||||
the wire protocol this note extends.
|
||||
- [file-system-development/file-system-hierarchy.md](file-system-development/file-system-hierarchy.md)
|
||||
— the tree the console surfaces in.
|
||||
- [os-development/protocol-namespace.md](os-development/protocol-namespace.md) —
|
||||
supersedes this note's device-node naming: the console lands as a protocol
|
||||
(`/protocol/console`, a protocol node), not a `/dev`-style device file. The
|
||||
stream semantics designed here (line discipline, cooked/raw modes) carry over
|
||||
unchanged.
|
||||
- [device-driver-development/input.md](device-driver-development/input.md) — the
|
||||
`InputEvent` stream the console cooks.
|
||||
@@ -67,7 +67,7 @@ Three, and only three.
|
||||
|
||||
**This exception is scoped to a file that *is* a foreign ABI, and nothing else.**
|
||||
danos has no such file today: the old `library/posix/` compatibility shim was retired
|
||||
once its callers moved to the danos-native `runtime.fs`, since a hand-rolled POSIX
|
||||
once its callers moved to the danos-native `file_system`, since a hand-rolled POSIX
|
||||
layer is premature until danos actually needs it (see
|
||||
[zig-self-hosting.md](zig-self-hosting.md)). The exception will apply again to the
|
||||
`std.os.danos` seam when danos becomes a real Zig target — that module *is* the C-ABI
|
||||
@@ -97,8 +97,9 @@ Three, and only three.
|
||||
|
||||
That's all — no Unix-abbreviation exception. The source directories are full words
|
||||
(`system`, `library`, not `src`/`lib`), and there is no daemon `d` suffix: a driver
|
||||
lives in `system/drivers/` and a service in `system/services/`, so the *location*
|
||||
already says what it is. Encoding the role in the name too (`busd`, `fatd`) is
|
||||
lives in `system/drivers/`, a service in `system/services/`, and a test fixture in
|
||||
`test/system/services/` (the repo path *is* its path on the boot volume), so the
|
||||
*location* already says what it is. Encoding the role in the name too (`busd`, `fatd`) is
|
||||
redundant — the program is just `ps2-bus`, `fat`. Don't put in a name what its directory
|
||||
already tells you.
|
||||
|
||||
@@ -135,15 +136,16 @@ Within those spelling rules, follow Zig's own conventions:
|
||||
`notify_badge_bit`.
|
||||
|
||||
**File names are `kebab-case`.** A file named for a multi-word thing hyphenates it:
|
||||
`device-tree.zig`, `ipc-synchronous.zig`, `vfs-protocol.zig`, `devices-broker.zig`. A
|
||||
`device-model.zig`, `ipc-synchronous.zig`, `vfs-protocol.zig`, `devices-broker.zig`. A
|
||||
single word or acronym needs no hyphen: `scheduler.zig`, `paging.zig`, `apic.zig`,
|
||||
`idt.zig`. (The module *alias* a file is imported under still follows the code
|
||||
conventions above — `snake_case` — because it's an identifier, not a filename.)
|
||||
|
||||
**A sub-project's entry point repeats its directory's name** — `init/init.zig`,
|
||||
`runtime/runtime.zig`, `ps2-bus/ps2-bus.zig` — and the sub-project is addressed by the
|
||||
*directory* (`system/services/init`, `library/runtime`), with the repeated leaf
|
||||
resolving away. See the repository-layout section of [README.md](README.md).
|
||||
`pci/pci.zig`, `ps2-bus/ps2-bus.zig` — and the sub-project is addressed by the
|
||||
*directory* (`system/services/init`, `library/device/pci`,
|
||||
`test/system/services/vfs-test`), with the repeated leaf resolving away. See the
|
||||
repository-layout section of [README.md](README.md).
|
||||
|
||||
## Named values, not magic numbers
|
||||
|
||||
|
||||
@@ -1,125 +0,0 @@
|
||||
# DanOS Filesystem Hierarchy Standard (DFHS)
|
||||
|
||||
Most modern Unix and Unix-like operating systems follow the FHS. DanOS has its own FHS structure which extends the unix FHS. Root path resolution is provided by the kernel-resident VFS root (`fs_resolve`, `system/kernel/vfs.zig`); mounted filesystem servers serve the subtrees they own.
|
||||
|
||||
## Directory structure
|
||||
|
||||
| Path | Description |
|
||||
|------------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| / | Primary hierarchy root and root directory of the entire file system hierarchy. |
|
||||
| /bin | Essential command binaries that need to be available in single-user mode, including to bring up the system or repair it, for all users (e.g., cat, ls, cp). |
|
||||
| /boot | Boot loader files (e.g., EFI, initial-ramdisk.img ). |
|
||||
| /dev | POSIX Device files (e.g., /dev/null, /dev/disk0, /dev/tty, /dev/random). |
|
||||
| /etc | Host-specific system-wide configuration files. |
|
||||
| /home | Users' home directories, containing saved files, personal settings, etc. |
|
||||
| /lib | Libraries essential for the binaries in /bin and /sbin. eg realtime, system, ipc etc. |
|
||||
| /sbin | Essential system binaries (e.g init) |
|
||||
| /srv | Site-specific data served by this system, such as data and scripts for web servers, data offered by FTP servers, and repositories for version control systems |
|
||||
| /system | DanOS operating system files (similar idea to C:\Windows). A true representation of danos — its layout mirrors the source tree, so `/system` is what danos *is*. |
|
||||
| /system/devices | danos virtual device tree e.g. similar to /sys on linux but with danos device tree conventions (the structures in the devices module) |
|
||||
| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/pci-bus, /system/drivers/ps2-bus) |
|
||||
| /system/services | system-service binaries — init, the FAT server, and other user-mode servers (e.g. /system/services/init, /system/services/fat) |
|
||||
| /system/kernel | the kernel image |
|
||||
| /tmp | Directory for temporary files (see also /var/tmp). Often not preserved between system reboots and may be severely size-restricted. |
|
||||
| /usr | Secondary hierarchy for read-only user data; contains the majority of (multi-)user utilities and applications. Should be shareable and read-only. |
|
||||
| /var | Variable files: files whose content is expected to continually change during normal operation of the system, such as logs, spool files, and temporary e-mail files. |
|
||||
|
||||
## File types
|
||||
|
||||
POSIX specifies the long format of the ls command to represent the Unix file type as the first letter for an entry.
|
||||
|
||||
| type | symbol | Description |
|
||||
|-------------------|--------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| regular | - | An ordinary file holding an uninterpreted byte stream. Reads and writes are positional, and the file grows on demand (e.g., a binary in /bin, a config file in /etc). |
|
||||
| directory | d | A container mapping names to other files. It may only be modified through directory operations, never written to directly. |
|
||||
| symbolic link | l | A file whose contents are a path that is resolved in its place. The target need not exist, and may cross mount points. |
|
||||
| FIFO special | p | A named pipe: an in-order byte stream between processes, where writers block until a reader opens the other end. |
|
||||
| block special | b | A device node addressed in fixed-size blocks with the kernel free to buffer and reorder access (e.g., /dev/disk0). |
|
||||
| character special | c | A device node addressed as an unbuffered byte stream, delivered to the driver in order (e.g., /dev/tty, /dev/null). |
|
||||
| socket | s | A named endpoint for bidirectional message-passing between processes, bound to a path rather than an address. |
|
||||
|
||||
## /dev
|
||||
|
||||
`/dev` holds the names through which processes reach devices. It is deliberately not
|
||||
the device tree: the tree — every node discovered by ACPI or PCI enumeration, with its
|
||||
resources and its parent — lives under [/system/devices](#directory-structure) and is
|
||||
addressed by device id. `/dev` is the much smaller set of devices that have a driver
|
||||
willing to serve them, addressed by name.
|
||||
|
||||
A device node is not a file the VFS can read. The bytes live in a driver process
|
||||
([drivers.md](drivers.md)), so opening a `/dev` name has to resolve to that driver's
|
||||
IPC endpoint, and subsequent reads and writes are calls against it. Resolve-to-endpoint
|
||||
is exactly what the kernel's `fs_resolve` already does for any mounted backend, and
|
||||
`FileStatus.kind` is the field that marks a device node; **what is not implemented today
|
||||
is `/dev` itself** — no service mounts it. (The flat eight-node ramfs this section once
|
||||
described is retired: the kernel-resident VFS root in `system/kernel/vfs.zig` serves the
|
||||
read-only `/system` initrd mount with real directories and node kinds, and filesystem
|
||||
backends such as the FAT server mount the rest.) The three sections below describe the
|
||||
intended shape, and are honest about which parts the kernel can already support.
|
||||
|
||||
### Character devices
|
||||
|
||||
A character device is a byte stream with no addressable position: bytes are delivered
|
||||
to the driver in the order written, and a read consumes what is there. Terminals,
|
||||
serial lines, keyboards and mice are all of this shape. These are the natural first
|
||||
device nodes in danos, because a character driver needs nothing the kernel doesn't
|
||||
already provide — it claims its device, maps its registers with `mmio_map`, and blocks
|
||||
on `replyWait` for either an interrupt or a client request. `system/drivers/ps2-bus/ps2-bus.zig`
|
||||
is already that program, minus the file-node client half.
|
||||
|
||||
The obstacle was never the file type; it is which hardware a ring-3 driver can reach.
|
||||
Direct `in`/`out` from user space is still a #GP (no TSS I/O bitmap, IOPL never raised),
|
||||
but a driver no longer needs it: **`io_read`/`io_write`** grant port access the same way
|
||||
`mmio_map` grants memory — gated by `device_claim` and the device's discovered `io_port`
|
||||
resource. So the 16550 UART at `0x3F8` and the PS/2 controller at `0x60`/`0x64` (and thus
|
||||
`/dev/ttyS0` and a keyboard node) are now writable as ordinary ring-3 drivers; the
|
||||
low-rate legacy hardware that needs port I/O is fine with a syscall per access. A
|
||||
memory-mapped device such as the framebuffer, needing no port I/O at all, remains the
|
||||
easiest first entry.
|
||||
|
||||
### Block devices
|
||||
|
||||
A block device is addressed in fixed-size blocks and, unlike a character device, the
|
||||
layer above is free to buffer, reorder, coalesce and retry requests against it. Disks
|
||||
and other persistent storage are the whole population of this class.
|
||||
|
||||
A block driver is now **writable, but not yet memory-safe.** Every storage controller
|
||||
worth naming is a bus master: it is programmed by handing it the physical address of a
|
||||
descriptor ring and left to read and write memory on its own. That ring is exactly what
|
||||
**`dma_alloc`** now provides — physically contiguous, pinned, uncacheable, with its
|
||||
physical address disclosed — and **`/lib/mmio`**'s barriers order the descriptor writes
|
||||
against the doorbell, and **`msi_bind`** delivers completions. So an AHCI or NVMe driver
|
||||
can be written today (the M14/M15 work in [driver-model.md](driver-model.md); the earlier
|
||||
"cannot host a block driver at all" is no longer true).
|
||||
|
||||
What is *not* yet true is that it is safe. A device programmed with an arbitrary physical
|
||||
address writes to arbitrary physical memory, and page tables do not sit between a device
|
||||
and RAM — an IOMMU does. The IOMMU is now *detected* (M16), but no translation domains
|
||||
are programmed, so granting a DMA-capable device to a driver process is still equivalent
|
||||
to granting ring 0. Until per-device domains confine a driver's DMA to the buffers it
|
||||
`dma_alloc`'d, a block driver works but forfeits the isolation that motivates user-space
|
||||
drivers — enforcement is the next step, and lands with that first driver. A ramdisk over
|
||||
the initial ramdisk remains the one block-shaped thing that needs no driver process at all.
|
||||
|
||||
### Pseudo-devices
|
||||
|
||||
A pseudo-device has the interface of a device and no hardware behind it: `/dev/null`
|
||||
discarding writes and reading as end-of-file, `/dev/zero` reading as an endless run of
|
||||
zero bytes, `/dev/full` failing writes with `ENOSPC`, `/dev/random` and `/dev/urandom`
|
||||
yielding unpredictable bytes.
|
||||
|
||||
These are the only `/dev` entries danos can implement immediately, and they are the
|
||||
sensible place to start, because they are exactly the entries that need no driver
|
||||
process, no `device_claim`, no MMIO grant and no interrupt. A future pseudo-device
|
||||
service would answer them out of its own address space — `null` and `zero` are a few
|
||||
lines each in its `read` and `write` handlers — and mount itself at `/dev` the way the
|
||||
FAT server mounts `/mnt/usb`. The two pieces of structure every later device node
|
||||
depends on (and that the flat ramfs of the time lacked) exist now: directories, so that
|
||||
`/dev/null` is a path rather than a name; and a populated `FileStatus.kind`, so that a
|
||||
caller can tell a character device from a regular file.
|
||||
|
||||
`/dev/random` is the one that is not free. It needs an entropy source, and the honest
|
||||
options on this kernel are `RDRAND`/`RDSEED` where CPUID advertises them, and the HPET
|
||||
counter's low bits as a poor fallback. Neither is a seeded CSPRNG, and a `/dev/random`
|
||||
that is merely unpredictable-looking is worse than none — nothing should be keyed from
|
||||
it until it is a real one.
|
||||
@@ -0,0 +1,344 @@
|
||||
# Native AMD GPU support — feasibility and roadmap
|
||||
|
||||
**Status: research snapshot, not implemented.** This records what a *minimal, display-only*
|
||||
native driver for a real discrete AMD GPU — specifically an **RX 6600-class card (Navi 23,
|
||||
RDNA2, DCN 3.0.2)**, the market analog of the RTX 3060 — would take, and how it slots into
|
||||
danos's pluggable scanout architecture. It is a survey of primary sources (the Linux
|
||||
[amdgpu Display Core](https://github.com/torvalds/linux/tree/master/drivers/gpu/drm/amd/display)
|
||||
driver and its [kernel documentation](https://docs.kernel.org/gpu/amdgpu/display/index.html),
|
||||
the AtomBIOS interpreter in
|
||||
[drivers/gpu/drm/amd](https://github.com/torvalds/linux/tree/master/drivers/gpu/drm/amd),
|
||||
linux-firmware's `LICENSE.amdgpu`, and Haiku's
|
||||
[radeon_hd](https://github.com/haiku/haiku/tree/master/src/add-ons/kernel/drivers/graphics/radeon_hd)),
|
||||
not an implementation. It completes the trilogy with [nvidia-gpus.md](nvidia-gpus.md) and
|
||||
[intel-igpu.md](intel-igpu.md) and should be read against both — AMD lands *between* them:
|
||||
NVIDIA-class discrete-card mechanics, but Intel-class (better, in one way) reference material.
|
||||
|
||||
Read [display.md](display.md) and [display-v2.md](display-v2.md) first — this doc assumes the
|
||||
v2 model where scanout is a **pluggable backend** and a native driver is just another `.scanout`
|
||||
service (like the virtio-gpu one), announcing to the compositor over `attach_scanout`.
|
||||
|
||||
## TL;DR
|
||||
|
||||
- **AMD's decisive advantage is that the vendor's own display driver is the register manual, and
|
||||
it's MIT-licensed.** The entire Display Core (DC) — hardware sequencer, per-block code for
|
||||
OTG/OPTC, HUBP, DPP, MPC, DIO/link encoders, plus the `asic_reg` register headers — ships in
|
||||
the Linux tree under MIT/X11, deliberately written OS-agnostic because AMD shares it across
|
||||
operating systems. You can study it, port it, even copy from it into a danos driver without
|
||||
license contamination. NVIDIA has no analog (nouveau is GPL); Intel has PRM prose but you
|
||||
still write the code yourself.
|
||||
- **The firmware wall is one small blob, not a GSP.** The only display-side firmware the Linux
|
||||
driver hard-requires is **DMCUB** (the display microcontroller), and only on **DCN 2.1
|
||||
through 4.x** — which includes Navi 23. It is redistributable from linux-firmware, and it is
|
||||
a display helper, not a full-card resource manager: on DCN 3.0.x, hardware init
|
||||
(`dcn30_init_hw`) is **host-driven direct register programming** — the only DMUB call in it
|
||||
is a capability query. All DCE generations, DCN 1.0 (Raven), and DCN 2.0 (Navi 10/12/14) run
|
||||
display with **no display firmware at all**.
|
||||
- **Whether the *silicon* (vs. the Linux driver) needs DMCUB for a bare GOP-inheriting modeset
|
||||
is unproven** — Linux fails init with `-EINVAL` if the blob is missing on a DMUB ASIC, but
|
||||
what it's *used for* at minimum scope (vs. PSR/ABM/offloaded DP link training) isn't
|
||||
documented. The safe plan ships the blob; it's legally and practically cheap to do so.
|
||||
- **Programming model is direct MMIO, not channel DMA.** DCN mode-set is ordered register-write
|
||||
sequences (the DC "hardware sequencer") against named, header-documented registers — no
|
||||
pushbuffers, no method streams, no RAMHT, no supervisor-interrupt handshake. This deletes the
|
||||
hardest structural layer of the NVIDIA path.
|
||||
- **Scanout is VRAM-only on discrete cards** — the claim that DCN can scan out of GTT/system
|
||||
memory was checked and *refuted* for dGPUs (Linux allows GTT scanout only on select APUs). So
|
||||
a small VRAM allocator + BAR CPU mapping is required, same as NVIDIA. Pitch-linear surfaces
|
||||
are supported; no DCC/tiling needed.
|
||||
- **danos's GOP boot helps here too, with a caveat.** DC explicitly models taking over a
|
||||
VBIOS/GOP-lit pipe (`dc_validate_boot_timing` reads back live DIG/OTG/pixel-clock state), so
|
||||
"repoint the surface on the running pipe" is demonstrably hardware-feasible — but Linux's
|
||||
seamless-boot path is **eDP-only and default-off on discrete cards**, so plan on a full
|
||||
self-owned modeset (including DP retrain) right after first light rather than living on the
|
||||
inherited link.
|
||||
- **AMD has real non-Linux prior art — but only for the old hardware.** Haiku's MIT `radeon_hd`
|
||||
mode-sets by executing VBIOS **AtomBIOS command tables** through AMD's own MIT interpreter;
|
||||
its compiled-in ceiling is **DCE 8.5 (Hawaii, ~2013)** — every Polaris/Vega/Navi entry sits
|
||||
in a `#if 0` block. There is zero non-Linux DCN precedent; a danos DCN driver would be first.
|
||||
- **Effort tier ≈ high-3 to 4** for a native DCN 3.0.x display-only driver on Navi 23 — the raw
|
||||
register surface is GA106-class (tier 4), but the MIT vendor reference, the one-blob firmware
|
||||
wall, and the absence of channel-DMA plumbing pull real risk out. The AtomBIOS-interpreter
|
||||
route is tier ≈ 3 but dead-ends at pre-2016 silicon.
|
||||
- **Recommendation:** the same sober conclusion as the other two docs — GOP already gives
|
||||
native-res scanout for zero code — but if danos ever does drive real discrete silicon
|
||||
natively, **an RDNA2 card is the best target of the three**: modern, mainstream, in-warranty
|
||||
hardware with a legally clean, vendor-authored reference. That combination exists nowhere
|
||||
else.
|
||||
|
||||
## The firmware wall (a fence, next to NVIDIA's wall)
|
||||
|
||||
AMD GPUs carry a zoo of firmware: PSP (security processor), SMU (power/clock management), CP/RLC
|
||||
(graphics), SDMA, VCN (media) — and, on the display side, DMCU (legacy) then **DMCUB**
|
||||
("Display Micro-Controller Unit, version B"), a per-generation blob in linux-firmware
|
||||
(`navi23_dmcub.bin` etc.). The display-only question is: which of these does a scanout driver
|
||||
actually need?
|
||||
|
||||
The Linux answer is precise and readable in
|
||||
[`amdgpu_dm.c`](https://github.com/torvalds/linux/blob/master/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c):
|
||||
`dm_init_microcode()` switches on the display IP version — **DCN 2.1 (Renoir) through DCN
|
||||
3.0.x / 3.1.x / 3.2 / 3.5 / 4.x** request a DMCUB blob as `AMDGPU_UCODE_REQUIRED`, and
|
||||
`dm_dmub_hw_init()` fails driver init with `-EINVAL` if it's absent. Everything earlier — **all
|
||||
of DCE (Southern Islands through Vega), DCN 1.0 (Raven), and DCN 2.0 (Navi 10/12/14)** — hits
|
||||
the `default:` case, `dmub_srv` stays NULL, and display runs with no display firmware at all
|
||||
([kernel display-manager doc](https://docs.kernel.org/6.2/gpu/amdgpu/display/display-manager.html)).
|
||||
|
||||
Two nuances survive verification:
|
||||
|
||||
- **The requirement is Linux-driver enforcement backed by real functional need, and it's
|
||||
version-sensitive.** AMD force-switched all Renoir ASICs to DMUB to fix a USB-C/resume bug
|
||||
(kernel commit `652de07addd2`, "with new dmub f/w dmcu is superseded"), which regressed users
|
||||
on old blobs and had to be patched with explicit `dmcub_fw_version` gating (`91adec9e0709`).
|
||||
What DMUB is *used for* varies with blob version. Ship a current blob.
|
||||
- **DMCUB is an architectural fixture, not a bolt-on** — the DCN hardware itself contains a DMU
|
||||
block housing the microcontroller
|
||||
([DCN overview](https://docs.kernel.org/gpu/amdgpu/display/dcn-overview.html)) — but it is
|
||||
**not a mediator of the programming model** on DCN 3.0: `dcn30_init_hw()` initializes clocks,
|
||||
disables power gating, and powers up link encoders via direct register writes; its sole DMUB
|
||||
interaction is `dc_dmub_srv_query_caps_cmd`. Firmware-*assisted* PHY/link bring-up appears
|
||||
from **DCN 3.1** onward — one more reason to target 3.0.x. Features like PSR and ABM are
|
||||
DMUB-offloaded on all generations; a minimal driver simply doesn't enable them.
|
||||
|
||||
**Contrast with NVIDIA's GSP:** the GSP is a full resource manager with a signed multi-stage
|
||||
boot chain and a firmware ABI that breaks every driver release. DMCUB is a display helper blob
|
||||
you copy onto the boot image once, load into a reserved buffer, and mostly ignore. There is no
|
||||
signature fuse-matching, no WPR carve-out, no RPC-only register access. The one genuinely open
|
||||
question — whether a GOP-inheriting minimal modeset could skip DMCUB entirely on DCN 3.0.2 —
|
||||
doesn't need answering, because shipping the blob costs nothing (see [Licensing](#licensing)).
|
||||
|
||||
**PSP/SMU remain the flagged risk.** Nothing display-only touches CP/RLC/SDMA (those gate the
|
||||
graphics rings, exactly like NVIDIA's PGRAPH — irrelevant here). But `dcn30_init_hw` calls into
|
||||
the clock manager, and on discrete cards the clock manager may message the SMU to change display
|
||||
clocks (DISPCLK/DPPCLK). Whether inherited GOP boot clocks suffice for a same-or-lower mode —
|
||||
avoiding SMU (and hence PSP firmware-load) entirely — is the largest unverified assumption in
|
||||
the milestone list below. The survey produced no confirmed claim either way.
|
||||
|
||||
## The display engine landscape
|
||||
|
||||
Two eras, one boundary that matters:
|
||||
|
||||
| Generation | Display IP | Cards | Display firmware | Route |
|
||||
|---|---|---|---|---|
|
||||
| GCN 1–4 (SI→Polaris) | DCE 6/8/10/11 | HD 7000 → RX 580 | none | AtomBIOS tables or direct DCE registers |
|
||||
| Vega / Raven | DCE 12 / DCN 1.0 | Vega 56/64, APUs | none | DC code (first DCN) |
|
||||
| Navi 1x (RDNA1) | DCN 2.0 | RX 5500–5700 | none | DC code |
|
||||
| Renoir APU | DCN 2.1 | 4000-series APUs | **DMCUB required** | DC code |
|
||||
| **Navi 2x (RDNA2)** | **DCN 3.0.x** | **RX 6600–6900** | **DMCUB required** | **DC code, host-driven init** |
|
||||
| RDNA3/RDNA4+ | DCN 3.1+/3.2/3.5/4.x | RX 7000/9000 | DMCUB required, fw-assisted PHY | DC code, more DMUB offload |
|
||||
|
||||
The best modern first-pixel target is **DCN 3.0.x**: it has the full MIT block stack from the
|
||||
June 2020 Sienna Cichlid patch series (207 patches, Linux 5.9; Navi 23 reuses the dcn30
|
||||
sequencer), host-driven hardware init, and sits *before* the DCN 3.1 shift toward
|
||||
firmware-assisted link management. Older DCE cards are even simpler (no firmware at all, plus
|
||||
the AtomBIOS escape hatch) but are 2013–2016 hardware; newer DCN 3.5/4.x pushes more into DMUB.
|
||||
|
||||
The DCN pipe, in one line each (the vocabulary the DC code speaks —
|
||||
[programming model](https://docs.kernel.org/next/gpu/amdgpu/display/programming-model-dcn.html)):
|
||||
**HUBP** fetches and unpacks the surface from memory (this is where the scanout address and
|
||||
pitch live), **DPP** scales/converts colors, **MPC** blends planes (bypassable for one plane),
|
||||
**OPP** packs output, **OTG/OPTC** generates raster timings (the CRTC), and the **DIO** block's
|
||||
DIG encoders + PHY drive the connector. Mode-set is the DC *hardware sequencer* walking these
|
||||
blocks with ordered register writes — plain MMIO with polling, no pushbuffer channels, no
|
||||
supervisor interrupts. Structurally this is Intel-shaped, not NVIDIA-shaped.
|
||||
|
||||
## Two routes: AtomBIOS interpreter vs. native DC-derived registers
|
||||
|
||||
**AtomBIOS** is AMD's VBIOS bytecode: every card's ROM carries *data tables* (connector
|
||||
topology, clock limits — the DCB equivalent) and *command tables* (`SetPixelClock`,
|
||||
`SetCRTC_Timing`, `EnableCRTC`, DIG encoder/transmitter control), executed by a small
|
||||
interpreter the driver embeds (`atom.c`, ~1.5k lines). The classic radeon driver and Haiku's
|
||||
`radeon_hd` mode-set this way: parse the tables, execute them, and the VBIOS does the
|
||||
register-level work for you — inherently per-board correct, since the tables come from the
|
||||
card's own ROM.
|
||||
|
||||
- **Where it's proven:** through DCE 8.5 (Haiku's ceiling, below) and in Linux's pre-DC code
|
||||
through Polaris (DCE 11.2). AMD's interpreter itself is MIT (Haiku ships AMD's own
|
||||
`atom.cpp`, "Copyright 2008 Advanced Micro Devices").
|
||||
- **Where it's unproven:** DCN. amdgpu's DC still *uses* AtomBIOS for init sub-steps
|
||||
(`bios_golden_init` in `dcn30_init_hw` executes host-interpreted tables) and reads the data
|
||||
tables for connector topology — but nobody drives a full DCN modeset from command tables, and
|
||||
whether RDNA2 VBIOSes still carry a complete modeset path or vestigial init-only tables is an
|
||||
open question no source answers. Do not bet on it.
|
||||
|
||||
**The native route** is: port the relevant slice of DC. Not wholesale — in-tree DC has
|
||||
accumulated Linux-isms (kernel-FPU guards around DML, the bandwidth-calculation library, which a
|
||||
single-plane fixed-mode driver can largely sidestep) — but the DC core is *designed* to be
|
||||
retargeted: the kernel docs state outright that DC "is shared with other OSes" and holds the
|
||||
OS-agnostic hardware programming behind a `dm_services` shim (register access, memory, delays,
|
||||
firmware loading). Dave Airlie initially rejected the DAL/DC merge in 2016 *because* it was
|
||||
AMD's cross-OS codebase — hostile-witness confirmation that this exact code runs outside Linux.
|
||||
Reimplement the shim in Zig, and the dcn30 sequences sit on top.
|
||||
|
||||
## The memory floor
|
||||
|
||||
Same shape as NVIDIA's, and the survey *hardened* one assumption:
|
||||
|
||||
- **VRAM-only scanout on discrete cards.** The documented DCN fetch path is VRAM → Data Fabric
|
||||
(SDP) → DCHUB → HUBP; the claim that display buffers can live in GTT/system memory was
|
||||
refuted for dGPUs in verification — Linux permits GTT scanout only on select APUs. The NVIDIA
|
||||
doc's "maybe sysmem ctxdma?" hope has a firm *no* here. Budget for a small VRAM allocator.
|
||||
- **Pitch-linear is fine.** HUBP programs a surface address + pitch; linear (untiled, no DCC)
|
||||
surfaces are first-class for scanout. No tiling math.
|
||||
- **CPU access via the VRAM BAR.** Compositing writes go through the PCI VRAM aperture;
|
||||
resizable BAR helps but isn't needed — one pitch-linear surface fits comfortably in a
|
||||
fixed 256 MB small-BAR window.
|
||||
- **No GPU VMM.** Display addresses are physical VRAM addresses programmed into HUBP; no page
|
||||
tables, no GEM/TTM, no eviction.
|
||||
|
||||
**Net:** (1) a contiguous aligned VRAM allocator, (2) a BAR CPU mapping, (3) a small reserved
|
||||
buffer for the DMCUB firmware regions. That's the whole memory story.
|
||||
|
||||
## Inheriting GOP state
|
||||
|
||||
danos's GOP boot pays off again, with sharper edges than on NVIDIA:
|
||||
|
||||
- **DC models pipe takeover explicitly.** `dc_validate_boot_timing()`
|
||||
([dc.c](https://github.com/torvalds/linux/blob/master/drivers/gpu/drm/amd/display/dc/core/dc.c))
|
||||
reads back *live* hardware — `is_dig_enabled` on the link encoder, OTG timing registers,
|
||||
pixel clock within tolerance — and keeps the VBIOS/GOP-lit pipe running until first flip.
|
||||
This is a vendor-blessed recipe for milestone 2 below: the exact register set that tells you
|
||||
which pipe is alive and how it's configured.
|
||||
- **But Linux's seamless path is eDP-only and APU-gated.** The code comment is blunt: "Support
|
||||
seamless boot on EDP displays only", and the enabling check requires an APU with DCN ≥ 3.0
|
||||
unless forced with `amdgpu.seamless=1`. On a discrete RX 6600 with DP/HDMI, Linux does a full
|
||||
modeset at takeover. Read that as a warning, not a prohibition: repointing HUBP at your own
|
||||
surface on the live pipe should still work (the readback code proves the state is
|
||||
inspectable), but plan the full self-owned modeset — **including DP retraining** — as the
|
||||
immediate next step, not a someday.
|
||||
- **No supervisor handshake exists to re-learn.** The NVIDIA doc's SV1/SV2/SV3 open question has
|
||||
no AMD counterpart; commit sequencing is ordered register writes + vblank/lock waits in the
|
||||
hwseq, all visible in MIT source.
|
||||
|
||||
## Licensing
|
||||
|
||||
The inverse of the NVIDIA situation, and the single strongest argument for AMD:
|
||||
|
||||
- **The reference code is MIT.** `amdgpu_dm.c` carries `SPDX-License-Identifier: MIT`;
|
||||
`dc/core/dc.c` and `dmub_srv.h` carry the full X11-style grant (use, copy, modify, merge,
|
||||
publish, distribute, sell). The `asic_reg` register headers ship under the same terms. One
|
||||
diligence note: SPDX tagging isn't uniform across the tree, so header-check each file before
|
||||
copying from it — but no GPL files are known inside `dc/`. Where nouveau forces a
|
||||
GPL-or-clean-room choice, here the *easy technical path and the permissive path are the same
|
||||
path*.
|
||||
- **Firmware redistribution is a solved problem.** linux-firmware's `LICENSE.amdgpu` grants
|
||||
anyone a royalty-free right to reproduce and distribute the blobs, binary-only, with the
|
||||
license text attached — no OSI-license gate like NVIDIA's, no AMD agreement needed. danos can
|
||||
ship `navi23_dmcub.bin` (and PSP/SMU blobs if ever needed) on its boot image today. The same
|
||||
license **prohibits reverse-engineering the blobs** — all programming knowledge must come
|
||||
from the MIT source, never from blob disassembly. (VBIOS images aren't in linux-firmware;
|
||||
they're read from the card's own ROM, as Haiku does.)
|
||||
- **Prose register docs are a DCE-era artifact.** AMD's classic X.Org-hosted PDFs cover the old
|
||||
families — and the famous `R6xx_3D_Registers.pdf` turns out to be 3D-only (verified: zero
|
||||
display content; the display material lives in the separate per-ASIC Register Reference
|
||||
Guides). For DCN there is **no prose display spec at all**: the MIT DC source plus the
|
||||
`asic_reg` headers *are* the register manual. Plan accordingly.
|
||||
|
||||
## Prior art
|
||||
|
||||
AMD, unlike NVIDIA, has genuine working non-Linux precedent — with a hard generational ceiling:
|
||||
|
||||
- **Haiku `radeon_hd`** (MIT, still in the tree): a real, shipping, from-scratch display driver
|
||||
that executes AtomBIOS command tables via AMD's own MIT interpreter. Verified ceiling:
|
||||
the last *enabled* device entry is **Hawaii (DCE 8.5, 0x67be)**; everything newer —
|
||||
Tonga/Fiji, Carrizo/Polaris, Vega/Raven, and every Navi/RDNA2 entry up to the RX 6900 XT —
|
||||
sits inside one `#if 0 /* disabled for R1/beta5 */` block under the comment "WARN: DCE
|
||||
versions below here get sketchy."
|
||||
- **AmigaOS/MorphOS RadeonHD drivers** (hdrlab): commercial non-Linux Radeon display drivers,
|
||||
again for the DCE era.
|
||||
- **FreeBSD** `drm-kmod`: a port of Linux amdgpu (DC and all), not independent prior art — but
|
||||
proof the DC codebase transplants.
|
||||
|
||||
**Nobody has driven DCN outside Linux-derived code.** A danos DCN 3.0.x driver would be a
|
||||
first — but a first with the vendor's MIT code as its map, which is a different proposition
|
||||
from nouveau-as-only-reference.
|
||||
|
||||
## Alternatives
|
||||
|
||||
| Option | What you get | The tradeoff |
|
||||
|---|---|---|
|
||||
| **Stay on GOP** (working today) | Native-res scanout, zero GPU code/firmware/maintenance | Resolution frozen at ExitBootServices; no runtime mode change, no hardware vsync, no multihead |
|
||||
| **AtomBIOS interpreter on an old DCE card** | Proven end-to-end (Haiku); interpreter is small + MIT; board-correct by construction | 2013–2016 hardware ceiling; tier ≈ 3; teaches AtomBIOS, not modern DCN |
|
||||
| **Native DCN 3.0.x on RX 6600** (this doc) | Runtime modeset, vsync, multihead on modern silicon; MIT vendor reference; one redistributable blob | Tier ≈ high-3–4; SMU/clock question open; no non-Linux precedent |
|
||||
| **Port DC wholesale** (reimplement `dm_services`) | Vendor-maintained sequences verbatim; designed-for-porting seam | Big codebase to carry (DML, abstractions); Linux-isms to shear off; overkill for one plane |
|
||||
| **RDNA3+/DCN 3.5+** | Newer cards | More DMUB offload (fw-assisted PHY from DCN 3.1); strictly harder than 3.0.x for no display-only gain |
|
||||
|
||||
## "First light" milestones (native DCN 3.0.x path)
|
||||
|
||||
Framed as a danos `.scanout` service, inheriting the GOP-initialized display:
|
||||
|
||||
1. **PCI/BAR bring-up** — enumerate Navi 23, map the register BAR and the VRAM BAR via danos
|
||||
MMIO grants; prove the pipe is GOP-live by writing pixels into the *existing* GOP
|
||||
framebuffer through the VRAM BAR.
|
||||
2. **Read back the live pipe** — port the `dc_validate_boot_timing` register set: which OTG is
|
||||
running, its timings, which DIG/link encoder is enabled, current HUBP surface address/pitch.
|
||||
This is pure reads — zero risk, high information.
|
||||
3. **Repoint the surface** — allocate a danos-owned pitch-linear VRAM surface, program the HUBP
|
||||
surface address/pitch on the live pipe at vblank. First self-owned pixel with **no modeset,
|
||||
no firmware, no clock changes**.
|
||||
4. **DMCUB bring-up** — load `navi23_dmcub.bin` (redistributed per `LICENSE.amdgpu`) into its
|
||||
reserved regions, minimal `dmub_srv` init, verify the caps query answers.
|
||||
5. **Full owned modeset** — port the dcn30 hwseq slice: OTG timing programming, MPC bypass
|
||||
(single plane), DIG/PHY enable, **DP link retrain** (or start on HDMI to defer it, exactly
|
||||
as the NVIDIA doc advises). This is where the SMU/clock question lands — first attempt:
|
||||
reuse inherited boot clocks for a same-or-lower mode.
|
||||
6. **EDID** — AUX (DP) / DDC (HDMI) over the DCN AUX engine registers; parse and build the mode
|
||||
list; connector topology from the VBIOS AtomBIOS data tables.
|
||||
7. **Wire into the compositor** — `attach_scanout`, vsync from the vblank/pageflip interrupt,
|
||||
then multihead.
|
||||
|
||||
Keep the GOP backend as the fallback the whole way — a stall at any step still leaves danos with
|
||||
a working display (the resilience v2 already provides via re-attach).
|
||||
|
||||
## Reading list
|
||||
|
||||
**The DC core (MIT — the register manual for DCN):**
|
||||
- `drivers/gpu/drm/amd/display/dc/hwss/dcn30/dcn30_hwseq.c` — hardware init + the modeset
|
||||
sequencer for the target generation (host-driven; one DMUB caps query).
|
||||
- `dc/dcn30/` + `dc/dcn302/` blocks: `dcn30_hubp.c` (surface address/pitch — milestone 3),
|
||||
`dcn30_optc.c` (OTG timings), `dcn30_dio_link_encoder.c` (DIG/PHY), `dcn30_mpc.c` (bypass),
|
||||
`clk_mgr/dcn30/` (the SMU question, read before milestone 5).
|
||||
- `dc/core/dc.c` — `dc_validate_boot_timing()`: the GOP-takeover readback recipe.
|
||||
- `asic_reg/dcn/dcn_3_0_0_{offset,sh_mask}.h` — every register name and bitfield.
|
||||
- `dmub/` (`dmub_srv.h`, `src/dmub_dcn30.c`) — firmware regions + bring-up for milestone 4.
|
||||
|
||||
**The Linux glue (for logic, not porting):**
|
||||
[`amdgpu_dm.c`](https://github.com/torvalds/linux/blob/master/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c)
|
||||
— `dm_init_microcode` / `dm_dmub_hw_init` (the firmware-wall switch), seamless-boot gating.
|
||||
|
||||
**Kernel docs (read first):**
|
||||
[DCN overview](https://docs.kernel.org/gpu/amdgpu/display/dcn-overview.html) (the block diagram
|
||||
+ VRAM→DF→DCHUB fetch path) ·
|
||||
[DC programming model](https://docs.kernel.org/next/gpu/amdgpu/display/programming-model-dcn.html)
|
||||
(dc_plane/dc_stream/dc_link objects, hwseq, block APIs) ·
|
||||
[display manager](https://docs.kernel.org/6.2/gpu/amdgpu/display/display-manager.html).
|
||||
|
||||
**AtomBIOS:** `drivers/gpu/drm/amd/amdgpu/atom.c` (the interpreter), `atombios.h` (table
|
||||
formats), [osdev AMD AtomBIOS](https://wiki.osdev.org/AMD_Atombios) (hobby-OS orientation),
|
||||
Haiku [`radeon_hd`](https://github.com/haiku/haiku/tree/master/src/add-ons/kernel/drivers/graphics/radeon_hd)
|
||||
(a complete worked example, MIT, through DCE 8.5).
|
||||
|
||||
**Licensing:** linux-firmware
|
||||
[`LICENSE.amdgpu`](https://github.com/endlessm/linux-firmware/blob/master/LICENSE.amdgpu);
|
||||
DCE-era prose specs at [x.org/docs/AMD](https://www.x.org/docs/AMD/) (Register Reference
|
||||
Guides — display; note `R6xx_3D_Registers.pdf` is 3D-only).
|
||||
|
||||
## Open questions (unresolved by the survey)
|
||||
|
||||
- **Does DCN 3.0.2 silicon need DMCUB for a bare inherit-and-modeset path**, or only for
|
||||
PSR/ABM/offloaded features? (Moot if the blob is shipped regardless — but it decides whether
|
||||
milestone 3 can precede milestone 4.)
|
||||
- **Can a display-only driver avoid PSP and SMU entirely** by inheriting GOP boot clocks — what
|
||||
does the dcn30 clock manager actually require from SMU messaging on Navi 23 for a
|
||||
same-or-lower mode? *The largest open risk in the plan.*
|
||||
- **Do RDNA2 VBIOS command tables still carry a complete modeset path**, or are they vestigial
|
||||
init-only tables? (Would open a Haiku-style route on modern cards; no source answers it.)
|
||||
- **Is the DCN 3.0 AUX/DDC engine and DP retrain fully host-drivable without DMUB**, as
|
||||
`dcn30_hwseq` implies?
|
||||
- Exact HUBP surface alignment/pitch constraints for linear scanout on Navi 23 (in the headers;
|
||||
not captured verbatim in the survey).
|
||||
|
||||
---
|
||||
|
||||
*Research snapshot (2026-07); findings pinned to Linux master and Haiku master as of the survey
|
||||
date. DMUB coverage only grows with new DCN generations — re-verify the firmware-wall switch in
|
||||
`amdgpu_dm.c` against current source before building.*
|
||||
@@ -1,6 +1,6 @@
|
||||
# Device interrupts
|
||||
|
||||
CPU exceptions ([interrupts.md](interrupts.md)) are the kernel reacting to its own
|
||||
CPU exceptions ([interrupts.md](../os-development/interrupts.md)) are the kernel reacting to its own
|
||||
mistakes. **Device interrupts** are the opposite: hardware asking for attention —
|
||||
a timer firing, a key pressed, a packet arriving. They share the IDT, but differ
|
||||
in one fundamental way: an exception here is terminal (we report and halt), while a
|
||||
@@ -10,7 +10,7 @@ back — the same mechanism a scheduler will later use to preempt tasks.
|
||||
|
||||
The first device we bring up is the **timer**, because it's the simplest: it lives
|
||||
entirely on the CPU's local interrupt controller, needing no external routing.
|
||||
It's all x86_64-specific, behind the [architecture](architecture.md) boundary.
|
||||
It's all x86_64-specific, behind the [architecture](../os-development/architecture.md) boundary.
|
||||
|
||||
## The APIC, not the PIC
|
||||
|
||||
@@ -40,7 +40,7 @@ count that becomes the reload value. From then on it fires vector 32 repeatedly,
|
||||
its own, forever.
|
||||
|
||||
The reload count isn't picked arbitrarily — it's **calibrated to real time**,
|
||||
which the [real-time](vision.md) scheduling guarantees depend on. Since the LAPIC
|
||||
which the [real-time](../vision.md) scheduling guarantees depend on. Since the LAPIC
|
||||
timer's raw rate is bus-clock dependent and unknown up front, `calibrate` runs the
|
||||
LAPIC timer one-shot from its maximum count while a **reference clock** counts out a
|
||||
known 10 ms, then sees how far the LAPIC got — its counts-per-millisecond, from which
|
||||
@@ -53,7 +53,7 @@ a missing PIT would hang the boot):
|
||||
|
||||
1. **CPUID leaf 0x15** — the CPU's TSC frequency directly, needing no external timer
|
||||
at all (the LAPIC is then measured against the TSC).
|
||||
2. The **HPET**, discovered via ACPI (see [discovery](discovery.md) / [acpi](acpi.md)).
|
||||
2. The **HPET**, discovered via ACPI (see [discovery](../os-development/discovery.md) / [acpi](../os-development/acpi.md)).
|
||||
3. The **ACPI PM timer** (a fixed 3.579545 MHz counter from the FADT).
|
||||
4. The **PIT** (legacy 8254, 1.193182 MHz) — last resort, and bounded so it can't hang.
|
||||
|
||||
@@ -100,7 +100,7 @@ values (a second socket, some firmware), so a thread migrating from a core readi
|
||||
check** as each application processor comes online (`checkWarpSource`, adapted from
|
||||
Linux's): the waking core and the BSP hammer a shared "highest seen" TSC under a lock,
|
||||
and if either ever reads below it, the cores' TSCs are skewed. It's pairwise because APs
|
||||
come up one at a time ([smp.md](smp.md)).
|
||||
come up one at a time ([smp.md](../os-development/smp.md)).
|
||||
|
||||
**The fallback.** When the TSC fails either test — non-invariant (a bare VM such as the
|
||||
default qemu64), or warped between cores — danos moves the monotonic clock onto the
|
||||
@@ -160,7 +160,7 @@ A device handler is a plain `fn () void` — a timer or keyboard handler doesn't
|
||||
the interrupted registers. (The stubs originally didn't save the SSE/vector
|
||||
registers, so a handler couldn't use them; `isr_common` now does an
|
||||
`fxsave`/`fxrstor` of the full SSE/x87 state around dispatch — see
|
||||
[interrupts.md](interrupts.md).)
|
||||
[interrupts.md](../os-development/interrupts.md).)
|
||||
|
||||
## Turning them on
|
||||
|
||||
@@ -168,11 +168,11 @@ Exceptions can't be masked, which is why they worked all along. Maskable device
|
||||
interrupts don't fire until the CPU's interrupt flag is set — so the final step is
|
||||
`sti` (`arch.enableInterrupts()`), after the APIC and timer are configured. From
|
||||
that instant the kernel has a heartbeat, and its idle `hlt` loop
|
||||
([halting.md](halting.md)) wakes on every tick and dozes off again.
|
||||
([halting.md](../os-development/halting.md)) wakes on every tick and dozes off again.
|
||||
|
||||
## Verifying it
|
||||
|
||||
The `timer` test (see [testing.md](testing.md)) is the proof that an interrupt both
|
||||
The `timer` test (see [testing.md](../testing.md)) is the proof that an interrupt both
|
||||
*fires* and *returns*: it records the tick count, busy-waits, and checks the count
|
||||
advanced on its own.
|
||||
|
||||
@@ -188,13 +188,13 @@ spinning in unrelated code — is the whole mechanism working end to end.
|
||||
## Since (done elsewhere)
|
||||
|
||||
- **Preemption**: the timer handler is where the scheduler decides to switch — the
|
||||
reason a *returning* interrupt matters. See [scheduling.md](scheduling.md).
|
||||
reason a *returning* interrupt matters. See [scheduling.md](../os-development/scheduling.md).
|
||||
- **`sleep()` / timeouts** built on the calibrated clock.
|
||||
- **The I/O APIC, routed**: external device lines now reach a vector, and the
|
||||
interrupt is delivered onward to a *user-space* driver as an IPC message. See
|
||||
[drivers.md](drivers.md).
|
||||
- **Uncacheable MMIO**: device grants are mapped `PCD|PWT` (strong-uncacheable) for
|
||||
user drivers — see [paging.md](paging.md).
|
||||
user drivers — see [paging.md](../os-development/paging.md).
|
||||
|
||||
## What's next (partly done since)
|
||||
|
||||
@@ -10,19 +10,19 @@ mirrors them and prunes a dead reporter's children, and the `usb-report`
|
||||
scenario proves report → prune → respawn → re-report. The application surface is built (M18.3, 2026-07-13):
|
||||
`enumerate` and `subscribe` over IPC, with `device-list` as the first client —
|
||||
the manager is now the one answer to "what devices exist" for applications.
|
||||
The primitives underneath are real ([process-management.md](process-management.md):
|
||||
The primitives underneath are real ([process-management.md](../os-development/process-management.md):
|
||||
spawn/supervise/kill/exit-notification; [driver-model.md](driver-model.md): the device
|
||||
table as a capability system; [drivers.md](drivers.md): claim/map/IRQ), and the first
|
||||
per-device driver spawn works (the device manager matches the xHCI controller by PCI
|
||||
class and spawns `usb-xhci-bus` with the device id as argv[1]). This document designs
|
||||
the rest: the device manager as **the tree, the matcher, and the supervisor** — the
|
||||
policy process that turns [resilience.md](resilience.md)'s restart goal into practice
|
||||
policy process that turns [resilience.md](../os-development/resilience.md)'s restart goal into practice
|
||||
for drivers.
|
||||
|
||||
How processes stop, reload, and report their deaths is deliberately **not** in this
|
||||
document: that is the universal lifecycle every danos process speaks —
|
||||
[process-lifecycle.md](process-lifecycle.md), signals over IPC and the stable
|
||||
`runtime.process` interface. The device manager is that design's first serious
|
||||
[process-lifecycle.md](../os-development/process-lifecycle.md), signals over IPC and the stable
|
||||
`process` interface. The device manager is that design's first serious
|
||||
customer, not its owner. Its own protocol contains nothing lifecycle-shaped; a
|
||||
driver is stopped, health-checked, and buried exactly like any other process.
|
||||
|
||||
@@ -35,7 +35,7 @@ The device tree is two things fused: *information* (what exists, how it nests) a
|
||||
claims, resource containment on `device_register`, the
|
||||
`mmio_map`/`irq_bind`/`msi_bind` gates — and **cleans all of it up when a process
|
||||
dies** (settled; it is increment 1 of
|
||||
[process-lifecycle.md](process-lifecycle.md)). The three invariants in
|
||||
[process-lifecycle.md](../os-development/process-lifecycle.md)). The three invariants in
|
||||
[driver-model.md](driver-model.md) stay exactly where they are. A device manager
|
||||
that could mint MMIO mappings by its own say-so would be a second kernel, and a
|
||||
buggy one would un-earn everything the microkernel bought.
|
||||
@@ -51,7 +51,7 @@ enumeration is a **pci-bus driver**: the manager spawns it against the host brid
|
||||
like any bus reports children. ACPI becomes an **acpi service** that interprets the
|
||||
tables and reports the namespace. The manager only orchestrates and merges. Moving
|
||||
AML interpretation out of ring 0 is its own project on its own track; nothing here
|
||||
depends on when it lands. (It landed: [discovery.md](discovery.md), M19–M20.)
|
||||
depends on when it lands. (It landed: [discovery.md](../os-development/discovery.md), M19–M20.)
|
||||
|
||||
`device_register` is **idempotent on exact match**: a re-registration with an
|
||||
identical (parent, class, identity, resources) tuple returns the existing id
|
||||
@@ -64,25 +64,54 @@ restarted instance to rebuild exactly the same ids.
|
||||
|
||||
## The protocol
|
||||
|
||||
A `device-manager-protocol` module (the vfs-protocol pattern): extern-struct
|
||||
messages, a version in the handshake, reserved fields everywhere. The manager is a
|
||||
well-known endpoint (`ipc.register(.device_manager)`); the badge tells it who is
|
||||
A `device-manager-protocol` module, defined through the
|
||||
[envelope](../os-development/protocol-namespace.md): every packet — request,
|
||||
reply, and pushed event alike — begins with the folded `Header`, and **the device
|
||||
id is `Header.target`**, the manager's object addressing. The contract is bound at
|
||||
`/protocol/device-manager`; the kernel-stamped badge tells the manager who is
|
||||
talking; the same endpoint receives its children's exit notifications — one loop,
|
||||
one world.
|
||||
|
||||
| Direction | Message | Purpose |
|
||||
| Direction | Packet | Purpose |
|
||||
|---|---|---|
|
||||
| driver → manager | `hello { version, role, device_id }` | confirms the argv assignment, starts the deadline clock |
|
||||
| bus → manager | `child_added { parent, bus_address, identity, device_id, hid }` | one node the bus discovered |
|
||||
| driver → manager | `hello { role, version }` @ the assigned device | confirms the argv assignment, starts the deadline clock |
|
||||
| bus → manager | `child_added { parent, bus_address, identity, bus, vendor, device, subsystem, hid }` @ the registered device id | one node the bus discovered |
|
||||
| bus → manager | `child_removed { parent, bus_address }` | unplug, or the bus lost it |
|
||||
| app → manager | `enumerate` | snapshot of the tree (read-only) |
|
||||
| app → manager | `subscribe` | receive published add/remove events |
|
||||
| app → manager | `enumerate` (reserved verb 1) | snapshot of the tree: one `ChildEntry` per record in the reply's tail |
|
||||
| app → manager | `subscribe` (reserved verb 2) | receive published add/remove events; the subscriber's endpoint rides as the call's capability |
|
||||
| manager → app | `child_added` / `child_removed` events | the same two structs, pushed rather than called |
|
||||
|
||||
The watcher table behind those last two rows is the **service harness's**
|
||||
(`service.Subscribers`, shared with input and power), not the manager's: it
|
||||
answers `subscribe`/`unsubscribe`, frames each event once for the fan-out, and
|
||||
sweeps a watcher on its exit notification — where the manager previously had no
|
||||
sweep for watchers at all. Its own supervised-driver exits are a different thing
|
||||
and unchanged, except that a driver's death now arrives twice (the manager is
|
||||
both its supervisor and a subscriber to published exits), so the manager retires
|
||||
a dead driver's process id as it handles the first and the second finds nothing
|
||||
to act on.
|
||||
|
||||
Two of what used to be the manager's own operations are the envelope's **reserved**
|
||||
verbs, which mean the same thing at every provider in the system, so this protocol
|
||||
numbers only three of its own (`hello` = 16, `child_added` = 17,
|
||||
`child_removed` = 18) and its two events in their own space (`child_added` = 16,
|
||||
`child_removed` = 17). No reply carries a status field: that is the `Status` every
|
||||
reply begins with.
|
||||
|
||||
`child_added` is the one struct that travels both ways — a bus *calls* it, the
|
||||
manager *pushes* it — which is why the operation and event numbering spaces are
|
||||
separate: one encoding, both directions, told apart by which way the packet went.
|
||||
Folding the operation byte and the device id out of it is also what makes it fit:
|
||||
a pushed event is 64 bytes at most, header included, and this one lands exactly on
|
||||
that floor. `child_removed` is the single message whose target stays 0, because it
|
||||
is addressed by the composite (parent, bus address) and no single `u64` carries a
|
||||
pair.
|
||||
|
||||
`hello` is the one deadline the manager enforces itself: spawned and silent past the
|
||||
deadline means wrong binary, wrong protocol version, or wedged before main — apply
|
||||
the stop sequence and the restart policy. Everything else lifecycle-shaped
|
||||
(terminate, the common `ping` liveness call, exit reasons) arrives through
|
||||
[process-lifecycle.md](process-lifecycle.md)'s vocabulary, not this protocol.
|
||||
[process-lifecycle.md](../os-development/process-lifecycle.md)'s vocabulary, not this protocol.
|
||||
|
||||
Assignment stays argv (`usb-xhci-bus <device id>`) for now — simple, and it works.
|
||||
The step after `hello` exists is delegation: the manager claims (or is granted) the
|
||||
@@ -91,13 +120,16 @@ mechanism), replacing first-come-first-served `device_claim` with policy. Identi
|
||||
`child_added` is per-bus: PCI children carry the class triple (`pci_class`, as the
|
||||
xHCI match already uses); USB children carry the (class, subclass, protocol) triple
|
||||
from usb-ids.zig — each bus's native language, decoded by the shared ids modules.
|
||||
(Since the registry landed, `child_added` also carries a `bus` discriminator and
|
||||
the numeric `vendor`/`device`/`subsystem` ids the finer match levels need —
|
||||
see [/etc/devices.csv](devices-csv.md).)
|
||||
|
||||
## Supervision and restart
|
||||
|
||||
Every driver is spawned with the manager's exit endpoint (`spawnSupervised` — built).
|
||||
On a death notification:
|
||||
|
||||
1. **Read the reason** ([process-lifecycle.md](process-lifecycle.md) increment 2).
|
||||
1. **Read the reason** ([process-lifecycle.md](../os-development/process-lifecycle.md) increment 2).
|
||||
Clean exit → it meant to; don't restart. Fault or missed `hello` deadline →
|
||||
restart with **backoff**, and a crash-loop cap (three fast deaths → mark failed,
|
||||
stop respawning, log loudly; a later `reload` to the manager can retry).
|
||||
@@ -136,8 +168,8 @@ way.
|
||||
## Increments
|
||||
|
||||
Increments 1–4 are the lifecycle prerequisites and live in
|
||||
[process-lifecycle.md](process-lifecycle.md) (claim cleanup on death, exit reasons,
|
||||
published exit events, signals + `runtime.process`). On top of those:
|
||||
[process-lifecycle.md](../os-development/process-lifecycle.md) (claim cleanup on death, exit reasons,
|
||||
published exit events, signals + `process`). On top of those:
|
||||
|
||||
5. **device-manager-protocol**: `hello`, supervised spawn with restart policy;
|
||||
usb-xhci-bus becomes the first conforming driver.
|
||||
@@ -147,7 +179,7 @@ published exit events, signals + `runtime.process`). On top of those:
|
||||
to a manager-internal seam.
|
||||
8. **Discovery migration** — DONE (M19–M20, 2026-07-13): enumeration moved to
|
||||
ring 3 as swappable per-firmware discoverers — the pci-bus driver (M19) then
|
||||
the acpi service (M20), see [discovery.md](discovery.md); of the enumerable
|
||||
the acpi service (M20), see [discovery.md](../os-development/discovery.md); of the enumerable
|
||||
devices, the kernel seeds only the host bridge and the acpi-tables node (the
|
||||
non-enumerable platform nodes — processors, interrupt controllers, the HPET,
|
||||
the loader's framebuffer — stay kernel-seeded too). Matching moved with it:
|
||||
@@ -169,9 +201,15 @@ published exit events, signals + `runtime.process`). On top of those:
|
||||
- **Manager death**: drivers survive the manager; the restarted manager re-learns
|
||||
the world (above). Checkpointing driver state with the manager is deferred until
|
||||
something demonstrates the need.
|
||||
- **Matching stays code until the third bus.** `driverFor`/`pciDriverFor` were
|
||||
honest at two bus types; the third was expected to trigger the manifest (a driver
|
||||
declares what it binds: a PCI class triple, a USB class triple, an ACPI `_HID`).
|
||||
(Since then: the third bus — USB — arrived and is matched in code too. Today's
|
||||
matchers are `pciDriverForIdentity`, `hidDriverFor`, and `usbDriverForIdentity`;
|
||||
the manifest waits until code matching actually hurts.)
|
||||
- **Matching is a registry, not code (resolved 2026-07-26).** `driverFor`/
|
||||
`pciDriverFor` were honest at two bus types; the third (USB) was matched in code
|
||||
too, and then the switch tables started to hurt — they keyed PCI matches on the
|
||||
class triple alone, so a virtio-gpu could only be matched as a generic display
|
||||
function and the driver had to re-confirm its `1AF4:1050` identity from config
|
||||
space after being spawned. The manifest the earlier note anticipated landed as a
|
||||
human-readable registry: **[/etc/devices.csv](devices-csv.md)**, parsed by the
|
||||
pure `device-registry` module and read by the manager at boot. A row binds a
|
||||
driver to a device by any of base / subclass / prog-IF / vendor / device /
|
||||
subsystem / `_HID`, most-specific match winning; it is authoritative (no
|
||||
compiled-in fallback — an unmatched device is logged, never guessed).
|
||||
`pciDriverForIdentity`, `hidDriverFor`, and `usbDriverForIdentity` are gone.
|
||||
@@ -0,0 +1,110 @@
|
||||
# /etc/devices.csv — the device registry
|
||||
|
||||
**Status: built (2026-07-26).** The device manager reads `/etc/devices.csv` at
|
||||
boot and binds every device a bus driver reports to the driver the registry
|
||||
names. It replaces the three hand-written `switch` tables that used to live in
|
||||
the manager (`pciDriverForIdentity`, `hidDriverFor`, `usbDriverForIdentity`) —
|
||||
the "manifest" [device-manager.md](device-manager.md) anticipated once code
|
||||
matching started to hurt. The parser and matcher are the pure, unit-tested
|
||||
`device-registry` module (`library/device/registry/device-registry.zig`).
|
||||
|
||||
## Why a registry
|
||||
|
||||
The switch tables keyed PCI matches on the 24-bit class/subclass/prog-IF triple
|
||||
alone. That is too coarse: a virtio-gpu is just "display / other" by class, so it
|
||||
could only be *class-matched* and the driver had to re-confirm its real
|
||||
`1AF4:1050` identity from config space **after** the manager had already spawned
|
||||
it. The registry lets a rule bind on the full identity — down to vendor, device,
|
||||
and subsystem — so the manager makes the precise decision itself, and the driver
|
||||
comes up already knowing it is the right one.
|
||||
|
||||
It is also **data, not code**: teaching the system new hardware is a line in a
|
||||
file, not an edit-and-recompile of the manager. And it is **greppable** — one
|
||||
place to read "what binds what," the same idea as Linux's `modules.alias`.
|
||||
|
||||
## The file
|
||||
|
||||
One rule per line, nine comma-separated fields; `#` starts a comment (whole-line
|
||||
or trailing); blank lines are ignored. Whitespace around a field is trimmed, so
|
||||
columns may be padded for readability.
|
||||
|
||||
```
|
||||
# bus base class prog_if vendor device subsystem hid driver
|
||||
pci, 0C, 03, 30, *, *, *, *, /system/drivers/usb-xhci-bus
|
||||
pci, 03, 00, 00, *, *, *, *, /system/drivers/display
|
||||
pci, 03, 80, *, 1AF4, 1050, *, *, /system/drivers/virtio-gpu
|
||||
usb, 03, 01, 01, *, *, *, *, /system/drivers/usb-hid-keyboard
|
||||
acpi, *, *, *, *, *, *, PNP0303, /system/drivers/ps2-bus
|
||||
```
|
||||
|
||||
| Field | Meaning | Notes |
|
||||
|---|---|---|
|
||||
| `bus` | `pci` \| `usb` \| `acpi` | which bus reported the device; picks the namespace for the id columns |
|
||||
| `base` | PCI base class / USB class | hex |
|
||||
| `class` | PCI subclass / USB subclass | hex |
|
||||
| `prog_if` | PCI prog-IF / USB protocol | hex |
|
||||
| `vendor` | PCI vendor / USB idVendor | hex |
|
||||
| `device` | PCI device / USB idProduct | hex |
|
||||
| `subsystem` | PCI subsystem, `(ssvid<<16)\|ssid` | hex; blank for usb/acpi |
|
||||
| `hid` | ACPI `_HID` (e.g. `PNP0303`) | blank for pci/usb |
|
||||
| `driver` | full ramdisk path to spawn | e.g. `/system/drivers/virtio-gpu` |
|
||||
|
||||
`*` or an empty field is a **wildcard** — it matches anything and adds nothing to
|
||||
a rule's specificity.
|
||||
|
||||
## Levels of detection: most-specific-wins
|
||||
|
||||
Several rows may match one device. The manager picks the **most specific** — the
|
||||
one that pins the finest-grained fields. Specificity weights double from the
|
||||
coarsest level so each outweighs all coarser levels combined:
|
||||
|
||||
```
|
||||
base(1) < class(2) < prog_if(4) < vendor(8) < subsystem(16) < device(32) ≈ hid(32)
|
||||
```
|
||||
|
||||
So the generic `pci, 03, 00, 00, …/display` rule and the precise
|
||||
`pci, 03, 80, *, 1AF4, 1050, …/virtio-gpu` rule coexist: the virtio card
|
||||
(vendor 1AF4, device 1050) takes the specific rule; a plain VGA adapter still
|
||||
falls to the generic one. Two rules that match a device with the *same*
|
||||
specificity are a registry authoring error — the manager logs it loudly and binds
|
||||
the first, so the shadowed rule is visible rather than silently dropped.
|
||||
|
||||
## Authoritative — no code fallback
|
||||
|
||||
There is no compiled-in default table behind the registry. A device that no row
|
||||
matches goes **unbound** and is logged; the manager never guesses. A missing or
|
||||
empty `/etc/devices.csv` therefore means nothing matches — which is loud at boot,
|
||||
not a silent half-working system.
|
||||
|
||||
## How the manager reads it
|
||||
|
||||
`/etc/devices.csv` is bundled into the initial ramdisk (`build.zig`'s `bundled`
|
||||
list). The kernel serves the initrd's `/etc` tree directly — the `fat` service is
|
||||
spawned *after* the device manager and is irrelevant to `/etc` — so the manager
|
||||
reads the file with a plain `fs.open("/etc/devices.csv")` + `read`, with no
|
||||
filesystem service running and no boot-ordering dependency. It parses the bytes
|
||||
once in `initialise`, before any bus driver can report a device to match.
|
||||
|
||||
## Feeding the matcher: the widened report
|
||||
|
||||
Finer-grained matching needs identity the old ABI threw away. Two things carry it
|
||||
now: `child_added` (and `DeviceDescriptor`) grew `vendor` / `device` /
|
||||
`subsystem` fields, filled by the PCI bus driver from config space (offsets
|
||||
0x00 and 0x2C); and each bus driver states its `bus` in the report (a `BusKind`),
|
||||
so the manager reads a PCI class triple and a USB class triple — the same 24 bits
|
||||
in different namespaces — against the right `bus` column.
|
||||
|
||||
## Adding a driver
|
||||
|
||||
(The step-by-step walkthrough with a worked example is
|
||||
[new-driver-checklist.md](new-driver-checklist.md).)
|
||||
|
||||
1. Create `system/drivers/<name>/` with the driver source plus a ~15-line
|
||||
package `build.zig` + `build.zig.zon` (copy an existing driver package,
|
||||
e.g. `system/drivers/pci-bus/`; per-driver extras go through
|
||||
`build_support.programModule`). Then bundle it at `/system/drivers/<name>`:
|
||||
one dependency + one bundled entry in the root `build.zig`, one line in the
|
||||
root `build.zig.zon`.
|
||||
2. Add a row to `etc/devices.csv` naming the identity it binds and its full path.
|
||||
|
||||
No device-manager change is required — the registry is the seam.
|
||||
@@ -18,18 +18,19 @@ Read [display.md](display.md) first for the *why*; this is the *what* and the *o
|
||||
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](coding-standards.md): spell out non-acronym abbreviations in
|
||||
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym abbreviations in
|
||||
full, kebab-case file names, no `Co-Authored-By` trailers on commits. New user binaries
|
||||
go through `addUserBinary` in [build.zig](../build.zig) and get packed into the
|
||||
initial-ramdisk; protocols are `b.addModule("…-protocol", …)` and imported into the
|
||||
`runtime` module.
|
||||
go through build-support's shared user-binary recipe and get packed into the
|
||||
initial-ramdisk; protocols are modules exported by the `library/protocol` package.
|
||||
(This section predates the build-packages split; see
|
||||
[build-packages-plan.md](../build-packages-plan.md) for the current build shape.)
|
||||
|
||||
## How to verify along the way
|
||||
|
||||
- `zig build test` — host unit tests (compositor math: layer clipping, damage merge,
|
||||
pitch/format blits are all host-testable with a fake framebuffer).
|
||||
- `python3 test/qemu_test.py <case>` — boots the real kernel in QEMU; assert on the
|
||||
serial log ([tests.zig](../system/kernel/tests.zig) is the registry).
|
||||
serial log ([tests.zig](../../system/kernel/tests.zig) is the registry).
|
||||
- The `run-efi` target renders to QEMU's display (`-device VGA,edid=on,xres=1280,yres=720`)
|
||||
— a screenshot confirms pixels for the milestones whose gate is visual.
|
||||
|
||||
@@ -39,22 +40,22 @@ initial-ramdisk; protocols are `b.addModule("…-protocol", …)` and imported i
|
||||
|
||||
Make the boot framebuffer reachable and mappable **write-combining** from user space.
|
||||
|
||||
- [x] [device-abi.zig](../system/devices/device-abi.zig): added `DeviceClass.display`; a
|
||||
- [x] [device-abi.zig](../../library/device/model/device-abi.zig): added `DeviceClass.display`; a
|
||||
`DisplayInfo{ width, height, pitch, format }` carried on the descriptor; a
|
||||
`flags` field on `ResourceDescriptor` + `resource_flag_write_combining`.
|
||||
- [x] [devices-broker.zig](../system/kernel/devices-broker.zig): `seedDisplay(base, w, h,
|
||||
- [x] [devices-broker.zig](../../system/kernel/devices-broker.zig): `seedDisplay(base, w, h,
|
||||
pitch, format)` publishes a root-level `display` node with one WC-flagged `memory`
|
||||
resource `[base, height*pitch]` + the `DisplayInfo`; `displayDevice()` /
|
||||
`displayClaimed()`. Seeded from `kmain` after `devices_broker.init`.
|
||||
- [x] [process.zig](../system/kernel/process.zig) `systemMmioMap` + paging
|
||||
- [x] [process.zig](../../system/kernel/process.zig) `systemMmioMap` + paging
|
||||
(`mapUserDeviceInto` gains a `write_combining` bool): a resource's WC flag maps it
|
||||
through the WC PAT slot (`setupPat`) instead of strong-uncacheable.
|
||||
- [x] [console.zig](../system/kernel/console.zig): `setSuppressed` quiesces `write` while
|
||||
- [x] [console.zig](../../system/kernel/console.zig): `setSuppressed` quiesces `write` while
|
||||
the display device is claimed (driven from `systemDeviceClaim` / release); the
|
||||
terminal panic + exception paths clear it first so a dying machine still draws.
|
||||
|
||||
**Gate (met, automated):** the `display` kernel test (`python3 test/qemu_test.py display`,
|
||||
`displayTest` in [tests.zig](../system/kernel/tests.zig)) asserts the seeded node's shape
|
||||
`displayTest` in [tests.zig](../../system/kernel/tests.zig)) asserts the seeded node's shape
|
||||
and geometry, then walks the real claim + `mmio_map` path into a throwaway address space
|
||||
and verifies the leaf is **write-combining** (PAT entry 4: PAT bit set, PCD/PWT clear) —
|
||||
with an uncacheable-still-uncacheable regression guard. Chosen over the original
|
||||
@@ -67,10 +68,10 @@ Regression-checked: `discovery`, `ioport`, `claim-release`, `supervision`, `devi
|
||||
|
||||
Stand up the named service and the double-buffer, no layers yet.
|
||||
|
||||
- [x] `system/services/display/protocol.zig`: `Operation{ info, create_layer,
|
||||
- [x] `library/protocol/display/display-protocol.zig`: `Operation{ info, create_layer,
|
||||
configure_layer, destroy_layer, fill_rect, blit_tile, damage, present }`; `extern`
|
||||
`Request`/`Reply`; size + `maximum_payload` consts. (Model: block/protocol.zig.)
|
||||
- [x] [abi.zig](../system/abi.zig): `ServiceId.display = 9`.
|
||||
- [x] [abi.zig](../../system/abi.zig): `ServiceId.display = 9`.
|
||||
- [x] `system/services/display/display.zig`: `main` → enumerate + claim + WC-map the LFB
|
||||
(front) → `mmap` a cacheable back buffer of `height*pitch` → `runtime.service.run`.
|
||||
`info` and a whole-screen `present` (back → front) are live; layer ops fail-stub
|
||||
@@ -78,8 +79,8 @@ Stand up the named service and the double-buffer, no layers yet.
|
||||
- [x] [library/runtime/display.zig](../library/runtime/runtime.zig) (+ barrel export of
|
||||
`display` and `display_protocol`): `info()` and `present()`, cached `.display`
|
||||
lookup with retry (model: block.zig).
|
||||
- [x] [init.zig](../system/services/init/init.zig): `"display"` added to `boot_services`.
|
||||
- [x] [build.zig](../build.zig): `display-protocol` module on the runtime; `display` exe
|
||||
- [x] [init.zig](../../system/services/init/init.zig): `"display"` added to `boot_services`.
|
||||
- [x] [build.zig](../../build.zig): `display-protocol` module on the runtime; `display` exe
|
||||
via `addUserBinary`; packed into the initial-ramdisk; installed to
|
||||
`/system/services/display`.
|
||||
- [x] **Kernel fix the back buffer surfaced:** `mmap` was capped at 256 pages (1 MiB) by
|
||||
@@ -105,7 +106,7 @@ The heart: composite an ordered layer stack, present only what changed.
|
||||
- [x] `create_layer` / `configure_layer` (damages old + new footprints) / `destroy_layer`,
|
||||
`fill_rect`, `blit_tile` (reads the inline tile from the IPC payload, unaligned-safe),
|
||||
`damage`, `present`.
|
||||
- [x] Pure, host-tested [compositor.zig](../system/services/display/compositor.zig): `Rect`
|
||||
- [x] Pure, host-tested [compositor.zig](../../system/services/display/compositor.zig): `Rect`
|
||||
(intersect/unite), `Surface`, `fillRect`, `composite` (opaque, clipped to a damage
|
||||
rect), `blitTile`. `present` clears the damaged region to the wallpaper, paints the
|
||||
visible layers bottom-to-top (z-sorted), and flushes just that rect back → front (WC).
|
||||
@@ -148,10 +149,10 @@ still pass, and the default `zig build` is clean.
|
||||
- [x] The three integration cases exist and pass: `display` (D1 handoff, kernel),
|
||||
`display-service` (D2/D3 compositor + self-check), and `display-demo` (D4 full
|
||||
pipeline: spawn `display` + `display-demo`, match `display-demo: ok`) —
|
||||
[tests.zig](../system/kernel/tests.zig) + [qemu_test.py](../test/qemu_test.py). Plus
|
||||
[tests.zig](../../system/kernel/tests.zig) + [qemu_test.py](../../test/qemu_test.py). Plus
|
||||
the pure host tests (`zig build test`).
|
||||
- [x] [display.md](display.md) updated to the built state (the "Verifying it" section names
|
||||
the real cases); [README index](README.md) entry present (#19); the `display-track`
|
||||
the real cases); [README index](../README.md) entry present (#19); the `display-track`
|
||||
memory marked DONE with the commits.
|
||||
|
||||
**Gate (met):** `python3 test/qemu_test.py display display-service display-demo` all pass,
|
||||
@@ -16,11 +16,13 @@ lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run,
|
||||
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](coding-standards.md): spell out non-acronym abbreviations,
|
||||
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym abbreviations,
|
||||
kebab-case file names, no `Co-Authored-By` trailers. New user binaries go through
|
||||
`addUserBinary` and get packed into the initial-ramdisk; protocols are
|
||||
`b.addModule("…-protocol", …)` imported into `runtime`; new syscalls extend
|
||||
[abi.zig](../system/abi.zig) `SystemCall` + a `library/runtime` wrapper.
|
||||
build-support's shared user-binary recipe and get packed into the initial-ramdisk;
|
||||
protocols are modules exported by the `library/protocol` package; new syscalls extend
|
||||
[abi.zig](../../system/abi.zig) `SystemCall` + a `library/kernel` wrapper.
|
||||
(This section predates the build-packages split; see
|
||||
[build-packages-plan.md](../build-packages-plan.md) for the current build shape.)
|
||||
|
||||
## How to verify along the way
|
||||
|
||||
@@ -59,7 +61,7 @@ is the only backend), and `zig build test` stays green.
|
||||
|
||||
## V2 — The shared-memory cross-process capability (kernel) ✅
|
||||
|
||||
- [x] [abi.zig](../system/abi.zig): `shared_memory_create` (34) / `shared_memory_map` (35) syscalls + a
|
||||
- [x] [abi.zig](../../system/abi.zig): `shared_memory_create` (34) / `shared_memory_map` (35) syscalls + a
|
||||
`shared_memory_test` service id. Handlers in process.zig: `shared_memory_create(len)` allocates contiguous,
|
||||
zeroed, **cacheable** frames, wraps them in a refcounted object, installs a capability
|
||||
handle, maps them into the caller's shared-memory arena → returns virtual_address + handle; `shared_memory_map(cap)`
|
||||
@@ -144,4 +144,4 @@ path in VMs**, where danos development happens. The framebuffer floor never goes
|
||||
- [display.md](display.md) — v1: the compositor, the GOP-vs-device split, the WC discipline.
|
||||
- [display-v2-plan.md](display-v2-plan.md) — the ordered build-out.
|
||||
- [driver-model.md](driver-model.md) — claim / `mmio_map` / MSI / capability passing (M13).
|
||||
- [resilience.md](resilience.md) — the restart machinery the hot-attach leans on.
|
||||
- [resilience.md](../os-development/resilience.md) — the restart machinery the hot-attach leans on.
|
||||
@@ -1,12 +1,12 @@
|
||||
# The display service: a framebuffer compositor
|
||||
|
||||
The [framebuffer](framebuffer.md) the loader hands over is a flat block of pixel
|
||||
memory, and the kernel's [bootstrap console](../system/kernel/console.zig) draws text
|
||||
The [framebuffer](../os-development/framebuffer.md) the loader hands over is a flat block of pixel
|
||||
memory, and the kernel's [bootstrap console](../../system/kernel/console.zig) draws text
|
||||
into it directly. That console is a stop-gap. The **display service**
|
||||
(`system/services/display/`) is the real thing: an ordinary ring-3 process that *owns*
|
||||
the framebuffer, composes a stack of **layers** into an off-screen back buffer, and
|
||||
**presents** finished frames to the screen — the display half of the GUI track
|
||||
([vision.md](vision.md)), the sibling of the [input service](input.md).
|
||||
([vision.md](../vision.md)), the sibling of the [input service](input.md).
|
||||
|
||||
This note is the architecture and the reasoning behind it. The concrete build order
|
||||
lives in [display-plan.md](display-plan.md).
|
||||
@@ -20,20 +20,20 @@ which one you're holding decides what you can do.
|
||||
|
||||
- **GOP is firmware's *temporary* driver** for the display controller. It gives you a
|
||||
linear framebuffer pointer and can set video modes — but only until
|
||||
`ExitBootServices`. The loader already leans on this: [`queryFramebuffer`](../boot/efi.zig)
|
||||
`ExitBootServices`. The loader already leans on this: [`queryFramebuffer`](../../boot/efi.zig)
|
||||
reads the monitor's EDID, picks the native mode, and calls `set_mode` **before**
|
||||
exiting ([gop.md](gop.md)). Once the kernel runs, GOP is **gone** — no `set_mode`, no
|
||||
exiting ([gop.md](../os-development/gop.md)). Once the kernel runs, GOP is **gone** — no `set_mode`, no
|
||||
mode list, no EDID. What survives is the frozen snapshot in
|
||||
[`BootInformation.framebuffer`](../system/boot-handoff.zig): `{base, width, height,
|
||||
[`BootInformation.framebuffer`](../../system/boot-handoff.zig): `{base, width, height,
|
||||
pitch, format, refresh_hz}`, and nothing more.
|
||||
|
||||
- **The PCI class-0x03 device is the raw controller** — BARs, config space, registers,
|
||||
IO ports. It is what you actually *own* after boot. On QEMU's emulated adapter
|
||||
([`-device VGA,edid=on`](../build.zig), the Bochs VBE/DISPI model) the `base` GOP handed
|
||||
([`-device VGA,edid=on`](../../build/qemu.zig), the Bochs VBE/DISPI model) the `base` GOP handed
|
||||
you *is* that device's linear-framebuffer BAR — the same physical memory, seen through
|
||||
a different door. On a real discrete GPU, GOP's `base` is an aperture inside the GPU's
|
||||
VRAM BAR. danos already decodes this device
|
||||
([pci-class.zig](../system/devices/pci-class.zig) has the full `display` namespace, and
|
||||
([pci-class.zig](../../library/device/pci/pci-class.zig) has the full `display` namespace, and
|
||||
`pci-bus` already reports it to the [device manager](device-manager.md) with its class
|
||||
triple) — but nothing binds it yet.
|
||||
|
||||
@@ -63,16 +63,16 @@ rest of the system hasn't had to face:
|
||||
|
||||
1. **The framebuffer is kernel-only today.** It arrives through the boot handoff, is
|
||||
mapped into the kernel's physmap, and is touched only by
|
||||
[`console.zig`](../system/kernel/console.zig). It is *not* a
|
||||
[devices-broker](../system/kernel/devices-broker.zig) node, so
|
||||
[`console.zig`](../../system/kernel/console.zig). It is *not* a
|
||||
[devices-broker](../../system/kernel/devices-broker.zig) node, so
|
||||
`device.claim`/`mmio_map` cannot reach it, and there is no framebuffer
|
||||
[syscall](syscall.md). A user-space display service needs a **new mechanism just to
|
||||
[syscall](../os-development/syscall.md). A user-space display service needs a **new mechanism just to
|
||||
touch the pixels**. (See "The handoff" below — this is built.)
|
||||
|
||||
2. **danos had no cross-process shared memory.** At v1 the memory syscalls were `mmap`
|
||||
(private, zeroed), `mmio_map` (a *claimed device's* MMIO), and `dma_alloc` (new
|
||||
pinned physical). The block driver's "pass a buffer by physical address" trick
|
||||
([block/protocol.zig](../system/services/block/protocol.zig)) works *only because its
|
||||
([block/protocol.zig](../../library/protocol/block/block-protocol.zig)) works *only because its
|
||||
consumer is DMA hardware*. A compositor that CPU-reads and blends client layers can't
|
||||
use it — it would have to *map* another process's memory, which nothing allowed. v1
|
||||
sidesteps it entirely (see "What v1 does not do"); v2 has since built the primitive
|
||||
@@ -87,27 +87,29 @@ rest of the system hasn't had to face:
|
||||
│ (ResourceKind.memory = [base, height*pitch], write-combining hint,
|
||||
│ plus DisplayInfo{width, height, pitch, format, refresh_hz})
|
||||
▼
|
||||
display service (system/services/display/, ServiceId.display) ← the compositor
|
||||
display service (system/services/display/, /protocol/display) ← the compositor
|
||||
│ device.claim(display node) → mmio_map(WRITE-COMBINING) = FRONT buffer (the LFB)
|
||||
│ mmap(cacheable) a BACK buffer of the same geometry
|
||||
│ owns: an ordered LAYER STACK + a per-frame DAMAGE tracker (rect list or tile grid)
|
||||
│ loop: composite dirty layers → back buffer → present dirty rects → front
|
||||
│ backend is an INTERNAL interface: {gop-fb} at boot; {virtio-gpu} on hot-attach (v2)
|
||||
▼ reached by name (ipc_lookup); clients drive it over the display protocol
|
||||
▼ reached by name (open /protocol/display); clients drive it over the display protocol
|
||||
┌────────────────────────────────────┬──────────────────────────────────────┐
|
||||
drawing clients (v1) surface clients (deferred)
|
||||
runtime.display commands: runtime.display surfaces:
|
||||
display commands: display surfaces:
|
||||
create_layer / configure_layer shared_memory_create → pass as a capability →
|
||||
fill_rect / blit_tile / damage the compositor maps & composites the
|
||||
present client-rendered bitmap directly
|
||||
```
|
||||
|
||||
The bring-up sequence mirrors a hardware driver's — it is the
|
||||
[`usb-xhci-bus` `initialise`](../system/drivers/usb-xhci-bus/usb-xhci-bus.zig) shape
|
||||
[`usb-xhci-bus` `initialise`](../../system/drivers/usb-xhci-bus/usb-xhci-bus.zig) shape
|
||||
(claim → `mmio_map` → run loop) — and the request/reply service shell is the
|
||||
[FAT](../system/services/fat/fat.zig) / [input](../system/services/input/input.zig) shape
|
||||
([`runtime.service.run`](../library/runtime/service.zig) with a `protocol.zig` of
|
||||
`extern struct` messages and an `Operation` tag).
|
||||
[FAT](../../system/services/fat/fat.zig) / [input](../../system/services/input/input.zig) shape
|
||||
([`service.run`](../../library/kernel/service.zig) over the dispatch table its
|
||||
protocol module generates through
|
||||
[`envelope.Define`](../os-development/protocol-namespace.md) — one request and
|
||||
reply type per verb, and the layer id in the packet header's `target`).
|
||||
|
||||
**One process, for now.** v1 is a *single* service that both owns the framebuffer and
|
||||
composites — it does not split a "framebuffer driver" from a "compositor" the way input
|
||||
@@ -119,12 +121,12 @@ second backend or a second monitor appears; until then it is complexity with no
|
||||
|
||||
The framebuffer crosses into user space through the machinery that already exists for
|
||||
every other device, rather than a bespoke syscall — so it inherits ownership,
|
||||
release-on-death, and re-claim-on-restart for free (the [resilience](resilience.md)
|
||||
release-on-death, and re-claim-on-restart for free (the [resilience](../os-development/resilience.md)
|
||||
story: a crashed display service returns the LFB to the kernel, and its restart
|
||||
re-claims it).
|
||||
|
||||
- The kernel seeds a synthetic **display-class** node into the
|
||||
[devices-broker](../system/kernel/devices-broker.zig) at init (`seedDisplay`), from
|
||||
[devices-broker](../../system/kernel/devices-broker.zig) at init (`seedDisplay`), from
|
||||
`BootInformation.framebuffer`: one `ResourceKind.memory` resource spanning
|
||||
`[base, height*pitch]`, tagged **write-combining**, plus a small
|
||||
`DisplayInfo{width, height, pitch, format, refresh_hz}` (the memory resource says *where*
|
||||
@@ -135,7 +137,7 @@ re-claims it).
|
||||
- The service `device.claim`s it and `mmio_map`s the resource. The map is
|
||||
**write-combining**, not the strong-uncacheable that `mmio_map` uses for register
|
||||
MMIO. The kernel already programs a WC PAT slot for its own console
|
||||
([`setupPat`](../system/kernel/architecture/x86_64/paging.zig)); this reaches it from
|
||||
([`setupPat`](../../system/kernel/architecture/x86_64/paging.zig)); this reaches it from
|
||||
the user mapping path. **This matters:** an uncacheable framebuffer makes the
|
||||
back→front blit unusably slow.
|
||||
- On `claim`, the kernel's bootstrap console goes quiet, so the two never fight over the
|
||||
@@ -143,7 +145,7 @@ re-claims it).
|
||||
panic on screen wins.
|
||||
|
||||
The display service is a **named boot service**: `init` spawns it by name alongside
|
||||
`input`/`device-manager`/`fat` ([init.zig](../system/services/init/init.zig)), and it
|
||||
`input`/`device-manager`/`fat` ([init.zig](../../system/services/init/init.zig)), and it
|
||||
self-discovers the display node with `device.enumerate` (matching on `DeviceClass.display`). The [device manager](device-manager.md)
|
||||
matching path (PCI class 0x03 → a driver) is reserved for the future *native* backend, not
|
||||
this singleton synthetic node.
|
||||
@@ -160,8 +162,8 @@ Two buffers, with deliberately different memory types:
|
||||
|
||||
So a frame is: compose every dirty layer into the cacheable back buffer, then **present**
|
||||
— copy the changed regions back→front in sequential, WC-friendly writes. Two details the
|
||||
[framebuffer](framebuffer.md) note already establishes carry over: step rows by `pitch`,
|
||||
not `width*4`; and handle both `rgbx` and `bgrx` [pixel formats](gop.md).
|
||||
[framebuffer](../os-development/framebuffer.md) note already establishes carry over: step rows by `pitch`,
|
||||
not `width*4`; and handle both `rgbx` and `bgrx` [pixel formats](../os-development/gop.md).
|
||||
|
||||
## Flicker vs. tearing — what double buffering does and doesn't buy
|
||||
|
||||
@@ -188,10 +190,10 @@ The compositor holds an **ordered stack of layers**. Each layer has a rectangle,
|
||||
z-order, a visibility flag, and a surface. Presenting walks the stack bottom-to-top,
|
||||
painting each dirty layer into the back buffer, then flushes the damage to the front.
|
||||
Damage is tracked by one of two interchangeable trackers behind a compile-time
|
||||
`damage_mode` A/B switch ([display.zig](../system/services/display/display.zig)): a
|
||||
`damage_mode` A/B switch ([display.zig](../../system/services/display/display.zig)): a
|
||||
free-form dirty-rectangle **list** (tight bounds, heuristic merging) or a fixed 64-px
|
||||
**tile grid** (exact O(1) merging, tile-quantized repaints) — the grid is the default;
|
||||
[compositor.zig](../system/services/display/compositor.zig) has both, with the trade-off
|
||||
[compositor.zig](../../system/services/display/compositor.zig) has both, with the trade-off
|
||||
discussion.
|
||||
|
||||
In v1 the surfaces are **server-owned**, and clients draw into them with a small
|
||||
@@ -209,8 +211,20 @@ shell, a terminal, a cursor, and a wallpaper:
|
||||
| `damage` | mark a region of a layer dirty |
|
||||
| `present` | request a repaint: composited at the next frame-clock tick |
|
||||
|
||||
**A layer belongs to the client that created it.** The id is a slot in a
|
||||
sixteen-entry table — small, dense, guessable — so every verb above that names one is
|
||||
answered only for the task whose `create_layer` produced it, and a layer that is
|
||||
somebody else's is refused exactly as one that never existed (`-ENOENT`), so a client
|
||||
cannot use the refusal to learn which ids are live
|
||||
([protocol-namespace.md](../os-development/protocol-namespace.md): handles are scoped
|
||||
per client, validated against the badge). The compositor's own layers — the cursor
|
||||
sprite and the startup self-check's pair — are marked service-owned and are created by
|
||||
direct call rather than over the protocol, so no client can move or destroy the
|
||||
cursor. A dead client's layers are released on its exit notification, the same sweep
|
||||
the FAT server runs for open files.
|
||||
|
||||
Text is intentionally *not* an operation — a client renders glyphs by blitting tiles
|
||||
(the [PSF font](../system/kernel/font.psf) path the console already uses can move into a
|
||||
(the [PSF font](../../system/kernel/font.psf) path the console already uses can move into a
|
||||
client). Keeping the protocol to rectangles and tiles keeps the compositor small and the
|
||||
policy in the client.
|
||||
|
||||
@@ -222,19 +236,19 @@ pacing on backends that have none (all of them today; see
|
||||
[display-v2.md](display-v2.md), "Fenced is not vsync"). Bring-up paths that must put
|
||||
pixels on screen synchronously (initialisation, the self-checks) bypass the clock.
|
||||
|
||||
## `runtime.display`
|
||||
## `display`
|
||||
|
||||
Clients speak the protocol through a new [`library/runtime/display.zig`](../library/runtime/runtime.zig),
|
||||
the [`runtime.block`](../library/runtime/block.zig) shape (a cached `.display` lookup
|
||||
Clients speak the protocol through a new [`library/client/display/display.zig`](../../library/client/display/display.zig),
|
||||
the [`block`](../../library/device/block/block.zig) shape (a cached `.display` lookup
|
||||
with a boot-race retry): `display.info()`, a `Layer` handle with `fill` / `blitTile` /
|
||||
`damage`, and `present()`. Application code never issues the raw syscalls — it calls the
|
||||
runtime, as with every other danos service.
|
||||
client module, as with every other danos service.
|
||||
|
||||
## The cursor: a mouse-listener thread feeding the compositor
|
||||
|
||||
The compositor is the single owner of the framebuffer — only the main `service.run` loop
|
||||
touches the backend and the layer stack. Tracking the mouse without breaking that
|
||||
ownership is the display's first use of [threads](threading.md): the service is built
|
||||
ownership is the display's first use of [threads](../os-development/threading.md): the service is built
|
||||
multi-threaded (`addThreadedUserBinary`) and, at startup, spawns a **mouse-listener
|
||||
thread** beside the compositor loop.
|
||||
|
||||
@@ -242,9 +256,9 @@ thread** beside the compositor loop.
|
||||
(`input.subscribeMouse()`), accumulates the relative `dx`/`dy` motion into an absolute
|
||||
cursor position clamped to the screen, and hands it to the compositor. It never touches
|
||||
the compositor — so no lock guards the framebuffer. A parked `next()` leaves its core
|
||||
free to halt ([halting.md](halting.md)).
|
||||
free to halt ([halting.md](../os-development/halting.md)).
|
||||
- **The channel.** A single-slot *latest-value* cell (`CursorChannel`) guarded by a
|
||||
`runtime.Thread.Mutex`: the renderer wants where the cursor *is now*, not a replay of
|
||||
`Thread.Mutex`: the renderer wants where the cursor *is now*, not a replay of
|
||||
every delta, so a new position overwrites the old. The listener also **pokes** the
|
||||
compositor awake — the main loop is parked in `replyWait`, so the listener posts a
|
||||
zero-payload `ipc.send` to the compositor's endpoint, which arrives as a
|
||||
@@ -254,13 +268,13 @@ thread** beside the compositor loop.
|
||||
which is just a top-z compositor layer — with the existing `configure` + `present` path
|
||||
(it damages the old and new footprints, so only those two rectangles repaint).
|
||||
|
||||
Two threading facts shape this (both in [threading.md](threading.md)). IPC **handles do
|
||||
Two threading facts shape this (both in [threading.md](../os-development/threading.md)). IPC **handles do
|
||||
not cross threads**, so the listener can't reuse the main loop's endpoint handle — it
|
||||
`ipc.lookup(.display)`s its *own* handle to the same endpoint to poke through. And a
|
||||
multi-threaded service doing concurrent IPC is why the kernel's endpoint-create / register
|
||||
/ lookup syscalls now serialize under the big kernel lock. Shared fate applies: a fault in
|
||||
the listener takes the whole display down, and the supervisor restarts the process
|
||||
([resilience.md](resilience.md)).
|
||||
([resilience.md](../os-development/resilience.md)).
|
||||
|
||||
## What v1 does not do (and why that's fine)
|
||||
|
||||
@@ -284,7 +298,7 @@ both are clean additions behind the interfaces v1 establishes.
|
||||
|
||||
## Verifying it
|
||||
|
||||
Four QEMU test cases ([tests.zig](../system/kernel/tests.zig), `python3
|
||||
Four QEMU test cases ([tests.zig](../../system/kernel/tests.zig), `python3
|
||||
test/qemu_test.py <case>`), each layering on the last:
|
||||
|
||||
- **`display`** — the kernel handoff: the seeded `display` device is shaped correctly and
|
||||
@@ -297,7 +311,7 @@ test/qemu_test.py <case>`), each layering on the last:
|
||||
layer — logging `display: compositor self-check ok`.
|
||||
- **`display-demo`** — the full pipeline from a separate process: the hardware-free
|
||||
[`display-demo`](../system/services/display-demo/) client (the
|
||||
[`input-source`](../system/services/input-source/) analog) drives layers — a wallpaper and
|
||||
[`input-source`](../test/system/services/input-source/) analog) drives layers — a wallpaper and
|
||||
a sliding rectangle — through the layer client API and heartbeats
|
||||
`display-demo: ok`, proving a frame travelled client → compositor → screen, exactly as
|
||||
the [input test](input.md) proves an event travels source → service → subscriber. It draws
|
||||
@@ -317,8 +331,8 @@ packing are additionally covered by pure host unit tests under `zig build test`.
|
||||
|
||||
## See also
|
||||
|
||||
- [framebuffer.md](framebuffer.md) — the linear framebuffer, pitch vs. width, `volatile`.
|
||||
- [gop.md](gop.md) — GOP, and why only linear RGBX/BGRX modes are paintable.
|
||||
- [framebuffer.md](../os-development/framebuffer.md) — the linear framebuffer, pitch vs. width, `volatile`.
|
||||
- [gop.md](../os-development/gop.md) — GOP, and why only linear RGBX/BGRX modes are paintable.
|
||||
- [input.md](input.md) — the sibling service; the async `ipc_send` fan-out.
|
||||
- [driver-model.md](driver-model.md) — claim / `mmio_map`, capability passing, the trust model.
|
||||
- [device-manager.md](device-manager.md) — matching and supervision (the native backend's route).
|
||||
@@ -34,7 +34,7 @@ plain bus driver with no controller — a USB hub — is also a real thing.
|
||||
|
||||
danos already has the right central structure. `system/kernel/devices-broker.zig` holds a table of
|
||||
`DeviceDescriptor`, each with a parent, a class, and a set of resources. Firmware discovery
|
||||
seeds it ([discovery.md](discovery.md)); `device_register` grows it.
|
||||
seeds it ([discovery.md](../os-development/discovery.md)); `device_register` grows it.
|
||||
|
||||
Three invariants make it a capability system rather than a directory:
|
||||
|
||||
@@ -95,40 +95,66 @@ A "family" is two modules, not one:
|
||||
- **A protocol module** — the IPC message types that let a class driver talk to
|
||||
*whatever* published its device. This is the part that makes class drivers portable.
|
||||
|
||||
danos already has one of each: `library/runtime/device.zig` is a logic module,
|
||||
[`system/vfs-protocol.zig`](system/vfs-protocol.zig) is a protocol module shared by the
|
||||
mount backends (today the fat server) and their clients. (The user-space VFS server it
|
||||
was originally written against has since retired — path routing moved into the kernel,
|
||||
`system/kernel/vfs.zig`'s `fs_resolve` — but the protocol module outlived it, which is
|
||||
rather the point.) The pattern generalises directly:
|
||||
danos already has one of each: `library/device/pci/pci.zig` is a logic module (the
|
||||
`Function` view of a claimed PCI function),
|
||||
[`library/protocol/vfs/vfs-protocol.zig`](../../library/protocol/vfs/vfs-protocol.zig) is a
|
||||
protocol module shared by the mount backends (today the fat server) and their clients.
|
||||
(The user-space VFS server it was originally written against has since retired — path
|
||||
routing moved into the kernel, `system/kernel/vfs.zig`'s `fs_resolve` — but the protocol
|
||||
module outlived it, which is rather the point.) The pattern generalises directly:
|
||||
|
||||
```
|
||||
library/
|
||||
runtime/ module "runtime" — syscalls, heap, ipc, device, stdio
|
||||
mmio/ module "mmio" — volatile register access + barriers [M14]
|
||||
bus/
|
||||
pci/ module "pci" — ECAM, BAR decode, capability walk
|
||||
usb/ module "usb" — descriptors, control transfers, hubs
|
||||
proto/
|
||||
vfs/ module "vfs-protocol" (today: system/vfs-protocol.zig)
|
||||
block/ module "block-protocol"
|
||||
hid/ module "hid-protocol"
|
||||
kernel/ the system library (kernel32-style): the syscall surface split by concern
|
||||
— ipc, memory (heap/dma/shared-memory), process, time, logging,
|
||||
file-system, thread, service, plus system-call stubs + start/root
|
||||
device/ device code grouped by domain; each domain splits into a shareable
|
||||
data module (enums/wire types, std-only) and a logic module (mmio/IPC)
|
||||
mmio/ module "mmio" — typed volatile register access + barriers [M14]
|
||||
model/ module "device-abi" — DeviceDescriptor, DeviceClass, ResourceKind
|
||||
pci/ "pci-class" (data) + "pci" — config/BAR/capability walk (Function)
|
||||
usb/ "usb-abi" + "usb-ids" (data) + "usb" — descriptors, control/interrupt/bulk client
|
||||
acpi/ "acpi-ids" (data) + "aml" — _HID names, the AML interpreter
|
||||
driver/ module "driver" — device-access syscalls + device-manager hello
|
||||
block/ module "block" — the block-device client (a device type)
|
||||
client/ userspace service clients — display, input (a program's view of a service)
|
||||
protocol/ driver <-> service wire contracts, one module per directory
|
||||
vfs/ block/ display/ scanout/ input/ power/ device-manager/ usb-transfer/
|
||||
|
||||
system/drivers/ one sub-project each → /system/drivers (no `d` suffix)
|
||||
xhci/ HCD + bus driver imports runtime, pci, usb, mmio
|
||||
usb-hid/ class driver imports runtime, usb, hid-protocol
|
||||
block/ class driver imports runtime, block-protocol
|
||||
usb-xhci-bus/ HCD + bus driver imports usb, mmio, usb-transfer-protocol (+ kernel modules)
|
||||
usb-hid/ class driver imports usb, input-protocol (+ kernel modules)
|
||||
virtio-gpu/ scanout driver imports pci, mmio, display-/scanout-protocol (+ kernel modules)
|
||||
```
|
||||
|
||||
The build side of this has since landed: [`addUserBinary`](build.zig) injects the
|
||||
default modules (`runtime`, `mmio`, `xkeyboard-config`, `acpi-ids`) into every user
|
||||
binary, and per-binary extras — protocol modules, bus logic — are added with
|
||||
`programModule(exe).addImport(...)`. That's the *entire* mechanism — Zig modules
|
||||
already give you everything else.
|
||||
The split by *dependency weight* is what lets the microkernel stay out of device
|
||||
business: it imports only the `device-abi` data module (the descriptor types its broker
|
||||
marshals across the syscall boundary) — never a logic module, never a taxonomy. That one
|
||||
pure-data import is the only edge from `system/kernel/` into `library/`; decoding a class
|
||||
code or `_HID` to a name is user space's job (the device manager owns those taxonomies).
|
||||
|
||||
The discipline that makes this work: **a class driver must not import a bus's logic
|
||||
module.** `usbhid` imports `proto.hid` and `usb` (for descriptor types), never `pci`.
|
||||
If a class driver needs `mmio`, it has become an HCD and should be one.
|
||||
A protocol lives in `library/protocol/` when it is the seam between a low-level driver and
|
||||
a higher-level service (block ↔ filesystem, a scanout driver ↔ the compositor). A driver's
|
||||
private wire to its *hardware* — virtio-gpu's command set — is not that; it stays a
|
||||
driver-private file, like the virtio-pci transport beside it.
|
||||
|
||||
The build side of this has since landed: every binary owns a package whose
|
||||
~15-line `build.zig` names EXACTLY the modules its source imports — the moral
|
||||
equivalent of a C file's include list — and the shared recipe in
|
||||
[`build-support/build.zig`](../../build-support/build.zig) (`userBinary`)
|
||||
resolves each name from the library domain that exports it (kernel's concern
|
||||
modules, the device driver libraries, the service clients, the protocols). An
|
||||
undeclared `@import` is a compile error, and a domain none of the imports come
|
||||
from never appears in the binary's manifest — a keyboard driver declares
|
||||
`xkeyboard-config`; nothing else does (see
|
||||
[build-packages-plan.md](../build-packages-plan.md)). That's the *entire*
|
||||
mechanism — Zig modules already give you everything else.
|
||||
|
||||
The discipline that makes this work: **a class driver must not import a bus's *hardware*
|
||||
logic module.** `usb-hid` imports `usb` (the transfer client) and `input-protocol`, never
|
||||
`pci` and never `mmio`. If a class driver needs `mmio`, it has become an HCD and should be
|
||||
one. The domain data modules (`usb-abi`, `usb-ids`, `pci-class`) carry no such weight — a
|
||||
class driver, the device manager, or the kernel may share them freely.
|
||||
|
||||
## What exists today
|
||||
|
||||
@@ -141,16 +167,16 @@ If a class driver needs `mmio`, it has become an HCD and should be one.
|
||||
and a `received_cap` return (r8): an endpoint travels with a message, installed into
|
||||
the receiver's handle table (shared, refcount-bumped — a copy, not a move). A full
|
||||
table fails `-ENOSPC` and does not half-deliver. This is the "open" primitive — a bus
|
||||
driver mints a per-device endpoint and hands it to a class driver. The runtime exposes
|
||||
`callCap` and `replyWait(..., send_cap)`, and class drivers consume them now: the
|
||||
PS/2 keyboard and mouse drivers attach to ps2-bus this way, and `runtime.usb` /
|
||||
`runtime.input` open their per-device and subscription channels with `callCap`.
|
||||
- **M14** — DMA memory + the memory-ordering layer. `/lib/mmio` gives drivers typed
|
||||
volatile access and `mb`/`rmb`/`wmb` (per-arch); `dma_alloc`/`dma_free` grant
|
||||
driver mints a per-device endpoint and hands it to a class driver. The `ipc` module
|
||||
exposes `callCap` and `replyWait(..., send_cap)`, and class drivers consume them now: the
|
||||
PS/2 keyboard and mouse drivers attach to ps2-bus this way, and the `usb` / `input`
|
||||
client modules open their per-device and subscription channels with `callCap`.
|
||||
- **M14** — DMA memory + the memory-ordering layer. `/lib/device/mmio` gives drivers typed
|
||||
volatile access and `memoryBarrier`/`readMemoryBarrier`/`writeMemoryBarrier` (per-arch); `dma_alloc`/`dma_free` grant
|
||||
physically-contiguous, pinned, uncacheable, reclaim-on-teardown buffers with the
|
||||
physical address exposed (`pmm.allocContiguous`, a DMA arena, `mapUserDmaInto`).
|
||||
`dma_below_4g` caps the address for legacy engines; `dma_write_combining` is accepted
|
||||
but falls back to coherent until PAT is programmed. The bus drivers use `/lib/mmio`,
|
||||
but falls back to coherent until PAT is programmed. The bus drivers use `/lib/device/mmio`,
|
||||
and `dma_alloc` has real consumers now: the xHCI driver's rings and contexts,
|
||||
usb-storage's command/status wrappers, virtio-gpu's virtqueue, and the fat
|
||||
service's bounce buffer.
|
||||
@@ -177,7 +203,7 @@ If a class driver needs `mmio`, it has become an HCD and should be one.
|
||||
`system_spawn(name, arguments)` loads a binary bundled in the initial-ramdisk as a
|
||||
fresh ring-3 process; `name` becomes the child's argv[0] and the optional
|
||||
NUL-separated `arguments` blob its argv[1..], delivered on a SysV entry stack
|
||||
([sysv.md](sysv.md)). This is what
|
||||
([sysv.md](../os-development/sysv.md)). This is what
|
||||
turned the device manager from "log the match" into "run the driver": the kernel now
|
||||
spawns only `init`, `init` spawns the services, and the **device-manager** discovers
|
||||
the hardware and spawns each driver ([drivers.md](drivers.md)). Ungated for now — a
|
||||
@@ -237,7 +263,7 @@ const dev_ep = ipc.callCap(h, // ... mint a per-device endpoint,
|
||||
|
||||
## M14 — DMA memory and the memory-ordering contract, for HCDs ✅ done
|
||||
|
||||
*Implemented: `/lib/mmio` (typed volatile access + `mb`/`rmb`/`wmb`, per-arch) and
|
||||
*Implemented: `/lib/device/mmio` (typed volatile access + `memoryBarrier`/`readMemoryBarrier`/`writeMemoryBarrier`, per-arch) and
|
||||
`dma_alloc`/`dma_free` (contiguous, pinned, uncacheable, reclaim-on-teardown, physical
|
||||
address exposed). `dma_write_combining` still falls back to coherent — real WC needs
|
||||
PAT, a small follow-up. The rest of this section is the original design note.*
|
||||
@@ -279,31 +305,31 @@ doorbell.* = i; // volatile store to UC MMIO
|
||||
// nothing stops the compiler reordering these; the device reads a stale descriptor
|
||||
```
|
||||
|
||||
So the rules, which belong in `library/mmio.zig` and behind `arch`:
|
||||
So the rules, which belong in `library/device/mmio/mmio.zig` and behind `arch`:
|
||||
|
||||
| Situation | Required |
|
||||
|---|---|
|
||||
| MMIO register read/write | `mmio.read` / `mmio.write` (volatile) |
|
||||
| Fill DMA descriptor, then ring doorbell | `wmb()` between them |
|
||||
| Woken by IRQ, then read what the device wrote | `rmb()` before the read |
|
||||
| MMIO write that must complete before the next read | `mb()` |
|
||||
| Fill DMA descriptor, then ring doorbell | `writeMemoryBarrier()` between them |
|
||||
| Woken by IRQ, then read what the device wrote | `readMemoryBarrier()` before the read |
|
||||
| MMIO write that must complete before the next read | `memoryBarrier()` |
|
||||
|
||||
And the per-arch lowering — the reason this must be an `arch` primitive and not a
|
||||
sprinkling of `asm volatile`:
|
||||
|
||||
| | x86_64 | aarch64 |
|
||||
|---|---|---|
|
||||
| `mb()` | `mfence` | `dsb sy` |
|
||||
| `rmb()` | `lfence` | `dsb ld` |
|
||||
| `wmb()` | `sfence` | `dsb st` |
|
||||
| `memoryBarrier()` | `mfence` | `dsb sy` |
|
||||
| `readMemoryBarrier()` | `lfence` | `dsb ld` |
|
||||
| `writeMemoryBarrier()` | `sfence` | `dsb st` |
|
||||
| DMA cache coherency | coherent; nothing to do | **not guaranteed**; needs non-cacheable buffers or cache maintenance |
|
||||
|
||||
x86 is forgiving here — TSO plus strong-uncacheable MMIO means you usually get away
|
||||
with a compiler barrier alone. ARM is not, and [vision.md](vision.md) makes ARM the win
|
||||
with a compiler barrier alone. ARM is not, and [vision.md](../vision.md) makes ARM the win
|
||||
condition. Build the abstraction while there is one caller to fix.
|
||||
|
||||
(Zig note: `@fence` was **removed in 0.16**. Use `@atomicRmw(..., .seq_cst)` for a full
|
||||
barrier, or per-arch inline asm — which is what `library/mmio.zig` should hide.)
|
||||
barrier, or per-arch inline asm — which is what `library/device/mmio/mmio.zig` should hide.)
|
||||
|
||||
## M15 — interrupts for PCI devices ✅ done (MSI)
|
||||
|
||||
@@ -376,6 +402,6 @@ from hand-rolling `*volatile` and getting ARM wrong.
|
||||
## See also
|
||||
|
||||
- [drivers.md](drivers.md) — how to write one, concretely.
|
||||
- [discovery.md](discovery.md) / [acpi.md](acpi.md) — where the device table comes from.
|
||||
- [discovery.md](../os-development/discovery.md) / [acpi.md](../os-development/acpi.md) — where the device table comes from.
|
||||
- [ipc.md](ipc.md) — endpoints, badges, and the notification path an IRQ arrives on.
|
||||
- [resilience.md](resilience.md) — restart, the reason any of this is worth the trouble.
|
||||
- [resilience.md](../os-development/resilience.md) — restart, the reason any of this is worth the trouble.
|
||||
@@ -4,13 +4,13 @@ In a monolithic kernel a driver is a function call away from everything: it runs
|
||||
ring 0, dereferences any physical address, and its interrupt handler *is* the ISR. In
|
||||
danos a driver is **an ordinary ring-3 process**. It has its own address space, it
|
||||
can crash without taking the kernel with it, and — the point of this document — it
|
||||
can be restarted ([resilience](resilience.md)).
|
||||
can be restarted ([resilience](../os-development/resilience.md)).
|
||||
|
||||
That leaves three questions the kernel has to answer, because a process can't answer
|
||||
them for itself:
|
||||
|
||||
1. **What hardware exists?** → `device_enumerate`, over the device table discovery built
|
||||
([discovery](discovery.md), [acpi](acpi.md)).
|
||||
([discovery](../os-development/discovery.md), [acpi](../os-development/acpi.md)).
|
||||
2. **How do I touch its registers?** → `device_claim` + `mmio_map`: the kernel maps the
|
||||
device's physical MMIO window into your address space, and from then on it's plain
|
||||
memory. No syscall per register access.
|
||||
@@ -51,7 +51,7 @@ the optional arguments its argv[1..], on a SysV entry stack, see sysv.md). Every
|
||||
is the **driver supervisor**. It does the three steps a monolithic kernel would do in
|
||||
its probe path, entirely from ring 3:
|
||||
1. **Discover** — `device_enumerate` snapshots the device table the kernel built from
|
||||
ACPI/PCI ([discovery](discovery.md)).
|
||||
ACPI/PCI ([discovery](../os-development/discovery.md)).
|
||||
2. **Match** — for each device it looks up a driver. The match policy is code, a few
|
||||
small per-bus tables: from the boot snapshot only the PCI host bridge matches
|
||||
(→ `pci-bus`); everything else arrives later as bus reports and matches on
|
||||
@@ -75,7 +75,7 @@ capability yet.
|
||||
## The capability: claim before touch
|
||||
|
||||
The driver syscall numbers (`system/abi.zig`) with the device types they carry
|
||||
(`system/devices/device-abi.zig`), dispatched in `system/kernel/process.zig`:
|
||||
(`library/device/model/device-abi.zig`), dispatched in `system/kernel/process.zig`:
|
||||
|
||||
| # | Call | Meaning |
|
||||
|---|------|---------|
|
||||
@@ -269,7 +269,7 @@ already owns.
|
||||
A device with **no resources** is legal and common. A USB device is reached through its
|
||||
controller, not by MMIO, so it gets `resource_count = 0`.
|
||||
|
||||
See [`system/drivers/pci-bus/pci-bus.zig`](../system/drivers/pci-bus/pci-bus.zig) for a
|
||||
See [`system/drivers/pci-bus/pci-bus.zig`](../../system/drivers/pci-bus/pci-bus.zig) for a
|
||||
real one — it claims a PCI host bridge, maps its ECAM window, and publishes each function
|
||||
it finds as a child — and [driver-model.md](driver-model.md) for how bus drivers, class
|
||||
drivers and host controller drivers fit together.
|
||||
@@ -290,10 +290,10 @@ Several things this list used to warn about are now available (see
|
||||
[driver-model.md](driver-model.md)): **port I/O** (`io_read`/`io_write`, claim-gated by
|
||||
the device's `io_port` resource — direct ring-3 `in`/`out` is still a #GP, so a PS/2 or
|
||||
16550 driver goes through these), **DMA memory** (`dma_alloc`: contiguous, pinned,
|
||||
uncacheable, physical address exposed), **memory barriers** (`library/mmio`'s
|
||||
`mb`/`rmb`/`wmb`, imported as the `mmio` module), **fault isolation** (a ring-3 fault kills only the faulting
|
||||
uncacheable, physical address exposed), **memory barriers** (`library/device/mmio`'s
|
||||
`memoryBarrier`/`readMemoryBarrier`/`writeMemoryBarrier`, imported as the `mmio` module), **fault isolation** (a ring-3 fault kills only the faulting
|
||||
process — `killCurrentProcess` — and the machine keeps running,
|
||||
[resilience](resilience.md)), and **reclaim + restart on death** (every path out of a
|
||||
[resilience](../os-development/resilience.md)), and **reclaim + restart on death** (every path out of a
|
||||
process releases its claims and IRQ/MSI bindings — `releaseAllOwnedBy`,
|
||||
`irq.releaseOwner` — and the device manager respawns the driver with backoff,
|
||||
[device-manager.md](device-manager.md)). What remains:
|
||||
@@ -391,10 +391,10 @@ the first DMA driver to protect and test against) and these smaller items:
|
||||
Claiming and mapping is half of being a danos driver; the other half is the
|
||||
**lifecycle and protocol contract**, and the runtime makes it nearly free:
|
||||
|
||||
- Build on `runtime.service.run` — one replyWait loop folding protocol
|
||||
- Build on `service.run` — one replyWait loop folding protocol
|
||||
requests, signals, and notifications into callbacks. The harness answers the
|
||||
universal zero-length ping and turns `terminate` into a clean exit for you
|
||||
([process-lifecycle.md](process-lifecycle.md)).
|
||||
([process-lifecycle.md](../os-development/process-lifecycle.md)).
|
||||
- A driver spawned with an assignment (its device id as argv[1]) sends the
|
||||
versioned `hello` to the device manager inside the deadline, and a **bus**
|
||||
driver reports what it discovers with `child_added`
|
||||
@@ -5,14 +5,14 @@ window server, a logger. None of them owns the hardware, and the driver should n
|
||||
who is listening. So between the drivers and the listeners sits the **input service**
|
||||
(`system/services/input/`): drivers **publish** events to it, programs **subscribe**, and
|
||||
it fans each event out to every interested subscriber. It is an ordinary ring-3 process
|
||||
reached over IPC, like the [FAT server](../system/services/fat/fat.zig) — no kernel knows
|
||||
reached over IPC, like the [FAT server](../../system/services/fat/fat.zig) — no kernel knows
|
||||
what a key is.
|
||||
|
||||
## One service, several device classes
|
||||
|
||||
The service carries three device classes today — **keyboard**, **mouse**, and
|
||||
**joystick/gamepad** — and is built to take more
|
||||
([protocol.zig](../system/services/input/protocol.zig)). Each class has its own typed
|
||||
([protocol.zig](../../library/protocol/input/input-protocol.zig)). Each class has its own typed
|
||||
event:
|
||||
|
||||
- `KeyEvent` — `key_down`/`key_up` (physical make/break) and `key_press` (a character was
|
||||
@@ -22,12 +22,22 @@ event:
|
||||
- `JoystickEvent` — `axis` moves (a signed value on a `control` index) and
|
||||
`button_down`/`button_up`.
|
||||
|
||||
All three travel in one **`InputEvent` envelope** tagged with a `DeviceKind`, so the
|
||||
fan-out is a single code path and a subscriber can take a mix of classes on one stream.
|
||||
Decode an envelope with `asKeyboard()` / `asMouse()` / `asJoystick()` (each returns null
|
||||
unless the tag matches). A subscriber names the classes it wants with a **`device_mask`**,
|
||||
and the service routes each event only to subscribers whose mask includes its class — so a
|
||||
mouse-only listener never wakes for keystrokes.
|
||||
A source publishes any of the three as one **`InputEvent`** tagged with a `DeviceKind`, so
|
||||
`publish` is a single verb; decode one with `asKeyboard()` / `asMouse()` / `asJoystick()`
|
||||
(each returns null unless the tag matches). On the *delivery* wire the class is the
|
||||
packet's own operation instead — the protocol declares one event per class
|
||||
([protocol-namespace.md](../os-development/protocol-namespace.md)), so a pushed packet is
|
||||
the 16-byte header plus the typed event and nothing carries a tag twice. The client
|
||||
helpers re-tag what arrives back into an `InputEvent`, so a subscriber can still take a
|
||||
mix of classes on one stream. A subscriber names the classes it wants with a
|
||||
**`device_mask`**, and the service routes each event only to subscribers whose mask
|
||||
includes its class — so a mouse-only listener never wakes for keystrokes.
|
||||
|
||||
**`subscribe` is not this protocol's verb.** Its shape — a synchronous call whose attached
|
||||
capability is the subscriber's own endpoint — is what the envelope's *reserved* subscribe
|
||||
means at every provider in the system, so the input protocol adopts it rather than
|
||||
defining a second spelling of the same thing. The interest mask rides as the packet's
|
||||
tail. `publish` is the one verb the protocol defines for itself.
|
||||
|
||||
## Why this needed a new kernel primitive
|
||||
|
||||
@@ -45,7 +55,7 @@ consequences decide the whole design:
|
||||
`ipc_call`, it would block until each subscriber replied. `ipc_call` has no timeout, and
|
||||
a subscriber's endpoint is an *unregistered* capability the kernel's death path cannot
|
||||
reach (since display v2's V6, `killOwnedEndpointsLocked` in
|
||||
[ipc-synchronous.zig](../system/kernel/ipc-synchronous.zig) marks a dead owner's
|
||||
[ipc-synchronous.zig](../../system/kernel/ipc-synchronous.zig) marks a dead owner's
|
||||
*registered* endpoints dead and wakes parked callers with `-EPEER` — but unregistered
|
||||
ones just drop with the task's handle table). One subscriber that exits mid-delivery
|
||||
would wedge input for everyone. That is the opposite of the resilience the microkernel
|
||||
@@ -66,7 +76,7 @@ the badge (distinguishing it from a bare IRQ/child-exit notification), the sende
|
||||
in the low bits, and the payload in the receive buffer, with no reply owed. The queue holds
|
||||
16 messages per endpoint; a full queue **drops the oldest**, because a buffered message is
|
||||
discrete data, not a coalescing "level" like an interrupt. See
|
||||
[ipc-synchronous.zig](../system/kernel/ipc-synchronous.zig) (`sendLocked`, `popPost`, and
|
||||
[ipc-synchronous.zig](../../system/kernel/ipc-synchronous.zig) (`sendLocked`, `popPost`, and
|
||||
the `replyWait` receive loop).
|
||||
|
||||
This is the async counterpart of `ipc_call`, and the input service is its first consumer.
|
||||
@@ -89,7 +99,7 @@ This is the async counterpart of `ipc_call`, and the input service is its first
|
||||
- A **subscriber** calls `input.subscribe(mask)` — or a typed helper: `subscribeKeyboard()`,
|
||||
`subscribeMouse()`, `subscribeJoystick()` (one class, `next()` returns the decoded event),
|
||||
or `subscribeAll()` (every class, `next()` returns a tagged `InputEvent`)
|
||||
([library/runtime/input.zig](../library/runtime/input.zig)). It creates its own endpoint
|
||||
([library/client/input/input.zig](../../library/client/input/input.zig)). It creates its own endpoint
|
||||
and hands it to the service as a **capability** (M13 capability passing — the input
|
||||
service is that feature's first real user), along with its `device_mask`. Then it loops on
|
||||
`next()`, a `replyWait` on that endpoint returning each pushed event.
|
||||
@@ -98,12 +108,25 @@ This is the async counterpart of `ipc_call`, and the input service is its first
|
||||
`publishJoystickEvent`. Publishing is a short synchronous `ipc_call` the service answers at
|
||||
once; the service's own fan-out is asynchronous, so publishing never blocks on a slow
|
||||
subscriber.
|
||||
- The **service** ([input.zig](../system/services/input/input.zig)) keeps a small subscriber
|
||||
table (endpoint handle + owning task id + `device_mask`). On `publish` it `ipc_send`s the
|
||||
event to every subscriber whose mask includes the event's device class. On `subscribe` it
|
||||
stores the passed capability and mask and, as housekeeping, prunes any slot whose owning
|
||||
process has exited (checked against `process_enumerate`) — not for correctness (an async
|
||||
send to an orphaned endpoint is harmless) but to reclaim the slot.
|
||||
- The **service** ([input.zig](../../system/services/input/input.zig)) owns none of that
|
||||
machinery any more: the subscriber table (endpoint handle + owning task + interest mask),
|
||||
the reserved `subscribe`/`unsubscribe` verbs, the fan-out, and the dead-subscriber sweep
|
||||
are the shared harness's (`service.Subscribers` in
|
||||
[service.zig](../../library/kernel/service.zig)), so every event stream in the system has
|
||||
identical semantics. What is left in this file is what is actually about input: which
|
||||
class an event belongs to, and which classes a subscriber asked for. On `publish` it names
|
||||
the event's class and the harness `ipc_send`s the packet — framed once — to every
|
||||
subscriber whose mask includes it.
|
||||
- **A dead subscriber goes away on its exit notification**, not on a poll. The service used
|
||||
to walk `process_enumerate` on every subscribe and drop slots whose owner had gone; it now
|
||||
subscribes to the kernel's published exits like the FAT server and the compositor do
|
||||
([process-lifecycle.md](../os-development/process-lifecycle.md)), which reclaims the slot
|
||||
*and* closes the endpoint capability in it promptly rather than at the next subscribe.
|
||||
(The fan-out also drops a subscriber whose `ipc_send` fails, as a backstop for a
|
||||
notification a full ring dropped.)
|
||||
- The service runs on the shared harness like every other, so it answers the universal ping
|
||||
and exits on `terminate`; it was the last hand-rolled receive loop in the tree, and the
|
||||
last service a shutdown had to kill rather than ask.
|
||||
|
||||
Publisher and subscriber must be **separate processes**: a single thread that both
|
||||
published and serviced its own subscription would deadlock (its `publish` call blocks until
|
||||
@@ -113,18 +136,18 @@ the service delivers to its endpoint, which only the same thread could receive).
|
||||
|
||||
- **The keyboard is real.** The `ps2-bus` driver owns PNP0303, which carries *both* the
|
||||
0x60/0x64 ports and IRQ1, so reading the hardware lives in the bus, not in
|
||||
[keyboard.zig](../system/drivers/ps2-bus/keyboard.zig): the bus binds IRQ1 and, on each
|
||||
[keyboard.zig](../../system/drivers/ps2-bus/keyboard.zig): the bus binds IRQ1 and, on each
|
||||
interrupt, drains port 0x60, routing every byte by the status register's
|
||||
auxiliary-output bit to whichever child driver **attached** for that device (an
|
||||
`AttachRequest` to the well-known `ps2_bus` service, carrying the child's endpoint as a
|
||||
capability; the bytes then arrive as asynchronous `ForwardedByte` messages, so the IRQ
|
||||
path never blocks on a child). The keyboard driver decodes the stream — scancode **set 2**,
|
||||
what the keyboard sends with the 8042's legacy translation off, decoded by
|
||||
[scancode.zig](../system/drivers/ps2-bus/scancode.zig) into USB HID usage keycodes with
|
||||
[scancode.zig](../../system/drivers/ps2-bus/scancode.zig) into USB HID usage keycodes with
|
||||
make/break, typematic-repeat, and modifier tracking (host-tested under `zig build test`) —
|
||||
and publishes real `key_down`/`key_press`/`key_up` events.
|
||||
- **Keycode → character** is wired in: the keyboard driver fills a `key_press` event's
|
||||
`character` through [`library/xkeyboard-config`](../library/xkeyboard-config/README.md)
|
||||
`character` through [`library/xkeyboard-config`](../../library/xkeyboard-config/README.md)
|
||||
(`xkb.map(layout, keycode, mods)` → keysym + Unicode character), synthesizing the ASCII
|
||||
control characters for Enter/Tab/Backspace/Escape, whose keysyms map to no Unicode. The
|
||||
layout defaults to `us`; the bus can pass another as the driver's argv[2] — the seam for
|
||||
@@ -132,9 +155,9 @@ the service delivers to its endpoint, which only the same thread could receive).
|
||||
- **The mouse is real too.** IRQ12 is enumerated on the auxiliary device's own ACPI node
|
||||
(PNP0F13), so the bus claims that node alongside the controller and routes both IRQs to
|
||||
its one endpoint, acking whichever line the notification's badge names.
|
||||
[mouse.zig](../system/drivers/ps2-bus/mouse.zig) attaches the way the keyboard does and
|
||||
[mouse.zig](../../system/drivers/ps2-bus/mouse.zig) attaches the way the keyboard does and
|
||||
assembles the forwarded bytes with
|
||||
[mouse-packet.zig](../system/drivers/ps2-bus/mouse-packet.zig) (three-byte stream-mode
|
||||
[mouse-packet.zig](../../system/drivers/ps2-bus/mouse-packet.zig) (three-byte stream-mode
|
||||
packets: sync/overflow handling, nine-bit movement, screen-convention `dy` — host-tested
|
||||
under `zig build test`) into `button_down`/`button_up` transitions and `motion` events.
|
||||
**Follow-up:** the IntelliMouse magic-knock for a scroll wheel (four-byte packets) and
|
||||
@@ -150,7 +173,7 @@ the service delivers to its endpoint, which only the same thread could receive).
|
||||
## Verifying it
|
||||
|
||||
The `input` case (`python3 test/qemu_test.py input`, in
|
||||
[tests.zig](../system/kernel/tests.zig) `inputTest`) boots the real kernel and spawns the
|
||||
[tests.zig](../../system/kernel/tests.zig) `inputTest`) boots the real kernel and spawns the
|
||||
service, the synthetic source (which cycles keyboard, mouse, and joystick events), and a
|
||||
subscriber that took all three classes. It passes only when the subscriber heartbeats
|
||||
`input-test: ok` — proof that an event travelled source → service → subscriber over IPC,
|
||||
@@ -160,5 +183,5 @@ serial line names the class received, so the log shows all three arriving on one
|
||||
## See also
|
||||
|
||||
- [ipc.md](ipc.md) — the synchronous rendezvous and the notification path `ipc_send` extends.
|
||||
- [syscall.md](syscall.md) — the system-call surface, including `ipc_send`.
|
||||
- [syscall.md](../os-development/syscall.md) — the system-call surface, including `ipc_send`.
|
||||
- [driver-model.md](driver-model.md) — class drivers, capability passing (M13), the trust model.
|
||||
@@ -0,0 +1,199 @@
|
||||
# IPC: the kernel-ipc transport
|
||||
|
||||
Inter-process communication is the **backbone of a microkernel**. Once drivers and
|
||||
services run isolated in their own address spaces ([vision](../vision.md)), they can't
|
||||
just call each other — a request becomes bytes on a wire. In a microkernel, whatever
|
||||
was a function call across a monolithic kernel is IPC, so it's a first-class
|
||||
concern, not an afterthought.
|
||||
|
||||
This document describes **one transport** — the bottom layer (L0) of the
|
||||
communication stack defined in
|
||||
[communication.md](../os-development/communication.md), which owns the model
|
||||
and the vocabulary (*protocol*, *channel*, *packet*, *signal*, *endpoint*).
|
||||
kernel-ipc is the **first** transport, not the only possible one: in
|
||||
buffer-plus-doorbell terms it is a kernel-owned mailbox with the scheduler as
|
||||
the doorbell. Its distinguishing properties, which the layers above may rely
|
||||
on where they say so:
|
||||
|
||||
- **Rendezvous.** A call is a synchronous meeting, copied sender-page to
|
||||
receiver-page — natural backpressure, no queue to size.
|
||||
- **Capability carriage.** The *only* transport that can move a handle
|
||||
between processes. Channels are therefore always established over
|
||||
kernel-ipc, and it remains every channel's control path even when bulk
|
||||
data is negotiated onto a fatter transport (a shared-memory ring).
|
||||
- **Verified source.** Every delivery carries the kernel-stamped badge — the
|
||||
identity the channel layer attaches to received packets.
|
||||
- **Bounded packets.** 256 bytes call/reply, 64 pushed — the floor every
|
||||
protocol may assume on any transport.
|
||||
|
||||
Three properties keep the networking analogy honest — kernel-ipc is
|
||||
networking-*shaped*, not TCP:
|
||||
|
||||
- **Channels over it are RPC-shaped, not streams.** Packets, call/reply,
|
||||
datagram pushes — closer to UDP plus RPC than to a byte stream. Ordering
|
||||
exists per exchange (a reply answers its call), not across a channel.
|
||||
- **Possession is the connection.** There is no handshake state in the
|
||||
kernel: holding the capability *is* having the channel. A provider's one
|
||||
endpoint terminates every client's channel at once, demultiplexed by badge
|
||||
— like every client sharing the server's listening socket, with
|
||||
per-connection state living in the provider, keyed by badge. A *private*
|
||||
channel (a dedicated endpoint pair) is built when wanted: that is exactly
|
||||
what `subscribe` does.
|
||||
- **Packets never fragment.** If it doesn't fit in a packet, it isn't a
|
||||
packet: bulk data lives in shared memory and a packet (or signal) is the
|
||||
doorbell. The display path already works this way.
|
||||
|
||||
The rest of this document is the implementation, bottom-up: the kernel-thread
|
||||
queue the blocking discipline was worked out on, then endpoints — this
|
||||
transport's termination points.
|
||||
|
||||
## The kernel-thread queue
|
||||
|
||||
The first form is a **bounded blocking queue** (`system/kernel/ipc.zig`): a
|
||||
fixed-size ring buffer of messages with a producer/consumer rendezvous, built
|
||||
on the scheduler's [wait queues](../os-development/scheduling.md). (Its type
|
||||
is still named `Channel(T, capacity)` — it predates the vocabulary above, and
|
||||
is a *queue between kernel threads in one address space*, not a channel in
|
||||
the model's sense; a rename can ride a later flag-day.)
|
||||
|
||||
`Channel(T, capacity)` is generic over the message type and buffer size. It holds a
|
||||
ring buffer, a count, and two wait queues:
|
||||
|
||||
- **`send(msg)`** — if the queue is full, block on the *not-full* queue; otherwise
|
||||
write the message, bump the count, and wake a waiting receiver.
|
||||
- **`receive()`** — if the queue is empty, block on the *not-empty* queue; otherwise
|
||||
take a message, drop the count, and wake a waiting sender.
|
||||
|
||||
Neither side busy-waits: a full queue parks the sender, an empty one parks the
|
||||
receiver, and each operation wakes the other side when it makes progress possible.
|
||||
|
||||
Two details make it correct:
|
||||
|
||||
- **Recheck in a loop.** A woken task re-tests the condition (`while (full) wait`)
|
||||
rather than assuming the slot is still available — another waiter may have taken
|
||||
it first. This is the standard guard against spurious or racing wakeups.
|
||||
- **One critical section.** `send`/`receive` run under the [big kernel
|
||||
lock](../os-development/smp.md) (`sync.enter` / `sync.leave`), which disables interrupts on this
|
||||
core *and* takes the kernel's one spinlock — since SMP, the interrupt flag alone
|
||||
is not atomicity, because `cli` on one core does nothing to another. So checking
|
||||
the condition and committing the block/enqueue happen atomically both with respect
|
||||
to the timer preempting mid-operation and to the other side running on another
|
||||
CPU. `waitLocked` / `wakeLocked` are the variants that assume the caller already
|
||||
holds that critical section.
|
||||
|
||||
### Verifying it
|
||||
|
||||
The `ipc` test (see [testing.md](../testing.md)) runs a producer and a consumer passing
|
||||
**100 messages through a 4-slot queue**. The small buffer means the queue goes
|
||||
full and empty over and over, so both the blocking-send and blocking-receive paths are
|
||||
exercised heavily. The messages arrive intact and in order (their sum is the
|
||||
expected `5050`), and neither task busy-waits — they block and wake each other.
|
||||
|
||||
## Endpoints: the termination points
|
||||
|
||||
A queue connects two kernel threads sharing one address space. Real providers are
|
||||
*processes*, so a packet has to cross an address-space boundary. That's
|
||||
`system/kernel/ipc-synchronous.zig`, and its shape is L4's: a synchronous **rendezvous** at an
|
||||
`Endpoint`, with the packet copied directly from the sender's pages to the receiver's
|
||||
(`copyAcross` walks both sets of page tables through the physmap — no CR3 switch, no
|
||||
bounce buffer).
|
||||
|
||||
Two syscalls carry the request/reply exchange:
|
||||
|
||||
- **`ipc_call(h, msg, reply)`** — copy the request packet to the provider, block
|
||||
until the reply packet comes back.
|
||||
- **`ipc_reply_wait(h, reply, recv)`** — reply to the client you're still holding (if
|
||||
any), then block for the next request. One syscall, because a provider's steady state
|
||||
is *always* "finish the last one, wait for the next".
|
||||
|
||||
An endpoint is reached by **handle** — a small integer index into the process's handle
|
||||
table (`Task.handles`), exactly like a file descriptor, and just as unforgeable.
|
||||
The provider never learns the client's identity beyond the **badge** delivered
|
||||
alongside each packet: the caller's task id, stamped by the kernel —
|
||||
unforgeable source addressing, a property a network's source field lacks.
|
||||
|
||||
The bootstrap problem — how a channel is first established — is the subject of
|
||||
[protocol-namespace.md](../os-development/protocol-namespace.md): a protocol is
|
||||
resolved by name and the channel arrives as a capability. (The mechanism it
|
||||
replaced — `ipc_register`/`ipc_lookup` under compile-time `ServiceId` integers —
|
||||
is gone: both syscalls and the enum were deleted when the registry landed, and
|
||||
their syscall numbers are left vacant.)
|
||||
|
||||
### Interrupts are signals
|
||||
|
||||
`notifyFromIsr` posts an *asynchronous* signal to an endpoint — no payload, no
|
||||
reply owed — and wakes whoever is blocked in `reply_wait`. Its badge has the top bit
|
||||
set (`notify_badge_bit`), which is how a driver's single event loop distinguishes "a
|
||||
client wants something" from "the hardware wants something". Signals sit in a
|
||||
small coalescing ring on the endpoint, so an interrupt taken while the driver was busy
|
||||
elsewhere is not lost — coalesced, never dropped, which is exactly a signal's
|
||||
contract (the *count* may collapse; the *fact* may not).
|
||||
|
||||
This is what makes a user-space driver possible at all, and it's the subject of
|
||||
[drivers.md](drivers.md).
|
||||
|
||||
## What's next (partly done since)
|
||||
|
||||
- **Priority inheritance** through IPC — still open: a high-priority client
|
||||
blocked on a low-priority provider suffers unbounded priority inversion.
|
||||
- **Handle transfer.** *Landed as cap-passing (M13)*: `ipc_call` and
|
||||
`ipc_reply_wait` carry an optional capability alongside the bytes (`send_cap`),
|
||||
copying an endpoint or shared-memory handle into the peer's table — the
|
||||
mechanism by which channels are established and private channels built. First
|
||||
user: [input](input.md) subscribers register by handing over their own
|
||||
endpoint, and class drivers get a private channel to one device.
|
||||
- **Asynchronous / buffered send** for the cases where a rendezvous is the wrong
|
||||
shape (logging, event fan-out). *Landed as `ipc_send`* — a
|
||||
non-blocking post of an event packet (≤ 64 bytes) to an endpoint's bounded
|
||||
queue, delivered through `reply_wait` (badge bit `notify_message_bit`). Built
|
||||
for, and first used by, the [input service](input.md)'s keyboard-event
|
||||
broadcast, where a synchronous push would let one dead subscriber hang the
|
||||
fan-out. A full queue drops the oldest — event packets are droppable by
|
||||
design ([protocol-namespace.md](../os-development/protocol-namespace.md)'s
|
||||
wiring section states the rule).
|
||||
- **A bounded reply** — half landed. The copy is still one packet
|
||||
(256 bytes) under the big kernel lock, but bulk transfer got its shared
|
||||
pages: `shared_memory_create`/`map`/`physical`, the region handle delegated as
|
||||
a capability (above) — the packets-never-fragment rule in practice.
|
||||
virtio-gpu's scanout surface is the first user
|
||||
([display-v2.md](display-v2.md)).
|
||||
|
||||
## Lifecycle conventions over IPC (M17)
|
||||
|
||||
Three conventions from [process-lifecycle.md](../os-development/process-lifecycle.md) ride the
|
||||
signal mechanism:
|
||||
|
||||
- **Process signals** arrive as endpoint signals on the endpoint a process
|
||||
nominated with `signal_bind` (`process.bindSignals`): badge = the signal bit
|
||||
plus the coalesced pending mask (`process.signalsFrom` decodes). Statements,
|
||||
never questions; no payload, no reply.
|
||||
- **One-shot timers** (`timer_bind`, `time.timerOnce`) land as a
|
||||
timer-bit signal — the timed wait: a service arms a deadline and keeps
|
||||
serving, instead of blocking in sleep.
|
||||
- **Kernel notifications go only to your own endpoint.** `signal_bind`,
|
||||
`timer_bind`, `process_subscribe`, `irq_bind`, `msi_bind`, and spawn's exit
|
||||
endpoint all *nominate where the kernel will speak*, and all of them refuse an
|
||||
endpoint the caller did not create (`-EPERM`; the check is `ipc.ownedBy`,
|
||||
normalized to the process, so any thread may nominate an endpoint a sibling
|
||||
created). Holding a handle is not enough, because holding a handle is cheap:
|
||||
`fs_resolve` installs a mounted backend's capability in *any* caller's table,
|
||||
so every process holds a handle to PID 1's mailbox. Without the rule, "bind
|
||||
init's endpoint, then signal yourself" is a genuine, kernel-stamped `terminate`
|
||||
badge in PID 1's queue — a shutdown a receiver has no way to disbelieve — and
|
||||
timers, which carry no identity at all, multiply any loop that re-arms on its
|
||||
own landing.
|
||||
- **A capability that arrives belongs to the turn.** The kernel installs a sent
|
||||
capability in the receiver's table whatever the message's length or kind, so a
|
||||
receive loop must dispose of one on *every* path — the ping, the notification,
|
||||
the malformed request. The service harness (`service.run`) and PID 1 both hold
|
||||
it in an `ipc.Arrival`, released by a `defer`, and a handler that means to keep
|
||||
it says `take()`: forgetting closes, keeping is explicit. The reverse
|
||||
arrangement leaks a handle-table slot per request, and thirty-two unauthorized
|
||||
zero-length pings then end a service's ability to accept any capability —
|
||||
no subscribe, no shared-memory handover — for the rest of the boot.
|
||||
- **The universal ping**: a **zero-length request is the liveness probe**,
|
||||
answered with a zero-length reply by the service harness itself
|
||||
(`service.run`). No protocol's requests start at length zero, so the
|
||||
encoding cannot collide, and a wedged service simply fails to answer — which
|
||||
is the diagnosis. Deep health ("can I reach my hardware?") stays a per-service
|
||||
protocol packet.
|
||||
@@ -0,0 +1,218 @@
|
||||
# New driver: the minimum steps
|
||||
|
||||
The shortest path from "a device shows up in the boot log" to "my process is
|
||||
running with its registers mapped". This is the checklist; the reasoning behind
|
||||
every step lives in [Writing a driver](drivers.md), the matching rules in
|
||||
[devices.csv](devices-csv.md), and interrupts in
|
||||
[device interrupts](device-interrupts.md).
|
||||
|
||||
Worked example throughout: the Intel UHD 750 iGPU, which the boot log reports as
|
||||
|
||||
```
|
||||
pci-bus: 0:2.0 bus=pci base=03 class=00 prog_if=00 vendor=8086 device=4C8A ...
|
||||
```
|
||||
|
||||
## 1. Create the source file
|
||||
|
||||
`system/drivers/<name>/<name>.zig` — kebab-case, abbreviations spelled out
|
||||
([coding standards](../coding-standards.md)). The directory name, the binary
|
||||
name, and the `devices.csv` driver path must all agree; a mismatch fails
|
||||
silently (the device-manager logs the spawn failure, nothing else happens).
|
||||
|
||||
The complete minimal driver — claims its device, logs every resource, maps the
|
||||
register window, then sleeps in the harness loop:
|
||||
|
||||
```zig
|
||||
//! /system/drivers/intel-uhd-graphics-750 — spawned by the device manager with
|
||||
//! the device-tree id as argv[1]; claims that device and no other.
|
||||
|
||||
const std = @import("std");
|
||||
const device = @import("driver");
|
||||
const ipc = @import("ipc");
|
||||
const memory = @import("memory");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
|
||||
/// No protocol yet: the kernel's IPC ceiling (MESSAGE_MAXIMUM) sizes the buffers.
|
||||
const message_maximum = 256;
|
||||
|
||||
var controller_id: u64 = 0;
|
||||
var register_base: usize = 0;
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
_ = endpoint; // needed later, for irq binding and timers
|
||||
|
||||
if (!device.claim(controller_id)) {
|
||||
std.log.err("unable to claim device {d}", .{controller_id});
|
||||
return false;
|
||||
}
|
||||
|
||||
// Fetch our own descriptor back for the device's resources.
|
||||
const buffer = memory.allocator().alloc(device.DeviceDescriptor, 64) catch return false;
|
||||
defer memory.allocator().free(buffer);
|
||||
const total = device.enumerate(buffer);
|
||||
const descriptor = for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||
if (d.id == controller_id) break d;
|
||||
} else {
|
||||
std.log.err("device {d} not in the device tree", .{controller_id});
|
||||
return false;
|
||||
};
|
||||
|
||||
// Log every resource BEFORE choosing one (see step 5).
|
||||
var register_index: u64 = 0;
|
||||
for (descriptor.resources[0..@intCast(descriptor.resource_count)], 0..) |resource, index| {
|
||||
std.log.info("resource {d}: kind={d} start=0x{x} len=0x{x}", .{
|
||||
index, resource.kind, resource.start, resource.len,
|
||||
});
|
||||
// The 16 MiB window is GTTMMADR, the register BAR (this device also has
|
||||
// a 256 MiB memory BAR, GMADR — "first memory resource" would be wrong).
|
||||
if (resource.kind == @intFromEnum(device.ResourceKind.memory) and
|
||||
resource.len == 16 * 1024 * 1024) register_index = index;
|
||||
}
|
||||
if (register_index == 0) {
|
||||
std.log.err("register BAR not found", .{});
|
||||
return false;
|
||||
}
|
||||
|
||||
register_base = device.mmioMap(controller_id, register_index) orelse {
|
||||
std.log.err("mmio_map failed", .{});
|
||||
return false;
|
||||
};
|
||||
std.log.info("registers mapped at 0x{x}", .{register_base});
|
||||
return true;
|
||||
}
|
||||
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||
_ = message;
|
||||
_ = reply;
|
||||
_ = sender;
|
||||
_ = capability;
|
||||
return 0; // no protocol yet; the zero-length ping is answered by the harness
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
const argument = init.arguments.get(1) orelse {
|
||||
std.log.err("missing device id (argv[1])", .{});
|
||||
return;
|
||||
};
|
||||
controller_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
std.log.err("malformed device id '{s}'", .{argument});
|
||||
return;
|
||||
};
|
||||
service.run(message_maximum, .{
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
// .on_notification only once an IRQ or timer is bound
|
||||
});
|
||||
}
|
||||
```
|
||||
|
||||
`claim` is the capability gate: MMIO mapping, DMA grants, and IRQ binding all
|
||||
require it, and it pins the IOMMU domain to this process
|
||||
([drivers.md — claim before touch](drivers.md#the-capability-claim-before-touch)).
|
||||
|
||||
## 2. Create the build package and register it in the root build
|
||||
|
||||
The driver directory is its own build package
|
||||
([build-packages-plan.md](../build-packages-plan.md)): a ~15-line `build.zig`
|
||||
plus a `build.zig.zon` beside the source. Copy both from an existing driver —
|
||||
`system/drivers/pci-bus/` is the template — and adjust the name, root source
|
||||
file, and the import list. The list names EXACTLY the modules the driver's
|
||||
source `@import`s (the moral equivalent of its include list; an undeclared
|
||||
import is a compile error):
|
||||
|
||||
```zig
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "intel-uhd-graphics-750",
|
||||
.root_source_file = b.path("intel-uhd-graphics-750.zig"),
|
||||
.imports = &.{ "driver", "ipc", "memory", "process", "service" },
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
}
|
||||
```
|
||||
|
||||
The zon declares `build-support`, `kernel` (implicit in every binary: the root
|
||||
shim lives there), and the homes of the listed imports — for the minimal
|
||||
driver above that is kernel alone plus `device` (for `driver`); add
|
||||
`protocol`, `client`, ... only when an import comes from them (again, copy
|
||||
pci-bus's zon and adjust). For the `.fingerprint` field, leave the copied
|
||||
value in place and `zig build` will reject it and suggest the fresh one to
|
||||
paste.
|
||||
|
||||
Then three one-liners in the root build register the package: the dependency
|
||||
and a row in the boot-tree array in `build.zig` (search for
|
||||
`virtio_gpu_package` to land in the right places),
|
||||
|
||||
```zig
|
||||
const intel_uhd_graphics_750_exe = b.dependency("intel-uhd-graphics-750", .{}).artifact("intel-uhd-graphics-750");
|
||||
```
|
||||
|
||||
```zig
|
||||
.{ .path = "system/drivers/intel-uhd-graphics-750", .binary = intel_uhd_graphics_750_exe.getEmittedBin() },
|
||||
```
|
||||
|
||||
and the path entry in the root `build.zig.zon`:
|
||||
|
||||
```zig
|
||||
.@"intel-uhd-graphics-750" = .{ .path = "system/drivers/intel-uhd-graphics-750" },
|
||||
```
|
||||
|
||||
Without the boot-tree row the binary never reaches the image and the
|
||||
device-manager has nothing to spawn. (The package also builds standalone:
|
||||
`cd system/drivers/intel-uhd-graphics-750 && zig build`.)
|
||||
|
||||
## 3. Add the match rule to `etc/devices.csv`
|
||||
|
||||
One row: bus, class triplet, vendor/device, driver path. **Copy the class
|
||||
triplet from the pci-bus boot log line, not from another row** — for the iGPU
|
||||
above the correct rule is
|
||||
|
||||
```
|
||||
pci, 03, 00, 00, 8086, 4C8A, *, *, /system/drivers/intel-uhd-graphics-750
|
||||
```
|
||||
|
||||
Field-by-field rules and the most-specific-wins policy: [devices.csv](devices-csv.md).
|
||||
The registry is authoritative: an unmatched device is logged unbound, never
|
||||
guessed — so a wrong nibble here means the driver simply never starts.
|
||||
|
||||
## 4. First contact: read, predict, verify
|
||||
|
||||
Before writing any register, read one whose value you can predict from state
|
||||
the firmware already programmed (for a display controller: the pipe source
|
||||
size of the live mode). Registers are volatile loads at `register_base +
|
||||
offset`, where `offset` is what the device's manual lists:
|
||||
|
||||
```zig
|
||||
fn read32(offset: usize) u32 {
|
||||
return @as(*volatile u32, @ptrFromInt(register_base + offset)).*;
|
||||
}
|
||||
```
|
||||
|
||||
A matching read proves the whole chain — CSV match, spawn, claim, BAR choice,
|
||||
mapping — with zero risk to the hardware.
|
||||
|
||||
## 5. Verify the plumbing
|
||||
|
||||
- `zig build test` still passes.
|
||||
- On the image: `/var/log/<boot-stamp>/system/services/device-manager.log`
|
||||
shows `spawned <name> for device <N>`, and
|
||||
`/var/log/<boot-stamp>/system/drivers/<name>.log` holds the resource list and
|
||||
your first read.
|
||||
- If the driver did not spawn, diagnose in this order: binary on the image
|
||||
(step 2) → CSV row matches the log line exactly (step 3) → path identical in
|
||||
both (step 1).
|
||||
|
||||
## Later, when the device needs them
|
||||
|
||||
- **Interrupts**: MSI/MSI-X via the `pci` module, delivered as notifications to
|
||||
`on_notification` — see [device interrupts](device-interrupts.md) and the
|
||||
xHCI driver's `setupMsi` (QEMU trap documented there: enable MSI-X before
|
||||
unmasking the device's own interrupt-enable bit).
|
||||
- **DMA**: grant-backed buffers, bounded by the IOMMU domain established at
|
||||
claim time ([driver model](driver-model.md)).
|
||||
- **Children**: a bus driver publishes what it finds via `device_register`
|
||||
([drivers.md — publishing children](drivers.md#publishing-children-device_register)).
|
||||
- **A protocol**: replace `message_maximum` with the protocol's own maximum and
|
||||
dispatch on the operation word in `onMessage` — every service under
|
||||
`system/services/` is an example.
|
||||
@@ -0,0 +1,124 @@
|
||||
# Dynamic libraries on danos
|
||||
|
||||
A design note and milestone plan for shared objects: building them, loading them
|
||||
with `dlopen`, and — the part that needs kernel work — actually *sharing* them
|
||||
between processes. Directional, post-P5 of
|
||||
[python-on-danos-milestones.md](python-on-danos-milestones.md); nothing on the
|
||||
CPython bring-up path depends on it.
|
||||
|
||||
## Reconciling the earlier "rejected"
|
||||
|
||||
Dynamic libraries were evaluated once before and rejected — but as an answer to a
|
||||
*different question*: whether they could claw back ReleaseSafe's measured ~2×
|
||||
code size. They cannot (the safety checks inline at every call site; no library
|
||||
scheme dedups them), and that verdict stands for that question. The reasons to
|
||||
build them now are the ones that investigation never weighed:
|
||||
|
||||
- **`ctypes` and runtime FFI** — Python calling into a danos library without
|
||||
rebuilding the interpreter. This is the piece that makes Python prototyping
|
||||
self-serve: drop a `.so` on the image, `ctypes.CDLL` it, iterate.
|
||||
- **Loadable CPython extension modules** — today every C extension means
|
||||
relinking the interpreter (`Modules/Setup`); with `dlopen`, an extension is a
|
||||
file.
|
||||
- **One interpreter image, many Python services** — a statically-linked CPython
|
||||
is tens of megabytes *per process*. A shared `libpython` mapped read-only once
|
||||
(milestone D3 below) makes Python services cheap enough to be the default way
|
||||
to prototype one.
|
||||
- **Plugin-shaped applications** — the UI toolkit and the terminal will want
|
||||
them eventually.
|
||||
|
||||
The scoping that dissolves the apparent contradiction is the **size doctrine**:
|
||||
leanness is an *operating-system* property — the kernel and system services stay
|
||||
small and statically linked, and none of them ever link the loader — while
|
||||
*applications* have their own budget and may be big. Dynamic libraries are an
|
||||
**application-layer facility**, full stop.
|
||||
|
||||
What also does **not** change: the public ABI stays the vDSO + the IPC
|
||||
protocols. Shared objects are artifacts *within* one system image, versioned by
|
||||
the build — not a new stable ABI surface for the OS.
|
||||
|
||||
## Design
|
||||
|
||||
- **Format and codegen are free.** ELF shared objects with position-independent
|
||||
code; `zig cc -fPIC -shared` against the [libdanos-c](c-library-compatibility.md)
|
||||
sysroot already emits them. The work is entirely on the loading side.
|
||||
- **The loader lives in userspace, inside the libc.** `dlopen` reads the `.so`
|
||||
through the VFS, maps its segments, applies relocations, resolves symbols
|
||||
against the process and the `DT_NEEDED` dependency graph, runs constructors,
|
||||
returns a handle. No kernel loader changes in v1 — segments land in anonymous
|
||||
`mmap` as private copies.
|
||||
- **Bind-now, always.** All relocations resolved at `dlopen` time
|
||||
(`RTLD_NOW` semantics only). Lazy PLT binding buys startup latency danos does
|
||||
not care about, at the price of a writable GOT dance and a much subtler
|
||||
loader. Not worth it; keep it out permanently.
|
||||
- **W^X from day one.** Map, relocate, then flip text pages read-execute —
|
||||
which requires memory-protection change (`mprotect`-shaped) in the danos
|
||||
`mmap` surface if it is not already there. No page is ever writable and
|
||||
executable at once.
|
||||
- **TLS in shared objects is deferred.** Thread-local storage models
|
||||
(initial-exec vs. general-dynamic) are the deep end of every dynamic linker.
|
||||
v1 refuses a `.so` with a TLS segment; revisit alongside the post-P5 pthread
|
||||
subset, which is when it could matter.
|
||||
- **Executables stay static until D4.** v1 is "a static binary that can
|
||||
`dlopen`" — no `PT_INTERP`, no program interpreter, no dynamically-linked
|
||||
`main` binaries. That keeps process startup untouched.
|
||||
|
||||
## Milestones
|
||||
|
||||
1. **D1 — dlopen in-process.** The `.so` build target; the loader in libdanos-c:
|
||||
map, relocate (`RELATIVE`/`GLOB_DAT`/`JUMP_SLOT`), resolve, constructors;
|
||||
`dlopen`/`dlsym`/`dlerror`/`dlclose`; private anonymous mappings; no TLS.
|
||||
*Test:* QEMU `dlopen-hello` — load a `.so`, call a symbol, unload, reload.
|
||||
2. **D2 — the FFI payoff.** `DT_NEEDED` dependency graphs; a **libffi port**
|
||||
(x86-64 SysV assembly is upstream; the port is its closure-allocation paths,
|
||||
which must respect W^X); CPython's `ctypes` enabled; extension modules
|
||||
loadable from file. *Test:* QEMU — a Python script `ctypes.CDLL`s a danos
|
||||
`.so` and round-trips a call; a `.so` extension module imports.
|
||||
3. **D3 — actual sharing (the kernel milestone).** Shared read-only file-backed
|
||||
mappings — a page-cache-shaped facility so N processes mapping `libpython`
|
||||
hold one physical copy. This is the memory-win milestone and the only one
|
||||
touching the kernel; design it with the existing shm machinery in view
|
||||
(the shared-fate walks already locked the relevant paths). *Test:* N Python
|
||||
services up; measure physical pages against N× the static baseline.
|
||||
4. **D4 — dynamically-linked executables** (optional, evaluate after D3):
|
||||
`PT_INTERP`, a danos program interpreter, and the spawn path teaching the
|
||||
loader about it. Only worth it if the image-size or update story demands it.
|
||||
|
||||
## Risks and gotchas
|
||||
|
||||
- **Scope creep is the failure mode.** Every dynamic linker grows toward glibc.
|
||||
The fences: bind-now only, no lazy binding ever, no TLS until pthreads demand
|
||||
it, no dlopen-from-memory, no versioned symbols. Each fence removed is a
|
||||
design discussion, not a patch.
|
||||
- **Code loading is a security event.** `dlopen` turns file bytes into executable
|
||||
code, so W^X discipline is table stakes and *what may be dlopened* is a
|
||||
capability question — the natural danos answer is that loadability follows VFS
|
||||
readability of the `.so`, and services' images are supervised like any other
|
||||
artifact. Revisit explicitly at D3 when mappings become shared.
|
||||
- **`dlclose` is where loaders go to die.** Constructors/destructors,
|
||||
dangling function pointers, re-open identity. Keep v1 semantics honest and
|
||||
simple: `dlclose` runs destructors and unmaps; holding pointers past it is
|
||||
undefined; no reference-counted deferral cleverness.
|
||||
- **The ReleaseSafe fact still applies to `.so`s** — a ReleaseSafe shared object
|
||||
carries its inlined checks like any static code; D3's sharing saves *copies*,
|
||||
not check overhead. Size expectations should be set accordingly.
|
||||
|
||||
## Decisions needing sign-off
|
||||
|
||||
- Dynamic libraries join the roadmap at all (this note exists because the
|
||||
earlier size-motivated rejection was re-opened for ABI/sharing reasons).
|
||||
- **Bind-now only; no lazy binding, permanently.**
|
||||
- **Loader in userspace libc; kernel involvement only at D3** (shared read-only
|
||||
mappings).
|
||||
- **Static executables until D4**, and D4 only on demonstrated need.
|
||||
|
||||
## Related
|
||||
|
||||
- [c-library-compatibility.md](c-library-compatibility.md) — the sysroot the
|
||||
loader ships in; its absence table gains `dlfcn.h` at D1.
|
||||
- [python-on-danos.md](python-on-danos.md) — the `ctypes` story this unlocks.
|
||||
- [python-on-danos-milestones.md](python-on-danos-milestones.md) — sequencing;
|
||||
this work is post-P5.
|
||||
- [os-development/memory-map.md](os-development/memory-map.md) /
|
||||
[os-development/paging.md](os-development/paging.md) — where W^X and shared
|
||||
mappings land.
|
||||
@@ -0,0 +1,90 @@
|
||||
# The danos file-system hierarchy
|
||||
|
||||
danos is not unix, and its tree does not follow the unix FHS. Paths are the
|
||||
system's universal namespace — files, the device inventory, and protocol
|
||||
endpoints all live in one tree — but what a path *yields* differs by subtree:
|
||||
bytes, facts, or a connection. Root path resolution is provided by the
|
||||
kernel-resident VFS root (`fs_resolve`, `system/kernel/vfs.zig`); mounted
|
||||
backends serve the subtrees they own.
|
||||
|
||||
Naming follows the codebase conventions: kebab-case, full words, no
|
||||
abbreviations. Every top-level name says what its subtree *is*.
|
||||
|
||||
## The tree
|
||||
|
||||
| Path | What it is |
|
||||
|-------------------------|---------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| `/` | The root of the one namespace. |
|
||||
| `/applications` | Installed applications, one directory per application — the directory is the identity, the same rule as source sub-projects. *(Planned; empty today.)* |
|
||||
| `/protocol` | The contract namespace: one protocol node per contract, grouped into directories by domain (`/protocol/display`, `/protocol/networking/ip`). Synthetic — no bytes; opening a name yields a connection to the current provider. See [protocol-namespace.md](../os-development/protocol-namespace.md). |
|
||||
| `/system` | The operating system — what danos *is*. Its program subtrees mirror the source tree exactly. |
|
||||
| `/system/kernel` | The kernel image. |
|
||||
| `/system/drivers` | Driver binaries, one per sub-project (`/system/drivers/pci-bus`, `/system/drivers/ps2-bus`). |
|
||||
| `/system/services` | System-service binaries (`/system/services/init`, `/system/services/fat`). |
|
||||
| `/system/devices` | The device inventory: every node hardware discovery found, with its resources and parent — the structures of the devices module, as a browsable virtual tree. Informational only; you *read about* hardware here and *talk to* it through `/protocol`. *(Planned; served by device-manager.)* |
|
||||
| `/system/configuration` | Machine configuration (`init.csv`, `devices.csv`). Writable, served from the boot volume. |
|
||||
| `/system/logs` | Per-boot logs: `/system/logs/<boot-stamp>/<binary-path>.log`. Writable, served from the boot volume. |
|
||||
| `/test` | Test fixtures for the QEMU integration suite. Read-only and initrd-backed like the program subtrees of `/system`, mirroring the repo's `test/` directory. Present on development and test images; a volume without it still boots. |
|
||||
| `/volumes` | Attached storage volumes, one directory per volume (`/volumes/usb`). A volume's own tree appears beneath its name. |
|
||||
|
||||
Read-only and writable halves of `/system`: the program subtrees (`kernel`,
|
||||
`drivers`, `services`) and the future `devices` are immutable at runtime —
|
||||
initrd-backed or synthetic — while `configuration` and `logs` are mutable
|
||||
machine state served by the boot-volume FAT backend. The kernel's
|
||||
reserved-prefix rule (no mount may shadow `/system`, `/test`, or `/protocol`)
|
||||
needs a carve-out for exactly these two writable subtrees; that lands with the
|
||||
path migration below.
|
||||
|
||||
Deliberately not defined yet: a temporary-files location and per-application
|
||||
mutable storage. Both belong to the `/applications` design and will be
|
||||
specified there, not guessed at here.
|
||||
|
||||
## Node kinds
|
||||
|
||||
What a path resolves to. These fill `FileStatus.kind` and
|
||||
`DirectoryEntry.kind` in the [vfs protocol](vfs-protocol.md)
|
||||
(`library/protocol/vfs/vfs-protocol.zig`); enum values are append-only.
|
||||
|
||||
| Kind | Meaning |
|
||||
|--------------------|-------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| `regular` | An ordinary file: an uninterpreted byte stream, positional reads and writes, grows on demand. |
|
||||
| `directory` | A container mapping names to nodes; modified only through directory operations. |
|
||||
| `character_device` | A node whose read/write have **stream semantics**: unseekable, reads block until bytes exist, size is meaningless. The console and every tty-shaped node ([character-devices-and-tty.md](../character-devices-and-tty.md)); what a POSIX layer's `isatty` detects. |
|
||||
| `block_device` | A node addressed in fixed-size sectors — a raw volume. Reserved: recognized, nothing serves one yet. |
|
||||
| `symbolic_link` | Reserved: a recognized value, not implemented by any backend. |
|
||||
| `fifo` | Reserved for the future pipe object (wanted by the POSIX compatibility layer); not implemented. |
|
||||
| `protocol` | A node naming a contract: `open` yields an IPC connection (an endpoint capability) instead of a file id — the kind of every leaf under `/protocol`. *(Being added; see protocol-namespace.md.)* |
|
||||
|
||||
Note the layering: `protocol` says what *opening the name* does (you get a
|
||||
conversation); `character_device`/`block_device` say what *read and write
|
||||
mean* on a node a provider serves you. The two compose — `/protocol/console`
|
||||
is a protocol node in the registry, and the node opened over that connection
|
||||
reports `character_device`, which is what gives it stream semantics. Only
|
||||
`socket` is retired (its value stays reserved for wire stability): a named
|
||||
rendezvous point is exactly what a protocol node is.
|
||||
|
||||
## What is deliberately absent
|
||||
|
||||
There is no `/bin`, `/boot`, `/dev`, `/etc`, `/home`, `/lib`, `/mnt`, `/sbin`,
|
||||
`/srv`, `/tmp`, `/usr`, or `/var`. These encode unix history — the
|
||||
binary/library split of small disks, configuration-as-scattered-text, devices
|
||||
as magic files — that danos does not carry. A POSIX compatibility layer (the
|
||||
Python track's mini-libc) may *present* whichever of these its programs
|
||||
expect, mapped onto the real tree; the tree itself stays danos-native.
|
||||
|
||||
## Migration
|
||||
|
||||
The tree above is the specification; some code still writes the unix paths it
|
||||
replaced. The flag-day converting them:
|
||||
|
||||
| Today (in code) | Becomes | Where |
|
||||
|------------------------------------------|-------------------------------------------|-----------------------------------------------------------------|
|
||||
| `/etc/init.csv` | `/system/configuration/init.csv` | `system/services/init/init.zig` |
|
||||
| `/etc/devices.csv` | `/system/configuration/devices.csv` | `system/services/device-manager/device-manager.zig` |
|
||||
| `/var/log/...` | `/system/logs/...` | `system/services/logger/logger.zig`, the FAT server's `/var` mount |
|
||||
| `/mnt/usb` | `/volumes/usb` | `system/services/fat/fat.zig`, the fat/vfs tests |
|
||||
| `ServiceId` lookup | resolve + open under `/protocol` | every service and client; [protocol-namespace.md](../os-development/protocol-namespace.md) |
|
||||
|
||||
The boot-image builder and the on-volume directory layout move in the same
|
||||
change, so a freshly written image and the paths the services expect never
|
||||
disagree.
|
||||
@@ -0,0 +1,251 @@
|
||||
# The VFS wire protocol
|
||||
|
||||
> **Status:** built and spoken today between `file_system` (the client) and the
|
||||
> filesystem BACKENDS (the FAT server). The mount router lives in the
|
||||
> **kernel** (`system/kernel/vfs.zig`): `fs_resolve` routes a path and either
|
||||
> serves it directly (the read-only /system initrd mount, via `fs_node`) or
|
||||
> redirects the caller to the owning backend's endpoint plus the rewritten
|
||||
> mount-relative path — after which the client speaks THIS protocol to the
|
||||
> backend, unchanged. Since P4a the contract is expressed through
|
||||
> `envelope.Define` (docs/os-development/protocol-namespace.md), so every
|
||||
> packet begins with the universal 16-byte prefix and the open-node id rides
|
||||
> in it. The Zig source of truth is `library/protocol/vfs/vfs-protocol.zig`
|
||||
> (the `vfs-protocol` module), whose unit test pins a sample of the sizes
|
||||
> and values below. This page is the **language-neutral wire specification**
|
||||
> of that contract — what a Rust or C client implements ([vdso.md](../os-development/vdso.md)
|
||||
> explains why the IPC protocols, not the syscall numbers, are danos's
|
||||
> public ABI).
|
||||
|
||||
## Transport
|
||||
|
||||
A VFS exchange is one synchronous IPC rendezvous (`ipc_call`,
|
||||
docs/ipc.md): the client sends one message and blocks; the server replies
|
||||
with one message. The endpoint comes from the kernel's `fs_resolve` — which
|
||||
also hands back the path rewritten relative to the mount — not from a
|
||||
registry lookup. (Service id 1, the old userspace router, is retired.)
|
||||
|
||||
- A message is at most **256 bytes** (`message_maximum`).
|
||||
- Every packet begins with the 16-byte **envelope prefix**
|
||||
([protocol-namespace.md](../os-development/protocol-namespace.md)): a
|
||||
`Header` on a request, a `Status` on a reply. The prefix is **folded, not
|
||||
stacked** — the verb and the object being addressed live in it, and no
|
||||
request or reply below repeats either.
|
||||
- A request is the header, then the verb's own fixed part (0–16 bytes), then
|
||||
an inline tail of at most **224 bytes** (`maximum_payload`) — a path, or
|
||||
write bytes. There is no multi-message request: paths and single
|
||||
reads/writes must fit, and larger transfers loop (see *read* / *write*).
|
||||
- A reply is the status, then the verb's own fixed part, then an inline tail
|
||||
— read bytes, or a directory entry's name.
|
||||
- All integers are **little-endian**; layouts are C layout for x86-64
|
||||
(`extern struct`), offsets given below so nothing need be inferred.
|
||||
|
||||
The kernel resolves NAMES (the mount table) but never parses these
|
||||
messages — it moves the bytes; file state is entirely the backend's affair.
|
||||
With clients holding backend node ids directly, a backend records each open
|
||||
handle's owner and sweeps a dead client's handles via the published process
|
||||
exit events.
|
||||
|
||||
## Request header — 16 bytes
|
||||
|
||||
The envelope's `Header`, identical in every danos protocol:
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
| 0 | 4 | `operation` | an **Operation** value (below); 0–15 are the reserved universal verbs |
|
||||
| 4 | 4 | — | padding |
|
||||
| 8 | 8 | `target` | **the open-node id** from a prior `open`; 0 for `open` itself and the path-based verbs |
|
||||
|
||||
## Reply header — 16 bytes
|
||||
|
||||
The envelope's `Status`:
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
| 0 | 4 | `status` | **0 = success**, negative = failure (signed) |
|
||||
| 4 | 4 | — | padding |
|
||||
| 8 | 4 | `len` | reply bytes following this header: the verb's fixed part plus its tail |
|
||||
| 12 | 4 | — | padding |
|
||||
|
||||
A failing backend replies with the status alone (`len` = 0) and no fixed
|
||||
part, and that reply reaches the client directly — there is no party between
|
||||
them on the wire. (Kernel-served paths produce no wire replies at all:
|
||||
`fs_resolve`/`fs_node` failures are syscall register statuses.) The errno
|
||||
vocabulary is the kernel's, continued by the envelope: `ENOENT` = 4 is what a
|
||||
backend answers for anything it cannot find or cannot do, `ENOSYS` = 10 for a
|
||||
verb it does not implement, `EPROTO` = 11 for a packet shorter than the verb
|
||||
it names. Clients must treat *any* negative status as failure rather than
|
||||
matching a particular one.
|
||||
|
||||
## Operations
|
||||
|
||||
Values number from 16 (`first_protocol_operation`) in declaration order, and
|
||||
are frozen once shipped. Values 0–15 are the envelope's reserved universal
|
||||
verbs, which mean the same thing at every provider in the system: `describe`
|
||||
(0) answers the protocol's name and version and is implemented by the
|
||||
envelope itself, so every backend answers it. A verb outside this table is
|
||||
answered `-ENOSYS`; it is never a safety check any more, because the
|
||||
dispatch compares numbers rather than decoding an enum.
|
||||
|
||||
Each row's *request* and *reply* name the bytes **after** the 16-byte prefix.
|
||||
|
||||
| value | operation | request | tail | reply | reply tail |
|
||||
|------:|-----------|---------|------|-------|-----------|
|
||||
| 16 | `open` | `flags` (4 bytes, below) | the path | `node` (8 bytes) = the open-node id | — |
|
||||
| 17 | `close` | — | — | — | — |
|
||||
| 18 | `read` | `offset` (8), `len` (4) = wanted count | — | — | the bytes read; `Status.len` 0 at end of file |
|
||||
| 19 | `write` | `offset` (8), `len` (4) = count | the bytes | `count` (4) = bytes accepted (may be short — loop) | — |
|
||||
| 20 | `status` | — | — | **FileStatus** (24 bytes) | — |
|
||||
| 21 | `readdir` | `cursor` (8) | — | one **DirectoryEntry** (16 bytes) | the name |
|
||||
| 22 | `mount` | — | the mount-point path; the backend endpoint rides as the call's **capability** | — | — |
|
||||
| 23 | `unmount` | — | the mount-point path | — | — |
|
||||
| 24 | `mkdir` | — | the path | — | — |
|
||||
| 25 | `unlink` | — | the path | — | — |
|
||||
| 26 | `rename` | — | old path, one `0x00`, new path | — | — |
|
||||
| 27 | `bind` | — | the contract name; the provider's endpoint rides as the call's **capability** | — | — |
|
||||
|
||||
Notes per operation:
|
||||
|
||||
- **open** — the path is the mount-relative path `fs_resolve` handed back
|
||||
(absolute-shaped: `/notes.txt` under fat's `/volumes/usb` mount). Bare names
|
||||
(`greeting`) resolve nowhere — the flat ramfs is retired, and `fs_resolve`
|
||||
refuses non-absolute paths. The returned `node` is the *backend's* own
|
||||
open-node id: with the router in the kernel there is no forwarding table,
|
||||
and clients hold backend ids directly (see *Lifetimes and trust*). Every
|
||||
later packet carries it in `Header.target` — the path is spoken once, here,
|
||||
and integers do the rest.
|
||||
- **read / write** — a single exchange moves at most 224 bytes
|
||||
(`maximum_payload`); the client loops, advancing its own offset by what
|
||||
came back, until done (read) or the slice is written (write). A `write`
|
||||
reply shorter than requested is progress, not an error; a count of 0 means
|
||||
no forward progress — stop rather than spin.
|
||||
- **readdir** — `cursor` is the **entry index**, not a byte position. Each
|
||||
call returns exactly one entry; the client increments the cursor by 1. **A
|
||||
`name_len` of 0 is end-of-directory** — the reply's own length cannot say
|
||||
so, because the envelope always sends the fixed reply part. The directory
|
||||
must have been opened with the `directory` flag.
|
||||
- **mount / unmount** — RETIRED from the wire: mounting is the `fs_mount`
|
||||
syscall now (a filesystem server passes its endpoint handle; possession is
|
||||
the capability, exactly the trust of the old cap-passing op). The verb
|
||||
numbers stay reserved. Mount-prefix semantics are unchanged: prefixes
|
||||
match at path boundaries only (`/volumes/usb` never captures
|
||||
`/volumes/usbextra`), the longest matching prefix wins, and an optional
|
||||
backend-side rewrite prefix maps a mount into the backend's namespace (fat
|
||||
serves `/volumes/usb` from its volume root and `/system/logs` from its
|
||||
`/system/logs` subtree).
|
||||
- **rename** — same-directory rename only: the backend compares the old and
|
||||
new parent paths and refuses a mismatch. The client (`file_system`) refuses
|
||||
earlier when the two paths resolve to different backend endpoints, but that
|
||||
check is coarser than "one mount" — one endpoint can serve several mounts
|
||||
(fat serves `/volumes/usb`, `/system/configuration` and `/system/logs`), so
|
||||
a cross-mount rename reaches the backend and fails on its same-directory
|
||||
check.
|
||||
- **bind** — the protocol registry's claim verb, implemented only by the
|
||||
synthetic `/protocol` backend inside PID 1
|
||||
([protocol-namespace.md](../os-development/protocol-namespace.md)). A file
|
||||
backend answers `-ENOSYS`.
|
||||
|
||||
## Open flags
|
||||
|
||||
Bitwise OR in `open`'s `flags`, meaningful for `open` only:
|
||||
|
||||
| bit | name | meaning |
|
||||
|----:|------|---------|
|
||||
| 1 | `create` | create the file if it does not exist |
|
||||
| 2 | `directory` | open a directory node for `readdir` rather than a file |
|
||||
| 4 | `truncate` | truncate an existing file to zero length on open (replace, don't overwrite in place) |
|
||||
|
||||
## FileStatus — 24 bytes (the `status` reply's fixed part)
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
| 0 | 8 | `size` | file size in bytes |
|
||||
| 8 | 4 | `kind` | a **NodeKind** value |
|
||||
| 12 | 4 | — | padding |
|
||||
| 16 | 8 | `mtime` | modification time, Unix epoch seconds UTC; 0 if the backend keeps none |
|
||||
|
||||
## DirectoryEntry — 16 bytes + name (the `readdir` reply)
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
| 0 | 4 | `kind` | a **NodeKind** value |
|
||||
| 4 | 4 | `name_len` | length of the name that follows; **0 means end of directory** |
|
||||
| 8 | 8 | `size` | the entry's size in bytes |
|
||||
| 16 | `name_len` | name | the entry's name, not NUL-terminated |
|
||||
|
||||
## NodeKind
|
||||
|
||||
Aligned to the node-kind table in the file-system hierarchy
|
||||
(docs/file-system-development/file-system-hierarchy.md):
|
||||
|
||||
| value | kind |
|
||||
|------:|------|
|
||||
| 0 | regular file |
|
||||
| 1 | directory |
|
||||
| 2 | character device |
|
||||
| 3 | block device |
|
||||
| 4 | symbolic link |
|
||||
| 5 | fifo |
|
||||
| 6 | socket |
|
||||
| 7 | protocol |
|
||||
|
||||
Clients should map unknown values to *regular* rather than reject — the
|
||||
table can grow. Kind 6 (`socket`) keeps its wire value but is retired from
|
||||
the design — a named rendezvous point is exactly what a `protocol` node is,
|
||||
landed as value 7 with the protocol namespace
|
||||
(docs/os-development/protocol-namespace.md). `character_device` (stream
|
||||
semantics — the tty/console shape) and `block_device` (raw sector-addressed
|
||||
volumes, reserved) remain part of the design.
|
||||
|
||||
## An open reply may carry a capability
|
||||
|
||||
`open` rides `ipc_call`, whose reply direction can hand back an endpoint
|
||||
capability alongside the reply. A file backend never uses it — FAT
|
||||
answers with a node id and nothing else — but a **synthetic** backend does:
|
||||
opening a `protocol` node returns the provider's endpoint, and possession of
|
||||
that endpoint *is* the channel. The convention is per-backend, not
|
||||
per-operation, so a client that opens an ordinary file simply receives no
|
||||
capability, exactly as before.
|
||||
|
||||
## Lifetimes and trust
|
||||
|
||||
Open-node ids live in the backend. A client that dies without closing leaks
|
||||
nothing permanently: the backend (the FAT server) subscribes to the kernel's
|
||||
published process-exit events (docs/process-lifecycle.md) and releases a dead
|
||||
client's handles. The kernel VFS root needs no sweep at all — its node tokens
|
||||
are permanent for a boot and carry no open state.
|
||||
|
||||
Ids are plain integers rather than capabilities, so the backend **scopes them
|
||||
to the caller's badge**: an open node belongs to the task that opened it, and
|
||||
every verb that names one — read, write, status, readdir, close — is answered
|
||||
only for that task. A node id is a small number drawn from a table of
|
||||
thirty-two, trivially guessable, and until this rule a backend honoured every
|
||||
client's ids from every other client
|
||||
(docs/os-development/protocol-namespace.md: *handles must be scoped per
|
||||
client — validated against the badge, or drawn from a per-client id
|
||||
namespace*).
|
||||
|
||||
The refusal is deliberately **identical to absence**: a node that is somebody
|
||||
else's answers `-ENOENT`, exactly as one that was never opened, so a prober
|
||||
learns nothing about which ids are live — the same discipline the protocol
|
||||
namespace applies to a refused open. The owner is a *task*, because the badge
|
||||
is: a threaded client uses a node from the thread that opened it, which is
|
||||
already the granularity of the exit sweep that releases it.
|
||||
|
||||
## Evolution rules
|
||||
|
||||
What a non-Zig implementation may rely on, and what it must not:
|
||||
|
||||
- Operation values, flag bits, `NodeKind` values, and struct layouts are
|
||||
**append-only and frozen once shipped**. The unit test in
|
||||
`library/protocol/vfs/vfs-protocol.zig` pins a sample of them (the
|
||||
`DirectoryEntry` size, `NodeKind` 0–1 and 6–7, `Operation` values 16–21, 26
|
||||
and 27); this page is the full record of the frozen values.
|
||||
*The one renumbering this contract has had was the rebase onto the envelope
|
||||
(P4a), which moved every verb above the reserved range — a deliberate
|
||||
flag-day across a system with no third-party clients yet, not a precedent.*
|
||||
- The 256-byte message ceiling is a property of the current IPC transport,
|
||||
not a promise; clients should read `maximum_payload`-shaped limits from the
|
||||
reply lengths they actually get (loop-until-done), not hard-code 224.
|
||||
- Success is exactly 0, and the negative statuses come from one system-wide
|
||||
errno vocabulary (the kernel's, continued by the envelope) rather than from
|
||||
this protocol.
|
||||
-132
@@ -1,132 +0,0 @@
|
||||
# IPC: message-passing channels
|
||||
|
||||
Inter-process communication is the **backbone of a microkernel**. Once drivers and
|
||||
services run isolated in their own address spaces ([vision](vision.md)), they can't
|
||||
just call each other — a request becomes a **message**. In a microkernel, whatever
|
||||
was a function call across a monolithic kernel is IPC, so it's a first-class
|
||||
concern, not an afterthought.
|
||||
|
||||
There are two layers, built a milestone apart:
|
||||
|
||||
- **`system/kernel/ipc.zig`** — a bounded blocking channel between *kernel threads*,
|
||||
described below. The primitive, and where the blocking discipline was worked out.
|
||||
- **`system/kernel/ipc-synchronous.zig`** — synchronous call/reply between *processes*, across
|
||||
address spaces. What user-space servers and drivers actually talk over. It's the
|
||||
second half of this document.
|
||||
|
||||
## The channel
|
||||
|
||||
The first form is a **bounded blocking channel** (`system/kernel/ipc.zig`): a fixed-size
|
||||
ring buffer of messages with a producer/consumer rendezvous, built on the
|
||||
scheduler's [wait queues](scheduling.md).
|
||||
|
||||
`Channel(T, capacity)` is generic over the message type and buffer size. It holds a
|
||||
ring buffer, a count, and two wait queues:
|
||||
|
||||
- **`send(msg)`** — if the channel is full, block on the *not-full* queue; otherwise
|
||||
write the message, bump the count, and wake a waiting receiver.
|
||||
- **`receive()`** — if the channel is empty, block on the *not-empty* queue; otherwise
|
||||
take a message, drop the count, and wake a waiting sender.
|
||||
|
||||
Neither side busy-waits: a full channel parks the sender, an empty one parks the
|
||||
receiver, and each operation wakes the other side when it makes progress possible.
|
||||
|
||||
Two details make it correct:
|
||||
|
||||
- **Recheck in a loop.** A woken task re-tests the condition (`while (full) wait`)
|
||||
rather than assuming the slot is still available — another waiter may have taken
|
||||
it first. This is the standard guard against spurious or racing wakeups.
|
||||
- **One critical section.** `send`/`receive` run under the [big kernel
|
||||
lock](smp.md) (`sync.enter` / `sync.leave`), which disables interrupts on this
|
||||
core *and* takes the kernel's one spinlock — since SMP, the interrupt flag alone
|
||||
is not atomicity, because `cli` on one core does nothing to another. So checking
|
||||
the condition and committing the block/enqueue happen atomically both with respect
|
||||
to the timer preempting mid-operation and to the other side running on another
|
||||
CPU. `waitLocked` / `wakeLocked` are the variants that assume the caller already
|
||||
holds that critical section.
|
||||
|
||||
## Verifying it
|
||||
|
||||
The `ipc` test (see [testing.md](testing.md)) runs a producer and a consumer passing
|
||||
**100 messages through a 4-slot channel**. The small buffer means the channel goes
|
||||
full and empty over and over, so both the blocking-send and blocking-receive paths are
|
||||
exercised heavily. The messages arrive intact and in order (their sum is the
|
||||
expected `5050`), and neither task busy-waits — they block and wake each other.
|
||||
|
||||
## Endpoints: call/reply across address spaces
|
||||
|
||||
A channel connects two kernel threads sharing one address space. Real servers are
|
||||
*processes*, so the payload has to cross an address-space boundary. That's
|
||||
`system/kernel/ipc-synchronous.zig`, and its shape is L4's: a synchronous **rendezvous** at an
|
||||
`Endpoint`, with the message copied directly from the sender's pages to the receiver's
|
||||
(`copyAcross` walks both sets of page tables through the physmap — no CR3 switch, no
|
||||
bounce buffer).
|
||||
|
||||
Two syscalls carry it:
|
||||
|
||||
- **`ipc_call(h, msg, reply)`** — copy `msg` to the server, block until it replies.
|
||||
- **`ipc_reply_wait(h, reply, recv)`** — reply to the client you're still holding (if
|
||||
any), then block for the next request. One syscall, because a server's steady state
|
||||
is *always* "finish the last one, wait for the next".
|
||||
|
||||
An endpoint is reached by **handle** — a small integer index into the process's handle
|
||||
table (`Task.handles`), exactly like a file descriptor, and just as unforgeable. The
|
||||
bootstrap problem (how do you get the first handle?) is solved by a tiny name registry:
|
||||
a server calls `ipc_register(service_id, h)` under a well-known small integer, and a
|
||||
client calls `ipc_lookup(service_id)`.
|
||||
|
||||
The server never learns the client's identity beyond a **badge**, delivered alongside
|
||||
the message: the caller's task id.
|
||||
|
||||
### Interrupts are messages too
|
||||
|
||||
`notifyFromIsr` posts an *asynchronous* notification to an endpoint — no payload, no
|
||||
reply owed — and wakes whoever is blocked in `reply_wait`. Its badge has the top bit
|
||||
set (`notify_badge_bit`), which is how a driver's single event loop distinguishes "a
|
||||
client wants something" from "the hardware wants something". Notifications sit in a
|
||||
small coalescing ring on the endpoint, so an interrupt taken while the driver was busy
|
||||
elsewhere is not lost.
|
||||
|
||||
This is what makes a user-space driver possible at all, and it's the subject of
|
||||
[drivers.md](drivers.md).
|
||||
|
||||
## What's next (partly done since)
|
||||
|
||||
- **Priority inheritance** through IPC — still open: a high-priority client
|
||||
blocked on a low-priority server suffers unbounded priority inversion.
|
||||
- **Handle transfer.** *Landed as cap-passing (M13)*: `ipc_call` and
|
||||
`ipc_reply_wait` carry an optional capability alongside the bytes (`send_cap`),
|
||||
copying an endpoint or shared-memory handle into the peer's table. First user:
|
||||
[input](input.md) subscribers register by handing over their own endpoint, and
|
||||
class drivers get a private channel to one device.
|
||||
- **Asynchronous / buffered send** for the cases where a rendezvous is the wrong
|
||||
shape (logging, notifications between servers). *Landed as `ipc_send`* — a
|
||||
non-blocking post to an endpoint's bounded payload queue, delivered through
|
||||
`reply_wait` as a buffered message (badge bit `notify_message_bit`). Built for, and
|
||||
first used by, the [input service](input.md)'s keyboard-event broadcast, where a
|
||||
synchronous push would let one dead subscriber hang the fan-out. A full queue drops
|
||||
the oldest (discrete messages, not a coalescing level like the notification ring).
|
||||
- **A bounded reply** — half landed. The copy is still 256 bytes
|
||||
(`MESSAGE_MAXIMUM`) under the big kernel lock, but bulk transfer got its shared
|
||||
pages: `shared_memory_create`/`map`/`physical`, the region handle delegated as
|
||||
a capability (above). virtio-gpu's scanout surface is the first user
|
||||
([display-v2.md](display-v2.md)).
|
||||
|
||||
## Lifecycle conventions over IPC (M17)
|
||||
|
||||
Three conventions from [process-lifecycle.md](process-lifecycle.md) ride the
|
||||
notification mechanism:
|
||||
|
||||
- **Signals** arrive as notifications on the endpoint a process nominated with
|
||||
`signal_bind` (`runtime.process.bindSignals`): badge = the signal bit plus the
|
||||
coalesced pending mask (`runtime.process.signalsFrom` decodes). Statements,
|
||||
never questions; no payload, no reply.
|
||||
- **One-shot timers** (`timer_bind`, `runtime.system.timerOnce`) land as a
|
||||
timer-bit notification — the timed wait: a service arms a deadline and keeps
|
||||
serving, instead of blocking in sleep.
|
||||
- **The universal ping**: a **zero-length request is the liveness probe**,
|
||||
answered with a zero-length reply by the service harness itself
|
||||
(`runtime.service.run`). No protocol's requests start at length zero, so the
|
||||
encoding cannot collide, and a wedged service simply fails to answer — which
|
||||
is the diagnosis. Deep health ("can I reach my hardware?") stays a per-service
|
||||
protocol message.
|
||||
@@ -1,9 +0,0 @@
|
||||
# OS Developer Guide
|
||||
|
||||
This document is for those who need to understand the architectural decisions behind the OS.
|
||||
|
||||
## Written in Zig?
|
||||
|
||||
The os was initially written in zig because it has excellent support for EFI. With zig, we could forgo using a third party bootloader, reducing the time to boot up the kernel. Following the "Zen of Zig", helped to produce the most readable codebase for an operating system ever created. So those, new to OS development could quickly get up to speed.
|
||||
|
||||
|
||||
@@ -0,0 +1,39 @@
|
||||
# OS Development
|
||||
|
||||
This document explains the architectural decisions behind the operating system.
|
||||
|
||||
## Written in Zig?
|
||||
|
||||
The OS is written in Zig because it has excellent EFI support, so the OS boots quickly without a third-party bootloader.
|
||||
|
||||
Zig comes batteries included for systems work — cross-compilation, a build system, and a test runner are all part of the toolchain. Building with `-Doptimize=ReleaseSafe` keeps runtime safety checks on in the shipped kernel, which removes entire classes of bugs. The built-in test suite, combined with a QEMU integration harness, means every feature is proven to work, before it is shipped.
|
||||
|
||||
The codebase of the OS prioritizes readability. The aim is a codebase where someone new to OS development can find their way around without a guide.
|
||||
|
||||
## A microkernel?
|
||||
|
||||
The kernel is a thin layer: it schedules processes and manages memory. Everything else — drivers, file systems, the display — runs in user space as separate, isolated processes.
|
||||
|
||||
The payoff is resilience. When a driver crashes, it doesn't take the OS down with it; it gets restarted. That makes this an ideal environment for *developing* an operating system, because a buggy driver is an ordinary bug: patch it, restart the service, and keep going.
|
||||
|
||||
There is a security benefit too. Processes are isolated and talk over Inter-Process Communication (IPC) channels, so compromising one service doesn't hand an attacker the whole machine. Vulnerabilities tend to stay contained in the process they started in.
|
||||
|
||||
Other operating systems choose to pack all of these duties into one binary as a Monolithic kernel, mostly for performance: a function call inside the kernel is faster than passing a message between isolated processes. That cost is real — an IPC round-trip is a few microseconds where a function call is nanoseconds — but it is also workload-shaped. Compute-bound programs don't notice it at all. For bulk data like file contents and pixels, the design moves data through shared memory and DMA so it is copied once, the same as a monolithic kernel; only small control messages cross the IPC boundary. What remains is the per-message cost on chatty paths, and the scheduler and memory management are designed to keep that small.
|
||||
|
||||
## Private ABI
|
||||
|
||||
The syscall layer is private. The numbers and structures in `abi.zig` are an internal detail shared between the kernel and the system's own libraries, and they are free to change between builds.
|
||||
|
||||
The public boundary sits one level up: the [vDSO](vdso.md) that programs call into, and the documented IPC protocols such as the [VFS protocol](../file-system-development/vfs-protocol.md). Programs that stick to those interfaces keep working while the kernel rearranges itself underneath. This is the opposite of the Linux approach, where raw syscall numbers are frozen forever; here, stability is promised at the library and protocol level, and nowhere below it.
|
||||
|
||||
## Steal the best bits and dump the legacy
|
||||
|
||||
The OS is Unix-like, but selectively. It borrows the ideas that have aged well — everything is a file, small services composed over clean interfaces — and skips the parts of POSIX that have caused decades of headaches.
|
||||
|
||||
Some concrete choices:
|
||||
|
||||
- **`spawn`, not `fork`.** Creating a process starts a fresh program and returns the child's id. There is no clone-the-whole-address-space-then-immediately-throw-it-away dance, and none of the subtle state-inheritance bugs that come with it.
|
||||
- **Time is a syscall.** The kernel owns the clock and timers directly. There is no time daemon to keep alive and no ambiguity about where the truth lives.
|
||||
- **Lifecycle events arrive as messages.** A supervisor learns that a child exited through an IPC message on an endpoint it already owns — delivered like any other message, not as an interrupt that can fire between any two instructions.
|
||||
|
||||
The test for keeping an idea is simple: does it still pull its weight, or is it only there because it was there in 1979?
|
||||
@@ -19,10 +19,10 @@ UEFI configuration table
|
||||
BootInformation.acpi_rsdp (u64, in the loader↔kernel handoff) system/boot-handoff.zig
|
||||
│ the kernel forwards the whole BootInformation
|
||||
▼
|
||||
platform.discover(boot_information, …) system/devices/platform.zig
|
||||
platform.discover(boot_information, …) system/kernel/platform.zig
|
||||
│ reads boot_information.acpi_rsdp, hands it to the ACPI backend
|
||||
▼
|
||||
acpi.discover(rsdp_phys, …) system/devices/acpi.zig
|
||||
acpi.discover(rsdp_phys, …) system/kernel/acpi.zig
|
||||
│ dereferences the RSDP, reads the pointer it contains
|
||||
▼
|
||||
RSDP ──(a field in the struct)──► RSDT / XSDT ──► SDTs (MADT, MCFG, FADT, HPET, DSDT…)
|
||||
@@ -75,9 +75,9 @@ There are really two independent questions, and it's worth not conflating them:
|
||||
- **`system/kernel/architecture/x86_64/paging.zig`** — the kernel's page tables and address-space
|
||||
management (see [paging.md](paging.md)).
|
||||
- **`system/kernel/architecture/x86_64/apic.zig`** / **`ioapic.zig`** — the Local APIC, its timer,
|
||||
and the I/O APIC for device interrupts (see [device-interrupts.md](device-interrupts.md)).
|
||||
and the I/O APIC for device interrupts (see [device-interrupts.md](../device-driver-development/device-interrupts.md)).
|
||||
- **`system/kernel/architecture/x86_64/serial.zig`** / **`io.zig`** — the COM1 UART (the kernel's
|
||||
machine-readable log channel, see [testing.md](testing.md)) and the shared port-I/O + MSR primitives.
|
||||
machine-readable log channel, see [testing.md](../testing.md)) and the shared port-I/O + MSR primitives.
|
||||
- **`system/kernel/architecture/x86_64/smp.zig`** / **`per-cpu.zig`** — application-processor bring-up
|
||||
and per-CPU state (GS base, system-call entry point, see [scheduling.md](scheduling.md)).
|
||||
- **`system/kernel/architecture/x86_64/isr.s`** — the exception stubs, the `lgdt`/`lidt`/`ltr` load
|
||||
@@ -87,7 +87,7 @@ The Pi is not a "standard" ARM platform — expect Broadcom-specific peripherals
|
||||
Pi 5. Everything below is an offset from it.
|
||||
- **UART**: a **PL011** (at base + `0x20_1000`) plus a mini-UART; on some boards the
|
||||
PL011 is wired to Bluetooth, so which one is the console varies. This is the
|
||||
`aarch64`/`arm` equivalent of our x86 [COM1 serial](testing.md).
|
||||
`aarch64`/`arm` equivalent of our x86 [COM1 serial](../testing.md).
|
||||
- **Interrupt controller**: *not* a standard ARM GIC on the older parts — the Zero W
|
||||
and Pi 3 use Broadcom's own ARMCTRL controller (Pi 3 adds a per-core "local"
|
||||
controller for timers/mailboxes). The **Pi 4 and 5 do have a GIC-400**. So the
|
||||
@@ -122,4 +122,4 @@ Two routes, mirroring how we test x86-64 with OVMF:
|
||||
- [architecture.md](architecture.md) — the arch-module boundary these targets plug into, and the
|
||||
CPU-arch vs boot-protocol "two axes".
|
||||
- [efi.md](efi.md) — the UEFI loader that carries over to aarch64-UEFI.
|
||||
- [vision.md](vision.md) — why isolated, portable-across-architectures is the goal.
|
||||
- [vision.md](../vision.md) — why isolated, portable-across-architectures is the goal.
|
||||
@@ -0,0 +1,120 @@
|
||||
# Communication: the four layers
|
||||
|
||||
*Design, agreed 2026-07-31. The model document — the vocabulary and layering
|
||||
every other communication document speaks.*
|
||||
|
||||
danos separates **what is said** from **how the bytes move**, so that the
|
||||
mechanism is replaceable. The shape is a network stack's, cut into four
|
||||
layers; a program only ever touches the top two.
|
||||
|
||||
```
|
||||
L3 namespace /protocol/... names establishment points protocol-namespace.md
|
||||
L2 protocol the language: packet schemas, verbs, targets the envelope, library/protocol/*
|
||||
L1 channel two ends exchanging packets and signals the client library's Channel
|
||||
L0 transport a buffer + a doorbell: moves the bytes ipc.md (kernel-ipc), later shm-ring, …
|
||||
```
|
||||
|
||||
## Vocabulary
|
||||
|
||||
| Term | Meaning |
|
||||
|---|---|
|
||||
| **protocol** | The language: which packets exist, what their fields mean, which verbs a provider answers. Defined transport-independently in a `library/protocol/*` module. |
|
||||
| **channel** | An open conversation between two processes, speaking one protocol. Established by opening a `/protocol/...` name; both ends can send and receive. |
|
||||
| **packet** | The unit a protocol transmits: a bounded, atomic header+payload. Never fragmented — if it doesn't fit, it isn't a packet; bulk data rides shared memory with a packet as the doorbell. |
|
||||
| **signal** | A payload-less poke below the packet layer: "something happened, come look." Coalescing — the count may collapse, the fact may not. |
|
||||
| **transport** | What moves the bytes of one channel: a buffer plus a doorbell. Chosen (and upgradable) at establishment, invisible above L1. |
|
||||
| **endpoint** | A termination point where a transport delivers. The kernel-ipc transport's endpoint is its kernel mailbox object. |
|
||||
|
||||
## Addressing: parties by channel, objects by target
|
||||
|
||||
There are no network-style addresses in a packet. The two questions addresses
|
||||
answer are answered at different layers:
|
||||
|
||||
- **Who am I talking to?** The **channel**, decided once at establishment.
|
||||
Opening `/protocol/input` yields a channel; every packet sent on it goes to
|
||||
the peer. Nothing to route per-packet — like TCP, where no HTTP request
|
||||
carries the server's IP.
|
||||
- **Who sent this?** Attached to every received packet **by the channel
|
||||
layer**, from identity the transport can verify — under kernel-ipc, the
|
||||
kernel-stamped badge. The sender never writes a source field, which is what
|
||||
makes source unforgeable (the property a network's spoofable source header
|
||||
lacks).
|
||||
- **Which of your things?** The packet's **`target`** field: *object*
|
||||
addressing within the already-chosen peer — the vfs protocol's node id, the
|
||||
display protocol's layer id, a block volume. `target = 0` addresses the
|
||||
provider itself; a protocol without objects never uses it.
|
||||
|
||||
`target` is how instance multiplicity stays out of the namespace. Ten USB
|
||||
sticks and the namespace still holds exactly one name, `/protocol/block`: a
|
||||
channel to the provider, `enumerate` lists the current volumes as targets, a
|
||||
`targets_changed` signal announces hotplug, and a read names its volume in
|
||||
`target`. The unix `/dev/sda`,`/dev/sdb` problem is dissolved, not renamed.
|
||||
|
||||
If a future transport genuinely routes between machines, *it* carries real
|
||||
source/destination addressing internally at L0 — the way IP runs under TCP —
|
||||
and none of it surfaces into the packet header. Protocols stay ignorant of
|
||||
distance.
|
||||
|
||||
## The transport (L0): a buffer and a doorbell
|
||||
|
||||
Strip any transport to its skeleton and the same two parts remain:
|
||||
|
||||
| Transport | Buffer | Doorbell | Status |
|
||||
|---|---|---|---|
|
||||
| **kernel-ipc** | kernel-owned mailbox (the `Endpoint`) | the scheduler (rendezvous wake) | the first transport — [ipc.md](../device-driver-development/ipc.md) |
|
||||
| **shm-ring** | user-owned shared-memory ring | a signal | exists ad hoc (display bulk); to be formalized — the unlock for the 256-byte ceiling |
|
||||
| network | NIC queue | an interrupt | someday, when danos networks |
|
||||
|
||||
Transports differ in their **properties**, which the channel layer exposes and
|
||||
the protocol layer may depend on:
|
||||
|
||||
- **packet ceiling** — kernel-ipc: 256 bytes request/reply, 64 pushed. An
|
||||
shm-ring's ceiling is its slot size. Kernel-ipc's 256 is the *floor* every
|
||||
protocol may assume everywhere.
|
||||
- **synchrony** — kernel-ipc's call is a rendezvous: natural backpressure, no
|
||||
queue to size. An asynchronous transport buffers, so a channel over one
|
||||
needs explicit flow control. Backpressure is a *transport property*, not a
|
||||
channel guarantee — protocols that rely on it say so.
|
||||
- **droppability** — pushed event packets may drop when a ring fills;
|
||||
request/reply may not.
|
||||
- **capability carriage** — **only kernel-ipc can move a capability.**
|
||||
Handles are kernel objects; a user-space ring cannot transfer one. So
|
||||
kernel-ipc is always the *establishment and control* transport — channels
|
||||
are born on it, capabilities ride it — even when a channel's data is
|
||||
negotiated onto something fatter.
|
||||
|
||||
That negotiation is the upgrade path: a channel starts on kernel-ipc; the
|
||||
protocol's handshake may then delegate a shared-memory region (as a
|
||||
capability, over kernel-ipc) and move its bulk traffic there. The display
|
||||
path already does exactly this by hand; formalizing it in the channel layer
|
||||
makes it every protocol's option.
|
||||
|
||||
## The channel (L1)
|
||||
|
||||
A channel has two ends, and **the ends are peers**: each may send packets,
|
||||
each may receive, each may signal. Request/reply is a *pattern* over the
|
||||
channel — a send with a correlated receive, which the kernel-ipc transport
|
||||
happens to accelerate as a single rendezvous — not the definition of it. The
|
||||
event stream (subscribe, then pushes) and the change signal (poke, then
|
||||
re-read) are the other two patterns; all three are catalogued in
|
||||
[protocol-namespace.md](protocol-namespace.md)'s wiring section.
|
||||
|
||||
The channel layer's obligations: deliver packets whole, attach the verified
|
||||
source to every receive, expose the transport's properties, and hide the
|
||||
transport's mechanics. The client library's `Channel` type is this layer made
|
||||
concrete — a program holds channels that speak protocols and never touches a
|
||||
raw handle.
|
||||
|
||||
## The protocol (L2) and the namespace (L3)
|
||||
|
||||
A protocol defines its packets through the envelope — every packet begins
|
||||
`{operation, target}`, reserved verbs (`describe`, `enumerate`, `subscribe`,
|
||||
`unsubscribe`) mean the same thing in every protocol, and `Define` checks
|
||||
every packet against the transport floor at compile time. The full treatment,
|
||||
including how names are granted, resolved, and restricted per process, is
|
||||
[protocol-namespace.md](protocol-namespace.md).
|
||||
|
||||
Establishment points are named by contract — `/protocol/display`, never
|
||||
`/protocol/ipc-1` — because the name must outlive the mechanism: a
|
||||
transport named in the namespace could never be swapped, which would defeat
|
||||
this document's premise.
|
||||
@@ -36,7 +36,7 @@ Two things drive the need, and they set the timing:
|
||||
**aarch64 it's required to boot at all**. The [aarch64 port](arm.md) is what
|
||||
forces the issue.
|
||||
|
||||
2. **Isolated user-space drivers need it.** In the [microkernel vision](vision.md),
|
||||
2. **Isolated user-space drivers need it.** In the [microkernel vision](../vision.md),
|
||||
drivers live in user space — but something has to enumerate the hardware and hand
|
||||
each driver its MMIO regions and IRQs. That enumeration *is* device discovery. So
|
||||
discovery is a prerequisite for real drivers, **not** for user mode itself.
|
||||
@@ -132,7 +132,7 @@ when*:
|
||||
- **User-space enumeration: a device-manager server.** Everything else — PCI devices,
|
||||
peripherals — is parsed (or queried from the kernel's parse) by a privileged
|
||||
user-space server that hands each driver process its MMIO regions and IRQ rights
|
||||
over [IPC](ipc.md). Combined with **interrupts-as-messages** (an IRQ delivered to a
|
||||
over [IPC](../device-driver-development/ipc.md). Combined with **interrupts-as-messages** (an IRQ delivered to a
|
||||
driver as a message on a channel — a natural extension of the wait queues and
|
||||
channels already built), that's what makes drivers genuinely isolated.
|
||||
|
||||
@@ -144,7 +144,7 @@ slice is unavoidably in-kernel.
|
||||
On ARMv8 the generic timer exposes its frequency directly via the `CNTFRQ` register —
|
||||
no calibration needed. That's cleaner than the x86 side, where we measure the LAPIC
|
||||
and TSC against the PIT because nothing tells us their frequency (see
|
||||
[device-interrupts.md](device-interrupts.md)). Discovery on ARM hands you more for
|
||||
[device-interrupts.md](../device-driver-development/device-interrupts.md)). Discovery on ARM hands you more for
|
||||
free; discovery on x86 is partly about *finding* what ARM just tells you.
|
||||
|
||||
## Suggested ordering
|
||||
@@ -166,18 +166,18 @@ free; discovery on x86 is partly about *finding* what ARM just tells you.
|
||||
- [arm.md](arm.md) — the aarch64 target that forces genuine discovery (DTB, GIC).
|
||||
- [memory-map.md](memory-map.md) — the same loader-captures / kernel-consumes seam,
|
||||
and the note about grabbing the RSDP before `ExitBootServices`.
|
||||
- [device-interrupts.md](device-interrupts.md) — the LAPIC/timer bring-up that
|
||||
- [device-interrupts.md](../device-driver-development/device-interrupts.md) — the LAPIC/timer bring-up that
|
||||
discovery will eventually feed (IOAPIC, real IRQ routing).
|
||||
- [ipc.md](ipc.md) — the channels that interrupts-as-messages and the device manager
|
||||
- [ipc.md](../device-driver-development/ipc.md) — the channels that interrupts-as-messages and the device manager
|
||||
will ride on.
|
||||
- [vision.md](vision.md) — why drivers belong in isolated user space at all.
|
||||
- [vision.md](../vision.md) — why drivers belong in isolated user space at all.
|
||||
|
||||
## Update (M19.3, 2026-07-13): PCI enumeration left the kernel
|
||||
|
||||
The kernel now seeds only the `pci_host_bridge` node (ECAM window, MMIO
|
||||
apertures derived from the memory map's holes, bus range, and the 16-bit I/O
|
||||
window). The per-function walk moved to the ring-3 `pci-bus` driver
|
||||
([device-manager.md](device-manager.md)): it claims the bridge, repeats the
|
||||
([device-manager.md](../device-driver-development/device-manager.md)): it claims the bridge, repeats the
|
||||
ECAM scan through its mmio grant, and `device_register`s what it finds, which
|
||||
the device manager mirrors and matches. The ACPI namespace walk follows in M20;
|
||||
the static tables (MADT, HPET, MCFG, FADT + `\\_S5`) stay kernel-side.
|
||||
@@ -190,7 +190,7 @@ for the host bridge, FADT); at this point it also still built the AML namespace
|
||||
but only to read the `\\_S5` sleep type for poweroff. (That remnant is gone too:
|
||||
the kernel now runs no AML at all — soft-off belongs to the acpi service, and the
|
||||
kernel keeps only the AML-free reboot path.) Device discovery is the ring-3 **acpi
|
||||
service** ([device-manager.md](device-manager.md)): it claims the `acpi-tables`
|
||||
service** ([device-manager.md](../device-driver-development/device-manager.md)): it claims the `acpi-tables`
|
||||
node the kernel publishes (the AML blobs, a broad io_port grant, the SCI),
|
||||
re-parses the same blobs with the shared AML module, evaluates `_STA`/`_CRS`,
|
||||
and registers + reports each `_HID` device — the device manager matches drivers
|
||||
@@ -206,7 +206,7 @@ ring 0.)
|
||||
Moving PCI and ACPI enumeration out of ring 0 was not just a relocation — it
|
||||
made discovery **firmware-neutral by construction**, which is the whole reason
|
||||
to do it before the second architecture rather than after. Everything at and
|
||||
above the [device-manager](device-manager.md) protocol — descriptors,
|
||||
above the [device-manager](../device-driver-development/device-manager.md) protocol — descriptors,
|
||||
containment, reports, matching, supervision — is generic and may never become
|
||||
x86-specific. Discovery is the single firmware-specific piece, and it is
|
||||
isolated as **one swappable process per firmware**:
|
||||
@@ -231,8 +231,8 @@ Two consequences of neutrality bind on later work:
|
||||
|
||||
- **Cross-firmware surfaces are named by domain, not firmware.** System power is
|
||||
a [`power`](power.md) protocol, not an "ACPI events" protocol: on x86 the acpi
|
||||
service registers it, on ARM a PSCI/mailbox service registers the same
|
||||
`ServiceId.power`, and subscribers never learn the difference.
|
||||
service binds it, on ARM a PSCI/mailbox service binds the same
|
||||
`/protocol/power`, and subscribers never learn the difference.
|
||||
- **Identity must widen before the fdt service exists.** `DeviceDescriptor`'s
|
||||
8-byte `hid` holds an EISA id but cannot hold an FDT `compatible` string
|
||||
(`"brcm,bcm2835-aux-uart"`); the identity field grows before the ARM path can
|
||||
@@ -241,7 +241,7 @@ Two consequences of neutrality bind on later work:
|
||||
Two supporting decisions keep the kernel's remaining slice honest:
|
||||
|
||||
- **The AML interpreter is a single build module**
|
||||
(`system/devices/aml/aml.zig`) — one source, no fork. During the ring-3 move
|
||||
(`library/device/acpi/aml/aml.zig`) — one source, no fork. During the ring-3 move
|
||||
it was compiled into both the kernel (which linked it just for the `\_S5`
|
||||
poweroff evaluation) and the acpi service, with the `acpi-parse` test
|
||||
asserting the two produce the same device count. Since soft-off followed
|
||||
@@ -24,10 +24,12 @@ EFI/BOOT/BOOTX64.efi <- the "removable media" default for x86-64
|
||||
```
|
||||
|
||||
The boot volume is **FHS-shaped** (see the repository-layout note in
|
||||
[README.md](README.md)): `build.zig` installs `boot/efi.zig` (built for the `uefi`
|
||||
target) at `EFI/BOOT/BOOTX64.efi` — the one path UEFI firmware fixes — and lays
|
||||
[README.md](../README.md)): the root `build.zig` compiles `boot/efi.zig` (built
|
||||
for the `uefi` target) and `build/images.zig` places it at
|
||||
`EFI/BOOT/BOOTX64.efi` — the one path UEFI firmware fixes — and lays
|
||||
the rest out by FHS path: the kernel at `system/kernel`, init at
|
||||
`system/services/init`, the pre-packed boot capsule at `boot/system.img`.
|
||||
`system/services/init`, the pre-packed boot capsule at `boot/system.img`
|
||||
([system-image.md](system-image.md)).
|
||||
`zig-out` mirrors that tree, but what a machine actually boots is the
|
||||
self-contained FAT32 image `tools/make-fat-image.py` builds from the same files
|
||||
(`danos-usb.img`). The `run-x86-64` step points QEMU at OVMF (UEFI firmware for
|
||||
@@ -66,8 +68,9 @@ captures the **ACPI RSDP** from the UEFI configuration table (while boot
|
||||
services are still up), loads the system binaries into an in-RAM
|
||||
**initial ramdisk** (`loadSystemTree` — normally a single read of the pre-packed
|
||||
`boot\system.img` capsule, which already *is* the ramdisk wire format; it falls
|
||||
back to opening each manifest-listed path, and walks the `/system` tree only as
|
||||
a last resort for hand-assembled sticks. Best-effort either way — a kernel-only
|
||||
back to opening each manifest-listed path, and walks the `/system` and `/test`
|
||||
trees only as a last resort for hand-assembled sticks — the capsule's format, builder, and
|
||||
fallback chain are documented in [system-image.md](system-image.md). Best-effort either way — a kernel-only
|
||||
volume still boots), and builds the **bootstrap page tables** the kernel starts
|
||||
life on (`buildBootstrapTables`), all before the jump:
|
||||
|
||||
@@ -171,7 +174,7 @@ The loader and kernel are two *separate* binaries built for two different target
|
||||
so everything they exchange must have an identically-defined memory layout. That's
|
||||
what `system/boot-handoff.zig` provides — imported by both as the `boot-handoff` module.
|
||||
It is *only* the handoff: the kernel↔user ABI (`system/abi.zig`) and the device types
|
||||
(`system/devices/device-abi.zig`) are separate contracts the bootloader never sees.
|
||||
(`library/device/model/device-abi.zig`) are separate contracts the bootloader never sees.
|
||||
|
||||
- `BootInformation` — the top-level struct passed to the kernel: the
|
||||
framebuffer, the memory map, the kernel's own `PT_LOAD` segments
|
||||
@@ -194,7 +197,7 @@ power on
|
||||
-> queryFramebuffer (via GOP: EDID native res, setMode, describe fb)
|
||||
-> loadKernel (read system/kernel ELF, load PT_LOAD segments low, .text at 0x100000)
|
||||
-> loadSystemTree (read the boot\system.img capsule as the in-RAM initial ramdisk;
|
||||
fallbacks: manifest-listed paths, then a /system tree walk)
|
||||
fallbacks: manifest-listed paths, then a /system + /test tree walk)
|
||||
-> buildBootstrapTables (identity + physmap + higher-half kernel mappings)
|
||||
-> exitBootServices (retry until the memory-map key holds)
|
||||
-> handoff: load bootstrap CR3, jump to e_entry, boot_information pointer in RDI
|
||||
@@ -51,7 +51,7 @@ rest — works directly on the kernel heap, no bespoke containers required.
|
||||
|
||||
## Verifying it
|
||||
|
||||
The `heap` test (see [testing.md](testing.md)) exercises the allocator end to end:
|
||||
The `heap` test (see [testing.md](../testing.md)) exercises the allocator end to end:
|
||||
|
||||
```
|
||||
[PASS] alloc 4096 bytes
|
||||
@@ -133,10 +133,10 @@ TSS/IST is wired up: the handler survived a completely broken stack.
|
||||
|
||||
Both items originally deferred here have landed:
|
||||
|
||||
- **The IO-APIC**: [ioapic.zig](../system/kernel/architecture/x86_64/ioapic.zig)
|
||||
- **The IO-APIC**: [ioapic.zig](../../system/kernel/architecture/x86_64/ioapic.zig)
|
||||
routes external device lines onto vectors — discovered via ACPI's MADT, every
|
||||
input masked at init, lines unmasked one at a time as user-space drivers bind
|
||||
them (see [device-interrupts.md](device-interrupts.md)). The keyboard followed
|
||||
them (see [device-interrupts.md](../device-driver-development/device-interrupts.md)). The keyboard followed
|
||||
exactly as predicted: the PS/2 bus driver (`system/drivers/ps2-bus/`) claims
|
||||
the 8042 controller and binds its IRQ 1 (and the aux mouse's IRQ 12) through
|
||||
this routing. USB HID keyboards arrive over xHCI instead, which interrupts via
|
||||
@@ -21,9 +21,9 @@ kernel log.print ─┘ │
|
||||
|
||||
1. **Emit.** A program calls `std.log.info("mounted {s}", .{path})` — the
|
||||
runtime's `logFn` (installed for every binary by the root shim,
|
||||
`library/runtime/log.zig`) formats one line and issues one `debug_write`
|
||||
`library/kernel/logging.zig`) formats one line and issues one `debug_write`
|
||||
carrying the level. The payload does NOT contain the process's name.
|
||||
`runtime.system.write` remains as the raw/bring-up path (panics, test
|
||||
`logging.write` remains as the raw/bring-up path (panics, test
|
||||
fixtures); raw bytes ride the same ring, attributed all the same.
|
||||
|
||||
2. **Stamp.** The kernel wraps every payload LINE in a record stamped with the
|
||||
@@ -112,7 +112,7 @@ kernel heap will build on to map pages on demand.
|
||||
|
||||
## Verifying it
|
||||
|
||||
Four tests (see [testing.md](testing.md)) pin down the guarantees:
|
||||
Four tests (see [testing.md](../testing.md)) pin down the guarantees:
|
||||
|
||||
- **`vmm`** — map a fresh frame at an unused virtual address, write and read it
|
||||
back. Proves `map` works end to end.
|
||||
@@ -138,8 +138,8 @@ Four tests (see [testing.md](testing.md)) pin down the guarantees:
|
||||
processes own the low half.
|
||||
- **Per-address-space tables** — done: each user process gets its own root with
|
||||
the kernel half shared, and refcounted shared-memory mappings exist
|
||||
([ipc.md](ipc.md)). Copy-on-write remains unbuilt — nothing has needed it yet.
|
||||
([ipc.md](../device-driver-development/ipc.md)). Copy-on-write remains unbuilt — nothing has needed it yet.
|
||||
- **Uncacheable MMIO** — half done: user-space device and DMA mappings are
|
||||
strong-uncacheable and the framebuffer is write-combining via the PAT, but the
|
||||
kernel's own `mapMmio` path is still writeback — the LAPIC included (see
|
||||
[device-interrupts.md](device-interrupts.md)).
|
||||
[device-interrupts.md](../device-driver-development/device-interrupts.md)).
|
||||
@@ -7,7 +7,7 @@ them owns the hardware that reported the event, and the reporter should not know
|
||||
who is listening. So system power is a **service**: an event source **publishes**
|
||||
button/lid/battery/AC events, interested processes **subscribe**, and one
|
||||
privileged caller — init — can ask it to power the machine off. It is the same
|
||||
publish/subscribe shape as the [input service](input.md), applied to power.
|
||||
publish/subscribe shape as the [input service](../device-driver-development/input.md), applied to power.
|
||||
|
||||
## Why a service, and why it is named for the domain, not the firmware
|
||||
|
||||
@@ -15,9 +15,9 @@ Where the events come from is firmware-specific — on x86 they ride the ACPI SC
|
||||
([acpi.md](acpi.md)); on a Raspberry Pi they would come from PSCI or a mailbox.
|
||||
What subscribers want is not: *the lid closed* means the same thing regardless of
|
||||
who noticed. So the surface is **domain-named**. There is a `power-protocol`
|
||||
module and a well-known `ServiceId.power = 5`; on x86 the **acpi service**
|
||||
registers it, and on ARM a PSCI/mailbox service will register the *same* id.
|
||||
Subscribers call `runtime.ipc.lookup(.power)` and never learn which firmware they
|
||||
module and a contract named `/protocol/power`; on x86 the **acpi service**
|
||||
binds it, and on ARM a PSCI/mailbox service will bind the *same* name.
|
||||
Subscribers open `/protocol/power` and never learn which firmware they
|
||||
are on — the neutrality the whole [discovery](discovery.md) migration exists to
|
||||
preserve, carried one layer up into a running-system surface.
|
||||
|
||||
@@ -27,29 +27,44 @@ unchanged.
|
||||
|
||||
## The protocol
|
||||
|
||||
The `power-protocol` module ([system/services/power/protocol.zig](../system/services/power/protocol.zig))
|
||||
follows the vfs-protocol pattern — extern-struct messages, a version, reserved
|
||||
fields. Three operations:
|
||||
The `power-protocol` module ([library/protocol/power/power-protocol.zig](../../library/protocol/power/power-protocol.zig))
|
||||
is defined through the [envelope](protocol-namespace.md), so every packet begins
|
||||
with the folded `Header`. `Header.target` is unused in both directions: the
|
||||
provider is the only object either side addresses.
|
||||
|
||||
| Direction | Operation | Purpose |
|
||||
| Direction | Packet | Purpose |
|
||||
|---|---|---|
|
||||
| subscriber → service | `subscribe` | receive published events; the subscriber's endpoint rides as the call's **capability** (the input/device-manager pattern) |
|
||||
| init → service | `shutdown` | orderly shutdown's last step: enter S5 (soft off) |
|
||||
| service → subscriber | `event` | a published `EventMessage`, delivered as a buffered message (never sent *to* the service) |
|
||||
| subscriber → service | `subscribe` (reserved verb 2) | receive published events; the subscriber's endpoint rides as the call's **capability** (the input/device-manager pattern) |
|
||||
| init → service | `shutdown` (verb 16) | orderly shutdown's last step: enter S5 (soft off) |
|
||||
| service → subscriber | one event per kind | a published `Notice`, `ipc_send`t as a buffered packet (never sent *to* the service) |
|
||||
|
||||
`subscribe` is not one of this protocol's own verbs: a synchronous call whose
|
||||
attached capability is the subscriber's endpoint is exactly what the envelope's
|
||||
reserved `subscribe` means everywhere, so power adopts it wholesale. And no
|
||||
packet carries a version — the reserved `describe` verb is the version handshake,
|
||||
asked once at connect time rather than out of every packet's budget.
|
||||
|
||||
Events are published, not polled: like the input service, the service holds
|
||||
subscriber endpoints as capabilities and `ipc_send`s each event as a buffered
|
||||
message, so a slow or dead subscriber can never wedge the source. The event
|
||||
packet, so a slow or dead subscriber can never wedge the source. The table, the
|
||||
reserved `subscribe`/`unsubscribe` verbs and the fan-out are the **service
|
||||
harness's** (`service.Subscribers`), shared with input and the device manager, so
|
||||
the acpi service's own code is the ACPI half only — and a subscriber that dies is
|
||||
now swept on its exit notification, where before this service had no sweep at
|
||||
all. **The kind is the packet's operation** — one declared event per named kind, exactly as the
|
||||
input service delivers one per device class — so a subscriber reads *what
|
||||
happened* out of the header rather than out of a tag inside the payload. The
|
||||
vocabulary is hardware-neutral:
|
||||
|
||||
- `power_button` — the button was pressed (a fixed ACPI event on x86).
|
||||
- `lid`, `ac`, `battery` — the named GPE-driven events.
|
||||
- `notify` — a device notification that maps to none of the above; its `code`
|
||||
(the ACPI `Notify` argument) and the notifying device's `hid` say which device
|
||||
and what happened.
|
||||
- `power_button` (event 16) — the button was pressed (a fixed ACPI event on x86).
|
||||
- `lid` (17), `ac` (18), `battery` (19) — the named GPE-driven events.
|
||||
- `notify` (20) — a device notification that maps to none of the above; its
|
||||
`code` (the ACPI `Notify` argument) and the notifying device's `hid` say which
|
||||
device and what happened.
|
||||
|
||||
An `EventMessage` carries the `event` tag plus `code` and an 8-byte `hid`, so a
|
||||
generic `notify` is fully described without a second round trip.
|
||||
The payload every one of them carries is a `Notice`: `code` plus an 8-byte `hid`,
|
||||
so a generic `notify` is fully described without a second round trip, and the
|
||||
four named kinds leave both fields zero because the verb already said it all.
|
||||
|
||||
**`shutdown` is authority, not information.** It is the only operation that
|
||||
*does* something irreversible, so it is gated: the contract is that only init
|
||||
@@ -58,7 +73,10 @@ sequence over everything else. The acpi service implements this as a **soft
|
||||
gate** — it honors `shutdown` only from a process that is a *subscriber*, and
|
||||
init is the one subscriber. That stands in for "only the system supervisor may
|
||||
power off" without hard-coding a pid, so it still holds under tests where PID 1
|
||||
is not init.
|
||||
is not init. The question is asked of the harness's table now
|
||||
(`Subscribers.has(sender)`), which is why the harness exposes it: the gate is
|
||||
unchanged, including the badge being the whole of it — the badge is
|
||||
kernel-stamped, so nothing inside a packet can claim to be init.
|
||||
|
||||
## Orderly shutdown
|
||||
|
||||
@@ -75,7 +93,7 @@ notifications, the lifecycle **signals** it can receive (`terminate`), and the
|
||||
On a `power_button` event or a `terminate` signal, init:
|
||||
|
||||
1. logs that it is shutting down,
|
||||
2. runs the standard stop sequence — `runtime.process.stop(child, deadline,
|
||||
2. runs the standard stop sequence — `process.stop(child, deadline,
|
||||
endpoint)` — over its children **in reverse spawn order**, so the VFS stops
|
||||
last (other services may flush through it), each child getting the
|
||||
*terminate → deadline → kill* escalation from
|
||||
@@ -92,9 +110,9 @@ fails rather than hangs.
|
||||
`acpi-tables` node ([discovery.md](discovery.md)) already put the PM1 control
|
||||
ports in the acpi service's hands, so writing S5 from ring 3 is something it
|
||||
could physically already do; formalizing it as a protocol operation added a
|
||||
contract, not authority. The kernel's `system/devices/power.zig` keeps only
|
||||
**reboot** (the FADT reset register plus the legacy fallbacks, which need no AML);
|
||||
it has no poweroff path at all — S5 is not a kernel operation.
|
||||
contract, not authority. The kernel keeps only **reboot** (`acpi.reboot` in
|
||||
`system/kernel/acpi.zig` — the FADT reset register plus the legacy fallbacks, which
|
||||
need no AML); it has no poweroff path at all — S5 is not a kernel operation.
|
||||
|
||||
## Verifying it
|
||||
|
||||
@@ -126,5 +144,5 @@ until laptop sleep), and thermal zones.
|
||||
firmware neutrality that makes a PSCI backend drop-in on ARM.
|
||||
- [process-lifecycle.md](process-lifecycle.md) — the stop sequence
|
||||
(`terminate → deadline → kill`) and signals init composes into shutdown.
|
||||
- [device-manager.md](device-manager.md) — the supervision model init mirrors for
|
||||
- [device-manager.md](../device-driver-development/device-manager.md) — the supervision model init mirrors for
|
||||
its own children.
|
||||
@@ -6,10 +6,10 @@ harness are all in — the interface below is as-built. The primitives underneat
|
||||
predate this design ([process-management.md](process-management.md):
|
||||
spawn, the supervision link, kill, child-exit notifications); this document designs
|
||||
the layer above them — the standard vocabulary a danos process speaks about its own
|
||||
life, and the stable `runtime.process` interface that carries it. Nothing here is
|
||||
life, and the stable `process` interface that carries it. Nothing here is
|
||||
device- or driver-specific: a driver, the VFS, and a user application all stop,
|
||||
reload, and die the same way. The device manager is simply this design's first
|
||||
serious customer ([device-manager.md](device-manager.md)).
|
||||
serious customer ([device-manager.md](../device-driver-development/device-manager.md)).
|
||||
|
||||
**"POSIX" in this document means the concepts, never the letter of the standard.**
|
||||
danos borrows the ideas and the hard-won lessons (what SIGTERM *means*, why SIGPIPE
|
||||
@@ -19,7 +19,7 @@ rule is danos's own and it is strict: plain words that communicate intent
|
||||
(`bind`, `subscribe`, `publish`, `endpoint`) — never `SIG*`, never a second word for
|
||||
a concept that already has one. Literal POSIX arrives later and lives elsewhere: the
|
||||
`std.os.danos` seam that makes danos a Zig target, and eventually a **musl-based C
|
||||
layer** on the same native surface (see [zig-self-hosting.md](zig-self-hosting.md)) —
|
||||
layer** on the same native surface (see [zig-self-hosting.md](../zig-self-hosting.md)) —
|
||||
musl's syscall surface retargeted at danos system calls and IPC protocols (files onto
|
||||
the VFS protocol, `sigaction`/`wait` onto this lifecycle, sockets onto whatever
|
||||
networking becomes). Ported programs see POSIX; the system underneath never does.
|
||||
@@ -55,6 +55,9 @@ pattern reused. Signals are the same pattern reused a third time.
|
||||
- **`process_signal(id, signal)`** — posts the signal as an asynchronous
|
||||
notification to the target's bound endpoint: badge = `notify_badge_bit |
|
||||
notify_signal_bit | pending signals`. Non-blocking for the sender, always.
|
||||
Signals address the *process*: `id` may name any member of a threaded process
|
||||
and resolves to its leader — whose endpoint the harness binds — with authority
|
||||
mirroring `process_kill` ([shared-fate-plan.md](shared-fate-plan.md)).
|
||||
- **Pending signals coalesce** in a per-process bitmask while the target has no
|
||||
signal endpoint bound, and the whole mask arrives as one notification at bind —
|
||||
POSIX's own semantics for non-realtime signals (two pending SIGTERMs are one
|
||||
@@ -169,7 +172,7 @@ zombie state or privileged snooping:
|
||||
the server's reply with `-EPEER`; a server that dies fails its waiting clients
|
||||
the same way. This covers the *synchronous* case only.
|
||||
3. **The subscribers** — the new piece, and it is the input service's
|
||||
publish/subscribe shape ([input.md](input.md)) applied to exits. A stateful
|
||||
publish/subscribe shape ([input.md](../device-driver-development/input.md)) applied to exits. A stateful
|
||||
service accumulates per-client state across many requests: a filesystem server
|
||||
(FAT today) holds a dead client's open file handles, the input service holds
|
||||
its subscriptions, a future network stack holds its sockets. None of these
|
||||
@@ -181,13 +184,19 @@ zombie state or privileged snooping:
|
||||
same encoding supervisors already decode, the IRQ-as-IPC pattern once more). The
|
||||
subscriber filters for ids it holds state for and releases what the dead client
|
||||
held. Correlating is free of bookkeeping: an IPC sender's badge already *is* its
|
||||
task id (`runtime.ipc.Received`), so the id a service has been keying client
|
||||
task id (`ipc.Received`), so the id a service has been keying client
|
||||
state by all along is the id the exit event carries.
|
||||
|
||||
Subscription, not broadcast-to-everyone: only processes that asked receive
|
||||
events, the kernel keeps a bounded subscriber table, and delivery is the same
|
||||
non-blocking coalescing notification as everything else — a dying process never
|
||||
waits on its mourners. Subscribing is ungated, like `process_enumerate`: what is
|
||||
events, the kernel keeps a bounded subscriber table (sixteen — a normal boot
|
||||
already fields six, since this is what *every* provider with per-client state
|
||||
releases on), and delivery is the same non-blocking coalescing notification as
|
||||
everything else — a dying process never waits on its mourners.
|
||||
A service does not usually write the sweep itself: the shared service harness
|
||||
subscribes for it and drops a dead task's event subscriptions
|
||||
(`service.Subscribers`), and a provider adds its own handler only for state the
|
||||
harness knows nothing about — open files, layers, device tokens.
|
||||
Subscribing is ungated, like `process_enumerate`: what is
|
||||
running (and dying) is not a secret between cooperating processes. Subscribers
|
||||
do not receive the exit reason — the filesystem server does not care *why*
|
||||
the client died.
|
||||
@@ -197,9 +206,9 @@ its clients cleaning up after themselves.** Handle release on client death is th
|
||||
service's job, triggered by the published exit event — never by a courtesy
|
||||
"closing now" message that a crashed client will never send.
|
||||
|
||||
## The stable interface: `runtime.process`
|
||||
## The stable interface: `process`
|
||||
|
||||
`runtime.process` already owns what a process receives at birth (`Init`, the
|
||||
`process` already owns what a process receives at birth (`Init`, the
|
||||
argv contract). It grows to own the other end of life.
|
||||
|
||||
**The runtime is the stable interface; the numbers are not.** danos applications do
|
||||
@@ -280,8 +289,12 @@ callbacks (`on_terminate`, `on_reload`) for programs that want defaults.
|
||||
|
||||
### The service harness
|
||||
|
||||
`runtime.service` owns the `replyWait` loop and folds every event source — signals,
|
||||
`service` owns the `replyWait` loop and folds every event source — signals,
|
||||
child exits, protocol messages — into callbacks, with the vocabulary's defaults:
|
||||
it also owns the **subscriber side** of any protocol that declares events
|
||||
(`service.Subscribers`: the table, the reserved `subscribe`/`unsubscribe` verbs,
|
||||
the fan-out, and the sweep on a subscriber's published exit), so every event
|
||||
stream in the system behaves identically.
|
||||
`terminate` returns from the loop (clean exit), the common `ping` is answered automatically,
|
||||
`reload` is ignored unless overridden. One loop, no locking, nothing reentrant. A
|
||||
service author writes domain logic; the lifecycle contract is satisfied by the
|
||||
@@ -309,15 +322,15 @@ get POSIX; danos-native programs never pay for it.
|
||||
kill a claiming driver, spawn it again, the claim succeeds.
|
||||
2. **Exit reason in the death notification** (`ExitReason` above).
|
||||
3. **Exit events**: `process_subscribe` in the kernel (bounded subscriber table,
|
||||
publishes on every death), `runtime.process.subscribeExits`; the userspace VFS
|
||||
publishes on every death), `process.subscribeExits`; the userspace VFS
|
||||
router was the first subscriber — releasing a dead client's handles was its
|
||||
proof test — and the FAT server inherited the role when the router moved into
|
||||
the kernel (clients now hold the filesystem server's node ids directly).
|
||||
4. **Signals**: `signal_bind` + `process_signal` + the pending mask in the kernel;
|
||||
`runtime.process` grows the interface above; the service harness handles
|
||||
`process` grows the interface above; the service harness handles
|
||||
`terminate` and answers the common `ping`; `stop()` for supervisors.
|
||||
|
||||
[device-manager.md](device-manager.md) builds directly on all four.
|
||||
[device-manager.md](../device-driver-development/device-manager.md) builds directly on all four.
|
||||
|
||||
## Settled questions (2026-07-12)
|
||||
|
||||
@@ -57,9 +57,13 @@ dangle even if the supervisor dies first.
|
||||
|
||||
### `process_kill(id) -> 0 / -ESRCH / -EPERM`
|
||||
|
||||
Only the supervisor may kill; kernel tasks are not killable processes. Like a
|
||||
signal, delivery is prompt but asynchronous — 0 means the kill is accepted and
|
||||
irrevocable; the exit notification confirms completion.
|
||||
Only the supervisor may kill; kernel tasks are not killable processes. The kill
|
||||
is a **whole-process** kill ([shared-fate-plan.md](shared-fate-plan.md)): `id`
|
||||
may name any member of a threaded process — it resolves to the group's leader,
|
||||
authorization is checked against the *leader's* supervisor, and every thread
|
||||
dies. Like a signal, delivery is prompt but asynchronous — 0 means the kill is
|
||||
accepted and irrevocable; the exit notification (badged with the leader, posted
|
||||
once the last member is gone) confirms completion.
|
||||
|
||||
## How a kill lands (the kernel mechanics)
|
||||
|
||||
@@ -110,7 +114,7 @@ the architecture layer calls up into `tick`.
|
||||
- ~~There is no exit status in the notification~~ Closed (M17.2): the kernel
|
||||
records how every process ends — exited, a fault class, or killed — before it
|
||||
posts the exit notification, and the supervisor reads it with
|
||||
`process_exit_reason` (`runtime.process.exitReason`). This is the input to
|
||||
`process_exit_reason` (`process.exitReason`). This is the input to
|
||||
restart policy ([process-lifecycle.md](process-lifecycle.md)); an exit *code*
|
||||
for the clean case can still ride alongside later.
|
||||
- Enumerate writes through the caller's raw pointer under the bring-up trust
|
||||
@@ -0,0 +1,474 @@
|
||||
# The protocol namespace
|
||||
|
||||
*Design, agreed 2026-07-31. Supersedes the `ServiceId` registry. P1–P3 of the
|
||||
migration plan at the end have landed (the envelope, the registry and the
|
||||
`ServiceId` flag-day, and restriction stage one); P4 and P5 are the remaining
|
||||
work list.*
|
||||
|
||||
How a program finds, connects to, and is restricted from the things it talks to.
|
||||
Three ideas, kept deliberately separate:
|
||||
|
||||
1. **Naming** — a path under `/protocol` names a *contract*, not a service.
|
||||
2. **Access** — resolving that path yields an endpoint *capability*; what a process
|
||||
cannot resolve, it cannot reach.
|
||||
3. **Transport** — unchanged: packets over channels, moved by whichever
|
||||
transport the channel rides (kernel-ipc first).
|
||||
This document is layers **L3** (the namespace) and **L2** (the protocol
|
||||
and its envelope) of the communication stack;
|
||||
[communication.md](communication.md) owns the model and the vocabulary
|
||||
(*protocol* the language, *channel* the conversation, *packet* the
|
||||
transmitted unit, *signal* the payload-less poke, *transport* the
|
||||
replaceable mechanism), and
|
||||
[ipc.md](../device-driver-development/ipc.md) is the first transport.
|
||||
|
||||
## Why ServiceId has to go
|
||||
|
||||
Today a service calls `ipc_register(service_id, endpoint)` and a client calls
|
||||
`ipc_lookup(service_id)`, where `ServiceId` is a compile-time enum in `abi.zig`
|
||||
backed by a flat 16-slot table in the kernel. Three defects, in rising order:
|
||||
|
||||
- **Static.** The id space is baked into the ABI at compile time. A third-party
|
||||
program can never introduce a service; the one place danos is *less* dynamic
|
||||
than its own design.
|
||||
- **Ungated.** `ipc_register` is callable by any process and *replaces* an
|
||||
existing registration. Any process can hijack `.fat` or `.display` and
|
||||
impersonate it. `ipc_lookup` is equally ambient.
|
||||
- **Unrestrictable.** Because lookup is a syscall available to everyone, there is
|
||||
no point at which "this process may not talk to the display" can be enforced.
|
||||
Any future file-access restriction would be bypassable by speaking to the FAT
|
||||
server directly.
|
||||
|
||||
## Naming: contracts, not services
|
||||
|
||||
`/protocol/<name>` names a protocol — the contract a conversation follows — and
|
||||
resolving it connects you to whatever process currently provides that contract.
|
||||
The client never cared *which* binary answers; it cares that its messages are
|
||||
understood. Naming the contract makes that explicit, and buys:
|
||||
|
||||
- **Swappable providers.** Replace the display server; `/protocol/display`
|
||||
routes to the new one; clients notice nothing.
|
||||
- **Test fakes.** Spawn a program whose namespace wires `/protocol/display` to a
|
||||
mock. The name promises the protocol; the mock speaks it.
|
||||
- **One vocabulary.** The names mirror `library/protocol/`: a program imports
|
||||
the `display-protocol` module, then opens `/protocol/display`. What you
|
||||
compiled against and what you ask the namespace for are the same word.
|
||||
|
||||
A leaf names one contract — kebab-case, full words, matching the
|
||||
`library/protocol/` module that defines its wire format — and related
|
||||
contracts group into directories: `/protocol/networking/ip`,
|
||||
`/protocol/networking/bluetooth`. Directories organize *contracts only*;
|
||||
they never encode addressing (see below), so a directory appears because a
|
||||
domain has several contracts, never because hardware multiplied. The module
|
||||
tree mirrors the namespace (`library/protocol/networking/ip` ↔
|
||||
`/protocol/networking/ip`), and registrar grants scope naturally to subtrees
|
||||
— an application installed at `/applications/foo` can be granted
|
||||
`/protocol/applications/foo/...` and nothing above it. `/protocol` is
|
||||
top level, beside `/system` and `/applications`, because the boundary it names
|
||||
is spoken on both sides: applications talk to protocols as much as the OS does
|
||||
(see [file-system-hierarchy.md](../file-system-development/file-system-hierarchy.md)).
|
||||
|
||||
**Addressing lives inside the protocol, never in the path.** Which volume, which
|
||||
layer, which input device — that is a destination field in the messages, the way
|
||||
TCP carries a destination address, and the way danos protocols already work (the
|
||||
display protocol multiplexes layer ids; the vfs protocol addresses node ids).
|
||||
The namespace answers exactly one question — *may this process speak this
|
||||
protocol at all* — so `/protocol/block` is one name no matter how many disks are
|
||||
attached. The source address is never in the message either: it is the IPC
|
||||
badge, stamped by the kernel per message, unforgeable — a property TCP's source
|
||||
address does not have.
|
||||
|
||||
`/system/devices` (the device inventory) stays purely informational: facts for
|
||||
diagnosis, never a routing mechanism. Unix conflated the two in `/dev`; danos
|
||||
does not. You *read about* hardware in `/system/devices`; you *talk to* it
|
||||
through `/protocol`.
|
||||
|
||||
## Resolution: a protocol node in the VFS
|
||||
|
||||
The kernel VFS router already does the hard part: `fs_resolve` matches a mount
|
||||
prefix and installs the backend's endpoint capability in the caller's handle
|
||||
table. The registry is just a backend mounted at `/protocol` — ring 3, like FAT.
|
||||
Connecting is a normal vfs-protocol `open` with one twist in the reply:
|
||||
|
||||
```
|
||||
client kernel router registry backend
|
||||
│ fs_resolve("/protocol/display") │
|
||||
│──────────────────────────▶│ prefix match: /protocol │
|
||||
│◀── registry endpoint ─────│ (capability installed) │
|
||||
│ vfs open("display") ──────────────────────────────────────▶│
|
||||
│◀───────────────── Reply + capability = provider endpoint ──│
|
||||
│ ipc_call(provider, display-protocol messages...) │
|
||||
```
|
||||
|
||||
Both capability moves use machinery the kernel already has: request-direction
|
||||
and reply-direction `send_cap` on `call`/`replyWait`. The vfs protocol needs two
|
||||
additions, both append-only:
|
||||
|
||||
- `NodeKind.protocol` — a node that names a contract; its `open` establishes
|
||||
a **channel** (delivered as an endpoint capability) instead of returning a
|
||||
file id. The node is the protocol, the channel is the conversation, and the
|
||||
addressing inside the packets decides where within the provider each one
|
||||
lands. `readdir` over `/protocol` lists protocol nodes like any others, so
|
||||
the tree stays browsable for diagnosis.
|
||||
- The convention that an `open` reply may carry a capability. File backends
|
||||
(FAT) never use it; synthetic backends (the registry, later the device
|
||||
inventory) do.
|
||||
|
||||
The path lookup happens once, at connect time. The hot path — `ipc_call` on the
|
||||
cached endpoint — is untouched. A provider crash turns the cached endpoint dead
|
||||
(`-EPEER`), and the client's recovery is to re-resolve: the restart story falls
|
||||
out of the naming layer for free.
|
||||
|
||||
## Registration: the registrar, held by init
|
||||
|
||||
The registry backend is **init**. It is already PID 1, already spawns every
|
||||
service from its manifest, and already holds the supervision link to each — it
|
||||
is the process that *knows* which binary is which. (If init grows
|
||||
uncomfortable, the same design lifts into a dedicated registry service that
|
||||
init spawns first and delegates to; nothing below changes.)
|
||||
|
||||
- **Binding.** A service creates its endpoint and sends the registry a `bind`
|
||||
request with the protocol name as payload and the endpoint attached as the
|
||||
call's capability.
|
||||
- **Authorization.** Init's manifest gains a column: the protocols each spawned
|
||||
binary may bind. A `bind` from any process not granted that name is refused
|
||||
(`-EPERM`) — the badge identifies the caller, the supervision records map
|
||||
badge to binary. This is the registrar authority; it never leaves init.
|
||||
- **Collision is an error.** A name already bound refuses a second bind — never
|
||||
last-writer-wins. When a provider dies, init (its supervisor) unbinds its
|
||||
names; the restarted instance binds again.
|
||||
- **Provenance.** The registry records name → task id → binary path, so a
|
||||
diagnostic listing answers "who serves this?" at a glance:
|
||||
|
||||
```
|
||||
/protocol/display pid 12 /system/services/display
|
||||
/protocol/input pid 7 /system/services/input
|
||||
```
|
||||
|
||||
`ipc_register` and `ipc_lookup` retire; the `ServiceId` enum leaves `abi.zig`.
|
||||
The kernel keeps one residual rule: `/protocol` becomes a reserved prefix like
|
||||
`/system` — `fs_mount` refuses to shadow it, and init's boot-time mount is the
|
||||
only one it will ever hold. (Full gating of `fs_mount` is a separate item on
|
||||
the security track; the reserved prefix closes the hole for this namespace
|
||||
without waiting for it.)
|
||||
|
||||
## Restriction: per-process namespaces, not ACLs
|
||||
|
||||
danos has no users and no principals, deliberately. Restriction is therefore
|
||||
**delegation**: what a process may open is decided by whoever spawned it, and
|
||||
enforcement is absence — a protocol you cannot resolve does not exist for you.
|
||||
"Permission denied" and "not found" are the same answer, which is the same
|
||||
discipline the device layer already follows: the claim is the capability; here,
|
||||
the resolvable name is the capability.
|
||||
|
||||
Two stages, deliberately ordered so the useful half lands first:
|
||||
|
||||
**Stage one — the registry filters by badge.** Init is both the spawner and the
|
||||
registry, so its manifest already knows which binary may *open* which protocols
|
||||
(a second manifest column, beside the bind grants). An `open` from a process
|
||||
whose binary is not granted that protocol is refused. No new kernel mechanism
|
||||
at all; the display driver's view can be narrowed to nothing, a future
|
||||
downloaded application's to `display` and `input`, today.
|
||||
|
||||
**Stage two — spawn passes the namespace.** `spawn` gains an initial
|
||||
capability: the child's connection to *its* registry view, chosen by the
|
||||
spawner. A newly spawned process starts with an empty handle table and this one
|
||||
handle — its world is whatever its parent wired in. This removes the last
|
||||
ambient reach (`fs_resolve` finding `/protocol` globally), lets any supervisor
|
||||
— not just init — narrow or fake a child's view (an application launcher
|
||||
granting an app only what its manifest declares; a test harness substituting
|
||||
every provider), and composes down the supervision tree. Stage one's manifest
|
||||
column becomes the *content* of the view init builds, so nothing is thrown
|
||||
away.
|
||||
|
||||
### A worked example: the microphone prompt
|
||||
|
||||
The scenario stage two exists for: an application opens
|
||||
`/protocol/audio-input`, and the user should be asked. The supervisor is an
|
||||
ordinary user process — an application launcher — and the flow needs no new
|
||||
security concepts:
|
||||
|
||||
1. The launcher spawned the app with a namespace channel that terminates at
|
||||
**the launcher itself**. The app's whole world is a conversation with its
|
||||
supervisor.
|
||||
2. The app's `open("audio-input")` packet lands in the launcher,
|
||||
badge-stamped. The launcher spawned the app, so badge → binary path
|
||||
(`/applications/foo`) is its own supervision record — "remember my choice"
|
||||
needs no identity system.
|
||||
3. Grant unknown → the launcher parks the request and shows a prompt (it is a
|
||||
user process with display access; init never does UI). Blocking an open on
|
||||
a human is architecturally fine: opens are connect-time, never hot-path.
|
||||
4. **Yes** → the launcher opens `/protocol/audio-input` in *its own*
|
||||
namespace and attaches the resulting channel to the parked reply. The app
|
||||
cannot tell a prompt happened — a consented open is indistinguishable from
|
||||
a direct one, merely slower.
|
||||
5. **No** → refuse the open, indistinguishable from "no such protocol" — or
|
||||
hand the app a **fake**: a silence-generating provider. The test-fake
|
||||
mechanism doubles as a privacy feature.
|
||||
|
||||
The capability discipline holds throughout: the launcher can only grant what
|
||||
it holds — if init never gave the launcher `audio-input`, no prompt can
|
||||
conjure it. Consent is delegation flowing down the supervision tree, never a
|
||||
global ACL edit. And the provider still sees the app's badge on every packet,
|
||||
so a coarser second check at the audio service remains possible.
|
||||
|
||||
Two mechanical requirements this scenario pins on stage two:
|
||||
|
||||
- **Parked replies.** A prompt takes seconds, and the service loop holds one
|
||||
outstanding reply today — the launcher must park request A, keep serving B
|
||||
and C, and reply to A later (by badge). The kernel already tracks owed
|
||||
replies (that is how death delivers `-EPEER`); multiple parked replies is
|
||||
the extension, in the harness and, if needed, the kernel.
|
||||
- **Granted channels are dedicated, hence revocable.** Once the app holds a
|
||||
channel capability, nobody reaches into its handle table — so a
|
||||
prompt-granted channel must be one that can be *killed*: a dedicated
|
||||
endpoint pair (or per-client session at the provider) whose death turns
|
||||
the app's capability into `-EPEER`. Revoking microphone access is then
|
||||
killing that channel, using machinery that already exists.
|
||||
|
||||
One adjacent problem, named and deferred: **trusted UI**. The prompt is only
|
||||
meaningful if the app cannot draw a convincing fake or overlay the real one —
|
||||
a display-layer question (a reserved surface for the supervisor chain), owned
|
||||
by the display track, not this one.
|
||||
|
||||
Fine-grained restriction *within* a protocol (this process may use volume A but
|
||||
not volume B) is not the namespace's job. The capability-shaped answer, when it
|
||||
is needed: the supervisor pre-opens a connection scoped to one target and passes
|
||||
that connection to the child, which never opens `/protocol/block` at all.
|
||||
Delegation again, not ACLs.
|
||||
|
||||
## The envelope: one addressing scheme for every protocol
|
||||
|
||||
Every protocol module today hand-rolls its `Request`/`Reply` with an
|
||||
`operation` first field. That convention becomes a library, so addressing is
|
||||
uniform and the rules are enforced by construction rather than by review. New
|
||||
module: **`library/protocol/envelope`** (the one protocol-layer module that is
|
||||
not itself a protocol).
|
||||
|
||||
```zig
|
||||
/// Every packet a danos protocol transmits begins with this header.
|
||||
pub const Header = extern struct {
|
||||
operation: u32, // the verb; values 0..15 are reserved universal verbs
|
||||
_padding: u32 = 0,
|
||||
/// Object addressing, never party addressing: which of the peer's
|
||||
/// objects this packet operates on — a volume, layer, node, device.
|
||||
/// 0 addresses the provider itself. Parties are addressed by the
|
||||
/// channel; the protocol defines target's meaning; the field's place
|
||||
/// and width are universal.
|
||||
target: u64 = 0,
|
||||
};
|
||||
|
||||
/// Reserved verbs, answered by every provider.
|
||||
pub const operation_describe: u32 = 0; // -> protocol name, version, target kinds
|
||||
pub const operation_enumerate: u32 = 1; // -> the current targets, one per reply page
|
||||
pub const operation_subscribe: u32 = 2; // capability = the subscriber's endpoint
|
||||
pub const operation_unsubscribe: u32 = 3;
|
||||
pub const first_protocol_operation: u32 = 16;
|
||||
|
||||
/// Every reply begins with this.
|
||||
pub const Status = extern struct {
|
||||
status: i32, // 0 or a negative errno
|
||||
_padding: u32 = 0,
|
||||
len: u32 = 0, // payload bytes following the header
|
||||
_padding2: u32 = 0,
|
||||
};
|
||||
```
|
||||
|
||||
A protocol is then *defined through* the envelope, not beside it:
|
||||
|
||||
```zig
|
||||
pub const Protocol = envelope.Define(.{
|
||||
.name = "display",
|
||||
.version = 1,
|
||||
.operations = &.{
|
||||
.{ .name = "configure_layer", .request = ConfigureLayer, .reply = void },
|
||||
.{ .name = "blit", .request = Blit, .reply = void },
|
||||
...
|
||||
},
|
||||
});
|
||||
```
|
||||
|
||||
`Define` is comptime and is where the enforcement lives:
|
||||
|
||||
- verbs are numbered automatically from `first_protocol_operation`, so no
|
||||
protocol can collide with the reserved range;
|
||||
- every packet is size-checked at compile time against the kernel-ipc floor
|
||||
— `packet_maximum` (256) for request/reply, `post_maximum` (64) for event
|
||||
packets. Ceilings are transport properties
|
||||
([communication.md](communication.md)); the floor is what every protocol
|
||||
may assume on any transport. The errors that today surface as runtime
|
||||
truncation become compile errors, and packets-never-fragment is enforced
|
||||
at the source;
|
||||
- the generated type carries encode/decode helpers and a provider-side dispatch
|
||||
table, so a provider answers `describe` automatically and unknown operations
|
||||
with `-ENOSYS` uniformly;
|
||||
- the service harness (`library/kernel/service.zig`) accepts the generated
|
||||
dispatch type, which is what makes the envelope *enforced*: a protocol that
|
||||
bypasses `Define` does not plug into the harness.
|
||||
|
||||
Universal conventions that ride on the reserved verbs:
|
||||
|
||||
- **`describe`** is the version handshake. Version lives in the handshake, not
|
||||
in every message — the 256-byte budget is too small to spend per call.
|
||||
- **`enumerate`** is how multi-target protocols expose their targets, and the
|
||||
standard `targets_changed` notification (a notify bit) tells subscribers to
|
||||
re-enumerate — arrival and removal of volumes, layers, devices all take the
|
||||
same shape. Hotplug fits the notification ring far better than a filesystem
|
||||
tree ever did.
|
||||
- **Source is the badge.** No protocol defines a "sender" field; the kernel's
|
||||
per-message badge is the only source identity, and providers key per-client
|
||||
state on it.
|
||||
|
||||
### Paths resolve once; integers do the work
|
||||
|
||||
A rule the envelope makes official: **a path appears in a conversation at most
|
||||
once — at resolve or open — and everything after it addresses integers.** The
|
||||
namespace resolves `/protocol/display` to an endpoint; a backend's `open`
|
||||
resolves a path payload to a node id; from then on every packet carries the
|
||||
integer in `target`. Integers compare in one instruction and fit the fixed
|
||||
header, and the 256-byte message budget never re-carries path strings on the
|
||||
hot path. This is already the system's shape — vfs node ids, display layer ids
|
||||
— and the envelope pins it as the required shape for every protocol.
|
||||
|
||||
Two integer identities, not to be confused:
|
||||
|
||||
- **An open handle** — what vfs `open` returns today: transient, meaningful
|
||||
only within one client's session with one provider, swept when the client
|
||||
exits. Cheap, and all a protocol usually needs. Handles must be **scoped per
|
||||
client** — validated against the badge, or drawn from a per-client id
|
||||
namespace. (Today the FAT server's node ids are guessable small integers
|
||||
honoured across clients; that hole closes with this rule.)
|
||||
- **A persistent node identity** — a unix inode number, stable across opens
|
||||
and renames. danos deliberately does not promise this, because FAT cannot
|
||||
deliver it: a FAT file's identity is its directory entry, and rename or
|
||||
truncation moves every candidate anchor. If a future filesystem or a cache
|
||||
layer needs stable identity, that is the backend's promise to make, never
|
||||
the protocol's assumption.
|
||||
|
||||
The five existing protocol modules (`vfs`, `display`, `input`, `power`,
|
||||
`block`, plus `scanout`, `usb-transfer`, `device-manager`) rebase onto the
|
||||
envelope during the migration flag-day. `input-protocol`'s subscribe/publish
|
||||
split and `vfs-protocol`'s node addressing both map cleanly (`node` and layer
|
||||
ids become `target`).
|
||||
|
||||
## Wiring: how conversations flow
|
||||
|
||||
The patterns below are channel-layer (L1) shapes; the delivery mechanics are
|
||||
the kernel-ipc transport's, described here because it is the transport every
|
||||
channel starts on. Kernel-ipc provides exactly three delivery shapes, and
|
||||
every one is unicast. An endpoint is a mailbox owned by one process — its
|
||||
creator receives; anyone holding its capability sends into it. That direction
|
||||
never reverses:
|
||||
|
||||
1. **Synchronous call** — request/reply. The kernel parks the caller and
|
||||
`replyWait` delivers the reply straight back, so the provider answers
|
||||
without holding any capability to the client. Badge-stamped, blocking, and
|
||||
the *only* shape that carries capabilities (in the request, and in the
|
||||
reply — which is how a reverse path is bootstrapped).
|
||||
2. **Asynchronous send** — an event packet pushed into the receiver's post
|
||||
ring, at most `post_maximum` (64) bytes, no reply owed, never blocks the
|
||||
sender. Strictly one-way: to be pushed to, you must first hand the pusher
|
||||
your endpoint.
|
||||
3. **Signals** — payload-less notification bits, below the packet layer,
|
||||
coalescing: "something changed, come look."
|
||||
|
||||
A bidirectional link is therefore always **a pair of endpoints**, one per
|
||||
direction, each delivered by cap-passing. Three conversation patterns are
|
||||
built from these, and the envelope names all three:
|
||||
|
||||
- **Request/response** — the synchronous call. The default, and the only
|
||||
place capabilities move.
|
||||
- **Event stream** — `subscribe` (a synchronous call whose attached
|
||||
capability is the subscriber's own endpoint), after which the provider
|
||||
pushes events asynchronously; `unsubscribe` or subscriber exit ends it.
|
||||
Listened-to, not blocked-on.
|
||||
- **Change signal** — a signal plus re-read: `targets_changed` →
|
||||
`enumerate`. For state whose truth lives with the provider.
|
||||
|
||||
**Broadcast is a provider pattern, never a kernel primitive.** The kernel
|
||||
does not know subscriber sets — a service does. The input service is the
|
||||
model: sources *publish* (a unicast call to the service), the service
|
||||
*broadcasts* (a fan-out loop of asynchronous sends over its subscriber list,
|
||||
so one dead subscriber can never stall the rest). One fan-out point per event
|
||||
domain, owned by the service that defines the event.
|
||||
|
||||
The harness owns the machinery: the subscriber table, the dead-subscriber
|
||||
sweep (via process-exit notifications), and the fan-out loop — all written by
|
||||
hand in `input.zig` today, lifted into the service harness so every protocol
|
||||
gets identical semantics. `Define` declares a protocol's events (`.events`),
|
||||
and each event type is checked against `post_maximum` at compile time,
|
||||
generalizing the assert `input-protocol` already carries.
|
||||
|
||||
**Event packets are droppable.** A slow subscriber's ring fills, and the
|
||||
provider must not block on it — so an event stream is a hint or a coalescing
|
||||
signal, never a ledger. Anything that must not be lost is either re-readable
|
||||
state (the change-signal pattern) or bulk data in shared memory with a
|
||||
packet as the doorbell, which is how the display path already works — the
|
||||
packets-never-fragment rule and this one are the same rule seen from two
|
||||
sides.
|
||||
|
||||
**Source direction (open point).** Today event sources are *clients*: an
|
||||
input driver resolves `/protocol/input` and delivers each event as a
|
||||
synchronous `publish` call — one capability, obtained by resolution, covers
|
||||
everything, and the badge tells the service exactly who each event came from.
|
||||
The inversion — the service subscribing to each driver — would require every
|
||||
driver to be individually discoverable and its endpoint ferried to the
|
||||
service, machinery whose payoff (the service choosing its sources) the
|
||||
namespace already provides more cheaply: only a process granted open on
|
||||
`/protocol/input` can publish into it. Sources stay clients for now;
|
||||
revisited at restriction stage two, when a supervisor can wire capabilities
|
||||
at spawn time.
|
||||
|
||||
## What this deliberately does not solve
|
||||
|
||||
The wider security track, for which this namespace is the foundation, not the
|
||||
whole:
|
||||
|
||||
- **File access restriction** — the point of the exercise. The same stage-two
|
||||
namespace mechanism extends from protocol names to file paths: the spawner
|
||||
decides which subtrees resolve. Designed separately once this lands.
|
||||
- `fs_mount` gating beyond the reserved prefixes; `system_spawn` gating;
|
||||
`klog_read` being world-readable; backends checking the badge on per-node
|
||||
operations (the FAT server honours node ids across clients today).
|
||||
- Kernel hardening items already noted in-tree: SMEP/SMAP and SYSRET
|
||||
canonical-RIP, now designed in [smep-smap.md](smep-smap.md).
|
||||
- Pipes/FIFOs for the POSIX layer — a byte-stream object *beside* message IPC,
|
||||
wanted by the Python track, unrelated to naming.
|
||||
- **Trusted UI** — a permission prompt an application cannot fake or overlay
|
||||
(see the microphone example). A display-track concern: the supervisor chain
|
||||
needs a reserved surface.
|
||||
|
||||
## Migration plan
|
||||
|
||||
Flag-day per phase, in the style of the DMA-capability conversion — no
|
||||
dual-stack periods, the QEMU suite green at each phase boundary.
|
||||
|
||||
**P1 — mechanics, no behavior change.** The `envelope` module with its comptime
|
||||
`Define`, unit tests; `NodeKind.protocol` and the open-reply-capability
|
||||
convention in `vfs-protocol`; existing protocols untouched.
|
||||
|
||||
**P2 — the registry.** Init serves `/protocol` (bind with manifest
|
||||
authorization, collision refusal, unbind on provider death, provenance);
|
||||
kernel reserves the `/protocol` prefix; every service converts from
|
||||
`ipc_register` to `bind`, every client from `ipc_lookup` to resolve-and-open;
|
||||
`ServiceId`, `ipc_register`, `ipc_lookup` deleted. Tests: unauthorized bind
|
||||
refused, collision refused, provider restart re-binds and a client re-resolves.
|
||||
|
||||
**P3 — restriction, stage one.** The open-grant column in init's manifest;
|
||||
registry refuses ungranted opens. Test: a fixture process denied a protocol its
|
||||
neighbour is granted.
|
||||
|
||||
**P4 — protocol rebase.** Existing protocol modules re-expressed through
|
||||
`Define`; providers move onto the generated dispatch; `describe`/`enumerate`
|
||||
answered everywhere; the conformance test fixture exercises the reserved verbs
|
||||
against every registered provider.
|
||||
|
||||
**P5 — restriction, stage two.** Spawn's initial capability; namespace views
|
||||
built by the spawner; ambient resolution of `/protocol` retired. Includes the
|
||||
two requirements the microphone example pins: **parked replies** (a
|
||||
supervisor parks an open, keeps serving, replies later by badge) and
|
||||
**dedicated, killable granted channels** (revocation = channel death →
|
||||
`-EPEER`). Scoped separately — it touches `spawn`, the loader contract, and
|
||||
every supervisor — and lands together with the file-path half of namespacing.
|
||||
|
||||
The unix-path migration ([file-system-hierarchy.md](../file-system-development/file-system-hierarchy.md#migration))
|
||||
is independent of P1–P5 and can land before or after.
|
||||
@@ -44,7 +44,7 @@ tables of contents, both pointing at the same embedded FAT image:
|
||||
same `BOOTX64.efi` off it.
|
||||
|
||||
Neither path involves the legacy BIOS boot-sector machinery: danos is
|
||||
UEFI-only ([system-requirements.md](system-requirements.md)), so the MBR holds
|
||||
UEFI-only ([system-requirements.md](../system-requirements.md)), so the MBR holds
|
||||
no boot code, just the partition entry, and the El Torito entry is EFI-class,
|
||||
not floppy emulation.
|
||||
|
||||
@@ -57,7 +57,7 @@ allocated from the front) either way. The USB path has no such cap.
|
||||
## The builder
|
||||
|
||||
`tools/make-iso-image.py` follows the house rule of
|
||||
[make-fat-image.py](../tools/make-fat-image.py): pure Python 3 standard
|
||||
[make-fat-image.py](../../tools/make-fat-image.py): pure Python 3 standard
|
||||
library, no external tools (no xorriso, mkisofs, or isohybrid), with a
|
||||
`--verify` mode the `check-iso-image` step runs — it checks that the MBR
|
||||
partition and the El Torito catalog agree on where the FAT image lives and
|
||||
@@ -5,7 +5,7 @@ isolation; fault → kill the process → keep the core (`onException`; the
|
||||
`fault-recovery` test); the supervisor notification **with exit reasons**
|
||||
([process-lifecycle.md](process-lifecycle.md) — clean exit, fault class, or
|
||||
killed, recorded before the notice posts); and the **restart policy itself**
|
||||
([device-manager.md](device-manager.md)): the device manager supervises every
|
||||
([device-manager.md](../device-driver-development/device-manager.md)): the device manager supervises every
|
||||
driver, restarts crashes with backoff, caps crash loops, and re-claims work
|
||||
because the kernel releases a dead process's claims. The `driver-restart` and
|
||||
`usb-report` scenarios prove kill → release → respawn → re-claim → re-report
|
||||
@@ -14,9 +14,9 @@ more of the system moved into restartable processes (the discovery migration,
|
||||
[discovery.md](discovery.md), is the next rung). This is the property danos is really chasing:
|
||||
**if a part of the OS breaks, isolate it, and re-initialise it — without rebooting.**
|
||||
A crashed driver gets restarted; a wedged service gets killed and brought back. It's
|
||||
the reason the [microkernel](vision.md) shape was chosen, and it's a *separate* goal
|
||||
the reason the [microkernel](../vision.md) shape was chosen, and it's a *separate* goal
|
||||
from [real-time](smp.md#does-the-right-choice-depend-on-real-time-vs-resilience) —
|
||||
one that's less pervasive to build (see [vision.md](vision.md)).
|
||||
one that's less pervasive to build (see [vision.md](../vision.md)).
|
||||
|
||||
## The idea: "let it crash" + supervision
|
||||
|
||||
@@ -44,7 +44,7 @@ down. **Keeping the kernel minimal is a resilience strategy, not just an aesthet
|
||||
## The building blocks
|
||||
|
||||
1. **Address-space isolation.** A fault in one component can't corrupt another or the
|
||||
kernel. This is the [user-mode milestone](vision.md) (ring 3, per-process page
|
||||
kernel. This is the [user-mode milestone](../vision.md) (ring 3, per-process page
|
||||
tables) — the shared prerequisite for *any* of this, and it's needed regardless.
|
||||
2. **Fault detection** — how the system notices a component is dead or sick:
|
||||
- **Crash**: a CPU fault in a user process (page fault, illegal instruction) traps
|
||||
@@ -81,7 +81,7 @@ Detecting and killing is the easy half. The genuinely tricky questions are about
|
||||
- **In-flight IPC**: messages sent to the dead component, or replies its clients are
|
||||
blocked waiting for. The channel has to break cleanly and unblock the waiters with
|
||||
an error rather than hang them forever (a design constraint that reaches back into
|
||||
[ipc.md](ipc.md) — channels need a "peer died" outcome).
|
||||
[ipc.md](../device-driver-development/ipc.md) — channels need a "peer died" outcome).
|
||||
- **Clients**: how does a client discover the service it was talking to is gone and
|
||||
has been replaced? Options: capability revocation makes stale handles fail; or a
|
||||
**name server** re-binds clients to the new instance; or clients retry through a
|
||||
@@ -137,7 +137,7 @@ Honest boundaries:
|
||||
|
||||
Resilience needs **structural** features (isolation + supervision + a resource
|
||||
model); real-time needs a **pervasive** timing invariant. They're separable, and
|
||||
resilience is the lighter commitment (see [smp.md](smp.md) and [vision.md](vision.md)).
|
||||
resilience is the lighter commitment (see [smp.md](smp.md) and [vision.md](../vision.md)).
|
||||
Note the overlap, though: **preemptive scheduling** and **priorities** — already
|
||||
built — serve resilience too (you can preempt and kill a misbehaving component, and
|
||||
run the supervisor at high priority). So danos keeps the useful *mechanisms* of the
|
||||
@@ -156,10 +156,10 @@ real-time work without owing anyone a timing *guarantee*.
|
||||
|
||||
## Related
|
||||
|
||||
- [vision.md](vision.md) — the goals this serves (learning by doing; resilience over
|
||||
- [vision.md](../vision.md) — the goals this serves (learning by doing; resilience over
|
||||
hard real-time).
|
||||
- [scheduling.md](scheduling.md) — preemption, which makes runaway components killable.
|
||||
- [ipc.md](ipc.md) — channels that need a "peer died" outcome for clean restart.
|
||||
- [ipc.md](../device-driver-development/ipc.md) — channels that need a "peer died" outcome for clean restart.
|
||||
- [interrupts.md](interrupts.md) — fault reporting that user mode turns into "kill and
|
||||
restart" instead of "halt".
|
||||
- [smp.md](smp.md) — the real-time-vs-resilience fork, in the SMP context.
|
||||
@@ -3,7 +3,7 @@
|
||||
The scheduler turns danos from a linear "boot then halt" kernel into a **running
|
||||
multitasking system**. It's **fixed-priority preemptive**: the highest-priority
|
||||
ready task always runs, and tasks at the same priority take turns. That model is
|
||||
chosen for [real-time](vision.md) — it's predictable (you can reason about which
|
||||
chosen for [real-time](../vision.md) — it's predictable (you can reason about which
|
||||
task runs when) and its decisions are O(1), unlike a fair-share scheduler.
|
||||
|
||||
The scheduler proper (`system/kernel/scheduler.zig`) is generic; the context switch and new-task
|
||||
@@ -37,7 +37,7 @@ down a return address pointing at `task_trampoline` and zeroed callee-saved slot
|
||||
`schedule()` — pick the best task and switch — runs from two places:
|
||||
|
||||
- **`yield()`** — a task voluntarily gives up the CPU.
|
||||
- **`tick()`** — the 1000 Hz [timer](device-interrupts.md) preempts the running
|
||||
- **`tick()`** — the 1000 Hz [timer](../device-driver-development/device-interrupts.md) preempts the running
|
||||
task. This is what lets a task that never yields still share the CPU.
|
||||
|
||||
The subtlety in mixing them is the **interrupt flag (IF)**. The rule: `switch_context`
|
||||
@@ -97,7 +97,7 @@ marks the task blocked with a wake deadline and switches away. On every tick the
|
||||
timer wakes any task whose deadline has passed (a bounded scan, so it stays
|
||||
deterministic), which makes it ready again; the scheduler then runs it when its
|
||||
priority comes up. `sleep` measures its deadline on the [calibrated
|
||||
clock](device-interrupts.md), so it's real time.
|
||||
clock](../device-driver-development/device-interrupts.md), so it's real time.
|
||||
|
||||
When *every* task is blocked, something still has to run — so there's an **idle
|
||||
task** at the lowest priority that just `hlt`s until the next interrupt (see
|
||||
@@ -111,7 +111,7 @@ The other form of blocking is waiting for an **event** rather than a duration. A
|
||||
the caller on it, `wake(wq)` moves the highest-priority waiter back to ready
|
||||
(preempting if it now outranks the running task). A task links into a wait queue
|
||||
through the same field the ready queues use — it's in exactly one queue at a time.
|
||||
These are the primitives locks, semaphores and [IPC](ipc.md) are built on.
|
||||
These are the primitives locks, semaphores and [IPC](../device-driver-development/ipc.md) are built on.
|
||||
|
||||
Blocking safely needs **composable critical sections**. A blanket `cli`/`sti` pair
|
||||
doesn't nest: an IPC channel that `cli`s and then calls `wait` would have `wait`'s
|
||||
@@ -124,7 +124,7 @@ caller's state.
|
||||
|
||||
## Verifying it
|
||||
|
||||
Three tests (see [testing.md](testing.md)) prove the guarantees:
|
||||
Three tests (see [testing.md](../testing.md)) prove the guarantees:
|
||||
|
||||
- **`sched`** spawns three tasks that busy-loop *without ever yielding*. They all
|
||||
make progress — which can only happen if the timer is **preempting** between them
|
||||
@@ -140,7 +140,7 @@ Three tests (see [testing.md](testing.md)) prove the guarantees:
|
||||
|
||||
- **Priority inheritance** — still open. Tasks now do block on shared resources
|
||||
(IPC rendezvous, the big kernel lock), and nothing yet bounds priority
|
||||
inversion — a [real-time](vision.md) requirement.
|
||||
inversion — a [real-time](../vision.md) requirement.
|
||||
- **Task exit / a reaper** — done. A dying task goes on its core's reap list in a
|
||||
`.reaping` state; the timer tick drains the list, frees the stack back to the
|
||||
heap, and recycles the task-table slot.
|
||||
@@ -0,0 +1,310 @@
|
||||
# Shared fate: whole-process death (plan)
|
||||
|
||||
**Status: implemented 2026-07-22 (branch shared-fate), M1–M4 all landed; leader
|
||||
`thread_exit` → `-EPERM` as decided. One scope addition forced by M4: the
|
||||
per-task DMA/shared-memory arena cursors moved to the per-space object (the
|
||||
`shm-mapping-ref` test could not distinguish corruption-by-remap from
|
||||
corruption-by-free while sibling threads overlapped the arena) — the same move
|
||||
the mmap/MMIO cursors made in threading M7.**
|
||||
|
||||
[threading.md](threading.md) promises that a process dies *whole* — a fault in any
|
||||
thread, or a kill, takes down every thread. The kernel doesn't do that yet: every
|
||||
death path (`exit`, `thread_exit`, a ring-3 fault, `process_kill`) tears down
|
||||
exactly one `Task`, and the address-space refcount keeps the space alive for the
|
||||
siblings — so a faulting worker orphans its threads, which keep running in the
|
||||
possibly-corrupted address space (`process.zig` `killCurrentProcess`;
|
||||
`scheduler.zig` `exitUserLocked`). This plan closes that gap the way Linux, Windows,
|
||||
and Fuchsia all did: **the process is the unit of fate; only a voluntary
|
||||
`thread_exit` is per-thread.** The supervisor contract — one exit notification,
|
||||
then `process_exit_reason` — is deliberately unchanged.
|
||||
|
||||
## The contract
|
||||
|
||||
| Event | Who dies | Reason the supervisor reads (leader's record) |
|
||||
|---|---|---|
|
||||
| CPU fault in **any** thread (recoverable vector) | the whole group | the fault class (`segmentation_fault`, …) |
|
||||
| `process_kill` on **any member id** | the whole group | `killed` |
|
||||
| `exit(code)` from **any** thread | the whole group | `exited` (0) / `aborted` (≠0) |
|
||||
| `thread_exit` from a worker | that worker only | — (per-task record: `exited`) |
|
||||
| `thread_exit` from the **leader** | nobody — refused, `-EPERM` (decided, below) | — |
|
||||
| NMI, double fault, machine check | the core halts (unchanged) | — |
|
||||
|
||||
`exit` gaining group semantics is the `exit_group` lesson from Linux: the runtime's
|
||||
main-return path calls `exit`, and a process whose main returned must not leave
|
||||
workers running. `thread_exit` (what the worker trampoline calls) keeps today's
|
||||
per-thread behavior, refcount and all.
|
||||
|
||||
**The leader-`thread_exit` rule (decided: refuse).** The syscall is reachable from
|
||||
the leader even though the runtime never does it. Options weighed: (i) **refuse
|
||||
with `-EPERM`** — cheapest and honest; the group ends only through
|
||||
`exit`/fault/kill; (ii) escalate to `exit(0)` — Linux-flavored, but silently turns
|
||||
a (buggy) library call into process death; (iii) a Linux-style zombie leader whose
|
||||
slot survives until the group ends — the most faithful, and by far the most
|
||||
machinery. **(i) chosen at sign-off**; rows and tests below follow it.
|
||||
|
||||
## Group identity: a leader id
|
||||
|
||||
Nothing on `Task` names a process today — threads are tied to their process only by
|
||||
an equal `address_space`, their name is `"thread"`, and their `supervisor` is
|
||||
whichever *task* spawned them (possibly another worker), so supervision links form a
|
||||
chain, not a group. Scanning by `address_space` is also fragile during teardown,
|
||||
because both death paths zero it.
|
||||
|
||||
So: **add `leader: u32` to `Task`** — the Linux tgid, in danos clothes.
|
||||
`spawnProcessSupervised` sets `leader = own id`; `spawnThreadSupervised` copies the
|
||||
*caller's* leader; kernel tasks keep `leader = 0`, which is never followed. The
|
||||
leader id is exactly the id `system_spawn` returned to the supervisor, so the
|
||||
outside world already speaks it. Group membership = equal `leader`, where a *live
|
||||
member* means `state` ∈ {`.ready`, `.blocked`, `.running`} — the same filter
|
||||
`taskByIdLocked` applies; `.reaping` corpses are excluded. While the field is being
|
||||
introduced, add `leader` to `ProcessDescriptor` too (the ABI is private, so this is
|
||||
cheap now and lets `process_enumerate` consumers group threads).
|
||||
|
||||
`process_kill` re-derives its authority through the leader — with the existing
|
||||
guard order preserved: a kernel task (`address_space == 0`) is `-ESRCH` *before*
|
||||
any leader resolution (the kernel test asserts exactly that). Then: resolve the
|
||||
target, follow `target.leader`, require `leader.supervisor == caller`. The kill
|
||||
capability becomes per-*process*, aimed at any member id, and the odd
|
||||
today-behavior where a worker can be individually killed by its spawning thread
|
||||
disappears. (Audited: nothing in-tree kills a worker tid or relies on
|
||||
thread-supervisor kill semantics.)
|
||||
|
||||
## The group-dying latch
|
||||
|
||||
`AddressSpaceRef` — one per space, refcounted by its member tasks, recycled with a
|
||||
full struct re-init — gains the group-death state:
|
||||
|
||||
```zig
|
||||
dying: bool = false, // set by the first trigger; never cleared
|
||||
group_reason: abi.ExitReason, // what the leader's record will say
|
||||
exit_endpoint: ?*ipc.Endpoint, // the leader's counted ref, moved here
|
||||
leader: u32,
|
||||
```
|
||||
|
||||
The latch answers three attacks the red team confirmed against a latch-free
|
||||
design:
|
||||
|
||||
- **The spawn gate.** A member already *inside* `thread_spawn` on another core
|
||||
when the fan-out runs (it passed the syscall-entry `kill_pending` check, then
|
||||
spun on the BKL) would otherwise complete the spawn after the fan-out's lock
|
||||
hold ends — a fresh, uncondemned member that escapes the kill and, worse, holds
|
||||
a space reference that keeps the group-death hook from ever firing. Fix:
|
||||
`retainAddressSpace` (equivalently `spawnUserLocked`) **refuses a dying
|
||||
space**; the in-flight `thread_spawn` fails with `-ESRCH` under the same lock
|
||||
that would have created the member.
|
||||
- **Concurrent triggers.** A second member faulting (or exiting) on another core
|
||||
while the first fan-out runs must not re-run the fan-out, double-bump
|
||||
`fault_kill_count`, or re-stamp reasons. Every kill path checks the latch first:
|
||||
already dying → skip straight to `terminateCurrentLocked`, no stamp, no count.
|
||||
First trigger wins, deterministically. `exit_reason` and `fault_kill_count`
|
||||
writes move under the BKL as part of this.
|
||||
- **Notification ownership.** The leader's `exit_endpoint` is a counted
|
||||
birth-to-death reference dropped at notify time. The stamp pass **moves** that
|
||||
reference onto the `AddressSpaceRef` and nulls `Task.exit_endpoint` in the same
|
||||
hold, so the leader's own `releaseTaskResourcesLocked` sees null (no early
|
||||
notify, no double drop); the group-death hook notifies and drops exactly once.
|
||||
|
||||
## The fan-out: `killGroupLocked`
|
||||
|
||||
One new function in `process.zig`, running under a **single BKL hold** (built from
|
||||
the `*Locked` primitives — the lock is non-recursive, and `terminateCurrentLocked`
|
||||
never returns, which forces the shape):
|
||||
|
||||
```
|
||||
killGroupLocked(leader: u32, reason: ExitReason, trigger: ?*Task)
|
||||
0. Latch: AddressSpaceRef.dying = true, stash {reason, leader,
|
||||
leader's exit_endpoint (moved)}.
|
||||
1. Stamp pass: the LEADER's exit_reason = reason — the leader's
|
||||
record is the one the supervisor can read, so it carries the
|
||||
group reason even when the trigger is a worker. The trigger
|
||||
also keeps `reason` (its own record tells the truth); every
|
||||
other live member gets .killed. All members get kill_pending.
|
||||
Stamping precedes any teardown, because recordExitLocked
|
||||
snapshots the reason first thing.
|
||||
2. Reap pass, to fixpoint: reap every member in .ready or .blocked
|
||||
via reapTaskLocked, re-reading Task.state each iteration — a
|
||||
member's teardown can wake another member (-EPEER wakes, joiner
|
||||
wakes), flipping it .blocked → .ready behind the scan cursor.
|
||||
Terminates in ≤ one pass per member: the scrub calls in
|
||||
releaseTaskResourcesLocked (abandonSenderLocked,
|
||||
removeFromWaitQueueLocked, forgetIpcClientLocked,
|
||||
killOwnedEndpointsLocked) run before destroy, so no wake path
|
||||
holds a pointer to a reaped member.
|
||||
3. Members .running on other cores stay condemned (kill_pending);
|
||||
a condemned member dies at its next syscall entry, at its own
|
||||
core's next tick while in user mode, or — once it blocks or is
|
||||
preempted — at any core's next tick reap. There is no kill IPI.
|
||||
(The entry check reads kill_pending unlocked; benign on
|
||||
x86-TSO — a missed read is caught by the next delivery point —
|
||||
but make the field atomic when touching it.)
|
||||
4. If the current task is a member (fault, exit, in-group kill):
|
||||
terminateCurrentLocked, last, because it switches away and the
|
||||
reap paths free the kernel stack being stood on.
|
||||
If the caller is outside the group (supervisor kill): return.
|
||||
```
|
||||
|
||||
The invariants this preserves, each load-bearing today:
|
||||
|
||||
- **Only `.ready`/`.blocked` tasks are reaped synchronously.** A member running on
|
||||
another core can only be condemned — it tears itself down after switching CR3
|
||||
off the dying page tables (the stack it stands on is freed later by the reap
|
||||
list), and its address-space reference protects the page tables its CR3 still
|
||||
points at. Force-destroying the space under a running sibling is the one
|
||||
unrecoverable mistake available here.
|
||||
- **The refcount decides when the space dies.** Reaping N members drops N
|
||||
references; the last drop — possibly on a condemned sibling's core, a tick
|
||||
later — destroys the space. No path forces it.
|
||||
- **`fault_kill_count` bumps once per group**, not per member (`fault-recovery`
|
||||
asserts `== 1` exactly); the latch is what enforces this under racing faults.
|
||||
- **Per-tid resource sweeps stay per-tid.** Each member's
|
||||
`releaseTaskResourcesLocked` releases what *that tid* owns — claims, GSI/MSI
|
||||
bindings, registered endpoints, handles. That keying is correct under shared
|
||||
fate (and is today's hazard: a lone worker death already yanks its claims out
|
||||
from under live siblings). A worker that *does* carry an `exit_endpoint` (the
|
||||
ABI allows it; the runtime passes `no_cap`) keeps today's per-task posting at
|
||||
its own teardown — only the leader's notification moves.
|
||||
|
||||
## When is the group dead? The notification
|
||||
|
||||
Today each task posts its own exit notification as the *last* step of its release,
|
||||
so a supervisor observes a fully-released child. For a group that guarantee must
|
||||
hold for the **whole group**: if the leader's notification fires while a condemned
|
||||
sibling still runs on another core, the device manager can respawn the driver into
|
||||
a claim conflict with a not-yet-dead sibling.
|
||||
|
||||
The clean fix falls out of the refcount: **the group is dead exactly when the
|
||||
address space is destroyed.** `releaseAddressSpace`'s last-drop path calls a new
|
||||
`group_exit_hook` (the scheduler already calls up through hooks —
|
||||
`terminate_current_hook` — precisely to keep this layering), which:
|
||||
|
||||
1. **re-stamps the leader's exit record** with the stashed `group_reason` — the
|
||||
record is written (again) at group-death time, so "the reason is recorded
|
||||
before the notification posts" stays true and a supervisor can never be
|
||||
notified and then read `-ESRCH` because the burst evicted an old record;
|
||||
2. posts the leader's exit notification (and subscriber broadcast) from the
|
||||
stashed endpoint, and drops that reference — exactly once.
|
||||
|
||||
Both `releaseAddressSpace` call sites (`exitUserLocked`, `destroyTaskLocked`) run
|
||||
under the BKL, so the hook does too; its wakes are safe at both (verified). For a
|
||||
single-threaded process the behavior is *externally indistinguishable* from
|
||||
today — the order of notify vs. destroy inverts, but both sit inside one lock
|
||||
hold, so no other core can observe the space destroyed but the notification
|
||||
unposted, or vice versa. That sentence is the correctness argument; it is also the
|
||||
first invariant to re-examine if the BKL is ever split, along with
|
||||
`killGroupLocked`'s single-hold atomicity. (Hand-built spaces that were never
|
||||
retained take the immediate-destroy path and are out of the hook's scope.)
|
||||
|
||||
Workers' `exit_subscribers` broadcasts still fire per task — the FAT server's
|
||||
dead-client sweep is keyed by tid and needs those.
|
||||
|
||||
**Signals.** `signal_bind` is per-task and the service harness binds on the main
|
||||
thread, so signals address the leader in practice; that stays. During a group
|
||||
death, `process_signal` may return `0` (accepted by a condemned member — never
|
||||
delivered, every delivery point kills first) or `-ESRCH` (member already reaped);
|
||||
init's stop sequence already tolerates both, and its timer escalation to
|
||||
`process_kill` covers the gap. `process_signal` follows `process_kill`'s
|
||||
leader re-key for consistency.
|
||||
|
||||
## The shared-memory frame hazard
|
||||
|
||||
`dropSharedMemoryReference` frees a region's physical frames when the last *handle*
|
||||
reference drops, but mappings die only with the address space. If the last handle
|
||||
lived in a torn-down member while any task still has the region mapped, that task
|
||||
holds a live mapping onto freed frames — and the red team showed this is **not**
|
||||
group-specific: a plain `thread_exit` of the handle-holding thread, or a last-ref
|
||||
drop by a task *outside* the dying group during the condemned window, hits the same
|
||||
use-after-free.
|
||||
|
||||
So the fix is a property of the **object**, not the dropper: give
|
||||
`SharedMemoryObject` a per-*mapping* reference — `shared_memory_map` (and create's
|
||||
self-map) retains; each space's destruction releases. "Last reference" then means
|
||||
*no handles and no mappings*, both hazard paths collapse into the existing
|
||||
refcount, and no group-kill special case is needed at all.
|
||||
|
||||
## Deliberately unchanged
|
||||
|
||||
- Worker `thread_exit`: per-thread, full per-tid resource sweep, refcount drop.
|
||||
- The condemned-but-running window: a member on another core can finish its
|
||||
in-flight syscall and run user code for up to a tick before dying — identical to
|
||||
today's single-task `process_kill` semantics ("prompt but asynchronous, like a
|
||||
Unix signal"). A kill IPI would shrink it; it is not part of this plan.
|
||||
- `thread_join` returns 0 for a killed thread; joiners inside a dying group are
|
||||
woken and then reaped like any member.
|
||||
- The `.reaping` state, reap lists, and stack reaper.
|
||||
|
||||
## Accepted limits (documented, not fixed here)
|
||||
|
||||
- **Notify-ring overflow**: a group death posts one subscriber badge per member
|
||||
into 8-slot rings; a >8-member group can drop badges. Group size is bounded by
|
||||
the 48-task table; today's largest production group is 2 (display) and
|
||||
thread-test already reaches 5.
|
||||
- **Exit-record ring pressure**: one 64-entry ring, one record per member — made
|
||||
harmless for the supervisor by the hook's group-death re-stamp.
|
||||
- **Enumerate shows a partial group** mid-death: reaped members vanish at once,
|
||||
condemned members linger up to a tick (audited: no in-tree consumer
|
||||
misbehaves; the `leader` field in `ProcessDescriptor` lets future consumers
|
||||
group correctly).
|
||||
- **Pre-existing reap race, not widened**: a task preempted *mid-syscall* is
|
||||
`.ready` with `in_system_call = true`, and the tick's reap loop will reap it —
|
||||
an existing hazard the fan-out inherits but must not add new instances of.
|
||||
Filed to investigate separately.
|
||||
- **Per-task DMA/shm cursors** — *fixed during M4 after all*: the
|
||||
`shm-mapping-ref` test tripped the overlap (the sibling's churn regions mapped
|
||||
over the worker's region), so both cursors moved to the `AddressSpaceRef`
|
||||
like the mmap/MMIO cursors before them. The post-implementation review then
|
||||
found the other half: the shm/DMA page-table walks and their pmm/heap calls
|
||||
ran *outside* the big kernel lock — pre-existing, but fatal once siblings
|
||||
were invited to race them (and a plausible root for the long-standing
|
||||
intermittent AP ring-3 fault at the shm base). All three paths now follow
|
||||
the mmap discipline: metadata and allocation under one hold, the map itself
|
||||
per-page under brief holds.
|
||||
- **Mapping-record slots are never recycled**: 16 per space, one per
|
||||
`shared_memory_create`/`map`, freed only at space destruction (there is no
|
||||
shm unmap). A long-lived compositor that churns surfaces will hit the cap;
|
||||
the failure is a clean refused create, and slot recycling can ride whatever
|
||||
adds `shared_memory_unmap`.
|
||||
- **Two properties lack direct tests**: the spawn gate (an in-flight
|
||||
`thread_spawn` racing the fan-out — inherently nondeterministic to arrange;
|
||||
covered by code inspection and the `-ESRCH` path) and the `process_signal`
|
||||
leader re-key (exercised only implicitly by the signals case).
|
||||
|
||||
## Milestones
|
||||
|
||||
- **M1 — the leader id.** `Task.leader` (kernel tasks: 0, never followed), set on
|
||||
both spawn paths; `leader` added to `ProcessDescriptor`; `process_kill` and
|
||||
`process_signal` re-keyed (kernel-task `-ESRCH` guard *before* leader
|
||||
resolution). No fan-out yet. Existing tests must pass untouched.
|
||||
- **M2 — the latch + fan-out.** `AddressSpaceRef.dying` + stash;
|
||||
`retainAddressSpace` refuses dying spaces; `killGroupLocked`; wire the fault
|
||||
path, `exit`, and `process_kill` into it; leader `thread_exit` → `-EPERM`;
|
||||
`exit_reason`/`fault_kill_count` writes under the BKL; `kill_pending` made
|
||||
atomic. Group notification via the `group_exit_hook` re-stamp + post.
|
||||
- **M3 — shared-memory mapping refs.** `SharedMemoryObject` counts mappings;
|
||||
space destruction releases them; frames free only at zero handles *and* zero
|
||||
mappings.
|
||||
- **M4 — tests + docs.** New `-Dtest-case`s (all `smp: 4` where cross-core
|
||||
matters), driving `thread-test` with new argv modes:
|
||||
- `thread-fault-group`: a worker faults; assert both tasks gone from
|
||||
`enumerate`, `fault_kill_count == 1`, `process_exit_reason(leader) ==
|
||||
segmentation_fault`, address-space and stack-bytes counters return to base.
|
||||
- `kill-threaded-group`: `process_kill(leader)` with a worker spinning on
|
||||
another core; assert the worker dies by the deferred path, exactly one exit
|
||||
badge, delivered only after both members are dead, and
|
||||
`process_exit_reason(leader) == .killed`.
|
||||
- `kill-via-worker-tid`: `process_kill(worker)` kills the whole group;
|
||||
`-EPERM` for a non-supervisor aiming at the worker.
|
||||
- `racing-triggers`: two members fault/exit simultaneously on different cores;
|
||||
assert a deterministic leader reason and `fault_kill_count == 1`.
|
||||
- `exit-group`: a *worker* calls `exit(3)`; group dies, leader reason
|
||||
`.aborted`.
|
||||
- `leader-thread-exit`: asserts the chosen rule (`-EPERM`, workers unaffected).
|
||||
- `thread-exit-solo`: regression — worker `thread_exit` still leaves siblings
|
||||
running.
|
||||
- `group-claim-release`: a member claims a device; assert the claim is free and
|
||||
the leader notification arrives only after every member is dead.
|
||||
- `shm-mapping-ref`: last handle dropped by a dying thread; sibling's mapping
|
||||
stays valid until space death (M3 regression).
|
||||
Then update [threading.md](threading.md) (the shared-fate gap note),
|
||||
[process-lifecycle.md](process-lifecycle.md),
|
||||
[process-management.md](process-management.md), and
|
||||
[ipc.md](../device-driver-development/ipc.md)/[drivers.md](../device-driver-development/drivers.md) mentions.
|
||||
@@ -0,0 +1,143 @@
|
||||
# SMEP and SMAP — supervisor-mode hardening
|
||||
|
||||
*Design, 2026-07-31. Not yet implemented. Companion to
|
||||
[protocol-namespace.md](protocol-namespace.md) on the security track — this is
|
||||
the hardware half; that is the namespace half.*
|
||||
|
||||
Two CR4 bits that make the CPU refuse the two things a kernel should never do
|
||||
with user memory:
|
||||
|
||||
- **SMEP** (Supervisor Mode Execution Prevention, CR4 bit 20): instruction
|
||||
fetch in ring 0 from a page whose U/S bit says *user* → #PF. Kills the
|
||||
classic ret2usr exploit shape — a kernel bug that redirects control flow
|
||||
can no longer land in attacker-prepared user code.
|
||||
- **SMAP** (Supervisor Mode Access Prevention, CR4 bit 21): data read/write
|
||||
in ring 0 to a user page → #PF, unless `EFLAGS.AC` is set. `stac`/`clac`
|
||||
open and close deliberate access windows; danos's design needs no windows
|
||||
at all (below).
|
||||
|
||||
Detection is CPUID leaf 7, subleaf 0, EBX bit 7 (SMEP) and bit 20 (SMAP).
|
||||
Both bits are per-core state: the BSP and every AP must set them.
|
||||
|
||||
## Why, in danos terms
|
||||
|
||||
Every syscall argument is an attacker-controlled integer, and several take
|
||||
pointers. A kernel bug that dereferences a crafted pointer reads, writes, or
|
||||
executes memory of the attacker's choosing — the exact bug class the
|
||||
isolation tracks exist to prevent. SMEP/SMAP turn that class from "silent
|
||||
compromise" into "immediate, attributable #PF with a kernel RIP in the log."
|
||||
|
||||
The second benefit matters as much as the first: **SMAP is a permanent
|
||||
tripwire.** Once it is on, any *future* syscall that touches user memory
|
||||
directly — instead of going through the checked copy layer — faults the
|
||||
first time the QEMU suite runs it. The discipline stops depending on review.
|
||||
|
||||
## Where danos already stands
|
||||
|
||||
The design is closer than it looks, because the IPC layer was built right:
|
||||
|
||||
- **The copy layer is already SMAP-proof.** `copyAcross` and `copyFromUser`
|
||||
(`system/kernel/ipc-synchronous.zig:305,333`) never dereference a user
|
||||
virtual address: they walk the page tables and move bytes through the
|
||||
physmap — kernel mappings throughout. SMAP cannot object.
|
||||
- **Syscall entry already clears AC.** `SFMASK = 0x4_0700` clears IF, TF,
|
||||
DF, **AC** on every `syscall`
|
||||
(`system/kernel/architecture/x86_64/per-cpu.zig:76`). The syscall path is
|
||||
SMAP-clean from day one.
|
||||
- **The interrupt path is not.** Hardware does *not* clear AC on IDT
|
||||
delivery, and ring 3 can set AC with `popfq` — so a hostile process could
|
||||
take an interrupt with AC=1 and have the handler run with SMAP suspended.
|
||||
`isr_common` (`system/kernel/architecture/x86_64/isr.s:366`) needs a
|
||||
`clac` beside its `swapgs`.
|
||||
- **CR4 today:** the BSP inherits firmware CR4 (no kernel write anywhere);
|
||||
APs set PAE/OSFXSR/OSXMMEXCPT in `trampoline.s:62-68`. Neither path sets
|
||||
SMEP/SMAP yet, and both must.
|
||||
- **The stragglers.** Nine syscalls still dereference user pointers raw
|
||||
after a bounds check — every one is a SMAP #PF waiting to happen, and
|
||||
every one is *already* a latent kernel fault today (an unmapped-but-in-
|
||||
range user page oopses the kernel instead of failing the call). The
|
||||
verified sweep of `system/kernel/process.zig` (2026-07-31; a
|
||||
whole-kernel `@ptrFromInt` audit found no user-address dereference
|
||||
outside this file):
|
||||
|
||||
| Syscall | Raw access | Direction |
|
||||
|---|---|---|
|
||||
| `system_spawn` | name + argument blob (`:972`, `:980`) | read |
|
||||
| `fs_resolve` | path in (`:1780`), result out (`:1797`) | read + write |
|
||||
| `fs_mount` | prefix + rewrite strings (`:1864`, `:1865`) | read |
|
||||
| `fs_unmount` | prefix string (`:1883`) | read |
|
||||
| `fs_node` | read buffer out (`:1820`) | write |
|
||||
| `debug_write` | message bytes (`:1700`; read twice — memcpy `:1710` and `log.append` `:1717`) | read |
|
||||
| `klog_read` | log bytes out (`:1741`) | write |
|
||||
| `klog_status` | status struct out (`:1758`) | write |
|
||||
| `process_enumerate` | descriptor array out (`:1132`) | write |
|
||||
| `device_enumerate` | descriptor array out (`:388`) | write |
|
||||
|
||||
For the write-direction rows the `@ptrFromInt` is in process.zig but the
|
||||
stores happen in callees (`scheduler.enumerate`
|
||||
`system/kernel/scheduler.zig:1209`, `devices_broker.enumerate`
|
||||
`devices-broker.zig:136`, `log.readAt` `log.zig:209`, the vfs node calls
|
||||
`vfs.zig:257/269/289`) — converting them means bounce buffers plus
|
||||
`copyToUser` around those calls, not just editing the process.zig lines.
|
||||
(Some paths already do it right — the futex word and the device-register
|
||||
descriptor go through `copyFromUser` (`:1087`, `:924`). The write
|
||||
direction has no public helper yet, but the mechanism exists:
|
||||
`copyAcross` with a kernel source is exactly how IPC replies reach user
|
||||
buffers, so `copyToUser` is a mechanical mirror.)
|
||||
|
||||
- **One known gap inside the copy layer itself:** the walk checks presence,
|
||||
not the leaf U/S and writable bits (`ipc-synchronous.zig:20-22` flags
|
||||
this). Today that is nearly moot — the user half contains only mappings
|
||||
the kernel itself created for that process — but it must close before
|
||||
shared or copy-on-write mappings exist, and closing it is part of making
|
||||
the copy layer the single trusted door.
|
||||
|
||||
## The plan
|
||||
|
||||
**H1 — copy discipline (the real work).** A `user-memory` kernel module:
|
||||
`copyFromUser` / `copyToUser` (the missing write direction) via the physmap
|
||||
walk, with U/S and writable leaf checks closing the in-tree TODO. Convert
|
||||
the nine stragglers. This fixes the latent unmapped-page kernel fault on
|
||||
its own — it is worth doing even if SMEP/SMAP never shipped. QEMU suite
|
||||
green; no behavior change visible to correct programs.
|
||||
|
||||
**H2 — SMEP.** A leaf-7 feature probe (the kernel has per-leaf `cpuid`
|
||||
helpers in `apic.zig` to generalize); set CR4.SMEP during per-CPU bring-up
|
||||
on BSP and APs — prefer the Zig-side per-CPU init over the trampoline
|
||||
assembly, so one code path covers every core and the trampoline stays
|
||||
minimal. Audit first that ring 0 never executes user-mapped pages: kernel
|
||||
text lives in the kernel half, `jump_to_user` is kernel code, and the AP
|
||||
trampoline page is kernel-mapped — expected clean, verify before flipping.
|
||||
|
||||
**H3 — SMAP.** Add `clac` at `isr_common` entry. `clac` is #UD on CPUs
|
||||
without SMAP, so the instruction is a 3-byte NOP in the image, patched to
|
||||
`clac` at boot when CPUID advertises SMAP (one-time patch beats a
|
||||
conditional branch in the hottest path in the kernel). Then set CR4.SMAP in
|
||||
the same per-CPU init. From this point the whole QEMU suite doubles as the
|
||||
enforcement test: any missed raw dereference is a vector-14 with a kernel
|
||||
RIP and a user CR2 — loud and attributable.
|
||||
|
||||
**H4 — keep it honest.** A line in the coding standards: kernel code
|
||||
touches user memory only through `user-memory`; there is no `stac` anywhere
|
||||
in the tree, and a PR that adds one is wrong by definition. SMAP enforces
|
||||
the rule mechanically at test time.
|
||||
|
||||
Feature-gating follows the timekeeping rule (work on any VM, real Intel,
|
||||
real AMD): both bits are probed, absence is logged and tolerated — like the
|
||||
IOMMU's fail-open, the machine still boots, just unhardened. QEMU: TCG
|
||||
implements both; KVM inherits the host (Intel Ivy Bridge+ for SMEP,
|
||||
Broadwell+ for SMAP; AMD Zen+ for both). The test images should run with
|
||||
`-cpu max` so the suite always exercises the enabled paths.
|
||||
|
||||
## Adjacent, deliberately separate
|
||||
|
||||
- **SYSRET canonical-RIP hardening** (`isr.s:192-194` documents it): a
|
||||
non-canonical return RIP makes `sysretq` #GP *in ring 0* on Intel. Same
|
||||
hardening bucket, independent fix (validate RCX before `sysretq`, fall
|
||||
back to `iretq`), should ride the same branch as H2/H3 but is not
|
||||
SMEP/SMAP.
|
||||
- **KPTI / Meltdown-class leaks are out of scope.** SMEP/SMAP police
|
||||
architectural accesses, not speculative ones. danos runs one kernel
|
||||
mapping in every address space and accepts that on affected hardware;
|
||||
revisit only if the threat model ever includes hostile native code on
|
||||
shared machines.
|
||||
@@ -112,7 +112,7 @@ Yes — and this is the branch that matters for danos right now.
|
||||
|
||||
These pull in different directions, so **picking the primary goal comes before
|
||||
picking the SMP design.** (danos's founding assumption was real-time; that's under
|
||||
active reconsideration in favour of resilience — see [vision.md](vision.md).)
|
||||
active reconsideration in favour of resilience — see [vision.md](../vision.md).)
|
||||
|
||||
## What this would mean for danos
|
||||
|
||||
@@ -169,7 +169,7 @@ next lands.
|
||||
the highest-priority ready task; per-core queues are a later optimisation.
|
||||
- **AP wake to long mode** — `architecture.startSecondary` drives INIT–SIPI–SIPI (via the
|
||||
LAPIC ICR) to wake each parked core one at a time. A woken core starts in 16-bit
|
||||
real mode at a low page and runs the [trampoline](../system/kernel/architecture/x86_64/trampoline.s)
|
||||
real mode at a low page and runs the [trampoline](../../system/kernel/architecture/x86_64/trampoline.s)
|
||||
up through protected mode into 64-bit long mode, then lands in `smp.zig:apEntry`,
|
||||
publishes its per-CPU pointer, and reports in. Verified in QEMU with `-smp 4`:
|
||||
all four cores report `online`.
|
||||
@@ -270,6 +270,6 @@ next lands.
|
||||
|
||||
- [scheduling.md](scheduling.md) — the single-core scheduler SMP would extend.
|
||||
- [discovery.md](discovery.md) — enumerating cores is a device-discovery problem.
|
||||
- [ipc.md](ipc.md) — the message passing cross-core coordination rides on.
|
||||
- [vision.md](vision.md) — the goals question (real-time vs resilience) this note
|
||||
- [ipc.md](../device-driver-development/ipc.md) — the message passing cross-core coordination rides on.
|
||||
- [vision.md](../vision.md) — the goals question (real-time vs resilience) this note
|
||||
keeps bumping into.
|
||||
@@ -51,7 +51,7 @@ Everything else---including`read()`,`write()`,`malloc()`, and`fork()`---will run
|
||||
3. **`Yield()`/`Thread_Ctrl()`**
|
||||
- **What it does:**Allows a thread to voluntarily give up its CPU time slice, or allows a root task to spawn/kill threads.
|
||||
4. **`ipc_send(endpoint, message_buffer)`(Asynchronous Send)**
|
||||
- **What it does:**Posts a small payload to an endpoint's bounded queue and returns *without* blocking — no rendezvous, no reply. The receiver picks it up through the same `IPC_ReplyWait`, as a buffered message. It is the async counterpart of `IPC_Call`, for one-to-many broadcasts where a synchronous rendezvous would let one dead or slow receiver hang the sender. The [input service](input.md) — keyboard-event fan-out — is its first user. A full queue drops the oldest message (a buffered message is discrete data, unlike a coalescing interrupt notification).
|
||||
- **What it does:**Posts a small payload to an endpoint's bounded queue and returns *without* blocking — no rendezvous, no reply. The receiver picks it up through the same `IPC_ReplyWait`, as a buffered message. It is the async counterpart of `IPC_Call`, for one-to-many broadcasts where a synchronous rendezvous would let one dead or slow receiver hang the sender. The [input service](../device-driver-development/input.md) — keyboard-event fan-out — is its first user. A full queue drops the oldest message (a buffered message is discrete data, unlike a coalescing interrupt notification).
|
||||
|
||||
* * * * *
|
||||
|
||||
@@ -92,4 +92,4 @@ Managing the Payload Challenge
|
||||
Because it is a microkernel, performance lives or dies by how fast your`IPC_Call`can move data from Client to Server. You have two minimal choices for handling the`message_buffer`pointer:[[1](https://anazimzada2020.medium.com/microkernel-architectural-pattern-5e4e9184170e)]
|
||||
|
||||
- **The Copy Method (Simplest to start):**Your kernel pauses the client, reads the data from the client's memory space, switches page tables to the server, and copies the data into the server's buffer.
|
||||
- **The Shared Memory Method (Fastest):**The kernel sets up a temporary, shared virtual memory page between the client and server. The client writes to it, calls`syscall`/`svc`, and the server reads it instantly without the kernel copying any bytes
|
||||
- **The Shared Memory Method (Fastest):**The kernel sets up a temporary, shared virtual memory page between the client and server. The client writes to it, calls`syscall`/`svc`, and the server reads it instantly without the kernel copying any bytes
|
||||
@@ -0,0 +1,131 @@
|
||||
# system.img — the boot capsule
|
||||
|
||||
## What it is
|
||||
|
||||
`boot/system.img` is the **boot capsule**: every bundled user binary — init, the
|
||||
services, the drivers, the test programs — packed into **one file** on the boot
|
||||
volume. It is not a filesystem image and it is not compressed; it is exactly the
|
||||
kernel's **initial-ramdisk wire format** (`system/initial-ramdisk.zig`, format
|
||||
v2), written to disk ahead of time. The EFI loader reads it in a single
|
||||
sequential pass and hands the bytes to the kernel unmodified.
|
||||
|
||||
The capsule is a *performance artifact*, not a source of truth. The boot
|
||||
volume's `/system` and `/test` file trees remain the canonical layout (see
|
||||
[file-system-hierarchy.md](../file-system-development/file-system-hierarchy.md));
|
||||
the capsule is a pre-baked snapshot of the same binaries, derived from the same
|
||||
build graph, so the running system is identical whether the loader read the
|
||||
capsule or walked the tree.
|
||||
|
||||
## Why it exists
|
||||
|
||||
Firmware file I/O has exactly one fast shape: **one open + one sequential
|
||||
read**. Everything else is a lottery. Loading the system per-file — dozens of
|
||||
opens, seeks, and short reads through the firmware's FAT driver — measured
|
||||
**minutes** on real hardware, against milliseconds in QEMU/OVMF. Packing the
|
||||
binaries into a single file turns the whole of user space into the shape
|
||||
firmware is good at.
|
||||
|
||||
Because the capsule already *is* the ramdisk wire format, the loader doesn't
|
||||
even repack it: `loadCapsule` (`boot/efi.zig`) validates the magic and passes
|
||||
the buffer straight through as `BootInformation.initial_ramdisk_base`/`len`.
|
||||
|
||||
## The format
|
||||
|
||||
The container is deliberately trivial — danos owns both producer and consumer,
|
||||
so it need be no fancier. Little-endian throughout:
|
||||
|
||||
```
|
||||
Header magic: u32 = "DNR2" (0x32524E44), count: u32
|
||||
Entry × count name: [64]u8 (NUL-padded hierarchy path), offset: u64, len: u64
|
||||
blobs... each entry's file bytes, at its offset within the image
|
||||
```
|
||||
|
||||
- **Names are full hierarchy paths** (`/system/services/init`), not basenames — that
|
||||
is what "v2" means. The 64-byte capacity matches `abi.maximum_process_name`,
|
||||
so a task named after its binary path is never truncated. Paths longer than
|
||||
63 bytes are a build error (`pack-system-image.py` rejects them).
|
||||
- **The v1 magic (`"DNRD"`, basename entries) is rejected**, not tolerated: a
|
||||
stale image should fail loudly at `Reader.init`, not misparse names.
|
||||
- `initial_ramdisk.Reader` is the one validated view over the bytes — magic
|
||||
check, table bounds, per-blob bounds — used by the kernel and shared with the
|
||||
loader. `Reader.find` resolves a binary by exact path first, then by unique
|
||||
basename, ASCII case-insensitively (the entries come from a FAT volume, whose
|
||||
name lookups are case-insensitive by definition).
|
||||
|
||||
## How it is built
|
||||
|
||||
`build.zig` maintains one `bundled` list — every user binary and its hierarchy home.
|
||||
Three artifacts are derived from that same list, in the same build graph, so
|
||||
they cannot drift apart:
|
||||
|
||||
1. **The tree**: each binary installed at its hierarchy path (`zig-out/system/...`
|
||||
and `zig-out/test/...`, mirrored onto the FAT boot volume by
|
||||
`tools/make-fat-image.py`).
|
||||
2. **The manifest** (`system/manifest`): the hierarchy path of every bundled binary,
|
||||
one per line — the loader's per-file fallback input.
|
||||
3. **The capsule**: `tools/pack-system-image.py` packs the same binaries into
|
||||
the v2 container, installed at `zig-out/boot/system.img` and placed on the
|
||||
boot volume at `boot/system.img`.
|
||||
|
||||
Note what the capsule does *not* contain: the kernel (`system/kernel` is loaded
|
||||
separately by `loadKernel`, as an ELF) and the EFI loader itself. It is user
|
||||
space only.
|
||||
|
||||
## How it is loaded
|
||||
|
||||
`loadSystemTree` (`boot/efi.zig`) tries three strategies, most portable first —
|
||||
the running system cannot tell which one ran, because all three produce the
|
||||
same in-RAM ramdisk image:
|
||||
|
||||
1. **The capsule** — open `boot\system.img`, read it whole, check the magic,
|
||||
hand it over as-is. The normal path on any build-produced volume.
|
||||
2. **The manifest** — read `system\manifest` and open each listed path *by
|
||||
name*. FAT name lookup is case-insensitive and firmware-portable, unlike
|
||||
directory enumeration. The loader assembles the v2 image in RAM itself.
|
||||
3. **The tree walk** — enumerate `/system` and `/test` recursively (`/test`
|
||||
is optional: a stick without fixtures still boots). Last resort for
|
||||
hand-assembled sticks with neither file: some firmware FAT drivers return
|
||||
bare 8.3 names uppercase from enumeration, which is why this is the
|
||||
fallback and not the primary path.
|
||||
|
||||
All three are best-effort: a **kernel-only volume still boots** — the kernel
|
||||
just has no user binaries to spawn and reports the absence.
|
||||
|
||||
One operational consequence of the ordering: the capsule *shadows* the tree.
|
||||
If you hand-edit binaries on a stick that also carries a `boot/system.img`,
|
||||
your edits are invisible — the loader boots the capsule's snapshot. Delete
|
||||
`boot/system.img` from the volume to force the manifest/tree path.
|
||||
|
||||
## What the kernel does with it
|
||||
|
||||
The loader records the image's physical base and length in `BootInformation`;
|
||||
the kernel (`kernel.zig`) then publishes the same bytes twice, to two
|
||||
consumers:
|
||||
|
||||
- **The process layer** (`process.zig`): `system_spawn` looks binaries up in
|
||||
the ramdisk via `Reader.find` — exact hierarchy path, or unique basename for
|
||||
pre-path callers — and loads them as fresh ring-3 processes. The stored path
|
||||
becomes the task's name.
|
||||
- **The VFS root** (`vfs.zig`, `setInitialRamdisk`): the image is mounted as
|
||||
kernel-backed, read-only mounts — one per top-level tree named by the entry
|
||||
paths, so `/system` and, when the fixtures are bundled, `/test`. Directory
|
||||
nodes are derived from the entry paths (the unique parents), so the trees
|
||||
are listable and their files readable over the normal VFS protocol — the
|
||||
boot tree every process sees comes straight out of the capsule bytes.
|
||||
|
||||
The image is never copied after the handoff and never mutated: the initrd is
|
||||
immutable, which is what makes the VFS's node serving lock-free.
|
||||
|
||||
## What it is not
|
||||
|
||||
- **Not `danos-usb.img`.** That is the 64 MiB FAT32 *boot volume* built by
|
||||
`tools/make-fat-image.py` — the thing a machine actually boots, which
|
||||
*contains* `boot/system.img` alongside the loader, kernel, manifest, and
|
||||
tree. See [efi.md](efi.md) and [release-iso.md](release-iso.md).
|
||||
- **Not a mountable filesystem.** No FAT, no block device, no driver — just a
|
||||
header, a table, and concatenated blobs, parsed by ~90 lines of
|
||||
`initial-ramdisk.zig`.
|
||||
- **Not required.** It is the fast path, with two slower equivalents behind
|
||||
it.
|
||||
- **Not a place where state lives.** It is regenerated on every build from the
|
||||
bundled binaries; nothing writes to it, at build time or runtime.
|
||||
@@ -59,9 +59,9 @@ danos's two binaries default to different conventions:
|
||||
When the loader jumps to the kernel passing the `BootInformation` pointer, both sides have
|
||||
to agree *which register that pointer lands in*. Left to their defaults, the loader
|
||||
would place it in RCX while the kernel looked in RDI — and the kernel would read
|
||||
garbage. So both sides reference the same `system.kernel_abi` (SysV): the loader's
|
||||
garbage. So both sides reference the same `boot_handoff.kernel_abi` (SysV): the loader's
|
||||
function-pointer type and the kernel's `_start` both carry
|
||||
`callconv(system.kernel_abi)`, and the pointer reliably arrives in RDI. That is the
|
||||
`callconv(boot_handoff.kernel_abi)`, and the pointer reliably arrives in RDI. That is the
|
||||
whole reason `kernel_abi` lives in the shared contract — see [efi.md](efi.md) for
|
||||
the handoff it governs.
|
||||
|
||||
@@ -90,9 +90,9 @@ process) instead of silently corrupting the image
|
||||
(`buildEntryStack` in `system/kernel/process.zig`); `argv[0]` is always the path
|
||||
or initial-ramdisk name the process was spawned as, and `system_spawn`'s optional
|
||||
argument blob becomes `argv[1..]`. The runtime's `_start`
|
||||
(`library/runtime/start.zig`) hands the block to `rt_start`, which builds a
|
||||
`runtime.process.Init` from it and passes that to the program's `main`
|
||||
(`pub fn main(init: runtime.process.Init)`; a parameterless `main()` is also
|
||||
(`library/kernel/start.zig`) hands the block to `rt_start`, which builds a
|
||||
`process.Init` from it and passes that to the program's `main`
|
||||
(`pub fn main(init: process.Init)`; a parameterless `main()` is also
|
||||
accepted). A C runtime's `crt0` would walk
|
||||
the identical layout unmodified — that's the compatibility being bought. The
|
||||
`args` test proves the round trip.
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
The ordered, checkpointable build-out for [threading.md](threading.md). Each milestone
|
||||
lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run, like
|
||||
[display-v2-plan.md](display-v2-plan.md). Read threading.md first for the *why*.
|
||||
[display-v2-plan.md](../device-driver-development/display-v2-plan.md). Read threading.md first for the *why*.
|
||||
|
||||
## Locked decisions (do not relitigate)
|
||||
|
||||
@@ -13,7 +13,7 @@ lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run,
|
||||
`single_threaded = false`.
|
||||
- **Blocking is futex-backed, never spin-backed** — waiters park in the kernel so an
|
||||
idle core still halts ([halting.md](halting.md)).
|
||||
- **New syscalls are private**: extend [abi.zig](../system/abi.zig) `SystemCall` after
|
||||
- **New syscalls are private**: extend [abi.zig](../../system/abi.zig) `SystemCall` after
|
||||
`shared_memory_physical = 36` (`thread_spawn = 37`, `thread_exit = 38`, `current_core = 39`,
|
||||
`futex_wait = 40`, `futex_wake = 41`) + a `library/runtime` wrapper; user code never names a number.
|
||||
- **Restart granularity stays the process** — a faulting thread kills its process; the
|
||||
@@ -21,12 +21,13 @@ lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run,
|
||||
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](coding-standards.md): spell out non-acronym abbreviations,
|
||||
kebab-case file names, no `Co-Authored-By` trailers. New user binaries go through
|
||||
`addUserBinary` (with the new `threaded` flag where a binary spawns threads) and get
|
||||
packed into the initial-ramdisk; new syscalls extend [abi.zig](../system/abi.zig)
|
||||
`SystemCall` + a `library/runtime` wrapper; test services live beside the code they
|
||||
exercise and register a `ServiceId` if they must be looked up.
|
||||
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym abbreviations,
|
||||
kebab-case file names, no `Co-Authored-By` trailers. New user binaries are
|
||||
packages whose build.zig calls `build_support.userBinary` (with `.threaded =
|
||||
true` where a binary spawns threads) and get packed into the initial-ramdisk;
|
||||
new syscalls extend [abi.zig](../../system/abi.zig) `SystemCall` + a
|
||||
`library/kernel` wrapper; test services live beside the code they exercise and
|
||||
bind a `/protocol/test/...` name if they must be reachable.
|
||||
|
||||
## How to verify along the way
|
||||
|
||||
@@ -58,7 +59,7 @@ fixed in *Locked decisions*; the checkboxes are the only state. A loop iteration
|
||||
5) so progress is backed up remotely; **do not push `main`** — merging Phase 2 into
|
||||
`main` stays a human step.
|
||||
3. **Implement** every unchecked item in that milestone, including adding its
|
||||
`-Dtest-case` to `CASES` in [test/qemu_test.py](../test/qemu_test.py) (with
|
||||
`-Dtest-case` to `CASES` in [test/qemu_test.py](../../test/qemu_test.py) (with
|
||||
`smp: true` / a `mem` bump where noted) so the gate is runnable.
|
||||
4. **Run the gate**: `python3 test/qemu_test.py <case>`, then the full **guardrail
|
||||
set**, then `zig build` (clean) and `zig build test` (green).
|
||||
@@ -67,7 +68,7 @@ fixed in *Locked decisions*; the checkboxes are the only state. A loop iteration
|
||||
the whole guardrail set passes, `zig build` is clean, and host tests are green.
|
||||
→ tick this milestone's boxes **and** its `**Gate:**`-referenced case, `git commit`
|
||||
(`threads(M<n>): <summary>`, no `Co-Authored-By` trailer per
|
||||
[coding-standards.md](coding-standards.md)), then **`git push` the working branch to
|
||||
[coding-standards.md](../coding-standards.md)), then **`git push` the working branch to
|
||||
`origin`** (use `-u` on the first push to set upstream). Continue to the next
|
||||
milestone in the same iteration if budget remains; otherwise let the loop re-fire.
|
||||
- **Red** = anything above fails. Diagnose from the captured serial log
|
||||
@@ -111,7 +112,7 @@ task's exit; make destruction happen on the **last** exit.
|
||||
- [x] A refcount keyed by the address-space root, held in `scheduler.zig`
|
||||
(`address_space_refs`): `retainAddressSpace` takes a reference in `spawnUserLocked` (on the
|
||||
success path, after the slot + stack are secured), all under the big kernel lock.
|
||||
- [x] Both task-teardown paths ([scheduler.zig](../system/kernel/scheduler.zig):
|
||||
- [x] Both task-teardown paths ([scheduler.zig](../../system/kernel/scheduler.zig):
|
||||
`exitUserLocked` and `destroyTaskLocked`) call `releaseAddressSpace`, which decrements
|
||||
and only `destroyAddressSpace`s at **zero**; an unretained space (hand-built test
|
||||
spaces) is destroyed directly, preserving prior behaviour.
|
||||
@@ -133,7 +134,7 @@ full guardrail set passes unchanged — 13/13 (`smoke`, `sched`, `priority`, `sm
|
||||
Spawn only — no join yet. Prove a second task executes in the **caller's** address
|
||||
space and exits cleanly.
|
||||
|
||||
- [x] [abi.zig](../system/abi.zig): `thread_spawn = 37`, `thread_exit = 38`. Handlers in
|
||||
- [x] [abi.zig](../../system/abi.zig): `thread_spawn = 37`, `thread_exit = 38`. Handlers in
|
||||
process.zig; `thread_spawn` calls `scheduler.spawnThread` (today, after M3, the
|
||||
handler goes `spawnThreadSupervised` → `scheduler.spawnUserLocked`; shares the caller's
|
||||
address space, `retainAddressSpace`); `thread_exit` ends the task like a process `exit(0)`
|
||||
@@ -195,7 +196,7 @@ plus `address-space-refcount`/`thread-spawn`; `zig build` clean, `zig build test
|
||||
|
||||
## M4 — Futex: the one blocking primitive ✅
|
||||
|
||||
- [x] [abi.zig](../system/abi.zig): `futex_wait = 40`, `futex_wake = 41`. A waiter is a
|
||||
- [x] [abi.zig](../../system/abi.zig): `futex_wait = 40`, `futex_wake = 41`. A waiter is a
|
||||
`.blocked` task tagged with `Task.futex_addr` (no queue linkage);
|
||||
`futex_wait(addr, expected, timeout_ns)` reads the user word under the big lock,
|
||||
parks iff `*addr == expected`, and returns on wake or timeout; `futex_wake(addr,
|
||||
@@ -261,7 +262,7 @@ green.
|
||||
restores it.
|
||||
- [x] `RwLock` / `WaitGroup` deferred (no consumer yet); they slot onto the same
|
||||
`Futex`/`Mutex`/`Condition` when wanted.
|
||||
- [x] All `thread-*` cases wired into [test/qemu_test.py](../test/qemu_test.py)
|
||||
- [x] All `thread-*` cases wired into [test/qemu_test.py](../../test/qemu_test.py)
|
||||
(`thread-spawn`/`-join`/`-futex`/`-mutex`/`-id`); threading.md + docs/README.md
|
||||
status updated to **built**; the worked example is threading.md's win-condition.
|
||||
- [x] `-Dtest-case=thread-id` (`smp: 4`): two workers read `getCurrentId`; the main
|
||||
@@ -405,7 +406,7 @@ guardrail 26/26 (incl. `process-kill`, `supervision`, `fault-recovery`, `task-re
|
||||
### M10 — Per-thread TLS: the thread-pointer mechanism ✅
|
||||
|
||||
Give each thread its own thread pointer and private TLS storage — the foundation
|
||||
self-hosting Zig ([zig-self-hosting.md](zig-self-hosting.md)) will build `threadlocal` on.
|
||||
self-hosting Zig ([zig-self-hosting.md](../zig-self-hosting.md)) will build `threadlocal` on.
|
||||
|
||||
- [x] **Kernel** stores `thread_pointer` on `Task` and restores it on every context switch
|
||||
**only when it changes** (the same conditional-load discipline as CR3;
|
||||
@@ -459,7 +460,7 @@ clean.
|
||||
## Deferred (explicitly not in this plan)
|
||||
|
||||
- **Cross-process shared-memory futex** — the `(address_space, virtual_address)` key can become a
|
||||
physical-address key so two processes share a futex through a [shared-memory](display-v2.md)
|
||||
physical-address key so two processes share a futex through a [shared-memory](../device-driver-development/display-v2.md)
|
||||
region. Not needed for intra-process threads.
|
||||
- **Per-thread priorities / affinity distinct from the process** — threads inherit the
|
||||
process priority ([scheduling.md](scheduling.md)); revisit only if it earns its keep.
|
||||
@@ -467,5 +468,5 @@ clean.
|
||||
([process-lifecycle.md](process-lifecycle.md)).
|
||||
- **A `pthread`/POSIX surface** — the API is `std.Thread`-shaped Zig, nothing more.
|
||||
- **A real `std.Thread` backend** — arrives with self-hosting
|
||||
([zig-self-hosting.md](zig-self-hosting.md)); it sits on these same primitives, so it
|
||||
([zig-self-hosting.md](../zig-self-hosting.md)); it sits on these same primitives, so it
|
||||
swaps the impl under `runtime.Thread`, not the call sites.
|
||||
@@ -1,8 +1,8 @@
|
||||
# Threading: `runtime.Thread`, a std-shaped API over a private thread ABI
|
||||
# Threading: `Thread`, a std-shaped API over a private thread ABI
|
||||
|
||||
A note on danos **threads** — several tasks sharing one address space — provided by a
|
||||
`runtime.Thread` type that mirrors the shape of Zig's `std.Thread` while keeping every
|
||||
kernel entry behind the [runtime](../library/runtime). **Built** (M1–M11, see
|
||||
`Thread` type that mirrors the shape of Zig's `std.Thread` while keeping every
|
||||
kernel entry behind the [runtime](../../library/kernel). **Built** (M1–M11, see
|
||||
[threading-plan.md](threading-plan.md)): `spawn`/`join`/`detach`, cross-core parallelism,
|
||||
a futex, `Mutex`/`Condition`/`Semaphore`/`RwLock`/`WaitGroup`, `getCurrentId`/`currentCore`,
|
||||
per-thread thread-pointer TLS, thread-safe allocation, and a task reaper that reclaims dead
|
||||
@@ -17,13 +17,13 @@ treat upstream shapes as "0.16.x."
|
||||
A danos service can write
|
||||
|
||||
```zig
|
||||
const t = try runtime.Thread.spawn(.{}, worker, .{ctx});
|
||||
const t = try Thread.spawn(.{}, worker, .{ctx});
|
||||
// ... do other work concurrently ...
|
||||
t.join();
|
||||
```
|
||||
|
||||
and get real parallelism across cores — with `runtime.Thread.Mutex`,
|
||||
`runtime.Thread.Condition`, and `runtime.Thread.Semaphore` available for
|
||||
and get real parallelism across cores — with `Thread.Mutex`,
|
||||
`Thread.Condition`, and `Thread.Semaphore` available for
|
||||
coordination — **without any code path reaching the kernel except through the
|
||||
runtime**. The call sites read exactly like `std.Thread`, so the day danos becomes a
|
||||
real Zig target (see [self-hosting](#the-self-hosting-endgame)) we swap the
|
||||
@@ -31,18 +31,18 @@ implementation underneath, not the API above.
|
||||
|
||||
## Locked decisions (do not relitigate)
|
||||
|
||||
- **We build `runtime.Thread`, not literal `std.Thread`.** It mirrors std's *API and
|
||||
- **We build `Thread`, not literal `std.Thread`.** It mirrors std's *API and
|
||||
features*; the implementation underneath is danos-native. See
|
||||
[Why not literal std.Thread](#why-not-literal-stdthread).
|
||||
- **Threads are a narrow, opt-in capability — not the default concurrency tool.** The
|
||||
default for resilience stays **process + IPC** ([resilience.md](resilience.md),
|
||||
[ipc.md](ipc.md)). See [Where threads fit](#where-threads-fit-the-resilience-tension).
|
||||
[ipc.md](../device-driver-development/ipc.md)). See [Where threads fit](#where-threads-fit-the-resilience-tension).
|
||||
- **Blocking synchronization is futex-backed, never spin-backed.** Waiters sleep in
|
||||
the kernel so an idle core still halts ([halting.md](halting.md)).
|
||||
- **Per-binary opt-in to multi-threaded codegen.** Only a service that asks for
|
||||
threads is built `single_threaded = false`; the rest stay lean and single-threaded.
|
||||
- **The thread ABI is private.** New syscalls extend [abi.zig](../system/abi.zig)
|
||||
`SystemCall` and are reached only through `library/runtime` wrappers, exactly like
|
||||
- **The thread ABI is private.** New syscalls extend [abi.zig](../../system/abi.zig)
|
||||
`SystemCall` and are reached only through `library/kernel` wrappers, exactly like
|
||||
every other danos syscall ([syscall.md](syscall.md)) — numbers stay renumberable.
|
||||
|
||||
## Why not literal `std.Thread`
|
||||
@@ -56,14 +56,15 @@ runtime — rebuilt in lockstep — knows the mapping.
|
||||
`std.Thread` is incompatible with that invariant on two counts:
|
||||
|
||||
1. **It selects its backend from `builtin.os.tag`, and issues syscalls directly.**
|
||||
danos targets `.os_tag = .freestanding` ([build.zig](../build.zig)), for which
|
||||
danos targets `.os_tag = .freestanding` ([build.zig](../../build.zig)), for which
|
||||
`std.Thread` resolves to an unsupported stub that `@compileError`s. Adding a real
|
||||
backend would either bake danos syscall numbers into std (breaking ABI privacy and
|
||||
renumbering) or fork std to route back through the runtime — a permanent rebase
|
||||
cost that buys nothing the native type doesn't.
|
||||
2. **Our user binaries are built `single_threaded = true`** ([build.zig](../build.zig)
|
||||
`addUserBinary`), which compiles threading out entirely and makes atomics and TLS
|
||||
single-threaded. Threads need this flipped per binary regardless.
|
||||
2. **Our user binaries are built `single_threaded = true`** (the shared recipe in
|
||||
[build-support/build.zig](../../build-support/build.zig)), which compiles threading
|
||||
out entirely and makes atomics and TLS single-threaded. Threads need this flipped
|
||||
per binary regardless.
|
||||
|
||||
So we take the *shape* of `std.Thread`, not the *type*. The cost of replicating the
|
||||
surface (spawn/join/Mutex/Condition) is small; the cost of the std type is the ABI
|
||||
@@ -99,7 +100,7 @@ processes. The isolation boundary stays at process granularity.
|
||||
|
||||
## The API surface (mirrors `std.Thread`)
|
||||
|
||||
Lives in `library/runtime/thread.zig`, re-exported as `runtime.Thread`.
|
||||
Lives in `library/kernel/thread.zig`, re-exported as `Thread`.
|
||||
|
||||
```zig
|
||||
pub const Thread = struct {
|
||||
@@ -129,14 +130,14 @@ Deviations from `std.Thread`, called out honestly:
|
||||
`void`). Return data through shared state or a `Semaphore`/`Condition`, not the
|
||||
return.
|
||||
- No `getCpuCount()` (a service rarely needs it) and no `Thread.yield()` — `yield`
|
||||
lives in `runtime.system`. Instead `currentCore()` exposes the calling core's dense
|
||||
lives in the `process` module. Instead `currentCore()` exposes the calling core's dense
|
||||
index ([smp.md](smp.md)), used to observe genuine cross-core parallelism.
|
||||
|
||||
## Kernel primitives (new private syscalls)
|
||||
|
||||
Five core entries extend [abi.zig](../system/abi.zig) `SystemCall` after
|
||||
Five core entries extend [abi.zig](../../system/abi.zig) `SystemCall` after
|
||||
`shared_memory_physical = 36` (plus small helpers `current_core`, `thread_self`, and
|
||||
`set_thread_pointer`), each with a `library/runtime` wrapper:
|
||||
`set_thread_pointer`), each with a `library/kernel` wrapper:
|
||||
|
||||
| Syscall | Signature | Purpose |
|
||||
|---|---|---|
|
||||
@@ -155,12 +156,12 @@ Plus one invariant change with no new syscall: **address-space reference countin
|
||||
Before this work an address space was 1:1 with a task: `spawnUserLocked` records
|
||||
`address_space` on the Task (as it still does), and teardown did
|
||||
`destroyAddressSpace(t.address_space)` when **any** user task exited
|
||||
([scheduler.zig](../system/kernel/scheduler.zig)). With threads, several tasks share
|
||||
([scheduler.zig](../../system/kernel/scheduler.zig)). With threads, several tasks share
|
||||
one `address_space`, so the first to exit would rip the address space out from under its
|
||||
siblings.
|
||||
|
||||
Fix: a small refcount keyed by the address-space root, kept in
|
||||
[scheduler.zig](../system/kernel/scheduler.zig): `retainAddressSpace` takes a
|
||||
[scheduler.zig](../../system/kernel/scheduler.zig): `retainAddressSpace` takes a
|
||||
reference for every user task `spawnUserLocked` starts (count 1 on the first take, so
|
||||
a thread sharing the caller's space increments it); task teardown calls
|
||||
`releaseAddressSpace`, which only calls `destroyAddressSpace` at **zero**. All of
|
||||
@@ -172,7 +173,7 @@ that must land and be proven before anything shares an address space.
|
||||
The scheduler already accepts an arbitrary `address_space` and does **not** smuggle
|
||||
values through scratch registers — `startUserTask` reads the entry/stack (and the
|
||||
thread's closure arg, delivered in `rdi` via `jumpToUserArg`) from the Task
|
||||
([scheduler.zig](../system/kernel/scheduler.zig)). That makes the thread path clean:
|
||||
([scheduler.zig](../../system/kernel/scheduler.zig)). That makes the thread path clean:
|
||||
|
||||
1. The runtime's `spawn` `mmap`s a stack (syscall `4`) and writes the closure —
|
||||
`{ tls_base, args }`, the std "Instance" pattern — at the **top of the new stack
|
||||
@@ -214,7 +215,7 @@ Keying: threads share an address space, so a **virtual address within that addre
|
||||
identifies a futex uniquely; the kernel keys its wait queue by `(address_space_root, virtual_address)`.
|
||||
Keying by the **physical** address instead (translate `virtual_address -> physical_address` on entry) is a
|
||||
deliberate forward door: it lets two *processes* share a futex through an
|
||||
[shared-memory](display-v2.md) region later, without changing the API. We start with the
|
||||
[shared-memory](../device-driver-development/display-v2.md) region later, without changing the API. We start with the
|
||||
private-per-address-space key and note the physical-key upgrade.
|
||||
|
||||
No spinning: a contended lock parks the task in the kernel and the core is free to run
|
||||
@@ -236,10 +237,11 @@ see the intro). Two scoped pieces, as built:
|
||||
|
||||
### Build: multi-threaded codegen, opt-in
|
||||
|
||||
A binary opts in by being added with `addThreadedUserBinary` — as `addUserBinary`,
|
||||
but the shared implementation builds it `single_threaded = false` — so atomics and
|
||||
(later) TLS are real. Threads and atomics are unsound in a `single_threaded` image,
|
||||
so a binary must opt in **before** it may call `runtime.Thread.spawn`. Everyone else
|
||||
A binary opts in with `.threaded = true` in its package's
|
||||
`build_support.userBinary` call — the shared recipe in build-support then builds it
|
||||
`single_threaded = false` — so atomics
|
||||
and (later) TLS are real. Threads and atomics are unsound in a `single_threaded` image,
|
||||
so a binary must opt in **before** it may call `Thread.spawn`. Everyone else
|
||||
stays single-threaded and lean.
|
||||
|
||||
## Interaction with the rest of the kernel
|
||||
@@ -252,31 +254,31 @@ stays single-threaded and lean.
|
||||
halts" property intact under lock contention — no busy-wait.
|
||||
- **Lifecycle** ([process-lifecycle.md](process-lifecycle.md)): the contract is that
|
||||
killing a process kills *all* its threads and only then drops the last address-space
|
||||
ref. **The kernel does not implement that fan-out yet**: `process_kill` reaps only
|
||||
the one task it resolves, and no death path loops over the tasks sharing an address
|
||||
space — the refcount keeps the space (and the sibling threads) alive and running.
|
||||
The gap is hit in practice: of the only threaded binaries (the `display` service and
|
||||
the `thread-test` harness), `display` is a boot service that handles no `.terminate`
|
||||
signal, so init's stop sequence escalates to `process_kill` on every orderly
|
||||
shutdown — benign only because poweroff follows. Whether to implement the fan-out or
|
||||
amend the contract is a decision still to be made.
|
||||
ref — and the kernel now implements exactly that
|
||||
([shared-fate-plan.md](shared-fate-plan.md)): every death path (`exit` from any
|
||||
thread, a fault, `process_kill` aimed at any member id) fans out through the whole
|
||||
group via a `dying` latch on the address space; the supervisor's one exit
|
||||
notification — badged with the leader — fires only when the last member is gone.
|
||||
A worker's voluntary `thread_exit` stays per-thread; the leader's is refused
|
||||
(`-EPERM`).
|
||||
- **Resilience** ([resilience.md](resilience.md)): by the same contract, a faulting
|
||||
thread kills its whole process (shared fate); the supervisor restarts the
|
||||
**process**, which respawns its threads from a known-good state — restart
|
||||
granularity stays the process. Today a CPU fault kills only the faulting task
|
||||
(`killCurrentProcess` tears down a single task), so sibling threads keep running —
|
||||
the same implementation gap as above.
|
||||
- **IPC — two consequences threads forced ([ipc.md](ipc.md)):**
|
||||
granularity stays the process. The leader's recorded exit reason carries the fault
|
||||
class even when a worker faulted, so restart policy is unchanged.
|
||||
- **IPC — two consequences threads forced ([ipc.md](../device-driver-development/ipc.md)):**
|
||||
- *Handles do not cross threads.* The handle table lives on the `Task`
|
||||
([scheduler.zig](../system/kernel/scheduler.zig)), so a handle number is meaningful
|
||||
([scheduler.zig](../../system/kernel/scheduler.zig)), so a handle number is meaningful
|
||||
only to the thread that created it — thread A's endpoint handle `3` is not thread B's.
|
||||
A thread that needs to reach an endpoint another thread owns looks it up
|
||||
(`ipc.lookup(service)`) to install its **own** handle to the same underlying endpoint.
|
||||
This is how the display's mouse-listener thread reaches the compositor loop's endpoint
|
||||
to poke it awake (docs/display.md).
|
||||
A thread that needs to reach an endpoint another thread owns opens the name
|
||||
(`channel.openEndpoint("display")`) to install its **own** handle to the same
|
||||
underlying endpoint — an ordinary client open, with no special mechanism for the
|
||||
fact that the provider happens to be this process. This is how the display's
|
||||
mouse-listener thread reaches the compositor loop's endpoint to poke it awake
|
||||
(docs/display.md).
|
||||
- *IPC syscalls that touch shared kernel state now serialize under the big kernel lock.*
|
||||
`create_ipc_endpoint`/`ipc_register`/`ipc_lookup` allocate from the kernel heap and
|
||||
mutate the global service registry, endpoint refcounts, and handle tables. Those paths
|
||||
`create_ipc_endpoint` allocates from the kernel heap and
|
||||
mutates endpoint refcounts and handle tables. Those paths
|
||||
were unlocked because a single-threaded process could not race itself; a multi-threaded
|
||||
one can, from two cores at once. They now take `sync.enter()` like `call`/`reply_wait`/
|
||||
`send` already did — the kernel heap has no lock of its own (heap.zig: "every kernel
|
||||
@@ -286,7 +288,7 @@ stays single-threaded and lean.
|
||||
|
||||
The ordered, `/loop`-runnable milestones live in
|
||||
**[threading-plan.md](threading-plan.md)** (shaped like
|
||||
[display-v2-plan.md](display-v2-plan.md)): every milestone lands on its own and ends in
|
||||
[display-v2-plan.md](../device-driver-development/display-v2-plan.md)): every milestone lands on its own and ends in
|
||||
a verifiable gate (`python3 test/qemu_test.py <case>`, asserting serial markers;
|
||||
`zig build test` for host unit tests). The stages below are the shape it expands.
|
||||
|
||||
@@ -309,15 +311,15 @@ a verifiable gate (`python3 test/qemu_test.py <case>`, asserting serial markers;
|
||||
the consumer blocked, e.g. via a low idle tick count).
|
||||
- **Stage 3 — polish.** Per-thread TLS / thread pointer and `threadlocal` (only if a
|
||||
consumer needs it), `RwLock`/`WaitGroup` as demanded, and this doc's cases wired
|
||||
into [test/qemu_test.py](../test/qemu_test.py).
|
||||
into [test/qemu_test.py](../../test/qemu_test.py).
|
||||
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](coding-standards.md): spell out non-acronym
|
||||
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym
|
||||
abbreviations, kebab-case file names, no `Co-Authored-By` trailers. New syscalls
|
||||
extend [abi.zig](../system/abi.zig) `SystemCall` + a `library/runtime` wrapper
|
||||
([syscall.md](syscall.md)). `runtime.Thread` is a first-class runtime module, the same
|
||||
way `runtime.process` ([process-lifecycle.md](process-lifecycle.md)) and `runtime.ipc`
|
||||
extend [abi.zig](../../system/abi.zig) `SystemCall` + a `library/kernel` wrapper
|
||||
([syscall.md](syscall.md)). `Thread` is a first-class runtime module, the same
|
||||
way `process` ([process-lifecycle.md](process-lifecycle.md)) and `ipc`
|
||||
are — user code never names a syscall.
|
||||
|
||||
## Non-goals
|
||||
@@ -333,19 +335,19 @@ are — user code never names a syscall.
|
||||
## The self-hosting endgame
|
||||
|
||||
When danos becomes a real Zig target and we (eventually) add a danos backend to std
|
||||
([zig-self-hosting.md](zig-self-hosting.md)), `std.Thread` can sit *on top of* these
|
||||
([zig-self-hosting.md](../zig-self-hosting.md)), `std.Thread` can sit *on top of* these
|
||||
same kernel primitives — the danos `std.Thread.Impl` would call the very
|
||||
`thread_spawn`/`futex_*` wrappers `runtime.Thread` already uses. Because
|
||||
`runtime.Thread` was built API-compatible from day one, that transition swaps the
|
||||
`thread_spawn`/`futex_*` wrappers `Thread` already uses. Because
|
||||
`Thread` was built API-compatible from day one, that transition swaps the
|
||||
implementation, not a single call site. Designing to the std shape now is what makes
|
||||
the later self-hosting lift cheap.
|
||||
|
||||
## Further reading
|
||||
|
||||
- [scheduling.md](scheduling.md), [smp.md](smp.md) — the task model these threads join.
|
||||
- [resilience.md](resilience.md), [vision.md](vision.md) — why isolation is the default
|
||||
- [resilience.md](resilience.md), [vision.md](../vision.md) — why isolation is the default
|
||||
and threads are the exception.
|
||||
- [syscall.md](syscall.md), [ipc.md](ipc.md) — the private ABI and the messaging model
|
||||
- [syscall.md](syscall.md), [ipc.md](../device-driver-development/ipc.md) — the private ABI and the messaging model
|
||||
threads sit beside.
|
||||
- [halting.md](halting.md) — the idle/halt property futex-backed blocking preserves.
|
||||
- [zig-self-hosting.md](zig-self-hosting.md) — the target this bends toward.
|
||||
- [zig-self-hosting.md](../zig-self-hosting.md) — the target this bends toward.
|
||||
@@ -7,7 +7,7 @@ Two different needs hide under the word "timer", and danos keeps them apart:
|
||||
|
||||
Both are answered by the **kernel**, because the kernel already owns a timer: it has
|
||||
to, to preempt tasks. The LAPIC heartbeat and the calibrated TSC that back all of this
|
||||
are built in [device-interrupts.md](device-interrupts.md); the scheduler's blocking and
|
||||
are built in [device-interrupts.md](../device-driver-development/device-interrupts.md); the scheduler's blocking and
|
||||
wait queues are in [scheduling.md](scheduling.md). This page is about the surface a
|
||||
ring-3 program actually uses, and one deliberate absence: **there is no user-space time
|
||||
service.**
|
||||
@@ -32,11 +32,11 @@ danos checks both — the invariant-TSC CPUID bit (`0x80000007` EDX[8], set on I
|
||||
AMD), and a cross-core "warp" check as the cores come up — and falls back to the HPET
|
||||
counter when either fails. So `now()` stays accurate on a real Intel box, a real AMD box,
|
||||
and inside a VM alike; only the source behind it differs. The mechanism is in
|
||||
[device-interrupts.md](device-interrupts.md).
|
||||
[device-interrupts.md](../device-driver-development/device-interrupts.md).
|
||||
|
||||
So the timer hardware lives in the kernel, and there is **no `hpet` driver and no time
|
||||
server** to consume. (An earlier HPET driver existed only to *demonstrate* the driver
|
||||
model; that role now lives in [drivers.md](drivers.md), as documentation.) The one place
|
||||
model; that role now lives in [drivers.md](../device-driver-development/drivers.md), as documentation.) The one place
|
||||
a user-space time service *is* justified — **wall-clock / calendar time** — is discussed
|
||||
at the end; it is deliberately not built yet.
|
||||
|
||||
@@ -55,20 +55,20 @@ Time and waiting are three entries in the small syscall table ([syscall.md](sysc
|
||||
service can keep answering messages on the same endpoint while a deadline is pending.
|
||||
This is the timed wait that stop-sequence escalation, hello deadlines, and restart
|
||||
backoff are built from ([process-lifecycle.md](process-lifecycle.md),
|
||||
[device-manager.md](device-manager.md)).
|
||||
[device-manager.md](../device-driver-development/device-manager.md)).
|
||||
|
||||
The kernel's own scheduling timer (the LAPIC, vector 32) is never exposed to user space;
|
||||
programs read the TSC through `clock` and get timed wakeups through `sleep`/`timer_bind`,
|
||||
both riding the scheduler tick.
|
||||
|
||||
## `runtime.time` — the generic interface
|
||||
## `time` — the generic interface
|
||||
|
||||
Applications don't call the syscalls directly; they use `runtime.time`
|
||||
(`library/runtime/time.zig`), a thin `Instant`/`Duration` layer over them — an ergonomic
|
||||
Applications don't call the syscalls directly; they use `time`
|
||||
(`library/kernel/time.zig`), a thin `Instant`/`Duration` layer over them — an ergonomic
|
||||
front door, not new mechanism.
|
||||
|
||||
```zig
|
||||
const time = @import("runtime").time;
|
||||
const time = @import("time");
|
||||
|
||||
const start = time.now(); // Instant — monotonic
|
||||
doWork();
|
||||
@@ -91,8 +91,9 @@ _ = time.after(endpoint, time.Duration.fromMillis(200));
|
||||
- `sleep(d)` wraps `sleep`; `spin(d)` busy-polls `now()` for the sub-millisecond delays
|
||||
the millisecond tick can't express; `after(endpoint, d)` wraps `timer_bind`.
|
||||
|
||||
The raw wrappers (`system.clock`, `system.sleep`, `system.timerOnce`) stay in
|
||||
`library/runtime/system.zig`; `runtime.time` is the layer meant for everyday use.
|
||||
The raw wrappers (`clock`, `sleepMillis`, `timerOnce`) and the ergonomic
|
||||
`Instant`/`Duration` layer both live in the `time` module
|
||||
(`library/kernel/time.zig`); the latter is what everyday code uses.
|
||||
|
||||
## Wall-clock time (not built)
|
||||
|
||||
@@ -106,10 +107,10 @@ owns covers every current use.
|
||||
|
||||
## Verifying it
|
||||
|
||||
`runtime.time`'s `Instant`/`Duration` arithmetic has unit tests that run on the host:
|
||||
`time`'s `Instant`/`Duration` arithmetic has unit tests that run on the host:
|
||||
|
||||
```
|
||||
$ zig build test # includes library/runtime/time.zig
|
||||
$ zig build test # includes library/kernel/time.zig
|
||||
```
|
||||
|
||||
End to end, the proof the clock is real is that it *advances*: read `now()`, `sleep` a
|
||||
@@ -1,7 +1,7 @@
|
||||
# The vDSO — the public system-call boundary
|
||||
|
||||
> **Status:** design note, not built. The runtime today issues raw `syscall`
|
||||
> instructions from `library/runtime/system-call.zig` using the numbers in
|
||||
> instructions from `library/kernel/system-call.zig` using the numbers in
|
||||
> `system/abi.zig`. This note designs the layer that replaces that arrangement:
|
||||
> a **kernel-supplied, C-ABI entry library** mapped into every process — the
|
||||
> only supported way into the kernel — so the raw numbers can stay private,
|
||||
@@ -25,8 +25,8 @@ ourselves:
|
||||
this mistake: it issued XNU syscalls directly instead of going through
|
||||
libSystem, and macOS updates repeatedly broke every Go binary until Go
|
||||
switched to the library like everyone else.
|
||||
2. **Not everything is Zig.** A Rust or C program can't import the `runtime`
|
||||
module. The public boundary has to be expressible in the one calling
|
||||
2. **Not everything is Zig.** A Rust or C program can't import the danos Zig
|
||||
modules. The public boundary has to be expressible in the one calling
|
||||
convention every language speaks: the C ABI.
|
||||
3. **Randomised syscall numbers** — a hardening option we want open — only
|
||||
work if no user binary anywhere knows a number at build time. The binding
|
||||
@@ -49,10 +49,10 @@ The public danos ABI then has exactly two layers, neither of which is
|
||||
|
||||
| Layer | Contract | Spoken by |
|
||||
|-------|----------|-----------|
|
||||
| **vDSO** | C-ABI functions, this note | every language's thin shim (`runtime.system` for Zig, a `-sys` crate for Rust, a header for C) |
|
||||
| **IPC wire protocols** | byte layouts over `ipc_call` ([vfs-protocol.md](vfs-protocol.md) is the first one documented) | any client that can lay out bytes |
|
||||
| **vDSO** | C-ABI functions, this note | every language's thin shim (the `system-call` module for Zig, a `-sys` crate for Rust, a header for C) |
|
||||
| **IPC wire protocols** | byte layouts over `ipc_call` ([vfs-protocol.md](../file-system-development/vfs-protocol.md) is the first one documented) | any client that can lay out bytes |
|
||||
|
||||
Everything above those — the heap, `runtime.fs`, the service harness — is
|
||||
Everything above those — the heap, `file_system`, the service harness — is
|
||||
per-language convenience, compiled into each binary from source, exactly as
|
||||
today. Nothing about the Zig runtime's shape changes; it just stops being the
|
||||
*only* door.
|
||||
@@ -107,7 +107,7 @@ convenience, not a requirement.)
|
||||
|
||||
The kernel already builds a System V entry block — argc, argv, envp
|
||||
terminator, **auxiliary vector** — on every new process's stack
|
||||
(`buildEntryStack`, read by `runtime.start`). The vDSO base rides in a new
|
||||
(`buildEntryStack`, read by the `start` module). The vDSO base rides in a new
|
||||
auxv entry, exactly Linux's `AT_SYSINFO_EHDR` move. No new syscall, no magic
|
||||
address, and a language shim finds it the same portable way on every
|
||||
architecture.
|
||||
@@ -137,15 +137,15 @@ Grouped as `abi.zig` groups them:
|
||||
| process | `danos_exit`, `danos_yield`, `danos_sleep`, `danos_spawn`, `danos_process_enumerate`, `danos_process_kill`, `danos_process_exit_reason`, `danos_process_subscribe`, `danos_process_signal`, `danos_signal_bind` |
|
||||
| threads | `danos_thread_spawn`, `danos_thread_exit`, `danos_current_core`, `danos_futex_wait`, `danos_futex_wake`, `danos_thread_self`, `danos_thread_join`, `danos_set_thread_pointer` |
|
||||
| memory | `danos_mmap`, `danos_munmap`, `danos_dma_alloc`, `danos_dma_free`, `danos_shared_memory_create`, `danos_shared_memory_map`, `danos_shared_memory_physical` |
|
||||
| ipc | `danos_endpoint_create`, `danos_ipc_register`, `danos_ipc_lookup`, `danos_ipc_call`, `danos_ipc_reply_wait`, `danos_ipc_send` |
|
||||
| ipc | `danos_endpoint_create`, `danos_ipc_call`, `danos_ipc_reply_wait`, `danos_ipc_send` (naming is not a syscall: a provider binds its contract at the registry and a client resolves `/protocol/<name>` — see [protocol-namespace.md](protocol-namespace.md)) |
|
||||
| devices | `danos_device_enumerate`, `danos_device_claim`, `danos_device_register`, `danos_mmio_map`, `danos_irq_bind`, `danos_irq_ack`, `danos_msi_bind`, `danos_io_read`, `danos_io_write` |
|
||||
| time | `danos_clock`, `danos_wall_clock`, `danos_timer_bind` |
|
||||
| diagnostics | `danos_debug_write` (leveled, kernel-stamped records), `danos_klog_read`, `danos_klog_status` |
|
||||
| filesystem naming | `danos_fs_resolve`, `danos_fs_node`, `danos_fs_mount`, `danos_fs_unmount` (naming only — file DATA still crosses the vfs-protocol IPC, see below) |
|
||||
|
||||
The constants that ride alongside the calls — mmap protection bits, DMA
|
||||
flags, notification badge bits, `ExitReason`, `Signal`, well-known service
|
||||
ids, `page_size`, the IPC message maximum — move to the public header too:
|
||||
flags, notification badge bits, `ExitReason`, `Signal`, `page_size`, the IPC
|
||||
message maximum — move to the public header too:
|
||||
they are wire values a Rust program needs verbatim. What stays private in
|
||||
`abi.zig` is exactly the thing the vDSO exists to hide: the `SystemCall`
|
||||
numbers and the trap convention.
|
||||
@@ -183,11 +183,11 @@ second — but the design should never be sold as more than that.
|
||||
Phased so every step ships alone (the M-milestone discipline):
|
||||
|
||||
1. **The blob + the table.** Build the vDSO, map it at spawn, deliver the
|
||||
base via auxv. `runtime.system-call.zig` binds through the table when the
|
||||
base via auxv. `library/kernel/system-call.zig` binds through the table when the
|
||||
auxv entry is present, falls back to raw `syscall` when absent — the whole
|
||||
tree keeps booting during the transition.
|
||||
2. **Cut the runtime over.** Delete the raw stubs; `runtime` no longer
|
||||
imports the `SystemCall` numbers at all (`abi.zig`'s enum becomes
|
||||
2. **Cut the system library over.** Delete the raw stubs; the `system-call`
|
||||
module no longer imports the `SystemCall` numbers at all (`abi.zig`'s enum becomes
|
||||
kernel-internal). The QEMU suite passing proves the table carries the
|
||||
whole system.
|
||||
3. **Enforce + randomise.** Add the `rip`-range check, then per-boot number
|
||||
@@ -0,0 +1,192 @@
|
||||
# Python on danos: the milestone plan
|
||||
|
||||
The execution plan for [python-on-danos.md](python-on-danos.md). That note holds
|
||||
the *why* and the design decisions; this one slices the work into milestones with
|
||||
concrete deliverables, tests, and exit criteria. Milestones are numbered **P0–P5**
|
||||
(track-local — the global M-series stays with the driver/lifecycle tracks).
|
||||
|
||||
Dependencies at a glance:
|
||||
|
||||
```
|
||||
P0 toolchain + mini-libc ──┐
|
||||
P1 streams + console + seam ─┴─→ P2 CPython minimal ─→ P3 terminal + REPL
|
||||
│ │
|
||||
└─→ P4 danos module │
|
||||
+ Python service│
|
||||
P5 process control + shell ←─────────────────────────────────┘
|
||||
```
|
||||
|
||||
P0 and P1 are independent of each other and can proceed in parallel. P1 is shared
|
||||
work — it is also Zig self-hosting Phase 1 and the first three slices of
|
||||
[character-devices-and-tty.md](character-devices-and-tty.md).
|
||||
|
||||
## P0 — Toolchain + the C library compatibility layer
|
||||
|
||||
**Goal:** a C hello-world, cross-compiled on the host with `zig cc`, runs on danos.
|
||||
|
||||
Design and slicing live in
|
||||
[c-library-compatibility.md](c-library-compatibility.md): the **libdanos-c**
|
||||
sysroot (hand-written danos-native headers + `libc.a`) as a `library/c/` build
|
||||
package — pure computation (string, libm, `strtod`, the printf/scanf engines)
|
||||
lifted from a vendored, pinned musl subtree; the OS plumbing written in Zig over
|
||||
the `runtime` surface (re-targeting `runtime.os` when the Zig track authors it);
|
||||
`malloc` over danos `mmap`; a `crt0` bridging the danos entry shim to C `main`.
|
||||
Driven by `zig cc -target x86_64-freestanding-none -isystem` (the triple becomes
|
||||
`x86_64-danos` if the Zig fork lands first; nothing else changes).
|
||||
|
||||
Its five slices (sysroot-skeleton, fd-plumbing, malloc, stdio,
|
||||
mathematics-and-time) carry their own tests — host-side oracle suites for the
|
||||
computation layer, QEMU cases (`c-hello`, `c-file-io`, `c-stdio`, `c-time`) for
|
||||
the plumbing.
|
||||
|
||||
**Exit:** `c-hello` and `c-file-io` green in the QEMU suite; host computation
|
||||
tests green.
|
||||
|
||||
## P1 — Stream nodes, console, and the seam pieces
|
||||
|
||||
**Goal:** the shared Phase-1 surface exists: byte-stream stdio, cwd, environment,
|
||||
entropy. Design and slicing live in
|
||||
[character-devices-and-tty.md](character-devices-and-tty.md); this milestone is
|
||||
its slices 1–3 plus three small seam pieces:
|
||||
|
||||
- **cwd/chdir** — per-process current directory used by path resolution (the
|
||||
kernel already anchors a VFS root per `fs_resolve`; the cwd is the same idea,
|
||||
process-scoped, with `getcwd`/`chdir` exposed through `runtime` and the libc).
|
||||
- **Environment** — spawn carries an environment block; the SysV entry stack's
|
||||
`envp` slot ([sysv.md](os-development/sysv.md)) stops being empty; `getenv`
|
||||
reads it. An empty block stays valid.
|
||||
- **Entropy** — a kernel `entropy` syscall (RDSEED/RDRAND with a jitter fallback,
|
||||
mirroring the TSC-reliability posture of not trusting one CPU feature blindly);
|
||||
the libc exposes `getentropy`.
|
||||
|
||||
- **Tests.** QEMU: the character-device tests from the tty note (offsetless
|
||||
read/write, blocking read, cooked/raw control round-trip), plus `cwd-basics`
|
||||
(chdir + relative open), `env-roundtrip` (spawn with env, child reads it),
|
||||
`entropy-sane` (nonzero, changing, correct length).
|
||||
|
||||
**Exit:** a C program reads a cooked line from fd 0 and echoes it to fd 1 —
|
||||
injected key events in, bytes read back through the console's in-memory sink,
|
||||
all under QEMU with no hardware involved — and `getcwd`/`getenv`/`getentropy`
|
||||
return real answers.
|
||||
|
||||
## P2 — CPython, minimal configuration
|
||||
|
||||
**Goal:** `python -c 'print(2**100)'` runs on danos under QEMU.
|
||||
|
||||
- Pin **CPython 3.13.x**; vendor as `third-party/cpython/` or fetch via the build
|
||||
(decide with the build-packages conventions).
|
||||
- Host build-Python of the same version (`--with-build-python`).
|
||||
- `config.site` cache for the cross answers; `config.sub` patch so
|
||||
`x86_64-unknown-danos` parses; a small `configure`/`pyconfig` patch set kept as
|
||||
rebasable diffs, WASI-style.
|
||||
- `--disable-shared`; static `Modules/Setup`: `posix errno _io _codecs _weakref
|
||||
time math _stat _collections itertools _functools _locale _sre` plus what the
|
||||
interpreter core insists on; threadless build (WASI precedent).
|
||||
- `Lib/` on the FAT image under the hierarchy (e.g. `/system/python/lib`);
|
||||
`PYTHONHOME` set accordingly; `.pyc` written with **checked-hash
|
||||
invalidation** (FAT's 2-second mtime granularity makes mtime-based validation
|
||||
lie during fast edit-run cycles).
|
||||
- `PYTHONHASHSEED` pinned only if P1's entropy slipped — otherwise real
|
||||
hash randomization from day one.
|
||||
- **Tests.** QEMU: `python-expr` (the exit criterion), `python-file` (run a
|
||||
script from FAT, write a file, read it back), then a curated slice of CPython's
|
||||
own suite (`test_int`, `test_float`, `test_io`, `test_dict`) as a
|
||||
longer-running target — the suite is the porting harness.
|
||||
|
||||
**Exit:** the four QEMU cases green; the CPython test slice green or with a
|
||||
short, documented skip list.
|
||||
|
||||
## P3 — Terminal + REPL: the first real application
|
||||
|
||||
**Goal:** an interactive `python` REPL in a graphical danos terminal — the
|
||||
milestone demo for the OS.
|
||||
|
||||
- Depends on the display track's font rendering (its stated next step) — until
|
||||
that lands, the REPL is exercised end-to-end through the pseudo-device
|
||||
harness from P1+P2 (scripted input in, output read back), so P2's exit is
|
||||
never blocked on graphics; the graphical terminal is the *interactive* debut.
|
||||
- The terminal application: draws with the UI toolkit / display client, consumes
|
||||
keyboard `InputEvent`s, and — per the tty note's load-bearing decision —
|
||||
**serves the VFS stream protocol itself** to its children, reusing the console's
|
||||
line-discipline library. Spawns `python` with its endpoints as fd 0/1/2.
|
||||
- Raw mode + the control set give the REPL line editing; window-size control
|
||||
gives it wrapping.
|
||||
- **Tests.** QEMU: scripted terminal session (inject key events, assert rendered
|
||||
or captured output). Real-hardware smoke on the Intel box joins the existing
|
||||
checklist.
|
||||
|
||||
**Exit:** typing `2+2` into the terminal on the QEMU GPU target prints `4`.
|
||||
|
||||
## P4 — The `danos` extension module + a Python service
|
||||
|
||||
**Goal:** Python can speak danos: IPC, capabilities, spawn.
|
||||
|
||||
- The `danos` module, **written in Zig against `Python.h`**, statically linked
|
||||
via `Modules/Setup`: endpoints (create/send/receive), capability passing,
|
||||
spawn + exit-notification, and the service bootstrap (announce, supervision
|
||||
handshake) — the same surface Zig services use, re-exposed.
|
||||
- UI-toolkit bindings as a second module once the toolkit's API settles.
|
||||
- Prototype **one real service in Python** — policy-shaped, not data-plane
|
||||
(candidates: hot-plug policy, a settings service) — speaking an existing wire
|
||||
protocol, supervised by the device manager like any service.
|
||||
- **Tests.** QEMU: `python-ipc-echo` (Python service echoes over an endpoint, a
|
||||
Zig client asserts), plus the prototype service's own protocol test.
|
||||
|
||||
**Exit:** a Python process runs as a supervised danos service exchanging IPC
|
||||
with Zig peers.
|
||||
|
||||
## P5 — Process control, then the shell
|
||||
|
||||
**Goal:** danos can spawn arbitrary programs with arguments and pipes; a small
|
||||
Python shell uses it.
|
||||
|
||||
The kernel/VFS cluster a shell forces (any shell, any language):
|
||||
|
||||
- **exec-of-path** — spawn an arbitrary VFS path, not a named ramdisk binary;
|
||||
- **argv/envp** — carried through spawn onto the child's entry stack (env from
|
||||
P1, argv new);
|
||||
- **numeric exit status** — extend the exit record beyond the categorical
|
||||
`ExitReason` (the gotcha the Zig roadmap flagged: `WEXITSTATUS` must be real);
|
||||
- **fd inheritance + pipes** — a kernel or service pipe (a character device by
|
||||
the tty note's definition) and spawn-time fd mapping.
|
||||
|
||||
Then, in order: `subprocess` enabled in CPython (maps onto spawn + the
|
||||
exit-notification endpoint — no fork, Windows-style); a **small Python shell** (a
|
||||
few hundred lines over `subprocess` + the console: prompt, argv parsing, pipes,
|
||||
cwd) as the forcing function that reveals what job control actually needs.
|
||||
|
||||
**Explicitly deferred past P5:** the pthread subset over `thread_spawn`/futex,
|
||||
signals-in-libc via M17, termios job control (Ctrl-C to foreground child), and
|
||||
**xonsh** — which wants all three and is the arc's endpoint, not a milestone.
|
||||
|
||||
**Tests.** QEMU: `spawn-argv-exit` (child echoes argv, exits 42, parent sees
|
||||
42), `pipe-through` (parent → child → parent), `python-subprocess`, and a
|
||||
scripted shell session.
|
||||
|
||||
**Exit:** the Python shell runs `program | program` typed at the terminal and
|
||||
reports the exit status.
|
||||
|
||||
## Post-P5 outlook
|
||||
|
||||
Two tracks continue past this plan, each with its own design doc rather than a
|
||||
P-number here:
|
||||
|
||||
- **Dynamic libraries** ([dynamic-libraries.md](dynamic-libraries.md), D1–D4) —
|
||||
an application-layer facility (the OS stays static and lean): `dlopen` in the
|
||||
libc, then libffi + `ctypes` + loadable extension modules, then shared
|
||||
read-only mappings so N Python services hold one physical `libpython`.
|
||||
- **The full C compatibility layer**
|
||||
([c-library-compatibility.md](c-library-compatibility.md), stages 2–3) — the
|
||||
standing rule that every system capability ships with its C spelling, draining
|
||||
the absence table toward "portable C builds on danos"; `fork` is the one
|
||||
permanent exception.
|
||||
|
||||
## Related
|
||||
|
||||
- [python-on-danos.md](python-on-danos.md) — the design note this executes.
|
||||
- [c-library-compatibility.md](c-library-compatibility.md) — P0's design.
|
||||
- [character-devices-and-tty.md](character-devices-and-tty.md) — P1's design.
|
||||
- [zig-self-hosting.md](zig-self-hosting.md) — shares P1; its fork makes P0's
|
||||
triple prettier but gates nothing here.
|
||||
- [os-development/process-management.md](os-development/process-management.md) —
|
||||
the spawn/exit surface P5 extends.
|
||||
@@ -0,0 +1,261 @@
|
||||
# Python on danos: the CPython milestone
|
||||
|
||||
A design note (not built yet) on bringing **CPython** to danos, compiled with the Zig
|
||||
toolchain (`zig cc`). Like [zig-self-hosting.md](zig-self-hosting.md), it is
|
||||
forward-looking: it sets a direction and the decisions that follow from it.
|
||||
|
||||
## Why Python, and why now
|
||||
|
||||
The Zig self-hosting road is gated on a compiler fork and a long std-library seam.
|
||||
Python is the **stop-gap that removes the wait**: a working CPython gives danos a way
|
||||
to write programs — services, tools, application prototypes — *without* the Zig
|
||||
compiler being self-hosted, and it brings the pure-Python package ecosystem along as
|
||||
a bonus. The intended division of labour:
|
||||
|
||||
- **Zig** — the kernel, drivers, and anything on a data plane (interrupt paths,
|
||||
DMA rings, block I/O). Unchanged.
|
||||
- **Python** — the control plane and the prototyping surface: services that are
|
||||
event loops over IPC, policy logic that changes often, application experiments,
|
||||
and eventually the shell.
|
||||
|
||||
Python is also the scripting language for the terminal-and-shell arc: the first
|
||||
real danos application is planned as a terminal, a terminal wants a shell, a shell
|
||||
wants a scripting language — and [xonsh](https://xon.sh) (a shell written in
|
||||
Python) marks where that road can end.
|
||||
|
||||
### Non-goals
|
||||
|
||||
- **No drivers in Python.** Interrupt handling, ring management, and DMA stay in
|
||||
Zig. Python may *supervise and configure* drivers; it does not sit in their hot
|
||||
paths (interpreter overhead and garbage-collection pauses in an interrupt path
|
||||
are disqualifying).
|
||||
- **No dynamic loading during bring-up, no `pip`.** The whole arc here ships
|
||||
statically linked. Dynamic libraries are a real *later* milestone
|
||||
([dynamic-libraries.md](dynamic-libraries.md)) — an application-layer
|
||||
facility that unlocks `ctypes` and loadable extension modules; the operating
|
||||
system itself stays static and lean regardless (the size doctrine below).
|
||||
`pip` stays out either way until a networking track exists.
|
||||
- **No fork.** `os.fork` will not exist. This costs almost nothing (see "The
|
||||
spawn model fits").
|
||||
|
||||
## The realization that shapes everything: the compiler is not the obstacle
|
||||
|
||||
`zig cc` is a full Clang-based C cross-compiler, and CPython is portable C with
|
||||
official precedent for stranger targets than danos — the WASI port is upstream
|
||||
tier-2, and it runs **without fork, without dynamic loading, and without working
|
||||
threads**. Every "CPython can't possibly run there" objection has already been
|
||||
answered upstream by a target *more* constrained than danos.
|
||||
|
||||
What CPython actually needs is a **C environment**: headers and a `libc.a`. danos
|
||||
has neither — and that is the whole project. In the language of the Zig roadmap's
|
||||
three doors, this is the **door-2-shaped work** (the deferred "musl door"), not the
|
||||
`std.os.danos` seam: CPython never touches Zig's std.
|
||||
|
||||
### The same surface, a third time
|
||||
|
||||
The Zig roadmap observed that door 1 (`std.os.danos`) and door 2 (a libc) implement
|
||||
the *same* ~30 danos-facing operations at different layers. CPython consumes that
|
||||
identical surface through C spellings. So nothing here is throwaway: the
|
||||
danos-native operations backing `runtime.os` are the same ones the libc bottoms out
|
||||
in, and the gaps this track must close (stdio byte streams, cwd, environment,
|
||||
entropy) are **exactly the Phase-1 gaps the Zig roadmap already lists**. The two
|
||||
tracks share a road until Python forks off at "build the libc."
|
||||
|
||||
## Where danos stands: coverage vs. the gaps
|
||||
|
||||
Judged against the minimal CPython configuration (static, WASI-like):
|
||||
|
||||
| CPython need | danos today | Gap |
|
||||
|--------------|-------------|-----|
|
||||
| open/read/write/close/lseek, readdir | VFS + FAT via `runtime.fs` | none — wrap in C |
|
||||
| mkdir / unlink / rename / truncate | done (self-hosting Phase 2) | none |
|
||||
| stat with mtime | done (`wall_clock` + FAT mtime) | none |
|
||||
| mmap/munmap (object allocator) | native syscalls | none |
|
||||
| monotonic + wall clock | `clock` + `wall_clock` syscalls | none |
|
||||
| a place for `Lib/` | FAT boot image | none — better than WASI has it |
|
||||
| fork / exec | not needed (subprocess disabled at first) | — |
|
||||
| dynamic loading | not needed (static extension modules) | — |
|
||||
| getcwd / chdir | — | **missing** (shared with Zig Phase 1) |
|
||||
| environment variables | `Init` has no env | **missing** (can start empty) |
|
||||
| entropy | — | **missing** (hash seed; `PYTHONHASHSEED` pins it meanwhile) |
|
||||
| byte-stream stdin/stdout (fd 0/1/2) | `debug_write` out; structured `InputEvent` in | **missing** (shared with Zig Phase 1; the REPL needs it) |
|
||||
| signals | — | stubs suffice (WASI precedent); M17 signals-over-IPC maps on later |
|
||||
| threads | native `thread_spawn`/futex | build threadless first; a pthread subset later (xonsh needs it) |
|
||||
|
||||
The clustering repeats the Zig roadmap's: **files, memory, and time are done; the
|
||||
work is the C packaging plus the small seam pieces** (tty bytes, cwd, env, entropy).
|
||||
|
||||
## The libc decision: hand-rolled in Zig, computation lifted from musl
|
||||
|
||||
Two viable shapes were considered:
|
||||
|
||||
| Option | What it is | Verdict |
|
||||
|--------|-----------|---------|
|
||||
| **Mini-libc in Zig** | C-ABI-exporting Zig library over `runtime.os`/`runtime.fs`, shipped as headers + `libc.a`. | **Take this.** Reuses the danos-native surface directly; no Linux assumptions to fight. |
|
||||
| **Port musl** | Full musl with a danos syscall backend. | Defer, again. musl assumes Linux syscall semantics in places; heavier than the need. |
|
||||
|
||||
The trick that makes the mini-libc tractable: musl's `string/`, `math/` (libm —
|
||||
CPython needs essentially all of it), and number-conversion layers are **pure
|
||||
computation with no syscalls**. Lift those wholesale (MIT-licensed, designed to
|
||||
compile standalone) and hand-write only:
|
||||
|
||||
- the OS-facing bottom: fds, `mmap`, clocks, `exit`, `getcwd` — thin C-ABI wrappers
|
||||
over `runtime.os`;
|
||||
- a `FILE*` stdio layer (buffered, over the fd layer);
|
||||
- `malloc` over danos `mmap` (a simple allocator is fine; CPython does its own
|
||||
small-object arena management above it);
|
||||
- the headers (`stdio.h`, `stdlib.h`, `string.h`, `math.h`, `errno.h`, …).
|
||||
|
||||
Estimate: **100–150 functions**, of which the hard 40% (libm, string, printf/strtod
|
||||
cores) are lifted, not written. Correctness hot spots are `strtod`/`dtoa` (Python's
|
||||
float repr round-trips through them) — another reason to lift musl's, not improvise.
|
||||
|
||||
## C interop: static extension modules, not ctypes
|
||||
|
||||
"Python can interface with C libraries" is true on danos with one important
|
||||
correction: **`ctypes` does not work at first** — it is built on `dlopen` + libffi,
|
||||
both of which arrive only with the [dynamic-libraries](dynamic-libraries.md)
|
||||
milestone (D2). Until then the interop story is the other, older one:
|
||||
|
||||
- **Extension modules statically linked into the interpreter** via CPython's
|
||||
`Modules/Setup` mechanism (the standard route for embedded/static builds).
|
||||
- **Zig speaks C ABI natively**, so danos extension modules are written in Zig
|
||||
against `Python.h` — no C required. Two modules are planned from the start:
|
||||
- **`danos`** — the system module: endpoints, send/receive, capability passing,
|
||||
spawn, exit notification. This is what makes a Python *service* possible: an
|
||||
event loop over IPC, speaking the same wire protocols as Zig services.
|
||||
- **UI toolkit bindings** — the in-progress danos UI toolkit exposed to Python,
|
||||
so application prototypes drive real windows.
|
||||
|
||||
The package story follows: **pure-Python packages work** (unpack into
|
||||
`Lib/site-packages` on the FAT image); packages with C extensions must be
|
||||
cross-compiled and baked into the interpreter — a curated set chosen per image,
|
||||
not `pip install`. That is the honest shape of the stop-gap.
|
||||
|
||||
## The roadmap
|
||||
|
||||
### Phase 0 — Toolchain + libc bring-up
|
||||
|
||||
`zig cc -target x86_64-freestanding-none` plus `-isystem` the danos headers and the
|
||||
mini-libc archive. No compiler fork required — this track deliberately avoids the
|
||||
Zig roadmap's Phase-0 gate (if the fork lands first, the triple becomes a clean
|
||||
`x86_64-danos`; nothing else changes). Exit criterion: a **hello-world C program**
|
||||
compiles on the host and runs on danos, printing via the libc's `write`.
|
||||
|
||||
### Phase 1 — The shared seam pieces
|
||||
|
||||
The same list as Zig self-hosting Phase 1, closed once for both tracks:
|
||||
|
||||
- fd 0/1/2 as console **byte** streams (output exists as `debug_write`; input is a
|
||||
new small thing — cooked line input first, raw mode when the REPL wants editing);
|
||||
- `getcwd`/`chdir`;
|
||||
- environment variables (an empty block is a valid start);
|
||||
- an entropy syscall or service (until then, builds pin `PYTHONHASHSEED`).
|
||||
|
||||
### Phase 2 — Cross-compile CPython, minimal configuration
|
||||
|
||||
Pin one CPython release (3.13 — strongest WASI-era cross-compile support). The
|
||||
mechanics are well-trodden upstream since 3.11:
|
||||
|
||||
- a same-version **build-Python on the host** (`--with-build-python`);
|
||||
- a `config.site` cache answering what configure cannot probe cross
|
||||
(`ac_cv_file__dev_ptmx=no` and friends);
|
||||
- a `config.sub` patch so `x86_64-unknown-danos` parses;
|
||||
- `--disable-shared`, static `Modules/Setup` with a minimal module set
|
||||
(`posix`, `errno`, `_io`, `_codecs`, `time`, `math`, …);
|
||||
- `Lib/` shipped on the FAT image; `PYTHONHOME` pointed at it.
|
||||
|
||||
Exit criterion: `python -c 'print(2**100)'` runs on danos under QEMU.
|
||||
|
||||
### Phase 3 — Terminal + REPL: the first real application
|
||||
|
||||
Depends on the display track's font rendering (already its stated next step) and
|
||||
Phase 1's tty. A terminal emulator drawing a `python` REPL is the milestone demo:
|
||||
interactive, self-evidently real, and it needs **zero** process-control machinery.
|
||||
|
||||
### Phase 4 — The `danos` module and Python services
|
||||
|
||||
Write the `danos` extension module and the UI-toolkit bindings; prototype one real
|
||||
service in Python (a policy-shaped one — e.g. hot-plug policy or a settings
|
||||
service) speaking the existing IPC protocols. This is the payoff phase for
|
||||
"prototyping a service or application."
|
||||
|
||||
### Phase 5 — Process control, then the shell
|
||||
|
||||
The shell — any shell, in any language — forces the surface danos has deferred so
|
||||
far: **exec-of-path, argv/envp passing, numeric exit status (`WEXITSTATUS`, not the
|
||||
categorical `ExitReason`), fd inheritance, and pipes.** That is a kernel/VFS
|
||||
milestone cluster of its own. Then, in order:
|
||||
|
||||
1. `subprocess` enabled in CPython (maps onto danos spawn — see below);
|
||||
2. a **small Python shell** (a few hundred lines over `subprocess` + line input, no
|
||||
job control) — the forcing function that reveals which process-control pieces
|
||||
actually matter;
|
||||
3. **explicitly deferred:** a pthread subset over `thread_spawn`/futex
|
||||
(create/join/mutex/condition/thread-locals), signals via M17 signals-over-IPC,
|
||||
termios job control — and then **xonsh**, which wants all three.
|
||||
|
||||
### The spawn model fits
|
||||
|
||||
One genuinely good alignment: **CPython does not need fork.** `subprocess` maps
|
||||
cleanly onto a posix_spawn-style model — exactly what danos has — and the existing
|
||||
exit-notification-via-endpoint is a *better* fit for `Popen.wait` than Unix's
|
||||
`wait` semantics. `os.fork` simply won't exist, as on Windows, and almost nothing
|
||||
in practice cares.
|
||||
|
||||
## Risks and gotchas
|
||||
|
||||
- **Binary size — and the size doctrine that makes it acceptable.** danos's
|
||||
leanness mandate applies to the **operating system**: the kernel and the system
|
||||
services stay small (the kernel is measured in kilobytes, not megabytes), and
|
||||
nothing in this track changes that — Python never enters the OS layer. An
|
||||
**application** budget is different: a statically-linked CPython with its
|
||||
module set will be tens of megabytes in ReleaseSafe (the measured ~2×
|
||||
safety-check factor compounds it), and that is *allowed* — applications live
|
||||
on the FAT image, not in the kernel's world. It still shapes the image, and it
|
||||
means every Python service shares one interpreter binary + per-service
|
||||
scripts, so the spawn model needs **argv** before "run this .py" works at all.
|
||||
- **FAT mtime granularity is 2 seconds.** CPython's `.pyc` cache validation is
|
||||
mtime-based by default; a rapid edit-run cycle can see stale bytecode. Use
|
||||
hash-based `.pyc` invalidation (PEP 552, `--invalidation-mode checked-hash` at
|
||||
freeze time) or accept the quirk during bring-up.
|
||||
- **FAT name lookups are case-insensitive.** Long file names preserve case but
|
||||
match insensitively — the same world Python inhabits on Windows/macOS, so
|
||||
importlib copes, but two modules differing only by case cannot coexist on the
|
||||
image.
|
||||
- **`strtod`/float repr correctness.** Python's float round-tripping is exacting;
|
||||
lift musl's conversions rather than writing them, and run CPython's float tests
|
||||
early.
|
||||
- **Threadless build is load-bearing, initially.** Like WASI, the first builds have
|
||||
no working `threading`. The escape hatch is real (danos has native threads and
|
||||
futexes; a pthread subset is Phase-5 work) but keep the configuration honestly
|
||||
single-threaded until then.
|
||||
- **The test suite is the porting harness.** CPython ships its own conformance
|
||||
suite; getting `test_builtin`, `test_int`, `test_float`, `test_io` running on
|
||||
danos early converts "it seems to work" into a checklist. Budget image space for
|
||||
the test `Lib/` tree during bring-up.
|
||||
- **Entropy before exposure.** `PYTHONHASHSEED=0` is fine for bring-up and wrong
|
||||
forever; hash randomization exists because attacker-controlled dict keys are a
|
||||
denial-of-service vector. Land the entropy source before any Python service
|
||||
parses external input.
|
||||
|
||||
## Related
|
||||
|
||||
- [python-on-danos-milestones.md](python-on-danos-milestones.md) — the execution
|
||||
plan (P0–P5) for this note.
|
||||
- [c-library-compatibility.md](c-library-compatibility.md) — the mini-libc
|
||||
(libdanos-c) design behind Phase 0.
|
||||
- [character-devices-and-tty.md](character-devices-and-tty.md) — the stream-node /
|
||||
console / no-pty design behind Phase 1.
|
||||
- [zig-self-hosting.md](zig-self-hosting.md) — the sibling track; shares Phase 1,
|
||||
diverges at the libc.
|
||||
- [os-development/syscall.md](os-development/syscall.md) — the kernel ABI the
|
||||
mini-libc bottoms out in.
|
||||
- [os-development/vdso.md](os-development/vdso.md) — the public ABI boundary the
|
||||
`danos` extension module wraps.
|
||||
- [os-development/sysv.md](os-development/sysv.md) — the entry stack (argv/envp)
|
||||
the spawn-argv work extends.
|
||||
- [device-driver-development/ipc.md](device-driver-development/ipc.md) — the IPC
|
||||
surface Python services speak.
|
||||
- [file-system-development/file-system-hierarchy.md](file-system-development/file-system-hierarchy.md)
|
||||
— where `Lib/` and `site-packages` land on the image.
|
||||
@@ -0,0 +1,545 @@
|
||||
# Security track execution plan: paths, protocol namespace, SMEP/SMAP
|
||||
|
||||
The design is settled in
|
||||
[communication.md](os-development/communication.md),
|
||||
[protocol-namespace.md](os-development/protocol-namespace.md),
|
||||
[file-system-hierarchy.md](file-system-development/file-system-hierarchy.md),
|
||||
and [smep-smap.md](os-development/smep-smap.md). This file is the build order
|
||||
— one phase at a time, each phase green before the next starts. Delete or
|
||||
archive this file when the last milestone lands.
|
||||
|
||||
**Context a fresh session should read first:** the four design docs above,
|
||||
then this plan's *Settled decisions* section — those decisions came out of a
|
||||
full-code grounding pass (2026-07-31) and must not be re-derived or reopened.
|
||||
|
||||
**Definition of green, every phase:** `zig build` clean, `zig build test`
|
||||
clean, `python3 test/qemu_test.py` passes (existing scenarios plus the
|
||||
phase's new ones — record the suite count in the checkbox), and the relevant
|
||||
design doc's status/known-gap lines updated in the same commit. Commit per
|
||||
green phase, style `area: lower-case declarative summary`, **no co-author
|
||||
trailers**. On a suite failure, read
|
||||
`zig-out/qemu-test/<case>-failed-serial.log` before changing anything.
|
||||
|
||||
**Workflow:** work in a dedicated git worktree on feature branches cut from
|
||||
`main` (one branch per milestone group as marked below); when a group's
|
||||
phases are all green, merge to `main` and push. The loop marks a phase `[x]`
|
||||
in the same commit that lands it.
|
||||
|
||||
**Numbering note:** milestones use the design docs' own names (PM, H1–H3,
|
||||
HS, P1–P4) — the M-number sequence is left alone (M19–M22 are reserved by
|
||||
the logging/USB-lifecycle track).
|
||||
|
||||
## Status
|
||||
|
||||
**Live state — updated on `main` after every phase, so this file read from a
|
||||
plain `main` checkout always tells the truth about where the work is.**
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| Working on | **H2** — SMEP (group 4) |
|
||||
| Branch carrying it | `feat/security-group-4` (cut next) |
|
||||
| On `main` | everything through P4c — groups 1, 2 and 3 merged |
|
||||
| Awaiting merge | nothing |
|
||||
| Suite | 111 cases, all passing |
|
||||
| Last updated | 2026-08-01 |
|
||||
|
||||
A checkbox below means the phase met its definition of green and was
|
||||
committed — on the branch named above, which reaches `main` at the next
|
||||
group boundary.
|
||||
|
||||
- [x] **Phase 0** — baseline: suite green on `main` (106/106, 2026-07-31; `zig build` + `zig build test` clean at 9a32380), plan committed
|
||||
- [x] **PM** — path-migration flag-day (`/etc`→`/system/configuration`, `/var/log`→`/system/logs`, `/mnt/usb`→`/volumes/usb`; vfs carve-out for the two writable `/system` subtrees, FAT's `/var` mount split in two; suite 106/106)
|
||||
- [x] **H1** — the `user-memory` module; nine stragglers converted; leaf U/S+W checks (plus physmap-coverage confirmation, so an `mmio_map`'d buffer cannot fault ring 0 — this also closes the same hazard on the IPC path; `fs_resolve`'s out-capacity bound made overflow-safe; suite 107/107)
|
||||
- [x] **merge** group 1 → main, push (f3bc23c, 2026-07-31)
|
||||
- [x] **P1** — envelope module + `Define`; vfs `NodeKind.protocol` + open-reply-capability; client `Channel` (mechanics only, nothing converted; suite unchanged at 107)
|
||||
- [x] **P2** — registry in init; `/protocol` reserved; ServiceId flag-day (11 binds, 17 lookups; `protocol.csv` grants, chain-attested identity, dead-owner rebind; the kernel's endpoint-death sweep generalized off the retired registry; suite 108/108). Three adversarial review rounds closed six defects a green suite had missed: a forged power event could shut the machine down; the ping path leaked a capability per call, first in init and then in the shared harness; supervisor attestation by name was defeated by a laundering deputy; and the kernel let any handle-holder bind signals, timers, exits and IRQs to an endpoint it did not own.
|
||||
- [x] **P3** — open grants: `protocol.csv` enforcement, denial test. `onOpen`
|
||||
consults the manifest with the same chain-attested identity a bind uses, and a
|
||||
refused caller gets the *same* answer as one naming a contract nobody bound —
|
||||
`-ENOENT`, no capability, the same reply bytes, no log line, and both questions
|
||||
asked on every open so there is nothing to time. Twenty-seven `open` rows cover
|
||||
the whole live client set. One wrinkle the plan had not foreseen: the driver
|
||||
tree is three deep (device manager → PS/2 bus → keyboard/mouse) and attestation
|
||||
is one hop, so a legitimate grandchild read exactly like a laundering deputy;
|
||||
the manifest gained a third permission, `supervise`, which names an authorized
|
||||
supervising task per contract and is deliberately **open-only**, leaving P2's
|
||||
bind attestation and every refusal it makes untouched (suite 109/109)
|
||||
- [x] **merge** group 2 → main, push
|
||||
- [x] **P4a** — clean protocols rebased onto `Define` (vfs, block, display, scanout, input; display's one overloaded request split per-operation and its field abuse ended, scanout's bogus 64-byte maximum deleted, directory EOF re-spelled as a nameless entry, input moved onto the service harness; new `protocol-conformance` case asks every reachable provider for `describe` and requires `-ENOSYS` for an undefined verb; suite 110/110)
|
||||
- [x] **P4b** — misfit protocols rebased (device-manager, power, usb-transfer; every leading operation byte folded into the header, and with it the `device_id`/`device_token` that followed it — `Header.target` now carries the device in all three. device-manager's own `enumerate`/`subscribe` became the reserved verbs and its three `{status, reserved}` reply structs the envelope's `Status`; `ChildAdded` is one struct under two numbers, a call and an event, landing exactly on the 64-byte push floor. power's kinds became one declared event each, the input protocol's shape, so init reads *what happened* from the header; usb-transfer's control data stage moved to the packet tail in both directions, which made `Status.len` the transferred length and `actual_length` redundant. The two silent-breakage sites — init's byte-offset power parse and acpi's `message[0]` dispatch — are gone, the shutdown badge gate unchanged; three more rows in the conformance table. Suite 110/110)
|
||||
- [x] **P4c** — harness subscriber lift + badge-scoped per-client integers (the
|
||||
subscriber table, the reserved subscribe/unsubscribe verbs, the fan-out and the
|
||||
dead-subscriber sweep are `service.Subscribers` now; input, acpi and
|
||||
device-manager deleted three hand-rolled variants and their three different
|
||||
ideas of when a subscriber goes away, standardizing on published exit
|
||||
notifications — acpi had no sweep at all and input polled the process list on
|
||||
every subscribe. The three guessable-id namespaces are scoped to the opening
|
||||
badge: FAT node ids on every verb that names one, xHCI device tokens on open,
|
||||
control, bulk and interrupt_subscribe, display layers on configure, fill, blit,
|
||||
damage and destroy — each refusing a wrong owner with the *same* answer as an id
|
||||
nobody holds. New `badge-scope` case, two processes of one fixture, every
|
||||
refusal paired with a control; suite 111/111)
|
||||
- [x] **merge** group 3 → main, push
|
||||
- [ ] **H2** — SMEP on every core
|
||||
- [ ] **HS** — SYSRET canonical-RIP guard
|
||||
- [ ] **H3** — SMAP + boot-patched `clac`; `-cpu max` in the harness; negative tests
|
||||
- [ ] **merge** group 4 → main, push
|
||||
|
||||
---
|
||||
|
||||
## Settled decisions (grounding pass, 2026-07-31 — do not reopen)
|
||||
|
||||
These resolve every open wrinkle the code inventory surfaced. Where one
|
||||
amends a design doc, the amendment lands in the same commit as the phase
|
||||
that implements it.
|
||||
|
||||
1. **Every packet — request, reply, and event — begins with the envelope
|
||||
`Header`, exactly as the design says; the header is FOLDED, never
|
||||
stacked.** It absorbs each protocol's existing operation/id fields
|
||||
rather than sitting on top of them, so the two apparent 64-byte-limit
|
||||
offenders fit: `ChildAdded` re-lays to 60 bytes (its packed operation
|
||||
byte and `device_id` become `Header.operation`/`.target`);
|
||||
`InterruptReport` puts `device_token` in `Header.target` and trims
|
||||
inline data 48 → 40 bytes (largest real report today is 8). A
|
||||
headerless-events variant was considered and REJECTED (2026-07-31): it
|
||||
re-invents per-protocol mini-headers and breaks uniform tooling. No
|
||||
design-doc amendment; `Define`'s event check stays ≤ 64 *including*
|
||||
the header.
|
||||
2. **Bind/open authorization is chain-attested identity: the
|
||||
kernel-stamped binary name PLUS the supervision chain**, both read from
|
||||
the kernel's process records (`ProcessDescriptor` carries `name` and
|
||||
`supervisor`; init walks the chain with `process_enumerate` — no new
|
||||
protocol). A grant row names the binary *and* the supervisor expected
|
||||
in its chain, so a malicious process re-spawning a granted binary
|
||||
(ungated `spawn`, hostile argv — the confused deputy) is refused: its
|
||||
chain roots at the attacker, not at init or device-manager. Name alone
|
||||
is NOT sufficient — that was considered and rejected (2026-07-31).
|
||||
Pure delegation (device-manager forwarding driver binds as
|
||||
capabilities — "option B") is deliberately deferred to P5, whose
|
||||
spawner-wired namespaces subsume it. Amends protocol-namespace.md's
|
||||
"Authorization" bullet in P2.
|
||||
3. **Grants live in a new manifest, `/system/configuration/protocol.csv`**
|
||||
(rows: `binary-path, supervisor, bind|open, protocol-name`, where
|
||||
`supervisor` is the binary expected in the caller's supervision chain —
|
||||
`init` for init's own children, `kernel` for harness-spawned fixtures),
|
||||
not in extra init.csv columns — today every post-path init.csv field is
|
||||
argv, and overloading that is ambiguous. init parses both files.
|
||||
*(P2 spelling: the supervisor column carries the binary exactly as the
|
||||
kernel stamped it, so init's own children say `/system/services/init` and
|
||||
the drivers say `/system/services/device-manager`; `kernel` stays a bare
|
||||
word because a kernel task has no binary. A trailing `*` on any field
|
||||
matches a subtree, which is how decision 4's `/test/` rule is expressed.)*
|
||||
*(Clarification, 2026-08-01: the supervisor column names **the authorized
|
||||
supervising task, matched by identity** — the binary is how the row spells
|
||||
it, but init checks the task id. `kernel` is satisfied only by supervisor
|
||||
id 0 (which only the kernel confers — user `system_spawn` always stamps the
|
||||
caller); init's own path only by this init's task id; any other path only by
|
||||
a task init spawned itself or one the kernel spawned. Matching the supervisor
|
||||
by *name* alone is defeated by a laundering deputy — an attacker runs its own
|
||||
instance of `/system/services/init`, has that spawn `/system/services/input`,
|
||||
and both stamped names satisfy the row while the chain is entirely the
|
||||
attacker's. Walking to the root of the chain does not fix it either, since
|
||||
the laundered chain still roots at the real PID 1.)*
|
||||
*(P3 amendment: a third permission, `supervise`, joins `bind|open`. One-hop
|
||||
attestation cannot express the one three-deep chain in the tree — the device
|
||||
manager starts the PS/2 bus, and the bus starts the keyboard and mouse
|
||||
drivers — and nothing structural tells that chain apart from the laundering
|
||||
deputy, since both are a granted binary spawned by a granted binary. Only
|
||||
policy can: a `supervise` row names the authorized supervising task the way
|
||||
every other row names a claimant (binary, its own supervisor, the contract it
|
||||
concerns), and an `open` row may then name that task in its supervisor
|
||||
column. The delegate is itself attested the ordinary strict way, so the chain
|
||||
still anchors in init or the kernel one hop above it and the recursion stops
|
||||
there. It is **open-only** on purpose — a delegate may vouch for what its
|
||||
children *reach*, never for what they *claim* — so the bind path is
|
||||
byte-for-byte P2's and the laundering-deputy refusal is untouched.)*
|
||||
4. **Test fixtures bind under `/protocol/test/...`**, granted to any
|
||||
binary whose path starts `/test/` — the subtree-scoping rule from the
|
||||
design doc, dogfooded. `shared_memory_test` (the borrowed-ServiceId
|
||||
hack) becomes `/protocol/test/shared-memory`; process-test's child gets
|
||||
`/protocol/test/process`.
|
||||
5. **Rebind after provider death:** a `bind` hitting an existing binding
|
||||
succeeds only if the current owner process is dead (init checks
|
||||
liveness); otherwise `-EBUSY`. Init also unbinds in `restartChild`
|
||||
before respawning its own children. This preserves collision-refusal
|
||||
while making restart work for providers init does not supervise.
|
||||
6. **Cross-thread service access** (the display mouse-listener's
|
||||
per-thread self-lookup, `display.zig:512`): threads resolve and open
|
||||
`/protocol/<name>` like any client — once, at thread startup. No
|
||||
special mechanism.
|
||||
7. **The envelope module is `library/protocol/envelope/envelope.zig`**
|
||||
(module name `envelope`) — the one protocol-package module not ending
|
||||
in `-protocol`, because it is not a protocol. Wired as a new
|
||||
`addModule` row in `library/protocol/build.zig` with its host tests in
|
||||
that package's test step.
|
||||
8. **The QEMU harness gains `-cpu max`** (in `qemu_args`,
|
||||
`test/qemu_test.py:66-83`) so TCG exposes SMEP/SMAP — without it the
|
||||
enabled paths never execute in CI. Landed in H2 so the flag soaks
|
||||
before H3 depends on it.
|
||||
9. **Scenario fixtures that need the registry are init-driven.** Kernel
|
||||
test cases that today spawn providers directly (shared-memory,
|
||||
process-test) either spawn init first or move to init.csv-driven
|
||||
scenario boots — resolved per-case in P2 with the suite as the
|
||||
arbiter.
|
||||
*(P2 resolution: init gained a `registry` argv role — it mounts
|
||||
`/protocol`, reads the grants, and starts no services — and each affected
|
||||
case calls `spawnRegistry(rd)` before its own providers. Every case keeps
|
||||
its own spawn set, so no scenario had to be re-shaped.)*
|
||||
10. **The capsule-staleness caveat is documented, not fixed.** On-volume
|
||||
edits to `/system/configuration/*.csv` do not reach the initrd copy
|
||||
the loader boots (capsule shadows tree). Same drift exists today with
|
||||
`/etc`; PM adds the note to file-system-hierarchy.md and moves on.
|
||||
|
||||
---
|
||||
|
||||
## PM — path-migration flag-day
|
||||
|
||||
One commit, everything moves together. The authoritative site inventory is
|
||||
the grounding pass; the checklist order:
|
||||
|
||||
1. Move repo `etc/` → `configuration/` sources; fix the three CSVs'
|
||||
self-referencing headers (`etc/init.csv:1,12`, `etc/devices.csv:1`,
|
||||
`etc/init-diagnose.csv:1`).
|
||||
2. `build.zig:309-311`: bundled entries `etc/...` →
|
||||
`system/configuration/...` (this alone re-shapes the image, manifest,
|
||||
and capsule — `tools/make-fat-image.py` and the EFI loader need
|
||||
nothing; the tree-walk fallback even starts picking the CSVs up, a
|
||||
bonus fix).
|
||||
3. `system/kernel/vfs.zig` `mountBackend` (`:332-340`): allow exactly
|
||||
`/system/configuration` and `/system/logs` as backend prefixes beneath
|
||||
the initrd `/system` mount; keep refusing everything else under
|
||||
`/system` and `/test`.
|
||||
4. `system/services/fat/fat.zig`: `mount_point` → `/volumes/usb` (`:25`);
|
||||
replace the `/var` mount (`:155`) with two `mountRewritten` calls for
|
||||
`/system/configuration` and `/system/logs`; update the mount log lines
|
||||
(the harness matches them).
|
||||
5. `system/services/init/init.zig:76` and
|
||||
`system/services/device-manager/device-manager.zig:48`: open the new
|
||||
CSV paths; update the message strings (`init.zig:77,92`,
|
||||
`device-manager.zig:49,61-63,454`).
|
||||
6. `system/services/logger/logger.zig:44`: `base = "/system/logs"`
|
||||
(buffers derive from `base.len` comptime — nothing else changes).
|
||||
7. `system/kernel/tests.zig:2808-2810`: exclude `/system/configuration/`
|
||||
from the spawn-everything sweep (the CSVs are not programs).
|
||||
8. Tests: `fat-test.zig` and `vfs-test.zig` `/mnt/usb` literals →
|
||||
`/volumes/usb`; harness regexes `test/qemu_test.py:175,211,632,717`.
|
||||
9. Comment sweep (init, device-manager, logger, fat, engine, vfs, abi,
|
||||
file-system, csv, device, protocol/device-manager, drivers, acpi,
|
||||
build.zig — full list in the grounding inventory); delete vestigial
|
||||
repo `var/`.
|
||||
|
||||
**Test:** no new case — the existing 106 are the test, since fat/logger/
|
||||
init/device-manager scenarios all assert the new paths through their
|
||||
regexes. Suite stays 106.
|
||||
|
||||
## H1 — user-memory copy discipline
|
||||
|
||||
New kernel module `system/kernel/user-memory.zig`:
|
||||
|
||||
- `copyFromUser` moves from ipc-synchronous.zig (which re-exports or
|
||||
imports it); new `copyToUser(user_as, user_va, source) bool` — the
|
||||
mechanical mirror (kernel-source `copyAcross` already does this for IPC
|
||||
replies at `ipc-synchronous.zig:431,460`).
|
||||
- The page walk gains leaf U/S and writable checks: `paging.translateIn`
|
||||
(`architecture/x86_64/paging.zig:513-525`) tests only `present` today —
|
||||
add a flags-accumulating variant (2 MiB leaves included); reads require
|
||||
U/S, writes require U/S+W. Closes the TODO at
|
||||
`ipc-synchronous.zig:20-22`.
|
||||
- Convert the nine stragglers (table in smep-smap.md). Read direction is
|
||||
local to `process.zig`; the write direction restructures callees with
|
||||
kernel bounce buffers: `scheduler.enumerate` (`scheduler.zig:1209`),
|
||||
`devices_broker.enumerate` (`devices-broker.zig:136`), `log.readAt`
|
||||
(`log.zig:209`), and the `fs_node` flows through
|
||||
`vfs.nodeRead/nodeStatus/nodeReaddir` (`vfs.zig:257/269/289`).
|
||||
|
||||
**Test:** kernel unit coverage in `system/kernel/tests.zig` for
|
||||
`copyToUser` bounds/permission refusals; one new QEMU case `user-memory` —
|
||||
a fixture passes an unmapped-but-in-range buffer to `klog_read`,
|
||||
`process_enumerate`, and `fs_resolve` and asserts `-EFAULT` returns with
|
||||
the system still alive (today each would oops the kernel). Suite 107.
|
||||
|
||||
## P1 — envelope, vfs additions, Channel
|
||||
|
||||
- `library/protocol/envelope/envelope.zig`: `Header` {operation:u32, pad,
|
||||
target:u64}, `Status`, reserved verbs (describe=0, enumerate=1,
|
||||
subscribe=2, unsubscribe=3, protocol verbs from 16), `packet_maximum`
|
||||
= 256 / `post_maximum` = 64 (the floor constants protocols compile
|
||||
against — nothing exports them today), and comptime
|
||||
`Define(.{name, version, operations, events})` generating request/reply
|
||||
types, encode/decode, a provider dispatch table (automatic `describe`,
|
||||
`-ENOSYS` for unknown verbs), and compile-time size checks:
|
||||
request/reply ≤ 256, each `.events` entry ≤ 64 *including* its Header
|
||||
(decision 1). Host unit tests in the protocol package's test step.
|
||||
- `library/protocol/vfs/vfs-protocol.zig`: `NodeKind.protocol = 7`; the
|
||||
open-reply-may-carry-capability convention documented in the module.
|
||||
Rewrite the value-pinning unit test (`:108-117`) to pin the *new*
|
||||
stable values.
|
||||
- `library/kernel/file-system.zig` + a new `Channel` type in
|
||||
`library/kernel` (or `library/client`): `open("/protocol/<name>")` →
|
||||
resolve, vfs open, receive the reply capability → a `Channel` wrapping
|
||||
the handle with `call`/typed helpers. Nothing uses it yet — P2 converts
|
||||
the world.
|
||||
- Docs: vfs-protocol.md's NodeKind table gains value 7 (no
|
||||
protocol-namespace.md amendment — decision 1 conforms to it as written).
|
||||
|
||||
**Test:** host unit tests only (envelope round-trips, size-check compile
|
||||
errors via `error` tests, Channel plumbing against a mock). Suite stays
|
||||
107.
|
||||
|
||||
## P2 — the registry; ServiceId flag-day
|
||||
|
||||
The single biggest phase; one branch, may be several commits, green at the
|
||||
end of each.
|
||||
|
||||
- **init as registry backend** (`system/services/init/init.zig`): a second
|
||||
endpoint (the supervision endpoint's reply-empty loop is unsuitable for
|
||||
a vfs backend); serve vfs `open`/`readdir` over `/protocol` plus the
|
||||
`bind` operation (name payload + capability). Mount `/protocol` before
|
||||
spawning children. Parse `/system/configuration/protocol.csv`
|
||||
(decision 3). Authorization by chain-attested identity (decision 2):
|
||||
badge → kernel process records → binary name **and** supervision chain
|
||||
(walk `supervisor` links) checked against the grant row's expected
|
||||
supervisor. Unbind on child death in `restartChild`; dead-owner rebind
|
||||
rule (decision 5).
|
||||
Provenance: readdir/diagnostics show name → pid → binary path.
|
||||
- **Kernel:** reserve `/protocol` — `mountBackend` refuses mounts at or
|
||||
under it once bound, `installMount`'s remount-replace path refuses it,
|
||||
and `fs_unmount` refuses it (`vfs.zig:164-181,332-351`,
|
||||
`process.zig:1879-1889`). First mount wins (init is PID 1).
|
||||
- **Harness:** `library/kernel/service.zig` `Callbacks.service:
|
||||
?abi.ServiceId` becomes a protocol name; the register call (`:49-51`)
|
||||
becomes bind-with-retry via the registry.
|
||||
- **Flag-day conversion** — all 11 registration sites and 17 lookup sites
|
||||
from the grounding inventory: providers (input:123, ps2-bus:223,
|
||||
device-manager:569, acpi:193, usb-xhci-bus:676, usb-storage:205,
|
||||
fat:307, display:699, virtio-gpu:550, shared-memory-server:43,
|
||||
process-test:130 → `/protocol/test/...` per decision 4); clients
|
||||
(input-client:53, display-client:28, driver.zig:173, usb.zig:139,
|
||||
block.zig:72+87, ps2-bus keyboard:35 + mouse:34, virtio-gpu:478,
|
||||
display:314+512 (decision 6), acpi:212, init:218+245 — init
|
||||
short-circuits its own registry, shared-memory-client:22,
|
||||
process-test:85, device-list:22, crash-test:32). Retry loops keep their
|
||||
cadence, wrapping resolve+open instead of lookup.
|
||||
- **Delete:** `abi.zig:36-37` (syscall ids — leave holes),
|
||||
`abi.zig:287-303` (enum), `process.zig:223-224,314-343`,
|
||||
`ipc-synchronous.zig:41-43,646-664` and the registry sweep in
|
||||
`:121-140`; the wrappers `library/kernel/ipc.zig:33-35,47-50`; comment
|
||||
sweep (irq.zig:50, tests.zig:3744, vdso.md's syscall table, the docs
|
||||
list in the inventory).
|
||||
- Kernel-spawned test scenarios made init-driven where they need the
|
||||
registry (decision 9).
|
||||
|
||||
**Test:** new QEMU case `protocol-registry`: a fixture asserts (a) bind of
|
||||
an ungranted name → `-EPERM`, (b) bind collision with a live owner →
|
||||
`-EBUSY`, (c) provider kill → re-resolve reaches the restarted instance.
|
||||
Every existing scenario doubles as conversion proof. Suite 108.
|
||||
|
||||
## P3 — open grants (restriction stage one)
|
||||
|
||||
- `protocol.csv` `open` rows enforced in the registry's `open` handler,
|
||||
same name-based identity as bind. Default rows grant what today's
|
||||
clients need (from the P2 conversion table); a deliberate hole for the
|
||||
test fixture.
|
||||
- Docs: protocol-namespace.md stage-one section gets its "landed" line.
|
||||
|
||||
**Test:** new QEMU case `protocol-denied`: a fixture granted
|
||||
`/protocol/test/shared-memory` but not `/protocol/display` asserts open of
|
||||
the first succeeds and the second fails identically to not-found. Suite
|
||||
109.
|
||||
|
||||
*Landed. Four things the plan did not foresee, recorded because P4 and P5
|
||||
inherit them:*
|
||||
|
||||
- *`supervise` — decision 3's amendment. The PS/2 keyboard and mouse drivers
|
||||
are started by the PS/2 bus driver, which the device manager started: the
|
||||
tree's one three-deep chain, and one hop deeper than attestation reaches.
|
||||
Nothing structural separates it from the laundering deputy, so the manifest
|
||||
says which delegate is authorized, per contract. Open-only, so P2's bind
|
||||
attestation is unchanged.*
|
||||
- *Indistinguishability is a claim about work, not only about bytes. `onOpen`
|
||||
refreshes the process table, identifies the caller, scans the grants and
|
||||
scans the bindings on **every** open and forms one verdict at the end; and
|
||||
it logs nothing on any branch, because `klog_read` is ungated (a line
|
||||
written on one branch is a line the refused caller can read) and a serial
|
||||
line is milliseconds it could time. The operator's diagnosis is the pair the
|
||||
namespace publishes anyway: `readdir /protocol` for what is bound, the
|
||||
manifest for who may reach it.*
|
||||
- *The fixture is `protocol-denied-test`, and its scenario boots the **input
|
||||
service** so the forbidden name is genuinely bound — the fixture reads the
|
||||
namespace listing to prove it before asking for it. Without a live provider
|
||||
the case would be comparing two boot races and asserting nothing.*
|
||||
- *Two channels stay open by design, named rather than papered over: `readdir`
|
||||
over `/protocol` lists every bound name to anyone (deliberate — the tree is
|
||||
diagnosable), and `/system/configuration/protocol.csv` is world-readable on
|
||||
the `/system` mount. Stage one hides neither the set of contracts nor the
|
||||
policy; what it removes is the **oracle in the reply**, which is what stage
|
||||
two's parked and faked opens depend on.*
|
||||
|
||||
## P4a — clean protocols onto Define
|
||||
|
||||
vfs, block, display, scanout, input — the modules whose shapes map
|
||||
directly (grounding inventory §1,3,4,6,8):
|
||||
|
||||
- vfs: `node` → `target`; `Reply.node` (open's result) moves to reply
|
||||
payload — `library/kernel/file-system.zig` decoders change; readdir
|
||||
stays a protocol verb.
|
||||
- block: pure renumber; `attach`'s DMA cap rides the call as today.
|
||||
- display: the overloaded 40-byte `Request` becomes per-operation structs
|
||||
(attach_scanout's field abuse dies); `layer` → `target`; blit payload
|
||||
grows to 224 bytes.
|
||||
- scanout: renumber; drop its bogus `message_maximum=64` (sync floor is
|
||||
256); fix virtio-gpu's hard-coded `service.run(256, …)` to the
|
||||
generated constant.
|
||||
- input: subscribe merges into reserved subscribe; publish renumbers;
|
||||
the event re-lays onto the Header folded (operation = event kind,
|
||||
target = 0; 16 + 28-byte payload = 44 ≤ 64); **input moves onto the
|
||||
service harness** (it is the last hand-rolled loop, no ping/terminate
|
||||
compliance today).
|
||||
|
||||
**Test:** new QEMU case `protocol-conformance`: a fixture opens every
|
||||
registered protocol and asserts `describe` answers (name, version) and an
|
||||
unknown verb returns `-ENOSYS`. Existing input/display/fat scenarios prove
|
||||
the rebase. Suite 110.
|
||||
|
||||
## P4b — misfit protocols onto Define
|
||||
|
||||
device-manager, power, usb-transfer (inventory §2,5,7 — the u8-operation
|
||||
re-layouts and raw-offset readers):
|
||||
|
||||
- device-manager: u8 operations → Header; its enumerate=4/subscribe=5
|
||||
merge into the reserved verbs; `ChildAdded` splits its dual role —
|
||||
request struct and event, both Header-first (folded to 60 B ≤ 64);
|
||||
`ChildRemoved`'s (parent, bus_address) addressing stays payload.
|
||||
- power: u8 operations → Header; subscribe merges; **init's raw
|
||||
byte-offset event parsing (`init.zig:171-173`) and acpi's
|
||||
`message[0]` dispatch (`acpi.zig:435-467`) are rewritten against the
|
||||
generated types** — the two silent-breakage sites, called out so the
|
||||
loop treats them as first-class conversions, not collateral.
|
||||
- usb-transfer: `device_token` → `target` (already layout-identical);
|
||||
`InterruptReport` re-lays onto the Header (`device_token` → `target`,
|
||||
inline data trimmed 48 → 40 — largest real report is 8); control/bulk
|
||||
budgets re-verified by `Define` (Status absorbs `actual_length`).
|
||||
|
||||
**Test:** existing scenarios are the proof (device hot-add, power button,
|
||||
USB storage/HID all exercise these wires); the conformance case now covers
|
||||
three more providers. Suite 110.
|
||||
|
||||
*Landed. Three judgment calls the plan left open, recorded because a reader of
|
||||
the wire formats will want them:*
|
||||
|
||||
- *`ChildAdded` is 48 bytes, not the 44 the "60 B" estimate assumed: three `u64`s
|
||||
give the struct eight-byte alignment, so 41 bytes of content round up whatever
|
||||
order the fields sit in. The packet is therefore **exactly** 64 — on the push
|
||||
floor, not under it — which `Define` accepts and the module pins in a test. The
|
||||
fields are ordered small-tail-last deliberately, so the slack the rounding pays
|
||||
for is where the small ones live.*
|
||||
- *power's events are declared **per kind** (`power_button`, `lid`, `ac`,
|
||||
`battery`, `notify`), not one `event` with the kind in the payload. That is the
|
||||
shape P4a gave input — "the class is the header's operation, so a subscriber
|
||||
reads the kind from the packet rather than from a tag inside the payload" — and
|
||||
it is what makes `Header.operation` carry information here at all. It also kept
|
||||
every call site's spelling: `Protocol.Event` is re-exported as the protocol's
|
||||
own `Event`, with the members it always had.*
|
||||
- *the conformance case still checks two providers, and the three new rows report
|
||||
as unbound. All three P4b contracts arrive with the device manager — it is the
|
||||
first, it spawns the discovery service that binds the second, and the xHCI
|
||||
driver that binds the third — so booting one means booting the driver tree, and
|
||||
the fixture takes **one snapshot** of `/protocol`: a scenario whose bound set
|
||||
depends on how far that tree got would make the case's own summary line a boot
|
||||
race. The rows still earn their place — a future scenario that binds one gets it
|
||||
checked with no edit here, and `/test/*` already holds the `open` grant for
|
||||
`device-manager`.*
|
||||
|
||||
## P4c — harness subscriber lift + badge scoping
|
||||
|
||||
- `library/kernel/service.zig` grows the subscriber table, exit-
|
||||
notification sweep, and fan-out loop declared via `Define(.events)`;
|
||||
input (:33-116), acpi (:67-68,393-406), and device-manager (:155-166)
|
||||
delete their hand-rolled variants. One sweep idiom: exit notifications
|
||||
(fat's pattern), replacing input's process-list polling and acpi's
|
||||
none-at-all.
|
||||
- Badge-scoped per-client integers (the guessable-id holes): fat node ids
|
||||
gain owner checks on every operation (`fat.zig:72-76`), xhci device
|
||||
tokens validate sender and sweep on exit (`usb-xhci-bus.zig:66-88,479`),
|
||||
display layers gain an owner field.
|
||||
|
||||
**Test:** extend the fat scenario: a second fixture guesses the first's
|
||||
node id and asserts refusal; kernel-side unit test for the harness sweep.
|
||||
Suite 111.
|
||||
|
||||
*Landed. Four things the plan had not foreseen:*
|
||||
|
||||
- *One sweep idiom means one more kernel subscriber per provider, and the kernel's
|
||||
published-exit table held **eight**. A normal boot now fields six (fat, input,
|
||||
power, device-manager, display, and one per xHCI controller), so the table grew
|
||||
to sixteen. It is not a table anyone notices until a service silently loses its
|
||||
sweep, which is exactly the failure the old ceiling was two subscriptions away
|
||||
from.*
|
||||
- *The device manager hears each of its drivers die **twice** now — it is both the
|
||||
supervisor its spawn named and, through the harness, a subscriber to published
|
||||
exits — and the notify ring delivers the two badges separately. Untreated, one
|
||||
death counted as two: the restart backoff doubled and the crash-loop cap fired
|
||||
at half the deaths it names. `onDriverExit` therefore retires the dead process
|
||||
id before it decides anything, and the second notification finds nothing to act
|
||||
on. (The `driver-restart` and `pci-scan` drills are what would have caught it.)*
|
||||
- *Refusal-equals-absence has a corollary for the verbs that **release**: FAT's
|
||||
`close` used to answer 0 for an unknown node, so scoping it had to change that
|
||||
too — a foreign node and a free one both answer `-ENOENT`, or the pair would
|
||||
have been an oracle for which ids are live. The same applies to the harness's
|
||||
`unsubscribe`.*
|
||||
- *Ownership is per **task**, not per process, because the badge is: the kernel
|
||||
stamps the sending thread's id, which is already the granularity of the exit
|
||||
sweep that releases the state (a worker thread's death releases the handles that
|
||||
worker opened). Nothing in the tree shares an id across its own threads today;
|
||||
a per-process notion would need the kernel to stamp the leader, and belongs with
|
||||
P5's spawner-wired namespaces if it is ever wanted.*
|
||||
|
||||
## H2 — SMEP
|
||||
|
||||
- Generalize the cpuid helper (`apic.zig:351-365`, private, subleaf-0) to
|
||||
a shared probe; gate on `cpuid(0).eax >= 7`.
|
||||
- Set CR4 bit 20 in `per-cpu.zig:initSystemCall` (or a sibling called
|
||||
from both `cpu.zig:148` and `smp.zig:181` — the one path both BSP and
|
||||
every AP already execute). Log enabled/absent (fail-open, IOMMU style).
|
||||
- Harness: add `-cpu max` to `qemu_args` (decision 8).
|
||||
|
||||
**Test:** new QEMU case `fault-smep` mirroring the `fault-*` injector
|
||||
pattern (`tests.zig:3906-3938`): ring-0 call through a pointer into a
|
||||
user-mapped page; expect `page fault (vector 14)` + `error code : 0x11` +
|
||||
kernel-half IP, machine reports the exception (deliberate-exception cases
|
||||
put the text in `expect`, per `qemu_test.py:189`). Suite 112.
|
||||
|
||||
## HS — SYSRET canonical-RIP guard
|
||||
|
||||
- `isr.s` syscall exit (`:256`): validate RCX canonicality before
|
||||
`sysretq`; non-canonical → `iretq` fallback (or kill), per the hazard
|
||||
note at `isr.s:192-194`.
|
||||
|
||||
**Test:** kernel unit case driving a thread whose return RIP is forged
|
||||
non-canonical via the syscall path if constructible cheaply; otherwise the
|
||||
review-level proof plus the existing fault cases regression. Suite 112.
|
||||
|
||||
## H3 — SMAP
|
||||
|
||||
- `clac` patch site at `isr_common` (`isr.s:367`, before the CPL test —
|
||||
ring-0 nesting inherits AC too): assemble a 3-byte NOP, patch to `clac`
|
||||
at boot through the physmap (the `process.zig:1990-1995` /
|
||||
`smp.zig:79-111` precedent), BSP-only before AP bring-up.
|
||||
- Set CR4 bit 21 in the same per-CPU init as SMEP.
|
||||
- Coding standards: kernel code touches user memory only through
|
||||
`user-memory`; no `stac` anywhere, ever.
|
||||
|
||||
**Test:** new QEMU case `fault-smap`: ring-0 deliberate read of a mapped
|
||||
user page; expect vector 14 + `error code : 0x1` + kernel IP. And the
|
||||
whole suite becomes the tripwire — any missed straggler now fails loudly.
|
||||
Suite 113.
|
||||
|
||||
---
|
||||
|
||||
**Explicitly out of scope** (own tracks, after this plan): P5 restriction
|
||||
stage two (spawn's initial capability, namespace views, parked replies,
|
||||
dedicated killable channels — needs a design session on the spawn
|
||||
contract), file-path namespacing, trusted UI (display track), pipes/FIFOs
|
||||
(Python track), `/applications` and its storage, `fs_mount`/`spawn`/
|
||||
`klog_read` gating beyond the `/protocol` reserved prefix, KPTI, IPC
|
||||
priority inheritance.
|
||||
@@ -95,9 +95,9 @@ hypervisor configured for UEFI firmware and an xHCI USB controller.
|
||||
|
||||
| Requirement | Detail | Source |
|
||||
|---|---|---|
|
||||
| **x86-64, 64-bit only** | Kernel and loader are built exclusively for `x86_64`; the loader rejects any non-x86-64 kernel ELF (`error.WrongArchitecture`). | `build.zig:481`, `boot/efi.zig:622` |
|
||||
| **x86-64, 64-bit only** | Kernel and loader are built exclusively for `x86_64`; the loader rejects any non-x86-64 kernel ELF (`error.WrongArchitecture`). | `build-support/build.zig` (`freestandingTarget`), `boot/efi.zig:622` |
|
||||
| **Long mode + PAE + NX** | AP trampoline sets `CR4.PAE`, `EFER.LME`, `EFER.NXE`; NX is used in kernel page-table entries. | `system/kernel/architecture/x86_64/trampoline.s:62` |
|
||||
| **SSE / SSE2** | Baseline: the compiler emits SSE for ordinary struct copies. Trampoline enables `CR4.OSFXSR` + `OSXMMEXCPT` and clears `CR0.EM`. | `build.zig:477`, `trampoline.s:62` |
|
||||
| **SSE / SSE2** | Baseline: the compiler emits SSE for ordinary struct copies. Trampoline enables `CR4.OSFXSR` + `OSXMMEXCPT` and clears `CR0.EM`. | `build-support/build.zig` (`freestandingTarget`), `trampoline.s:62` |
|
||||
| **`syscall` / `sysret`** | Primary user↔kernel entry path. `EFER.SCE` enabled; `STAR`/`LSTAR`/`SFMASK` programmed per core. (`int 0x80` exists as a parallel gate.) | `architecture/x86_64/per-cpu.zig:71`, `isr.s:196` |
|
||||
| **Local APIC (xAPIC)** | LAPIC accessed via MMIO at `0xFEE00000`. LAPIC ID read as a `u8` — classic xAPIC. **x2APIC is not supported** (no MSR path). | `apic.zig:67`, `apic.zig:646` |
|
||||
| **CPUID + RDTSC** | CPUID leaf `0x15` for TSC frequency; RDTSC is the monotonic clock. | `apic.zig:333`, `apic.zig:113` |
|
||||
@@ -108,18 +108,20 @@ hypervisor configured for UEFI firmware and an xHCI USB controller.
|
||||
- **UEFI only.** A custom UEFI application loader is installed to
|
||||
`\EFI\BOOT\BOOTX64.efi`. There is **no BIOS, multiboot, or limine** path. The
|
||||
loader tolerates UEFI Class-3 machines with no legacy PIC/PIT.
|
||||
(`build.zig:246`, `boot/efi.zig`)
|
||||
(`build/images.zig` — the EFI/BOOT install — and `boot/efi.zig`)
|
||||
- **ACPI is the hardware-discovery mechanism.** The RSDP is taken from the UEFI
|
||||
configuration table (ACPI 2.0 GUID preferred, 1.0 fallback). Without a valid
|
||||
RSDP there is **no device discovery** — no SMP, no IOAPIC routing, no PCI/USB.
|
||||
(`efi.zig:790`, `boot-handoff.zig:149`)
|
||||
- **Required ACPI tables:** MADT (interrupt topology), MCFG (PCIe ECAM base),
|
||||
FADT (power / PM timer). Optionally consumed: HPET, DMAR, SPCR.
|
||||
(`system/devices/acpi.zig:3`)
|
||||
(`system/kernel/acpi.zig:3`)
|
||||
- The loader reads `/system/kernel` off the FAT boot volume, then loads user
|
||||
space: a prebuilt `boot\system.img` capsule when present, otherwise it walks
|
||||
the volume's `/system` tree (init included) into the initial ramdisk. The
|
||||
kernel can boot "kernel-only" without either. (`efi.zig:16`, `efi.zig:68`)
|
||||
space: a prebuilt `boot\system.img` capsule
|
||||
([system-image.md](os-development/system-image.md)) when present, otherwise it walks
|
||||
the volume's `/system` and optional `/test` trees (init included) into the
|
||||
initial ramdisk. The kernel can boot "kernel-only" without either.
|
||||
(`efi.zig:16`, `efi.zig:68`)
|
||||
|
||||
## Interrupt controller
|
||||
|
||||
@@ -220,7 +222,7 @@ named for reporting only; internal SATA / NVMe / IDE disks have no driver.
|
||||
first VT-d DRHD unit and its capabilities are exposed via `PlatformInformation`
|
||||
(`iommu_present`, `iommu_base`, `iommu_version`). No DMA-remapping tables are
|
||||
programmed and no translation is enforced. An IOMMU is therefore **not required**
|
||||
and does not currently constrain devices. (`system/devices/acpi.zig:96`)
|
||||
and does not currently constrain devices. (`system/kernel/acpi.zig:96`)
|
||||
|
||||
## What is explicitly NOT supported
|
||||
|
||||
|
||||
+6
-4
@@ -9,10 +9,12 @@ There are two layers:
|
||||
|
||||
- **Host unit tests** (`zig build test`) — for pure, platform-independent logic.
|
||||
What began as the three shared contracts (`system/boot-handoff.zig`,
|
||||
`system/abi.zig`, `system/devices/device-abi.zig`) now spans ~26 modules:
|
||||
`system/abi.zig`, `library/device/model/device-abi.zig`) now spans ~26 modules:
|
||||
protocol and on-wire definitions (VFS, USB, virtio-gpu), the FAT engine, the
|
||||
display compositor, PS/2 and HID decoding, the kernel log ring, and the
|
||||
runtime's `time`/`thread` — the full list is the test step in `build.zig`.
|
||||
runtime's `time`/`thread` — the list is distributed across the library-domain
|
||||
and binary packages' own `test` steps, which the root `zig build test`
|
||||
aggregates (docs/build-packages-plan.md).
|
||||
These compile for the host and run natively.
|
||||
- **QEMU integration tests** (`python3 test/qemu_test.py`) — boot the real kernel
|
||||
and check its behaviour. This is the interesting part.
|
||||
@@ -27,7 +29,7 @@ boot log, memory summary, exception reports — appears on serial as plain text.
|
||||
|
||||
QEMU captures that with `-serial file:serial.log`, giving a machine-readable
|
||||
transcript. Serial is per-architecture (x86 uses port I/O; an ARM board uses a
|
||||
memory-mapped UART), so it lives behind the [architecture](architecture.md) boundary — and adding
|
||||
memory-mapped UART), so it lives behind the [architecture](os-development/architecture.md) boundary — and adding
|
||||
a new architecture's UART is what makes the same tests run there.
|
||||
|
||||
The serial log sink is **compiled in only under `-Dserial`** (off by default).
|
||||
@@ -88,7 +90,7 @@ table in `test/qemu_test.py`):
|
||||
| `fault-recovery` | a ring-3 process that faults is killed and reaped while init keeps heartbeating — the OS survives | `DANOS-TEST-RESULT: PASS` |
|
||||
|
||||
The faulting cases don't print a result line — they deliberately raise a CPU
|
||||
exception, and the harness asserts on the [exception report](interrupts.md) the
|
||||
exception, and the harness asserts on the [exception report](os-development/interrupts.md) the
|
||||
handler prints (which also reaches serial). This reuses the real fault path as the
|
||||
test oracle: if the IDT/TSS weren't wired up, `fault-df` would triple-fault and the
|
||||
marker would never appear.
|
||||
|
||||
@@ -1,192 +0,0 @@
|
||||
# The VFS wire protocol
|
||||
|
||||
> **Status:** built and spoken today between `runtime.fs` (the client) and the
|
||||
> filesystem BACKENDS (the FAT server). The mount router lives in the
|
||||
> **kernel** (`system/kernel/vfs.zig`): `fs_resolve` routes a path and either
|
||||
> serves it directly (the read-only /system initrd mount, via `fs_node`) or
|
||||
> redirects the caller to the owning backend's endpoint plus the rewritten
|
||||
> mount-relative path — after which the client speaks THIS protocol to the
|
||||
> backend, unchanged. The Zig source of truth is `system/vfs-protocol.zig`
|
||||
> (the `vfs-protocol` module), whose unit test pins a sample of the sizes
|
||||
> and values below. This page is the **language-neutral wire specification**
|
||||
> of that contract — what a Rust or C client implements ([vdso.md](vdso.md)
|
||||
> explains why the IPC protocols, not the syscall numbers, are danos's
|
||||
> public ABI).
|
||||
|
||||
## Transport
|
||||
|
||||
A VFS exchange is one synchronous IPC rendezvous (`ipc_call`,
|
||||
docs/ipc.md): the client sends one message and blocks; the server replies
|
||||
with one message. The endpoint comes from the kernel's `fs_resolve` — which
|
||||
also hands back the path rewritten relative to the mount — not from a
|
||||
registry lookup. (Service id 1, the old userspace router, is retired.)
|
||||
|
||||
- A message is at most **256 bytes** (`message_maximum`).
|
||||
- A request is a fixed 32-byte **Request** header followed by an inline
|
||||
payload of at most **224 bytes** (`maximum_payload`) — a path, or write
|
||||
bytes. There is no multi-message request: paths and single reads/writes
|
||||
must fit, and larger transfers loop (see *read* / *write*).
|
||||
- A reply is a fixed 24-byte **Reply** header followed by an inline payload —
|
||||
read bytes, a `FileStatus`, or a `DirectoryEntry`.
|
||||
- All integers are **little-endian**; layouts are C layout for x86-64
|
||||
(`extern struct`), offsets given below so nothing need be inferred.
|
||||
|
||||
The kernel resolves NAMES (the mount table) but never parses these
|
||||
messages — it moves the bytes; file state is entirely the backend's affair.
|
||||
With clients holding backend node ids directly, a backend records each open
|
||||
handle's owner and sweeps a dead client's handles via the published process
|
||||
exit events.
|
||||
|
||||
## Request header — 32 bytes
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
| 0 | 4 | `operation` | an **Operation** value (below) |
|
||||
| 4 | 4 | — | padding |
|
||||
| 8 | 8 | `node` | the server-side open-node id from a prior `open`; 0 for path-based operations |
|
||||
| 16 | 8 | `offset` | byte position for read/write; entry index (cursor) for readdir; else 0 |
|
||||
| 24 | 4 | `len` | payload length for path/write operations; requested byte count for read |
|
||||
| 28 | 4 | `flags` | open flags (below); else 0 |
|
||||
|
||||
## Reply header — 24 bytes
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
| 0 | 4 | `status` | **0 = success**, negative = failure (signed) |
|
||||
| 4 | 4 | — | padding |
|
||||
| 8 | 8 | `node` | the new open-node id (for `open`); else 0 |
|
||||
| 16 | 4 | `len` | reply payload length in bytes |
|
||||
| 20 | 4 | — | padding |
|
||||
|
||||
On failure the backend replies `status = -1`, and that reply reaches the
|
||||
client directly — there is no party between them on the wire. (Kernel-served
|
||||
paths produce no wire replies at all: `fs_resolve`/`fs_node` failures are
|
||||
syscall register statuses.) A richer errno vocabulary is future work —
|
||||
clients must treat *any* negative status as failure, not match on -1.
|
||||
|
||||
## Operations
|
||||
|
||||
Values are append-only and never renumbered (the same evolution rule every
|
||||
danos protocol follows). Send only values from this table: the shipped server
|
||||
decodes the operation into an exhaustive enum, so an out-of-range value is
|
||||
not answered with a `status = -1` reply — it trips a safety check in safe
|
||||
builds and is undefined otherwise. (The `-1` replies cover recognised but
|
||||
refused operations, such as `mount` sent to a backend.)
|
||||
|
||||
| value | operation | request payload | reply |
|
||||
|------:|-----------|-----------------|-------|
|
||||
| 0 | `open` | the path (`len` = its length), `flags` as below | `node` = open-node id |
|
||||
| 1 | `close` | — (`node` set) | status only |
|
||||
| 2 | `read` | — (`node`, `offset`, `len` = wanted count) | `len` bytes read, payload = the bytes; `len` 0 at end of file |
|
||||
| 3 | `write` | the bytes (`node`, `offset`, `len` = count) | `len` = bytes accepted (may be short — loop) |
|
||||
| 4 | `status` | — (`node` set) | payload = **FileStatus** (24 bytes) |
|
||||
| 5 | `readdir` | — (`node` = a directory, `offset` = cursor) | payload = one **DirectoryEntry** + name; `len` 0 at end |
|
||||
| 6 | `mount` | the mount-point path; the backend endpoint rides as the call's **capability** | status only |
|
||||
| 7 | `unmount` | the mount-point path | status only |
|
||||
| 8 | `mkdir` | the path | status only |
|
||||
| 9 | `unlink` | the path | status only |
|
||||
| 10 | `rename` | old path, one `0x00`, new path (`len` = total) | status only |
|
||||
|
||||
Notes per operation:
|
||||
|
||||
- **open** — the path is the mount-relative path `fs_resolve` handed back
|
||||
(absolute-shaped: `/notes.txt` under fat's `/mnt/usb` mount). Bare names
|
||||
(`greeting`) resolve nowhere — the flat ramfs is retired, and `fs_resolve`
|
||||
refuses non-absolute paths. The returned `node` is the *backend's* own
|
||||
open-node id: with the router in the kernel there is no forwarding table,
|
||||
and clients hold backend ids directly (see *Lifetimes and trust*).
|
||||
- **read / write** — a single exchange moves at most 224 bytes
|
||||
(`maximum_payload`); the client loops, advancing `offset` by the returned
|
||||
`len`, until done (read) or the slice is written (write). A `write` reply
|
||||
shorter than requested is progress, not an error; a `len` of 0 means no
|
||||
forward progress — stop rather than spin.
|
||||
- **readdir** — `offset` is a **cursor: the entry index**, not a byte
|
||||
position. Each call returns exactly one entry; the client increments the
|
||||
cursor by 1. A reply with `len` 0 is end-of-directory. The directory must
|
||||
have been opened with the `directory` flag.
|
||||
- **mount / unmount** — RETIRED from the wire: mounting is the `fs_mount`
|
||||
syscall now (a filesystem server passes its endpoint handle; possession is
|
||||
the capability, exactly the trust of the old cap-passing op). The op
|
||||
numbers stay reserved. Mount-prefix semantics are unchanged: prefixes
|
||||
match at path boundaries only (`/mnt/usb` never captures `/mnt/usbextra`),
|
||||
the longest matching prefix wins, and an optional backend-side rewrite
|
||||
prefix maps a mount into the backend's namespace (fat serves `/mnt/usb`
|
||||
from its volume root and `/var` from its `/var` subtree).
|
||||
- **rename** — same-directory rename only: the backend compares the old and
|
||||
new parent paths and refuses a mismatch. The client (`runtime.fs`) refuses
|
||||
earlier when the two paths resolve to different backend endpoints, but that
|
||||
check is coarser than "one mount" — one endpoint can serve several mounts
|
||||
(fat serves `/mnt/usb` and `/var`), so a cross-mount rename reaches the
|
||||
backend and fails on its same-directory check.
|
||||
|
||||
## Open flags
|
||||
|
||||
Bitwise OR in `Request.flags`, meaningful for `open` only:
|
||||
|
||||
| bit | name | meaning |
|
||||
|----:|------|---------|
|
||||
| 1 | `create` | create the file if it does not exist |
|
||||
| 2 | `directory` | open a directory node for `readdir` rather than a file |
|
||||
| 4 | `truncate` | truncate an existing file to zero length on open (replace, don't overwrite in place) |
|
||||
|
||||
## FileStatus — 24 bytes (the `status` reply payload)
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
| 0 | 8 | `size` | file size in bytes |
|
||||
| 8 | 4 | `kind` | a **NodeKind** value |
|
||||
| 12 | 4 | — | padding |
|
||||
| 16 | 8 | `mtime` | modification time, Unix epoch seconds UTC; 0 if the backend keeps none |
|
||||
|
||||
## DirectoryEntry — 16 bytes + name (the `readdir` reply payload)
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
| 0 | 4 | `kind` | a **NodeKind** value |
|
||||
| 4 | 4 | `name_len` | length of the name that follows |
|
||||
| 8 | 8 | `size` | the entry's size in bytes |
|
||||
| 16 | `name_len` | name | the entry's name, not NUL-terminated |
|
||||
|
||||
## NodeKind
|
||||
|
||||
Aligned to the FSH file-type table
|
||||
(docs/danos-file-system-hierarchy-FSH.md):
|
||||
|
||||
| value | kind |
|
||||
|------:|------|
|
||||
| 0 | regular file |
|
||||
| 1 | directory |
|
||||
| 2 | character device |
|
||||
| 3 | block device |
|
||||
| 4 | symbolic link |
|
||||
| 5 | fifo |
|
||||
| 6 | socket |
|
||||
|
||||
Clients should map unknown values to *regular* rather than reject — the
|
||||
table can grow.
|
||||
|
||||
## Lifetimes and trust
|
||||
|
||||
Open-node ids live in the backend. A client that dies without closing leaks
|
||||
nothing permanently: the backend (the FAT server) subscribes to the kernel's
|
||||
published process-exit events (docs/process-lifecycle.md) and releases a dead
|
||||
client's handles. The kernel VFS root needs no sweep at all — its node tokens
|
||||
are permanent for a boot and carry no open state. Ids are plain integers, not
|
||||
capabilities — a backend trusts its callers with each other's ids today, which
|
||||
is acceptable while every client is part of the system image and worth
|
||||
revisiting (per-client id namespaces) before third-party binaries arrive.
|
||||
|
||||
## Evolution rules
|
||||
|
||||
What a non-Zig implementation may rely on, and what it must not:
|
||||
|
||||
- Operation values, flag bits, `NodeKind` values, and struct layouts are
|
||||
**append-only and frozen once shipped**. The unit test in
|
||||
`system/vfs-protocol.zig` pins a sample of them (the `DirectoryEntry`
|
||||
size, `NodeKind` 0–1, `Operation` values 0, 4 and 5); this page is the
|
||||
full record of the frozen values.
|
||||
- The 256-byte message ceiling is a property of the current IPC transport,
|
||||
not a promise; clients should read `maximum_payload`-shaped limits from the
|
||||
reply lengths they actually get (loop-until-done), not hard-code 224.
|
||||
- Negative statuses beyond -1 will appear (an errno vocabulary); success is
|
||||
exactly 0.
|
||||
-118
@@ -1,118 +0,0 @@
|
||||
# Vision: a microkernel, built to learn
|
||||
|
||||
danos exists first and foremost as a **learning-by-doing project**: the point is to
|
||||
build a real operating system, bump into the hard constraints for real, and research
|
||||
them from a position of having actually hit them. The docs in this folder are part of
|
||||
that — they're where a constraint gets understood once it's been met.
|
||||
|
||||
That framing sets the priorities. danos is not chasing a spec or a product; it's
|
||||
chasing understanding, with a concrete, motivating **win condition** to aim at.
|
||||
|
||||
## The win condition
|
||||
|
||||
danos is a "win" when it:
|
||||
|
||||
- **boots and runs on real hardware** — the author's **PC** (x86-64) and **both
|
||||
Raspberry Pis**: the **Zero 2 W** and the **Pi 5** (both `aarch64`, one backend —
|
||||
see [arm.md](arm.md)),
|
||||
- **has a graphical user interface**, ideally — building on the framebuffer it
|
||||
already draws to.
|
||||
|
||||
Everything below serves that, or serves the curiosity that the project runs on.
|
||||
|
||||
## Why a microkernel: resilience
|
||||
|
||||
The kernel stays **minimal** — only what genuinely must run privileged:
|
||||
|
||||
- scheduling,
|
||||
- inter-process communication (IPC),
|
||||
- memory management (address spaces, page tables),
|
||||
- low-level interrupt dispatch.
|
||||
|
||||
Everything else — device drivers, filesystems, the GUI, the network stack — runs as
|
||||
an **isolated user-space server**, each in its own address space with only the
|
||||
privileges it needs.
|
||||
|
||||
The reason for this shape is **resilience**: the ability to **re-initialise parts of
|
||||
the OS while it runs**. A driver bug can't corrupt the kernel or another driver; a
|
||||
crashed or wedged component is contained, killed, and **restarted** — "if I break
|
||||
something, I can just fix it," without rebooting. Keeping the kernel tiny is part of
|
||||
that strategy: the one thing that *can't* be restarted is the trusted base, so the
|
||||
less code in it, the less that can take the whole system down. This is the project's
|
||||
real motivation, and it has its own design note: [resilience.md](resilience.md).
|
||||
|
||||
The cost is that **IPC becomes the backbone**: what used to be a function call inside
|
||||
a monolithic kernel is now a message between address spaces. In a microkernel, IPC
|
||||
performance essentially *is* system performance (the lesson of L4), so it's a
|
||||
first-class concern. Hardware interrupts become IPC too: the kernel turns an IRQ into
|
||||
a message to the driver that owns the device.
|
||||
|
||||
## On real-time: an option, not a commitment
|
||||
|
||||
danos was originally framed as a hard **real-time** OS. That's now held as **one
|
||||
interesting constraint to explore, not a requirement** — because real-time is a
|
||||
*pervasive* invariant (every operation must be provably time-bounded, everywhere)
|
||||
that would slow every milestone, whereas resilience is a set of *structural* features
|
||||
that's lighter to build and is what the project actually wants. The trade-off is
|
||||
written up in [smp.md](smp.md#does-the-right-choice-depend-on-real-time-vs-resilience).
|
||||
|
||||
What danos keeps from the real-time direction, because it's cheap and useful anyway:
|
||||
|
||||
- **Fixed-priority preemptive scheduling** — the highest-priority ready task runs, and
|
||||
preemption lets a runaway component be interrupted and killed (which *serves
|
||||
resilience*). Already built ([scheduling.md](scheduling.md)).
|
||||
- **A calibrated, deterministic clock** — already built ([device-interrupts.md](device-interrupts.md)).
|
||||
|
||||
What danos does *not* owe anyone unless it deliberately chooses real-time later:
|
||||
timing *guarantees*, priority inheritance, bounded allocators, tickless timers, MCS
|
||||
scheduling contexts. Concretely, the current [heap](heap.md) is a first-fit free list
|
||||
with unbounded allocation time — fine here, and only a problem *if* a hard-real-time
|
||||
path is ever added. Note that **QNX is both** a real-time and a restartable
|
||||
microkernel, so choosing resilience now doesn't close the real-time door — it just
|
||||
doesn't pay the tax yet.
|
||||
|
||||
## The roadmap — tracks, not a strict line
|
||||
|
||||
Because the driver is curiosity plus the win condition, the roadmap is a set of
|
||||
**tracks** with dependencies, not a rigid sequence. Pick by interest; mind the
|
||||
prerequisites.
|
||||
|
||||
**Done:** UEFI boot, framebuffer + [serial](testing.md), [physical frames](frame-allocator.md)
|
||||
(with boot-services memory reclaimed), [paging](paging.md) with W^X, [exceptions and
|
||||
interrupts](interrupts.md), a [calibrated timer + ns clock](device-interrupts.md), a
|
||||
[heap](heap.md), a [fixed-priority preemptive scheduler](scheduling.md) with blocking,
|
||||
in-kernel [IPC channels](ipc.md), SMP (all cores scheduling, with affinity), a
|
||||
**higher-half kernel** with a physmap, and **user space**: per-process address
|
||||
spaces, `syscall`/`sysret` with the `swapgs` discipline, a user-ELF loader, and
|
||||
`/system/services/init` — a real user ELF built from `system/services/init/`, running at CPL 3 as PID 1 on its
|
||||
own page tables — plus a [test harness](testing.md).
|
||||
|
||||
- **Isolation track** — **user mode + address-space isolation**. *Done: a
|
||||
higher-half kernel with a physmap (the low half is user space), per-process
|
||||
address spaces with CR3 switched on context switch, the `swapgs` discipline,
|
||||
`syscall`/`sysret`, a user-ELF loader, an address-space/stack reaper for exited
|
||||
tasks, and `/system/services/init` running as a real preemptive ring-3 process
|
||||
(PID 1). Remaining polish: SMAP + fault-recovering copy-in/out, and TLB shootdown
|
||||
once a process has more than one thread. (The real IPC syscalls —
|
||||
`ipc_call`/`ipc_reply_wait` — have since been built and are the backbone every
|
||||
driver and service speaks; see [ipc.md](ipc.md).)*
|
||||
- **Resilience track** — fault → kill → notify, a supervisor/reincarnation server,
|
||||
resource cleanup on death, then a restartable driver as proof. Needs isolation.
|
||||
See [resilience.md](resilience.md).
|
||||
- **ARM track** — the `aarch64` port so danos runs on the Zero 2 W and Pi 5. Largely
|
||||
independent of the others (it's the [architecture layer](architecture.md)); directly serves the win
|
||||
condition. Likely via aarch64-UEFI first (QEMU `virt` + AAVMF), then real boards.
|
||||
See [arm.md](arm.md), and [discovery.md](discovery.md) for the device tree it needs.
|
||||
- **GUI track** — a framebuffer-based windowing/compositor, and the input + display
|
||||
drivers under it. Builds on the neutral framebuffer (so it's arch-independent), and
|
||||
on the driver model from the isolation/resilience tracks. The visible payoff.
|
||||
|
||||
The natural spine is **isolation → (resilience + drivers) → GUI**, with the **ARM
|
||||
track** pursued alongside whenever the itch to see it boot on a Pi wins out.
|
||||
|
||||
## How to use this page
|
||||
|
||||
Read it before adding anything structural. When a design decision comes up, the
|
||||
question is: does it serve the **win condition** (runs on the three machines, with a
|
||||
GUI), or the **learning** (a constraint worth meeting)? If it serves neither — e.g.
|
||||
paying the full real-time tax with no payoff in sight — it can wait.
|
||||
@@ -108,7 +108,7 @@ localised (below).
|
||||
## The architecture decision: `runtime.os` + `runtime.fs`, and retire `posix`
|
||||
|
||||
danos already has the right split ([the private-ABI boundary](../README.md)): the
|
||||
kernel exposes a minimal syscall ABI ([syscall.md](syscall.md)); the **`runtime`**
|
||||
kernel exposes a minimal syscall ABI ([syscall.md](os-development/syscall.md)); the **`runtime`**
|
||||
library is the stable, danos-native application ABI. What this roadmap adds:
|
||||
|
||||
- **`runtime.os` — the seam.** A C-ABI-shaped module of the ~30 operations
|
||||
@@ -165,7 +165,7 @@ What the seam needs, and what danos already provides:
|
||||
| mmap / munmap | native syscalls ([abi.zig](../system/abi.zig)) | none |
|
||||
| page allocator | over `mmap`, via `root.os.heap.page_allocator` override | ~30-line hook |
|
||||
| monotonic clock | `clock` syscall | none |
|
||||
| args / argv | SysV entry stack ([sysv.md](sysv.md)), `runtime.process.Init` | none |
|
||||
| args / argv | SysV entry stack ([sysv.md](os-development/sysv.md)), `runtime.process.Init` | none |
|
||||
| stdout / stderr | `debug_write` today | wire fd 1/2 to a console **byte** stream |
|
||||
| mkdir / unlink / rename / truncate | done — engine + VFS + `runtime.fs` (Phase 2) | — |
|
||||
| stat fields | `{size, kind, mtime}` | **mode / inode** still missing (cache validity) |
|
||||
@@ -209,7 +209,7 @@ build); point danos's `build.zig`/CI at the resulting binary. Four localised pat
|
||||
plan9/serenity;
|
||||
- add `danos` to the freestanding/other **no-op `_start` list** in `std`'s `start.zig`,
|
||||
so std does *not* emit its own System-V `_start` — danos keeps owning the entry shim
|
||||
and `Init`/argv construction it already builds ([sysv.md](sysv.md));
|
||||
and `Init`/argv construction it already builds ([sysv.md](os-development/sysv.md));
|
||||
- wire the `system` selector `.danos => std.os.danos` in `std.posix`;
|
||||
- add `std/os/danos.zig` — **the seam itself**, promoted near-verbatim from the
|
||||
`runtime.os` developed first in Phase 1 (against the stock toolchain, so the fork is
|
||||
@@ -243,7 +243,7 @@ readdir/isatty/args/exit) exists. Those are downstream and out of scope here.
|
||||
danos's biggest genuine gap, and the correctness-critical one:
|
||||
|
||||
- Add **mkdir / unlink / rename / truncate** to *both* the VFS wire protocol
|
||||
([vfs-protocol.zig](../system/vfs-protocol.zig)) and the FAT engine
|
||||
([vfs-protocol.zig](../library/protocol/vfs/vfs-protocol.zig)) and the FAT engine
|
||||
([engine.zig](../system/services/fat/engine.zig)), then expose them via `runtime.os`.
|
||||
- Extend `stat` beyond `{size, kind}` to carry **mtime + inode + mode** — `std`'s file
|
||||
stat needs them for build-cache validity — which in turn needs **wall-clock** time
|
||||
@@ -344,10 +344,10 @@ Two current decisions fall out of this roadmap:
|
||||
## Related
|
||||
|
||||
- [vision.md](vision.md) — the north star this serves.
|
||||
- [syscall.md](syscall.md) — the kernel↔runtime ABI `runtime.os` is built on.
|
||||
- [sysv.md](sysv.md) — the entry stack (`argc/argv/envp/auxv`) danos already constructs.
|
||||
- [ipc.md](ipc.md) — the IPC the VFS/FAT operations travel over.
|
||||
- [danos-file-system-hierarchy-FSH.md](danos-file-system-hierarchy-FSH.md) — the
|
||||
- [syscall.md](os-development/syscall.md) — the kernel↔runtime ABI `runtime.os` is built on.
|
||||
- [sysv.md](os-development/sysv.md) — the entry stack (`argc/argv/envp/auxv`) danos already constructs.
|
||||
- [ipc.md](device-driver-development/ipc.md) — the IPC the VFS/FAT operations travel over.
|
||||
- [file-system-hierarchy.md](file-system-development/file-system-hierarchy.md) — the
|
||||
filesystem layout the file surface serves.
|
||||
- [coding-standards.md](coding-standards.md) — danos naming (why the compat spellings
|
||||
are confined, and now retired).
|
||||
|
||||
@@ -0,0 +1,50 @@
|
||||
//! The "client" library domain (library/client): userspace-service clients —
|
||||
//! they talk to services over IPC, not to the kernel. Client modules end in
|
||||
//! `-client` the way wire protocols end in `-protocol`, so a service, its
|
||||
//! protocol, and its client never share a name (`display` the service,
|
||||
//! `display-protocol` the wire contract, `display-client` a program's view).
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const kernel = b.dependency("kernel", .{});
|
||||
const protocol = b.dependency("protocol", .{});
|
||||
|
||||
const ipc = kernel.module("ipc");
|
||||
const time = kernel.module("time");
|
||||
// Every client reaches its service by name now: resolve `/protocol/<name>`,
|
||||
// open it, and take the provider's endpoint out of the reply
|
||||
// (docs/os-development/protocol-namespace.md).
|
||||
const channel = kernel.module("channel");
|
||||
|
||||
// A client frames its own packets, so it needs the envelope alongside the
|
||||
// protocol whose verbs it speaks.
|
||||
const envelope = protocol.module("envelope");
|
||||
|
||||
_ = b.addModule("display-client", .{
|
||||
.root_source_file = b.path("display/display-client.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "channel", .module = channel },
|
||||
.{ .name = "envelope", .module = envelope },
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "time", .module = time },
|
||||
.{ .name = "display-protocol", .module = protocol.module("display-protocol") },
|
||||
},
|
||||
});
|
||||
_ = b.addModule("input-client", .{
|
||||
.root_source_file = b.path("input/input-client.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "channel", .module = channel },
|
||||
.{ .name = "envelope", .module = envelope },
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "time", .module = time },
|
||||
.{ .name = "input-protocol", .module = protocol.module("input-protocol") },
|
||||
},
|
||||
});
|
||||
|
||||
// Standalone `zig build test`, kept for uniformity across the domains (the
|
||||
// root aggregate depends on every domain's test step). The clients have no
|
||||
// host-runnable unit tests yet — they are thin IPC conversation wrappers —
|
||||
// so the step is empty until one grows some.
|
||||
_ = b.step("test", "Run the client unit tests (none yet)");
|
||||
}
|
||||
@@ -0,0 +1,13 @@
|
||||
.{
|
||||
.name = .client,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0xc74404553e73d4ff, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// The clients converse over ipc with time-bounded waits.
|
||||
.kernel = .{ .path = "../kernel" },
|
||||
// Each client speaks its service's wire protocol.
|
||||
.protocol = .{ .path = "../protocol" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -0,0 +1,165 @@
|
||||
//! User-space display client: talk to the display service (query the mode, and — from D3
|
||||
//! — create layers, draw, and present) without hand-rolling the IPC. The `runtime.block`
|
||||
//! shape: a cached `/protocol/display` open with a boot-race retry, then extern-struct request/
|
||||
//! reply marshalling. See system/services/display/ and docs/display.md.
|
||||
|
||||
const std = @import("std");
|
||||
const channel = @import("channel");
|
||||
const envelope = @import("envelope");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const display_protocol = @import("display-protocol");
|
||||
|
||||
const Protocol = display_protocol.Protocol;
|
||||
|
||||
/// The display's current mode, as `info()` reports it.
|
||||
pub const Info = struct {
|
||||
width: u32,
|
||||
height: u32,
|
||||
pitch: u32, // bytes per row (may exceed width*4; see docs/framebuffer.md)
|
||||
format: u32, // a device-abi DisplayFormat value (0 = rgbx, 1 = bgrx)
|
||||
};
|
||||
|
||||
/// The service endpoint, looked up once and cached.
|
||||
var handle: ?ipc.Handle = null;
|
||||
|
||||
/// Open `/protocol/display`, retrying while it comes up (a client races the
|
||||
/// service's bind at boot). Returns the endpoint, or null if it never appears.
|
||||
fn service() ?ipc.Handle {
|
||||
if (handle) |h| return h;
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 100) : (attempts += 1) {
|
||||
if (channel.openEndpoint("display")) |h| {
|
||||
handle = h;
|
||||
return h;
|
||||
}
|
||||
time.sleepMillis(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// A reply the compositor answered with, kept whole so the caller can decode the
|
||||
/// verb's own fixed part out of it.
|
||||
const Answered = struct {
|
||||
packet: [display_protocol.message_maximum]u8,
|
||||
len: usize,
|
||||
|
||||
fn bytes(self: *const Answered) []const u8 {
|
||||
return self.packet[0..self.len];
|
||||
}
|
||||
};
|
||||
|
||||
/// Send one request (`target` addresses a layer, or 0 for the compositor itself)
|
||||
/// and keep the reply. Null when the transport failed or the compositor refused.
|
||||
fn transact(
|
||||
comptime operation: Protocol.Operation,
|
||||
target: u64,
|
||||
request: Protocol.RequestOf(operation),
|
||||
tail: []const u8,
|
||||
) ?Answered {
|
||||
const h = service() orelse return null;
|
||||
var packet: [display_protocol.message_maximum]u8 = undefined;
|
||||
const framed = Protocol.encodeRequest(operation, target, request, tail, &packet) orelse return null;
|
||||
var answered: Answered = .{ .packet = undefined, .len = 0 };
|
||||
answered.len = ipc.call(h, framed, &answered.packet) catch return null;
|
||||
const status = envelope.statusOf(answered.bytes()) orelse return null;
|
||||
if (status.status != 0) return null;
|
||||
return answered;
|
||||
}
|
||||
|
||||
/// The display's current mode, or null if the service never came up.
|
||||
pub fn info() ?Info {
|
||||
const answered = transact(.info, 0, {}, &.{}) orelse return null;
|
||||
const reply = Protocol.decodeReply(.info, answered.bytes()) orelse return null;
|
||||
return .{ .width = reply.width, .height = reply.height, .pitch = reply.pitch, .format = reply.format };
|
||||
}
|
||||
|
||||
/// Composite the dirty layers and flush the frame to the screen.
|
||||
pub fn present() bool {
|
||||
return transact(.present, 0, {}, &.{}) != null;
|
||||
}
|
||||
|
||||
/// One selectable display mode.
|
||||
pub const Mode = display_protocol.Mode;
|
||||
|
||||
/// Fill `out` with the resolutions the display can switch to; returns how many were written
|
||||
/// (zero on the GOP floor, or if the service never came up).
|
||||
pub fn modes(out: []Mode) usize {
|
||||
const answered = transact(.get_modes, 0, {}, &.{}) orelse return 0;
|
||||
const offered = Protocol.decodeReply(.get_modes, answered.bytes()) orelse return 0;
|
||||
const count = @min(@min(offered.count, display_protocol.max_modes), out.len);
|
||||
for (0..count) |i| out[i] = offered.modes[i];
|
||||
return count;
|
||||
}
|
||||
|
||||
/// Change the display resolution. Only a native backend that supports mode-setting honours it
|
||||
/// (on the GOP floor it returns false); on success the display's `info()` reports the new mode.
|
||||
pub fn setMode(width: u32, height: u32) bool {
|
||||
const changed = transact(.set_mode, 0, .{ .width = width, .height = height }, &.{}) != null;
|
||||
if (changed) mode = null; // the cached mode is stale now
|
||||
return changed;
|
||||
}
|
||||
|
||||
/// The mode, cached after the first `info()` so `color()` doesn't round-trip per pixel.
|
||||
var mode: ?Info = null;
|
||||
|
||||
fn cachedInfo() ?Info {
|
||||
if (mode) |m| return m;
|
||||
const i = info() orelse return null;
|
||||
mode = i;
|
||||
return i;
|
||||
}
|
||||
|
||||
/// The native pixel value for an 8-bit-per-channel colour, in the display's format. A
|
||||
/// client packs colours through this so it never has to know the byte order itself.
|
||||
pub fn color(r: u8, g: u8, b: u8) u32 {
|
||||
const format = if (cachedInfo()) |i| i.format else 0;
|
||||
return display_protocol.pack(format, r, g, b);
|
||||
}
|
||||
|
||||
/// A handle to a server-owned layer: a positioned, z-ordered surface the client draws
|
||||
/// into by command. Create with `createLayer`; drawing and moves take effect on the next
|
||||
/// `present`. Coordinates are signed (a layer may sit partly off-screen).
|
||||
///
|
||||
/// The id is the packet header's `target` on every call below, so it is named once
|
||||
/// per request rather than repeated inside one.
|
||||
pub const Layer = struct {
|
||||
id: u32,
|
||||
|
||||
/// Fill a rectangle of this layer (layer-local coordinates) with a native `colour`.
|
||||
pub fn fill(self: Layer, x: i32, y: i32, w: u32, h: u32, colour: u32) bool {
|
||||
return transact(.fill_rect, self.id, .{ .x = x, .y = y, .width = w, .height = h, .colour = colour }, &.{}) != null;
|
||||
}
|
||||
|
||||
/// Copy a `w`×`h` tile of native pixels (row-major, little-endian bytes) into this
|
||||
/// layer at (`x`, `y`). The tile rides inline as the request's tail, so `w*h*4` must
|
||||
/// fit `display_protocol.maximum_payload` — the bound the protocol derives from this
|
||||
/// verb's own fixed part, so the check here can never drift from what fits.
|
||||
pub fn blitTile(self: Layer, x: i32, y: i32, w: u32, h: u32, pixels: []const u8) bool {
|
||||
if (pixels.len > display_protocol.maximum_payload) return false;
|
||||
return transact(.blit_tile, self.id, .{ .x = x, .y = y, .width = w, .height = h }, pixels) != null;
|
||||
}
|
||||
|
||||
/// Move / restack / show or hide the layer.
|
||||
pub fn configure(self: Layer, x: i32, y: i32, z: u32, visible: bool) bool {
|
||||
return transact(.configure_layer, self.id, .{ .x = x, .y = y, .z = z, .visible = if (visible) 1 else 0 }, &.{}) != null;
|
||||
}
|
||||
|
||||
/// Mark a rectangle of this layer (layer-local) dirty for the next present — for when
|
||||
/// the layer's pixels changed without a drawing call the compositor already tracked.
|
||||
pub fn damage(self: Layer, x: i32, y: i32, w: u32, h: u32) bool {
|
||||
return transact(.damage, self.id, .{ .x = x, .y = y, .width = w, .height = h }, &.{}) != null;
|
||||
}
|
||||
|
||||
/// Release the layer and its surface.
|
||||
pub fn destroy(self: Layer) bool {
|
||||
return transact(.destroy_layer, self.id, {}, &.{}) != null;
|
||||
}
|
||||
};
|
||||
|
||||
/// Create a server-owned layer of `w`×`h` pixels at screen (`x`, `y`) with stacking order
|
||||
/// `z` (higher is nearer the front), initially visible. Returns a handle, or null.
|
||||
pub fn createLayer(x: i32, y: i32, w: u32, h: u32, z: u32) ?Layer {
|
||||
const answered = transact(.create_layer, 0, .{ .x = x, .y = y, .width = w, .height = h, .z = z, .visible = 1 }, &.{}) orelse return null;
|
||||
return .{ .id = (Protocol.decodeReply(.create_layer, answered.bytes()) orelse return null).layer };
|
||||
}
|
||||
@@ -21,37 +21,39 @@
|
||||
//! if (event.asKeyboard()) |k| { ... } else if (event.asMouse()) |m| { ... }
|
||||
//! }
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const ipc = @import("ipc.zig");
|
||||
const system = @import("system.zig");
|
||||
const protocol = @import("input-protocol");
|
||||
const channel = @import("channel");
|
||||
const envelope = @import("envelope");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const input_protocol = @import("input-protocol");
|
||||
|
||||
pub const DeviceKind = protocol.DeviceKind;
|
||||
pub const InputEvent = protocol.InputEvent;
|
||||
pub const KeyEvent = protocol.KeyEvent;
|
||||
pub const MouseEvent = protocol.MouseEvent;
|
||||
pub const JoystickEvent = protocol.JoystickEvent;
|
||||
pub const EventKind = protocol.EventKind;
|
||||
pub const MouseEventKind = protocol.MouseEventKind;
|
||||
pub const JoystickEventKind = protocol.JoystickEventKind;
|
||||
pub const Keycode = protocol.Keycode;
|
||||
const Protocol = input_protocol.Protocol;
|
||||
|
||||
pub const DeviceKind = input_protocol.DeviceKind;
|
||||
pub const InputEvent = input_protocol.InputEvent;
|
||||
pub const KeyEvent = input_protocol.KeyEvent;
|
||||
pub const MouseEvent = input_protocol.MouseEvent;
|
||||
pub const JoystickEvent = input_protocol.JoystickEvent;
|
||||
pub const EventKind = input_protocol.EventKind;
|
||||
pub const MouseEventKind = input_protocol.MouseEventKind;
|
||||
pub const JoystickEventKind = input_protocol.JoystickEventKind;
|
||||
pub const Keycode = input_protocol.Keycode;
|
||||
|
||||
/// Interest masks re-exported so a caller can `subscribe(input.device_keyboard |
|
||||
/// input.device_mouse)`.
|
||||
pub const device_keyboard = protocol.device_keyboard;
|
||||
pub const device_mouse = protocol.device_mouse;
|
||||
pub const device_joystick = protocol.device_joystick;
|
||||
pub const device_all = protocol.device_all;
|
||||
pub const device_keyboard = input_protocol.device_keyboard;
|
||||
pub const device_mouse = input_protocol.device_mouse;
|
||||
pub const device_joystick = input_protocol.device_joystick;
|
||||
pub const device_all = input_protocol.device_all;
|
||||
|
||||
/// Look up the input service, retrying while it is still coming up. Both a subscriber and
|
||||
/// a source race the service's registration at boot, so both wait for it here rather than
|
||||
/// failing. Returns the service endpoint handle, or null if it never appears.
|
||||
/// Open `/protocol/input`, retrying while it is still coming up. Both a subscriber and
|
||||
/// a source race the service's bind at boot, so both wait for it here rather than
|
||||
/// failing. Returns the provider's endpoint handle, or null if it never appears.
|
||||
fn lookupService() ?ipc.Handle {
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 100) : (attempts += 1) {
|
||||
if (ipc.lookup(.input)) |handle| return handle;
|
||||
system.sleep(50);
|
||||
if (channel.openEndpoint("input")) |handle| return handle;
|
||||
time.sleepMillis(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
@@ -66,31 +68,44 @@ pub const Subscriber = struct {
|
||||
/// The endpoint the service delivers events to (created and owned by us; its handle
|
||||
/// was handed to the service as a capability at subscribe time).
|
||||
endpoint: ipc.Handle,
|
||||
receive: [protocol.event_size]u8 = undefined,
|
||||
/// A pushed packet is the folded header plus one typed event, so the buffer is
|
||||
/// the push floor rather than any one event's size.
|
||||
receive: [envelope.post_maximum]u8 = undefined,
|
||||
|
||||
/// Block until the next event is pushed, and return it. Events arrive as asynchronous
|
||||
/// buffered messages (`ipc_send` from the service), so nothing is owed in reply — the
|
||||
/// empty reply this issues is a harmless no-op. Returns null for any non-event wake-up
|
||||
/// (there should be none), so callers can loop.
|
||||
///
|
||||
/// The device class is the packet's operation, so it is read from the header and
|
||||
/// re-tagged into an `InputEvent` here — one decoded type for a caller that took
|
||||
/// several classes on one stream.
|
||||
pub fn next(self: *Subscriber) ?InputEvent {
|
||||
const got = ipc.replyWait(self.endpoint, &.{}, &self.receive, null);
|
||||
if (!got.isMessage() or got.len < protocol.event_size) return null;
|
||||
return std.mem.bytesToValue(InputEvent, self.receive[0..protocol.event_size]);
|
||||
if (!got.isMessage()) return null;
|
||||
const packet = self.receive[0..@min(got.len, self.receive.len)];
|
||||
return switch (Protocol.eventOf(packet) orelse return null) {
|
||||
.keyboard => InputEvent.fromKeyboard(Protocol.decodeEvent(.keyboard, packet) orelse return null),
|
||||
.mouse => InputEvent.fromMouse(Protocol.decodeEvent(.mouse, packet) orelse return null),
|
||||
.joystick => InputEvent.fromJoystick(Protocol.decodeEvent(.joystick, packet) orelse return null),
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
/// Subscribe to the input classes named in `device_mask` (an OR of `device_*`, or
|
||||
/// `device_all`). Creates an endpoint for the service to push to and hands it over as a
|
||||
/// capability. Returns a `Subscriber` to loop `next` on, or null on failure.
|
||||
/// capability — the envelope's reserved `subscribe`, whose shape this is exactly. Returns
|
||||
/// a `Subscriber` to loop `next` on, or null on failure.
|
||||
pub fn subscribe(device_mask: u32) ?Subscriber {
|
||||
const service = lookupService() orelse return null;
|
||||
const endpoint = ipc.createIpcEndpoint() orelse return null;
|
||||
|
||||
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.subscribe), .device_mask = device_mask };
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
const result = ipc.callCap(service, std.mem.asBytes(&request), &reply, endpoint) catch return null;
|
||||
if (result.len < protocol.reply_size) return null;
|
||||
if (std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status != 0) return null;
|
||||
var packet: [input_protocol.message_maximum]u8 = undefined;
|
||||
const framed = input_protocol.encodeSubscribe(device_mask, &packet) orelse return null;
|
||||
var reply: [input_protocol.message_maximum]u8 = undefined;
|
||||
const result = ipc.callCap(service, framed, &reply, endpoint) catch return null;
|
||||
const status = envelope.statusOf(reply[0..result.len]) orelse return null;
|
||||
if (status.status != 0) return null;
|
||||
return .{ .endpoint = endpoint };
|
||||
}
|
||||
|
||||
@@ -149,11 +164,12 @@ pub const Publisher = struct {
|
||||
service: ipc.Handle,
|
||||
|
||||
fn publish(self: Publisher, event: InputEvent) bool {
|
||||
var request = protocol.Request{ .operation = @intFromEnum(protocol.Operation.publish), .event = event };
|
||||
var reply: [protocol.reply_size]u8 = undefined;
|
||||
const len = ipc.call(self.service, std.mem.asBytes(&request), &reply) catch return false;
|
||||
if (len < protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(protocol.Reply, reply[0..protocol.reply_size]).status == 0;
|
||||
var packet: [input_protocol.message_maximum]u8 = undefined;
|
||||
const framed = Protocol.encodeRequest(.publish, 0, event, &.{}, &packet) orelse return false;
|
||||
var reply: [input_protocol.message_maximum]u8 = undefined;
|
||||
const len = ipc.call(self.service, framed, &reply) catch return false;
|
||||
const status = envelope.statusOf(reply[0..len]) orelse return false;
|
||||
return status.status == 0;
|
||||
}
|
||||
|
||||
/// Broadcast a keyboard event to every subscriber that took keyboard events.
|
||||
@@ -204,8 +220,8 @@ pub fn syntheticKeyEvent(step: usize) KeyEvent {
|
||||
pub fn syntheticMouseEvent(step: usize) MouseEvent {
|
||||
return switch (step % 3) {
|
||||
0 => .{ .kind = @intFromEnum(MouseEventKind.motion), .button = 0, .dx = 1, .dy = 1, .scroll_x = 0, .scroll_y = 0, .buttons = 0 },
|
||||
1 => .{ .kind = @intFromEnum(MouseEventKind.button_down), .button = protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = protocol.mouse_button_left },
|
||||
else => .{ .kind = @intFromEnum(MouseEventKind.button_up), .button = protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = 0 },
|
||||
1 => .{ .kind = @intFromEnum(MouseEventKind.button_down), .button = input_protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = input_protocol.mouse_button_left },
|
||||
else => .{ .kind = @intFromEnum(MouseEventKind.button_up), .button = input_protocol.mouse_button_left, .dx = 0, .dy = 0, .scroll_x = 0, .scroll_y = 0, .buttons = 0 },
|
||||
};
|
||||
}
|
||||
|
||||
@@ -0,0 +1,20 @@
|
||||
//! The "csv" library domain: shared CSV helpers (comment stripping, field
|
||||
//! iteration) for the /system/configuration/*.csv config files — the device registry and the
|
||||
//! init service list both parse them.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
_ = b.addModule("csv", .{ .root_source_file = b.path("csv.zig") });
|
||||
|
||||
// Standalone `zig build test` for this domain alone; the root build keeps
|
||||
// its aggregate test step.
|
||||
const test_step = b.step("test", "Run the csv unit tests");
|
||||
const csv_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("csv.zig"),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(csv_tests).step);
|
||||
}
|
||||
@@ -0,0 +1,8 @@
|
||||
.{
|
||||
.name = .csv,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x8a4525791f4e5b6, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -0,0 +1,56 @@
|
||||
//! Minimal CSV helpers shared by the `/system/configuration/*.csv` config files — the device
|
||||
//! registry (`/system/configuration/devices.csv`) and the init service list (`/system/configuration/init.csv`).
|
||||
//! Freestanding, no allocator: returned fields are slices into the source line,
|
||||
//! so the source must outlive them. `#` starts a comment (whole-line or trailing);
|
||||
//! whitespace around a field is trimmed, so columns may be padded for alignment.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// Strip a trailing `#` comment and surrounding whitespace from one raw line.
|
||||
/// A blank or comment-only line returns "" (length 0) — the caller's skip signal.
|
||||
pub fn stripComment(raw: []const u8) []const u8 {
|
||||
const body = if (std.mem.indexOfScalar(u8, raw, '#')) |hash| raw[0..hash] else raw;
|
||||
return std.mem.trim(u8, body, " \t\r\n");
|
||||
}
|
||||
|
||||
/// Iterate the comma-separated fields of a line body, each trimmed of spaces and
|
||||
/// tabs. Build it from a `stripComment`ed body.
|
||||
pub const Fields = struct {
|
||||
inner: std.mem.SplitIterator(u8, .scalar),
|
||||
|
||||
/// The next field, trimmed, or null when the row is exhausted.
|
||||
pub fn next(self: *Fields) ?[]const u8 {
|
||||
const field = self.inner.next() orelse return null;
|
||||
return std.mem.trim(u8, field, " \t");
|
||||
}
|
||||
};
|
||||
|
||||
pub fn fields(body: []const u8) Fields {
|
||||
return .{ .inner = std.mem.splitScalar(u8, body, ',') };
|
||||
}
|
||||
|
||||
// --- tests -------------------------------------------------------------------
|
||||
|
||||
const testing = std.testing;
|
||||
|
||||
test "stripComment trims and drops comments" {
|
||||
try testing.expectEqualStrings("a, b", stripComment(" a, b # trailing\r\n"));
|
||||
try testing.expectEqualStrings("", stripComment(" # whole-line comment"));
|
||||
try testing.expectEqualStrings("", stripComment(" \t "));
|
||||
try testing.expectEqualStrings("x", stripComment("x"));
|
||||
}
|
||||
|
||||
test "fields splits and trims each column" {
|
||||
var it = fields(stripComment("pci, 03 , 80 , /system/drivers/x # note"));
|
||||
try testing.expectEqualStrings("pci", it.next().?);
|
||||
try testing.expectEqualStrings("03", it.next().?);
|
||||
try testing.expectEqualStrings("80", it.next().?);
|
||||
try testing.expectEqualStrings("/system/drivers/x", it.next().?);
|
||||
try testing.expect(it.next() == null);
|
||||
}
|
||||
|
||||
test "a single field yields one column then null" {
|
||||
var it = fields(stripComment("/system/services/input"));
|
||||
try testing.expectEqualStrings("/system/services/input", it.next().?);
|
||||
try testing.expect(it.next() == null);
|
||||
}
|
||||
@@ -0,0 +1,100 @@
|
||||
//! Block-device client: the helper a filesystem uses to read and write a block
|
||||
//! device (a USB stick, via usb-storage) without hand-rolling the block-protocol
|
||||
//! IPC. Layered over `ipc` and the shared `block-protocol` wire format, like
|
||||
//! `runtime.usb` over the transfer protocol.
|
||||
//!
|
||||
//! Transfers name a caller-owned DMA buffer by physical address (from
|
||||
//! `runtime.dma.alloc`), so whole sectors move without crossing the IPC size
|
||||
//! limit — the same handoff usb-storage uses toward the controller.
|
||||
|
||||
const channel = @import("channel");
|
||||
const envelope = @import("envelope");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const block_protocol = @import("block-protocol");
|
||||
|
||||
const Protocol = block_protocol.Protocol;
|
||||
|
||||
pub const Geometry = struct { block_size: u32, block_count: u64 };
|
||||
|
||||
pub const Device = struct {
|
||||
endpoint: ipc.Handle,
|
||||
|
||||
/// The device's block size and total block count.
|
||||
pub fn geometry(self: Device) ?Geometry {
|
||||
var reply: [block_protocol.message_maximum]u8 = undefined;
|
||||
const answered = self.call(.geometry, {}, null, &reply) orelse return null;
|
||||
const result = Protocol.decodeReply(.geometry, answered) orelse return null;
|
||||
return .{ .block_size = result.block_size, .block_count = result.block_count };
|
||||
}
|
||||
|
||||
/// Hand the block server a DMA-region capability (`handle` — from a `shareable`
|
||||
/// dma_alloc) so it forwards it to the controller and the buffer's physical
|
||||
/// addresses become reachable by the device. Call once per buffer before naming it
|
||||
/// in `read`/`write`. Harmless success when no IOMMU is enforcing.
|
||||
pub fn attach(self: Device, handle: ipc.Handle) bool {
|
||||
var reply: [block_protocol.message_maximum]u8 = undefined;
|
||||
return self.call(.attach, {}, handle, &reply) != null;
|
||||
}
|
||||
|
||||
/// Read `count` blocks starting at `lba` into the DMA buffer at `physical`.
|
||||
pub fn read(self: Device, lba: u64, count: u32, physical: u64) bool {
|
||||
var reply: [block_protocol.message_maximum]u8 = undefined;
|
||||
return self.call(.read, .{ .lba = lba, .count = count, .physical = physical }, null, &reply) != null;
|
||||
}
|
||||
|
||||
/// Write `count` blocks starting at `lba` from the DMA buffer at `physical`.
|
||||
pub fn write(self: Device, lba: u64, count: u32, physical: u64) bool {
|
||||
var reply: [block_protocol.message_maximum]u8 = undefined;
|
||||
return self.call(.write, .{ .lba = lba, .count = count, .physical = physical }, null, &reply) != null;
|
||||
}
|
||||
|
||||
/// Commit any device write cache to stable media (SCSI SYNCHRONIZE CACHE), so
|
||||
/// prior writes survive a power-off. A filesystem calls this before the machine
|
||||
/// goes down; no data transfer, so the buffer arguments are unused.
|
||||
pub fn flush(self: Device) bool {
|
||||
var reply: [block_protocol.message_maximum]u8 = undefined;
|
||||
return self.call(.flush, {}, null, &reply) != null;
|
||||
}
|
||||
|
||||
/// One request at the driver. `target` is always 0: one endpoint per device, so
|
||||
/// there is no object within the peer to address.
|
||||
fn call(
|
||||
self: Device,
|
||||
comptime operation: Protocol.Operation,
|
||||
request: Protocol.RequestOf(operation),
|
||||
capability: ?ipc.Handle,
|
||||
reply: []u8,
|
||||
) ?[]u8 {
|
||||
var packet: [block_protocol.message_maximum]u8 = undefined;
|
||||
const framed = Protocol.encodeRequest(operation, 0, request, &.{}, &packet) orelse return null;
|
||||
const answer = ipc.callCap(self.endpoint, framed, reply, capability) catch return null;
|
||||
const status = envelope.statusOf(reply[0..answer.len]) orelse return null;
|
||||
if (status.status != 0) return null;
|
||||
return reply[0..answer.len];
|
||||
}
|
||||
};
|
||||
|
||||
/// One open attempt, no waiting — for a server that retries on its own
|
||||
/// timer (the fat service) instead of blocking its harness in here.
|
||||
pub fn tryOpen() ?Device {
|
||||
if (channel.openEndpoint("block")) |handle| return .{ .endpoint = handle };
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Open `/protocol/block`, retrying generously while the USB storage chain
|
||||
/// (controller reset, enumeration, mass-storage bring-up) comes up.
|
||||
pub fn open() ?Device {
|
||||
// Patient: the whole USB storage chain (firmware discovery, xHCI reset and
|
||||
// enumeration, mass-storage bring-up) must complete first, which can take
|
||||
// tens of seconds under emulation.
|
||||
var attempts: usize = 0;
|
||||
// 30 s covers the slowest observed healthy chain (a flaky QEMU enumeration
|
||||
// completed at ~24 s); a machine whose stick genuinely failed setup should
|
||||
// not sit a further minute pretending otherwise.
|
||||
while (attempts < 600) : (attempts += 1) {
|
||||
if (channel.openEndpoint("block")) |handle| return .{ .endpoint = handle };
|
||||
time.sleepMillis(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
@@ -0,0 +1,147 @@
|
||||
//! The "device" library domain (library/device): what a driver author imports.
|
||||
//! The flat reference data (device-abi, pci-class, acpi-ids, usb-abi, usb-ids),
|
||||
//! typed MMIO access, the driver-side client libraries (driver, pci, usb,
|
||||
//! block), the AML interpreter, and the data-driven device registry.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const kernel = b.dependency("kernel", .{});
|
||||
const protocol = b.dependency("protocol", .{});
|
||||
const csv = b.dependency("csv", .{});
|
||||
|
||||
const abi = kernel.module("abi");
|
||||
const system_call = kernel.module("system-call");
|
||||
const ipc = kernel.module("ipc");
|
||||
const time = kernel.module("time");
|
||||
// A driver finds the bus it attaches to by name — `/protocol/device-manager`,
|
||||
// `/protocol/usb-transfer`, `/protocol/block`
|
||||
// (docs/os-development/protocol-namespace.md).
|
||||
const channel = kernel.module("channel");
|
||||
|
||||
// The devices sub-project's public interface (the flat wire types),
|
||||
// importable by user space, unlike the kernel-internal device model it
|
||||
// also feeds (system/kernel/device-model.zig).
|
||||
const device_abi = b.addModule("device-abi", .{
|
||||
.root_source_file = b.path("model/device-abi.zig"),
|
||||
});
|
||||
// PCI class-code decoding (class/subclass/prog-IF -> names). Pure reference
|
||||
// data, shared by kernel discovery and any user-space PCI tool.
|
||||
const pci_class = b.addModule("pci-class", .{
|
||||
.root_source_file = b.path("pci/pci-class.zig"),
|
||||
});
|
||||
// ACPI/PnP hardware-ID (_HID) names — the flat analog of pci-class.
|
||||
_ = b.addModule("acpi-ids", .{
|
||||
.root_source_file = b.path("acpi/acpi-ids.zig"),
|
||||
});
|
||||
// The AML interpreter, a build module so the ring-3 acpi service can run
|
||||
// the same parser the kernel does (docs/discovery.md). Pure Zig, no kernel
|
||||
// imports — one source, two builds.
|
||||
_ = b.addModule("aml", .{
|
||||
.root_source_file = b.path("acpi/aml/aml.zig"),
|
||||
});
|
||||
// The USB device-framework wire ABI (chapter-9 set-up packets, standard +
|
||||
// class requests, descriptors) and the USB class-code taxonomy.
|
||||
const usb_abi = b.addModule("usb-abi", .{
|
||||
.root_source_file = b.path("usb/usb-abi.zig"),
|
||||
});
|
||||
const usb_ids = b.addModule("usb-ids", .{
|
||||
.root_source_file = b.path("usb/usb-ids.zig"),
|
||||
});
|
||||
// Typed volatile MMIO register access + memory-ordering barriers, for
|
||||
// drivers on top of an mmio_map grant. Depends only on `builtin`.
|
||||
const mmio = b.addModule("mmio", .{
|
||||
.root_source_file = b.path("mmio/mmio.zig"),
|
||||
});
|
||||
// The driver author's interface: device access + the device-manager hello
|
||||
// handshake, folded together.
|
||||
const driver = b.addModule("driver", .{
|
||||
.root_source_file = b.path("driver/driver.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi },
|
||||
.{ .name = "channel", .module = channel },
|
||||
.{ .name = "device-abi", .module = device_abi },
|
||||
.{ .name = "envelope", .module = protocol.module("envelope") },
|
||||
.{ .name = "system-call", .module = system_call },
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "time", .module = time },
|
||||
.{ .name = "device-manager-protocol", .module = protocol.module("device-manager-protocol") },
|
||||
},
|
||||
});
|
||||
// A device driver's view of its claimed PCI function: config-space header
|
||||
// fields, BAR decode + map, capability walks (legacy + extended), MSI/MSI-X
|
||||
// programming, power state, and function-level reset — the generic PCI
|
||||
// mechanics every leaf PCI driver used to re-derive inline.
|
||||
_ = b.addModule("pci", .{
|
||||
.root_source_file = b.path("pci/pci.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "driver", .module = driver },
|
||||
.{ .name = "mmio", .module = mmio },
|
||||
.{ .name = "pci-class", .module = pci_class },
|
||||
.{ .name = "time", .module = time },
|
||||
},
|
||||
});
|
||||
// The USB class-driver transfer client: open a device on the xHCI bus and
|
||||
// drive it (control / interrupt / bulk). Re-exports usb-abi / usb-ids as
|
||||
// usb.abi / usb.ids for a single USB import.
|
||||
_ = b.addModule("usb", .{
|
||||
.root_source_file = b.path("usb/usb.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "channel", .module = channel },
|
||||
.{ .name = "envelope", .module = protocol.module("envelope") },
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "time", .module = time },
|
||||
.{ .name = "usb-transfer-protocol", .module = protocol.module("usb-transfer-protocol") },
|
||||
.{ .name = "usb-abi", .module = usb_abi },
|
||||
.{ .name = "usb-ids", .module = usb_ids },
|
||||
},
|
||||
});
|
||||
// The block-device client — a device type, so it lives here.
|
||||
_ = b.addModule("block", .{
|
||||
.root_source_file = b.path("block/block.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "channel", .module = channel },
|
||||
.{ .name = "envelope", .module = protocol.module("envelope") },
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "time", .module = time },
|
||||
.{ .name = "block-protocol", .module = protocol.module("block-protocol") },
|
||||
},
|
||||
});
|
||||
// The device registry: parse /system/configuration/devices.csv into match rules and bind a
|
||||
// reported device to a driver. Pure logic (no hardware, no syscalls), so it
|
||||
// unit-tests on the host; the device manager imports it.
|
||||
_ = b.addModule("device-registry", .{
|
||||
.root_source_file = b.path("registry/device-registry.zig"),
|
||||
.imports = &.{.{ .name = "csv", .module = csv.module("csv") }},
|
||||
});
|
||||
|
||||
// Standalone `zig build test` for this domain alone; the root build keeps
|
||||
// its aggregate test step.
|
||||
const test_step = b.step("test", "Run the device library unit tests");
|
||||
for ([_][]const u8{
|
||||
"model/device-abi.zig", // wire-type sizes
|
||||
"pci/pci-class.zig", // class/subclass/prog-IF name decoding
|
||||
"acpi/acpi-ids.zig", // _HID name decoding
|
||||
"acpi/aml/aml.zig", // AML parse + interpret, incl. Notify dispatch
|
||||
"usb/usb-abi.zig", // wire sizes + bit packings + set-up packet encodings
|
||||
"usb/usb-ids.zig", // class/subclass/protocol code assignments
|
||||
"mmio/mmio.zig", // barriers assemble + registers round-trip
|
||||
}) |root| {
|
||||
const device_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(root),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(device_tests).step);
|
||||
}
|
||||
// The registry needs its csv import wired, so it doesn't fit the loop.
|
||||
const registry_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("registry/device-registry.zig"),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
.imports = &.{.{ .name = "csv", .module = csv.module("csv") }},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(registry_tests).step);
|
||||
}
|
||||
@@ -0,0 +1,15 @@
|
||||
.{
|
||||
.name = .device,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x92fb68eace23a4f, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// driver/block/usb/pci build on the kernel library's concern modules.
|
||||
.kernel = .{ .path = "../kernel" },
|
||||
// driver speaks device-manager-protocol; block/usb their transfer protocols.
|
||||
.protocol = .{ .path = "../protocol" },
|
||||
// device-registry parses /system/configuration/devices.csv with the shared csv helpers.
|
||||
.csv = .{ .path = "../csv" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -1,12 +1,17 @@
|
||||
//! User-space device access: enumerate the kernel's device table, claim a device,
|
||||
//! map its MMIO, and bind its interrupt. A driver uses these to find and take
|
||||
//! ownership of its hardware; the claim is the capability the kernel checks before
|
||||
//! mapping registers or routing an IRQ.
|
||||
//! library/device/driver — the driver author's interface: enumerate the kernel's device
|
||||
//! table, claim a device, map its MMIO, bind its interrupt (the claim is the capability the
|
||||
//! kernel checks before mapping registers or routing an IRQ), and say `hello` to the device
|
||||
//! manager at startup. The whole kernel + manager surface a driver needs, in one import.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const device_abi = @import("device-abi");
|
||||
const sc = @import("system-call.zig");
|
||||
const sc = @import("system-call");
|
||||
const channel = @import("channel");
|
||||
const envelope = @import("envelope");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const device_manager_protocol = @import("device-manager-protocol");
|
||||
|
||||
pub const DeviceDescriptor = device_abi.DeviceDescriptor;
|
||||
pub const ResourceDescriptor = device_abi.ResourceDescriptor;
|
||||
@@ -96,6 +101,27 @@ pub fn msiBind(device_id: u64, endpoint: usize) ?Msi {
|
||||
return .{ .address = rax, .data = @intCast(rdx) };
|
||||
}
|
||||
|
||||
/// Map a delegated DMA-region (or shared-memory) capability into a claimed device's
|
||||
/// IOMMU domain, so the device may DMA to that buffer. The caller must own `device_id`
|
||||
/// and hold `handle` (received over IPC or from its own `dma.alloc(.. | shareable)`).
|
||||
/// Idempotent. Returns true on success (and trivially when no IOMMU is present).
|
||||
pub fn dmaBind(device_id: u64, handle: usize) bool {
|
||||
return !failed(sc.systemCall2(.dma_bind, device_id, handle));
|
||||
}
|
||||
|
||||
/// Unmap a previously `dmaBind`'d buffer from the device's domain.
|
||||
pub fn dmaUnbind(device_id: u64, handle: usize) bool {
|
||||
return !failed(sc.systemCall2(.dma_unbind, device_id, handle));
|
||||
}
|
||||
|
||||
/// Drain and log any pending IOMMU translation faults, returning the count seen. A
|
||||
/// diagnostic: a driver that suspects its device attempted an out-of-domain DMA (or a
|
||||
/// test proving enforcement) forces the hardware's fault records to the log now. Returns
|
||||
/// 0 when no IOMMU is present.
|
||||
pub fn iommuFaultDrain() usize {
|
||||
return sc.systemCall0(.iommu_fault_drain);
|
||||
}
|
||||
|
||||
/// Read `width` bytes (1, 2, or 4) from a port in a claimed device's `io_port`
|
||||
/// resource, at byte `offset` within it. Ring 3 has no direct `in`/`out`, so a legacy
|
||||
/// driver (PS/2, 16550 UART) reaches its ports through this claim-gated call — each
|
||||
@@ -129,3 +155,55 @@ pub fn findDeviceDescriptorByHid(buffer: []DeviceDescriptor, hid_needle: []const
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
// --- device-manager handshake (folded in from the former device-manager.zig) ---
|
||||
|
||||
/// What kind of driver is announcing itself (a bus that reports children, or a leaf
|
||||
/// device). Re-exported so callers name it without importing the protocol.
|
||||
pub const Role = device_manager_protocol.Role;
|
||||
|
||||
const lookup_attempts: u32 = 100;
|
||||
const lookup_pause_ms: u64 = 20;
|
||||
|
||||
/// Say hello to the device manager and return its endpoint, or null if there is no manager
|
||||
/// (best-effort standalone bring-up) or it refused the handshake. Bus drivers keep the handle
|
||||
/// to report children through; a driver that runs fine unsupervised discards it with `_ =`,
|
||||
/// and one that requires supervision bails on null. Logs the outcome itself.
|
||||
///
|
||||
/// The device this driver was assigned is the packet's `Header.target` — the manager's
|
||||
/// object addressing, so `no_device` here is a driver that serves none.
|
||||
pub fn hello(role: Role, device_id: u64) ?ipc.Handle {
|
||||
var attempts: u32 = 0;
|
||||
const manager = while (attempts < lookup_attempts) : (attempts += 1) {
|
||||
if (channel.openEndpoint("device-manager")) |handle| break handle;
|
||||
time.sleepMillis(lookup_pause_ms);
|
||||
} else {
|
||||
std.log.info("no device manager to hello", .{});
|
||||
return null;
|
||||
};
|
||||
|
||||
var packet: [device_manager_protocol.message_maximum]u8 = undefined;
|
||||
const framed = device_manager_protocol.Protocol.encodeRequest(
|
||||
.hello,
|
||||
device_id,
|
||||
.{ .role = @intFromEnum(role) },
|
||||
&.{},
|
||||
&packet,
|
||||
) orelse return null;
|
||||
|
||||
var reply: [device_manager_protocol.message_maximum]u8 = undefined;
|
||||
const length = ipc.call(manager, framed, &reply) catch {
|
||||
std.log.info("hello call failed", .{});
|
||||
return null;
|
||||
};
|
||||
const status = envelope.statusOf(reply[0..length]) orelse {
|
||||
std.log.info("hello answered nothing readable", .{});
|
||||
return null;
|
||||
};
|
||||
if (status.status != 0) {
|
||||
std.log.info("hello refused", .{});
|
||||
return null;
|
||||
}
|
||||
std.log.info("hello acknowledged", .{});
|
||||
return manager;
|
||||
}
|
||||
@@ -1,4 +1,4 @@
|
||||
//! /lib/mmio — typed volatile MMIO register access, plus the memory-ordering
|
||||
//! /lib/device/mmio — typed volatile MMIO register access, plus the memory-ordering
|
||||
//! barriers a device driver needs. Used by drivers on top of an `mmio_map` grant.
|
||||
//!
|
||||
//! **`volatile` is not a barrier.** In Zig it means only: don't elide this access, and
|
||||
@@ -11,13 +11,13 @@
|
||||
//! doorbell.* = i; // volatile store to UC MMIO
|
||||
//! // nothing orders these; the device can read a stale descriptor
|
||||
//!
|
||||
//! Put a `wmb()` between them. The barriers lower per-architecture — which is the whole
|
||||
//! reason they are a named primitive and not scattered `asm volatile`:
|
||||
//! Put a `writeMemoryBarrier()` between them. The barriers lower per-architecture — which
|
||||
//! is the whole reason they are a named primitive and not scattered `asm volatile`:
|
||||
//!
|
||||
//! x86_64 aarch64
|
||||
//! mb() mfence dsb sy
|
||||
//! rmb() lfence dsb ld
|
||||
//! wmb() sfence dsb st
|
||||
//! x86_64 aarch64
|
||||
//! memoryBarrier() mfence dsb sy
|
||||
//! readMemoryBarrier() lfence dsb ld
|
||||
//! writeMemoryBarrier() sfence dsb st
|
||||
//!
|
||||
//! x86 is forgiving (TSO + strong-uncacheable MMIO), so a compiler barrier usually
|
||||
//! suffices; ARM is not, and ARM is the win condition (docs/vision.md) — so the
|
||||
@@ -29,52 +29,52 @@ const builtin = @import("builtin");
|
||||
/// Read a register of type `T` at absolute virtual address `addr` — a location inside
|
||||
/// a device's `mmio_map` grant. `volatile`: never elided, never reordered against
|
||||
/// another volatile access.
|
||||
pub inline fn read(comptime T: type, addr: usize) T {
|
||||
pub inline fn readRegister(comptime T: type, addr: usize) T {
|
||||
return @as(*const volatile T, @ptrFromInt(addr)).*;
|
||||
}
|
||||
|
||||
/// Write `value` of type `T` to the register at absolute virtual address `addr`.
|
||||
pub inline fn write(comptime T: type, addr: usize, value: T) void {
|
||||
pub inline fn writeRegister(comptime T: type, addr: usize, value: T) void {
|
||||
@as(*volatile T, @ptrFromInt(addr)).* = value;
|
||||
}
|
||||
|
||||
/// Full barrier: all loads and stores before it are globally visible before any after
|
||||
/// it. Use when an MMIO write must complete before a following read.
|
||||
pub inline fn mb() void {
|
||||
/// Full memory barrier: all loads and stores before it are globally visible before any
|
||||
/// after it. Use when an MMIO write must complete before a following read.
|
||||
pub inline fn memoryBarrier() void {
|
||||
switch (builtin.target.cpu.arch) {
|
||||
.x86_64 => asm volatile ("mfence" ::: .{ .memory = true }),
|
||||
.aarch64 => asm volatile ("dsb sy" ::: .{ .memory = true }),
|
||||
else => @compileError("mmio.mb: unsupported architecture"),
|
||||
else => @compileError("mmio.memoryBarrier: unsupported architecture"),
|
||||
}
|
||||
}
|
||||
|
||||
/// Read barrier: loads before it complete before loads after it. Use after an IRQ
|
||||
/// Read memory barrier: loads before it complete before loads after it. Use after an IRQ
|
||||
/// wake, before reading what the device wrote to shared memory.
|
||||
pub inline fn rmb() void {
|
||||
pub inline fn readMemoryBarrier() void {
|
||||
switch (builtin.target.cpu.arch) {
|
||||
.x86_64 => asm volatile ("lfence" ::: .{ .memory = true }),
|
||||
.aarch64 => asm volatile ("dsb ld" ::: .{ .memory = true }),
|
||||
else => @compileError("mmio.rmb: unsupported architecture"),
|
||||
else => @compileError("mmio.readMemoryBarrier: unsupported architecture"),
|
||||
}
|
||||
}
|
||||
|
||||
/// Write barrier: stores before it become visible before stores after it. Use between
|
||||
/// filling a DMA descriptor in RAM and ringing the device's doorbell.
|
||||
pub inline fn wmb() void {
|
||||
/// Write memory barrier: stores before it become visible before stores after it. Use
|
||||
/// between filling a DMA descriptor in RAM and ringing the device's doorbell.
|
||||
pub inline fn writeMemoryBarrier() void {
|
||||
switch (builtin.target.cpu.arch) {
|
||||
.x86_64 => asm volatile ("sfence" ::: .{ .memory = true }),
|
||||
.aarch64 => asm volatile ("dsb st" ::: .{ .memory = true }),
|
||||
else => @compileError("mmio.wmb: unsupported architecture"),
|
||||
else => @compileError("mmio.writeMemoryBarrier: unsupported architecture"),
|
||||
}
|
||||
}
|
||||
|
||||
test "barriers emit and registers round-trip through a RAM cell" {
|
||||
// The barriers must at least assemble for the host arch; ordering can't be unit
|
||||
// tested, but a missing/mistyped mnemonic is caught here.
|
||||
wmb();
|
||||
rmb();
|
||||
mb();
|
||||
writeMemoryBarrier();
|
||||
readMemoryBarrier();
|
||||
memoryBarrier();
|
||||
var cell: u64 = 0;
|
||||
write(u64, @intFromPtr(&cell), 0xDEAD_BEEF);
|
||||
try @import("std").testing.expectEqual(@as(u64, 0xDEAD_BEEF), read(u64, @intFromPtr(&cell)));
|
||||
writeRegister(u64, @intFromPtr(&cell), 0xDEAD_BEEF);
|
||||
try @import("std").testing.expectEqual(@as(u64, 0xDEAD_BEEF), readRegister(u64, @intFromPtr(&cell)));
|
||||
}
|
||||
@@ -6,7 +6,7 @@
|
||||
//! by name, and neither reaches into the other's files.
|
||||
//!
|
||||
//! It is also the **single source of truth** for `DeviceClass` and `ResourceKind`:
|
||||
//! the kernel's rich, pointer-based device tree (system/devices/device-model.zig,
|
||||
//! the kernel's rich, pointer-based device tree (system/kernel/device-model.zig,
|
||||
//! which user space must never import) re-exports these, so the enum that a driver
|
||||
//! matches on and the enum the kernel classifies with are the *same* type — no
|
||||
//! hand-kept "mirror in order" to drift. The core kernel↔user ABI is [[abi]]; the
|
||||
@@ -125,6 +125,15 @@ pub const DeviceDescriptor = extern struct {
|
||||
// `pci_device` is (an xHCI controller, an AHCI controller) — decode the triple into
|
||||
// names with the pci-class module.
|
||||
pci_class: u64,
|
||||
// Numeric identity beyond the class triple, mirrored in the bus report's
|
||||
// ChildAdded so /system/configuration/devices.csv can bind on it: `vendor`/`device` are the PCI
|
||||
// vendor/device (or USB idVendor/idProduct), `subsystem` is the PCI subsystem id
|
||||
// packed `(subsystem_vendor << 16) | subsystem_device`. Zero where the bus has no
|
||||
// such concept. Defaulted so existing descriptor literals keep compiling and lay
|
||||
// out identically until they choose to set them.
|
||||
vendor: u16 = 0,
|
||||
device: u16 = 0,
|
||||
subsystem: u32 = 0,
|
||||
hid_len: u64,
|
||||
resource_count: u64,
|
||||
hid: [8]u8,
|
||||
@@ -41,6 +41,153 @@ pub const ClassCode = struct {
|
||||
}
|
||||
};
|
||||
|
||||
// --- Configuration-space layout ---------------------------------------------------------
|
||||
// The offsets and bit layouts of the PCI configuration header (PCI spec; see
|
||||
// https://wiki.osdev.org/PCI). Pure data — named here so both a device driver's view of
|
||||
// its own claimed function (library/device/pci/pci.zig) and the bus enumerator name the
|
||||
// same bytes instead of scattering bare 0x04/0x34/0xFFFF_FFF0 magic across the tree.
|
||||
|
||||
/// Header field offsets (byte offsets into the 256-byte configuration space).
|
||||
pub const config_vendor_id: usize = 0x00;
|
||||
pub const config_device_id: usize = 0x02;
|
||||
pub const config_command: usize = 0x04;
|
||||
pub const config_status: usize = 0x06;
|
||||
pub const config_revision_id: usize = 0x08;
|
||||
pub const config_class_code: usize = 0x09; // 3 bytes: prog-IF 0x09, subclass 0x0A, base class 0x0B
|
||||
pub const config_bar0: usize = 0x10; // BAR0; BAR n is at config_bar0 + n*4
|
||||
pub const config_subsystem_vendor_id: usize = 0x2C;
|
||||
pub const config_subsystem_id: usize = 0x2E;
|
||||
pub const config_expansion_rom: usize = 0x30;
|
||||
pub const config_capabilities_pointer: usize = 0x34;
|
||||
pub const config_interrupt_line: usize = 0x3C;
|
||||
pub const config_interrupt_pin: usize = 0x3D; // 0 = none, 1..4 = INTA..INTD
|
||||
|
||||
/// Command register bits.
|
||||
pub const command_io_space: u16 = 0x0001; // bit 0: I/O-space decode enable
|
||||
pub const command_memory_space: u16 = 0x0002; // bit 1: memory-space decode enable
|
||||
pub const command_bus_master: u16 = 0x0004; // bit 2: bus-master (DMA) enable
|
||||
pub const command_interrupt_disable: u16 = 0x0400; // bit 10: suppress legacy INTx (MSI/MSI-X unaffected)
|
||||
/// The pair a bus-mastering driver enables together: decode my BARs, let me DMA.
|
||||
pub const command_memory_and_bus_master: u16 = command_memory_space | command_bus_master;
|
||||
|
||||
/// Status register bit 3: legacy INTx is asserted (upstream of the command bit-10 gate).
|
||||
pub const status_interrupt: u16 = 0x0008;
|
||||
/// Status register bit 4: a capability list is present at config_capabilities_pointer.
|
||||
pub const status_capabilities_list: u16 = 0x10;
|
||||
/// Capability pointers are dword-aligned; the low two bits are reserved.
|
||||
pub const capability_pointer_mask: u8 = 0xFC;
|
||||
|
||||
/// Capability IDs — the first byte of each entry in the legacy capability list.
|
||||
/// Non-exhaustive: hardware may report IDs not named here.
|
||||
pub const CapabilityId = enum(u8) {
|
||||
power_management = 0x01,
|
||||
msi = 0x05,
|
||||
vendor_specific = 0x09,
|
||||
pci_express = 0x10,
|
||||
msix = 0x11,
|
||||
_,
|
||||
};
|
||||
|
||||
/// MSI capability (id 0x05) register layout. Offsets are relative to the capability
|
||||
/// header; whether the address is one or two dwords (and therefore where the data word
|
||||
/// sits) depends on `control_64bit_capable`.
|
||||
pub const msi = struct {
|
||||
pub const control: usize = 0x02; // u16 Message Control
|
||||
pub const control_enable: u16 = 0x0001;
|
||||
pub const control_multiple_message_capable_mask: u16 = 0x000E; // bits 3:1, log2(vectors requested)
|
||||
pub const control_multiple_message_enable_mask: u16 = 0x0070; // bits 6:4, log2(vectors granted)
|
||||
pub const control_64bit_capable: u16 = 0x0080; // bit 7: address is 64-bit (layout shifts)
|
||||
pub const control_per_vector_masking: u16 = 0x0100; // bit 8
|
||||
pub const address: usize = 0x04; // u32 low address dword (both layouts)
|
||||
pub const address_high: usize = 0x08; // u32, present only when 64-bit capable
|
||||
pub const data_32: usize = 0x08; // u16 message data, 32-bit layout
|
||||
pub const data_64: usize = 0x0C; // u16 message data, 64-bit layout
|
||||
pub const mask_bits_32: usize = 0x0C; // u32, only with per-vector masking
|
||||
pub const mask_bits_64: usize = 0x10;
|
||||
};
|
||||
|
||||
/// MSI-X capability (id 0x11) register layout, plus the 16-byte vector table entry that
|
||||
/// lives in BAR space (not configuration space) at the decoded (BIR, offset).
|
||||
pub const msix = struct {
|
||||
pub const control: usize = 0x02; // u16 Message Control
|
||||
pub const control_table_size_mask: u16 = 0x07FF; // bits 10:0, encoded as N-1
|
||||
pub const control_function_mask: u16 = 0x4000; // bit 14: mask every vector
|
||||
pub const control_enable: u16 = 0x8000; // bit 15
|
||||
pub const table_offset_word: usize = 0x04; // u32: BIR in bits 2:0, table offset in bits 31:3
|
||||
pub const pba_offset_word: usize = 0x08; // u32: same encoding, pending-bit array
|
||||
pub const bir_mask: u32 = 0x0000_0007;
|
||||
pub const offset_mask: u32 = 0xFFFF_FFF8;
|
||||
pub const entry_size: usize = 16; // table entry stride; offsets within an entry:
|
||||
pub const entry_address: usize = 0x0; // u32 low
|
||||
pub const entry_address_high: usize = 0x4; // u32 high
|
||||
pub const entry_data: usize = 0x8; // u32
|
||||
pub const entry_vector_control: usize = 0xC; // u32
|
||||
pub const entry_vector_control_masked: u32 = 0x1; // bit 0; entries reset to masked
|
||||
|
||||
/// Where the table (or pending-bit array) lives, decoded from its offset/BIR dword.
|
||||
pub const TableLocation = struct { bar: u8, offset: u32 };
|
||||
pub fn tableLocation(word: u32) TableLocation {
|
||||
return .{ .bar = @intCast(word & bir_mask), .offset = word & offset_mask };
|
||||
}
|
||||
/// Number of table entries (the control field encodes N-1).
|
||||
pub fn tableSize(control_value: u16) u16 {
|
||||
return (control_value & control_table_size_mask) + 1;
|
||||
}
|
||||
};
|
||||
|
||||
/// Power-management capability (id 0x01) register layout.
|
||||
pub const power_management = struct {
|
||||
pub const capabilities: usize = 0x02; // u16 PMC (read-only: version, D-state support)
|
||||
pub const control_status: usize = 0x04; // u16 PMCSR
|
||||
pub const control_status_power_state_mask: u16 = 0x0003; // bits 1:0
|
||||
pub const power_state_d0: u16 = 0x0;
|
||||
pub const power_state_d3_hot: u16 = 0x3;
|
||||
pub const control_status_pme_enable: u16 = 0x0100; // bit 8: plain RW — preserve on writes
|
||||
pub const control_status_pme_status: u16 = 0x8000; // bit 15: RW1C — write 0 or you clear it
|
||||
};
|
||||
|
||||
/// PCI Express capability (id 0x10) register layout — the slice function-level reset
|
||||
/// needs; the full capability is much larger.
|
||||
pub const pci_express = struct {
|
||||
pub const capabilities: usize = 0x02; // u16 PCIe Capabilities register
|
||||
pub const device_capabilities: usize = 0x04; // u32
|
||||
pub const device_capabilities_flr: u32 = 1 << 28; // Function Level Reset supported
|
||||
pub const device_control: usize = 0x08; // u16
|
||||
pub const device_control_initiate_flr: u16 = 1 << 15;
|
||||
pub const device_status: usize = 0x0A; // u16
|
||||
pub const device_status_transactions_pending: u16 = 1 << 5;
|
||||
};
|
||||
|
||||
/// Extended (PCI Express) capabilities start here in the 4 KiB configuration space; a
|
||||
/// conventional-PCI function has nothing there (the space reads as all-ones).
|
||||
pub const extended_capability_start: usize = 0x100;
|
||||
/// Extended-capability next pointers are dword-aligned within the 4 KiB space.
|
||||
pub const extended_capability_pointer_mask: u16 = 0xFFC;
|
||||
|
||||
/// The 32-bit header at the start of each extended capability: ID in bits 15:0,
|
||||
/// version in 19:16, next offset in 31:20 (0 = end of list).
|
||||
pub const ExtendedCapabilityHeader = struct {
|
||||
id: u16,
|
||||
version: u4,
|
||||
next: u16,
|
||||
|
||||
pub fn decode(word: u32) ExtendedCapabilityHeader {
|
||||
return .{
|
||||
.id = @truncate(word),
|
||||
.version = @truncate(word >> 16),
|
||||
.next = @intCast((word >> 20) & extended_capability_pointer_mask),
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
/// BAR bit layout: bit 0 selects I/O (1) vs memory (0) space; for a memory BAR, bits 2:1
|
||||
/// give the type (00 = 32-bit, 10 = 64-bit spanning the next BAR), and the base address is
|
||||
/// the dword with the low 4 flag bits masked off.
|
||||
pub const bar_io_space: u32 = 0x1;
|
||||
pub const bar_type_mask: u32 = 0x6;
|
||||
pub const bar_type_64bit: u32 = 0x4;
|
||||
pub const bar_memory_base_mask: u32 = 0xFFFF_FFF0;
|
||||
|
||||
/// Base class (config byte 0x0B). Non-exhaustive: an unlisted code is a real but
|
||||
/// unnamed class, decoded as "Unknown" rather than rejected.
|
||||
pub const BaseClass = enum(u8) {
|
||||
@@ -566,3 +713,37 @@ test "named parts pack to the raw triple" {
|
||||
};
|
||||
try std.testing.expectEqual(@as(u24, 0x0C_03_30), xhci.pack());
|
||||
}
|
||||
|
||||
test "MSI-X table word decodes to BIR and offset" {
|
||||
const eq = std.testing.expectEqual;
|
||||
// BIR 3, table at 0x2000 within that BAR.
|
||||
try eq(msix.TableLocation{ .bar = 3, .offset = 0x2000 }, msix.tableLocation(0x0000_2003));
|
||||
// BIR 0, offset 0 — the degenerate-but-common "table at BAR start" case.
|
||||
try eq(msix.TableLocation{ .bar = 0, .offset = 0 }, msix.tableLocation(0));
|
||||
// Table size encodes N-1 in bits 10:0; enable/function-mask bits must not leak in.
|
||||
try eq(@as(u16, 11), msix.tableSize(msix.control_enable | 0x000A));
|
||||
try eq(@as(u16, 1), msix.tableSize(0));
|
||||
try eq(@as(u16, 2048), msix.tableSize(msix.control_table_size_mask));
|
||||
}
|
||||
|
||||
test "extended capability header unpacks id, version, next" {
|
||||
const eq = std.testing.expectEqual;
|
||||
// AER (id 0x0001), version 1, next capability at 0x140.
|
||||
const aer = ExtendedCapabilityHeader.decode(0x1401_0001);
|
||||
try eq(@as(u16, 0x0001), aer.id);
|
||||
try eq(@as(u4, 1), aer.version);
|
||||
try eq(@as(u16, 0x140), aer.next);
|
||||
// A zero header is the "nothing here" terminator.
|
||||
const none = ExtendedCapabilityHeader.decode(0);
|
||||
try eq(@as(u16, 0), none.id);
|
||||
try eq(@as(u16, 0), none.next);
|
||||
}
|
||||
|
||||
test "command bits and capability ids compose" {
|
||||
const eq = std.testing.expectEqual;
|
||||
try eq(command_memory_space | command_bus_master, command_memory_and_bus_master);
|
||||
try eq(@as(u8, 0x05), @intFromEnum(CapabilityId.msi));
|
||||
try eq(@as(u8, 0x11), @intFromEnum(CapabilityId.msix));
|
||||
try eq(@as(u8, 0x01), @intFromEnum(CapabilityId.power_management));
|
||||
try eq(@as(u8, 0x10), @intFromEnum(CapabilityId.pci_express));
|
||||
}
|
||||
@@ -0,0 +1,384 @@
|
||||
//! library/device/pci/pci.zig — a device driver's view of the ONE PCI function it has
|
||||
//! claimed. Config space is mapped as resource 0 (a full 4 KiB ECAM page); this gives
|
||||
//! header-field accessors, BAR decode + map, capability walks (legacy and extended),
|
||||
//! MSI/MSI-X programming, power-state handling, and function-level reset, so a driver
|
||||
//! never re-derives the config-space layout by hand.
|
||||
//!
|
||||
//! This is the *device-owned* view: read my own function's live config, map my own BARs.
|
||||
//! The bus enumerator's view — probing arbitrary, not-yet-claimed functions and sizing
|
||||
//! their BARs — is a different mechanism and lives in the pci-bus driver. The pure
|
||||
//! config-space layout both need (offsets, BAR bit fields) is named once in the `pci-class`
|
||||
//! data module; this logic module adds the parts that need `mmio` + the `driver` client.
|
||||
|
||||
const std = @import("std");
|
||||
const mmio = @import("mmio");
|
||||
const pci_class = @import("pci-class");
|
||||
const device = @import("driver");
|
||||
const time = @import("time");
|
||||
|
||||
/// Spec recovery time after a D3hot -> D0 transition.
|
||||
const d0_recovery_millis: u64 = 10;
|
||||
/// How long to wait for in-flight transactions to drain before a function-level reset
|
||||
/// (then reset anyway — resetting a stuck function is the point of FLR).
|
||||
const flr_pending_timeout_millis: u64 = 100;
|
||||
/// The spec's maximum FLR completion time.
|
||||
const flr_settle_millis: u64 = 100;
|
||||
/// How long to wait for the function to become readable again after an FLR.
|
||||
const flr_ready_timeout_millis: u64 = 1000;
|
||||
const flr_poll_interval_millis: u64 = 10;
|
||||
|
||||
/// A claimed PCI function whose configuration space is mapped (resource 0). `descriptor`
|
||||
/// must outlive the Function — the driver's `device.enumerate` buffer does, for the whole
|
||||
/// bring-up. Header reads and the capability walk hit live config space; `mapBar` caches.
|
||||
pub const Function = struct {
|
||||
device_id: u64,
|
||||
descriptor: *const device.DeviceDescriptor,
|
||||
config: usize, // virtual base of mapped resource 0
|
||||
bar_virtual: [6]usize = .{ 0, 0, 0, 0, 0, 0 }, // per-BAR mmio_map cache
|
||||
|
||||
/// Map config space (resource 0) of the already-claimed `device_id`. null if the map
|
||||
/// fails (not claimed, or no config resource).
|
||||
pub fn map(device_id: u64, descriptor: *const device.DeviceDescriptor) ?Function {
|
||||
const base = device.mmioMap(device_id, 0) orelse return null;
|
||||
return .{ .device_id = device_id, .descriptor = descriptor, .config = base };
|
||||
}
|
||||
|
||||
pub fn vendorId(self: *const Function) u16 {
|
||||
return mmio.readRegister(u16, self.config + pci_class.config_vendor_id);
|
||||
}
|
||||
pub fn deviceId(self: *const Function) u16 {
|
||||
return mmio.readRegister(u16, self.config + pci_class.config_device_id);
|
||||
}
|
||||
pub fn command(self: *const Function) u16 {
|
||||
return mmio.readRegister(u16, self.config + pci_class.config_command);
|
||||
}
|
||||
pub fn status(self: *const Function) u16 {
|
||||
return mmio.readRegister(u16, self.config + pci_class.config_status);
|
||||
}
|
||||
pub fn revisionId(self: *const Function) u8 {
|
||||
return mmio.readRegister(u8, self.config + pci_class.config_revision_id);
|
||||
}
|
||||
/// Subsystem vendor ID (config 0x2C) — with `subsystemId`, the standard key for
|
||||
/// board-level quirk matching.
|
||||
pub fn subsystemVendorId(self: *const Function) u16 {
|
||||
return mmio.readRegister(u16, self.config + pci_class.config_subsystem_vendor_id);
|
||||
}
|
||||
pub fn subsystemId(self: *const Function) u16 {
|
||||
return mmio.readRegister(u16, self.config + pci_class.config_subsystem_id);
|
||||
}
|
||||
/// Interrupt pin (config 0x3D): 0 = none, 1..4 = INTA..INTD.
|
||||
pub fn interruptPin(self: *const Function) u8 {
|
||||
return mmio.readRegister(u8, self.config + pci_class.config_interrupt_pin);
|
||||
}
|
||||
/// The live class-code triple (config 0x09..0x0B), same shape discovery records.
|
||||
pub fn classCode(self: *const Function) pci_class.ClassCode {
|
||||
return .{
|
||||
.prog_if = mmio.readRegister(u8, self.config + pci_class.config_class_code),
|
||||
.subclass = mmio.readRegister(u8, self.config + pci_class.config_class_code + 1),
|
||||
.base = mmio.readRegister(u8, self.config + pci_class.config_class_code + 2),
|
||||
};
|
||||
}
|
||||
|
||||
fn commandSetBits(self: *const Function, bits: u16) void {
|
||||
const at = self.config + pci_class.config_command;
|
||||
mmio.writeRegister(u16, at, mmio.readRegister(u16, at) | bits);
|
||||
}
|
||||
fn commandClearBits(self: *const Function, bits: u16) void {
|
||||
const at = self.config + pci_class.config_command;
|
||||
mmio.writeRegister(u16, at, mmio.readRegister(u16, at) & ~bits);
|
||||
}
|
||||
|
||||
/// Set Memory-Space + Bus-Master Enable in the command register. Firmware only enables
|
||||
/// memory decode on devices it used at boot; any other device has dead BARs until its
|
||||
/// driver sets it. Bus mastering is separately required for the device to do DMA.
|
||||
pub fn enableMemoryAndBusMaster(self: *const Function) void {
|
||||
self.commandSetBits(pci_class.command_memory_and_bus_master);
|
||||
}
|
||||
|
||||
/// Clear Bus-Master Enable — stop the device initiating DMA. The quiesce half of a
|
||||
/// driver's shutdown (or a supervisor restart): after this the device can no longer
|
||||
/// write memory the process is about to stop owning.
|
||||
pub fn disableBusMaster(self: *const Function) void {
|
||||
self.commandClearBits(pci_class.command_bus_master);
|
||||
}
|
||||
|
||||
/// Set command bit 10: suppress legacy INTx assertion. MSI/MSI-X are unaffected —
|
||||
/// set this when enabling either, so the device cannot also raise the shared pin.
|
||||
pub fn setInterruptDisable(self: *const Function) void {
|
||||
self.commandSetBits(pci_class.command_interrupt_disable);
|
||||
}
|
||||
/// Clear command bit 10, re-allowing legacy INTx assertion.
|
||||
pub fn clearInterruptDisable(self: *const Function) void {
|
||||
self.commandClearBits(pci_class.command_interrupt_disable);
|
||||
}
|
||||
|
||||
/// Decode BAR `bar` (0..5) and map it: read the BAR register, reject I/O-space BARs,
|
||||
/// combine the high dword for a 64-bit BAR, mask the base, then correlate that physical
|
||||
/// base with one of the descriptor's memory resources and `mmio_map` it — a BAR names a
|
||||
/// *number*, while `mmio_map` takes a *resource index*, and gaps/config-space shift the
|
||||
/// numbering. Cached per BAR. null if the BAR is I/O-space or is not a mapped resource.
|
||||
pub fn mapBar(self: *Function, bar: u8) ?usize {
|
||||
if (bar >= 6) return null;
|
||||
if (self.bar_virtual[bar] != 0) return self.bar_virtual[bar];
|
||||
|
||||
const low = mmio.readRegister(u32, self.config + pci_class.config_bar0 + @as(usize, bar) * 4);
|
||||
if (low & pci_class.bar_io_space != 0) return null; // an I/O-space BAR
|
||||
var base: u64 = low & pci_class.bar_memory_base_mask;
|
||||
if ((low & pci_class.bar_type_mask) == pci_class.bar_type_64bit) { // 64-bit: high half is the next dword
|
||||
const high = mmio.readRegister(u32, self.config + pci_class.config_bar0 + (@as(usize, bar) + 1) * 4);
|
||||
base |= @as(u64, high) << 32;
|
||||
}
|
||||
|
||||
for (self.descriptor.resources[0..@intCast(self.descriptor.resource_count)], 0..) |resource, index| {
|
||||
if (resource.kind == @intFromEnum(device.ResourceKind.memory) and resource.start == base) {
|
||||
const v = device.mmioMap(self.device_id, index) orelse return null;
|
||||
self.bar_virtual[bar] = v;
|
||||
return v;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Iterate the capability list. Empty when the function advertises none.
|
||||
pub fn capabilities(self: *const Function) CapabilityIterator {
|
||||
const present = self.status() & pci_class.status_capabilities_list != 0;
|
||||
const first = if (present)
|
||||
mmio.readRegister(u8, self.config + pci_class.config_capabilities_pointer) & pci_class.capability_pointer_mask
|
||||
else
|
||||
0;
|
||||
return .{ .config = self.config, .cursor = first };
|
||||
}
|
||||
|
||||
/// First capability with `id`, or null.
|
||||
pub fn findCapability(self: *const Function, id: pci_class.CapabilityId) ?Capability {
|
||||
var walk = self.capabilities();
|
||||
while (walk.next()) |capability| {
|
||||
if (capability.id == @intFromEnum(id)) return capability;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Program the MSI capability with the kernel's `msi_bind` result and enable it —
|
||||
/// one vector (multiple-message-enable 0, matching the kernel's single-vector
|
||||
/// grant), INTx suppressed. false if the function has no MSI capability.
|
||||
pub fn programMsi(self: *const Function, message: device.Msi) bool {
|
||||
const cap = self.findCapability(.msi) orelse return false;
|
||||
const control_at = cap.offset + pci_class.msi.control;
|
||||
const control = mmio.readRegister(u16, control_at);
|
||||
// Program the registers while the capability is disabled.
|
||||
mmio.writeRegister(u16, control_at, control & ~pci_class.msi.control_enable);
|
||||
mmio.writeRegister(u32, cap.offset + pci_class.msi.address, @truncate(message.address));
|
||||
const data_offset = if (control & pci_class.msi.control_64bit_capable != 0) offset: {
|
||||
mmio.writeRegister(u32, cap.offset + pci_class.msi.address_high, @intCast(message.address >> 32));
|
||||
break :offset pci_class.msi.data_64;
|
||||
} else pci_class.msi.data_32;
|
||||
// Message data is a 16-bit register in both layouts.
|
||||
mmio.writeRegister(u16, cap.offset + data_offset, @truncate(message.data));
|
||||
mmio.writeRegister(u16, control_at, (control & ~pci_class.msi.control_multiple_message_enable_mask) | pci_class.msi.control_enable);
|
||||
self.setInterruptDisable();
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Clear the MSI enable bit. No-op if the function has no MSI capability.
|
||||
pub fn disableMsi(self: *const Function) void {
|
||||
const cap = self.findCapability(.msi) orelse return;
|
||||
const control_at = cap.offset + pci_class.msi.control;
|
||||
mmio.writeRegister(u16, control_at, mmio.readRegister(u16, control_at) & ~pci_class.msi.control_enable);
|
||||
}
|
||||
|
||||
/// The function's MSI-X capability with its vector table mapped: the table's BIR is
|
||||
/// resolved through `mapBar` (a free cache hit when it is a BAR the driver already
|
||||
/// mapped). null if the capability is absent or the table's BAR cannot be mapped.
|
||||
pub fn msix(self: *Function) ?MsiX {
|
||||
const cap = self.findCapability(.msix) orelse return null;
|
||||
const control = mmio.readRegister(u16, cap.offset + pci_class.msix.control);
|
||||
const word = mmio.readRegister(u32, cap.offset + pci_class.msix.table_offset_word);
|
||||
const location = pci_class.msix.tableLocation(word);
|
||||
const bar_base = self.mapBar(location.bar) orelse return null;
|
||||
return .{
|
||||
.capability = cap.offset,
|
||||
.table = bar_base + location.offset,
|
||||
.entry_count = pci_class.msix.tableSize(control),
|
||||
};
|
||||
}
|
||||
|
||||
/// Bring the function to D0. Firmware can leave a non-boot device in D3hot, where
|
||||
/// its BARs and MSI registers do not decode; call this before touching either. No
|
||||
/// power-management capability means the function is always at D0: nothing to do.
|
||||
/// Preserves PME-Enable and never clears the write-1-to-clear PME-Status bit.
|
||||
pub fn ensurePowerStateD0(self: *const Function) void {
|
||||
const cap = self.findCapability(.power_management) orelse return;
|
||||
const at = cap.offset + pci_class.power_management.control_status;
|
||||
const pmcsr = mmio.readRegister(u16, at);
|
||||
if (pmcsr & pci_class.power_management.control_status_power_state_mask == pci_class.power_management.power_state_d0) return;
|
||||
// PME-Status is RW1C: echoing a read 1 back would clear it, so write it as 0.
|
||||
mmio.writeRegister(u16, at, (pmcsr & ~pci_class.power_management.control_status_power_state_mask & ~pci_class.power_management.control_status_pme_status) | pci_class.power_management.power_state_d0);
|
||||
time.sleepMillis(d0_recovery_millis);
|
||||
}
|
||||
|
||||
/// Function Level Reset via the PCI Express capability: return the hardware to a
|
||||
/// known state (a supervisor re-claiming a device after its driver died, or a driver
|
||||
/// recovering a wedged function). The six BAR dwords are saved and restored — FLR
|
||||
/// clears them, and the bus enumerator's assignment must survive for the descriptor
|
||||
/// correlation and `mapBar` cache to stay valid. Everything else is reset: command
|
||||
/// enables and MSI/MSI-X programming are gone, so the caller re-runs its whole
|
||||
/// bring-up afterwards. false if the function has no PCI Express capability, does
|
||||
/// not advertise FLR (conventional-PCI Advanced Features FLR is a possible
|
||||
/// follow-up), or never became readable again. Blocks for at least 100 ms.
|
||||
pub fn functionLevelReset(self: *const Function) bool {
|
||||
const cap = self.findCapability(.pci_express) orelse return false;
|
||||
const device_capabilities = mmio.readRegister(u32, cap.offset + pci_class.pci_express.device_capabilities);
|
||||
if (device_capabilities & pci_class.pci_express.device_capabilities_flr == 0) return false;
|
||||
|
||||
// Stop new DMA, then give in-flight transactions a bounded chance to drain —
|
||||
// and reset anyway on timeout, since resetting a stuck function is the point.
|
||||
self.disableBusMaster();
|
||||
var waited: u64 = 0;
|
||||
while (mmio.readRegister(u16, cap.offset + pci_class.pci_express.device_status) & pci_class.pci_express.device_status_transactions_pending != 0) {
|
||||
if (waited >= flr_pending_timeout_millis) break;
|
||||
time.sleepMillis(flr_poll_interval_millis);
|
||||
waited += flr_poll_interval_millis;
|
||||
}
|
||||
|
||||
var bars: [6]u32 = undefined;
|
||||
for (&bars, 0..) |*bar, index| bar.* = mmio.readRegister(u32, self.config + pci_class.config_bar0 + index * 4);
|
||||
|
||||
const control_at = cap.offset + pci_class.pci_express.device_control;
|
||||
mmio.writeRegister(u16, control_at, mmio.readRegister(u16, control_at) | pci_class.pci_express.device_control_initiate_flr);
|
||||
time.sleepMillis(flr_settle_millis);
|
||||
|
||||
waited = 0;
|
||||
while (self.vendorId() == 0xFFFF) {
|
||||
if (waited >= flr_ready_timeout_millis) return false;
|
||||
time.sleepMillis(flr_poll_interval_millis);
|
||||
waited += flr_poll_interval_millis;
|
||||
}
|
||||
for (bars, 0..) |bar, index| mmio.writeRegister(u32, self.config + pci_class.config_bar0 + index * 4, bar);
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Iterate the extended (PCI Express) capability list at 0x100.. in the 4 KiB ECAM
|
||||
/// page. Empty on a conventional-PCI function (the space reads as all-ones).
|
||||
pub fn extendedCapabilities(self: *const Function) ExtendedCapabilityIterator {
|
||||
return .{ .config = self.config };
|
||||
}
|
||||
|
||||
/// First extended capability with `id`, or null.
|
||||
pub fn findExtendedCapability(self: *const Function, id: u16) ?ExtendedCapability {
|
||||
var walk = self.extendedCapabilities();
|
||||
while (walk.next()) |capability| {
|
||||
if (capability.id == id) return capability;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
};
|
||||
|
||||
/// One capability header. `offset` is the ABSOLUTE virtual address of the header, so the
|
||||
/// caller reads its body with `mmio.readRegister(T, cap.offset + n)`.
|
||||
pub const Capability = struct { id: u8, offset: usize };
|
||||
|
||||
pub const CapabilityIterator = struct {
|
||||
config: usize,
|
||||
cursor: u8,
|
||||
guard: u32 = 0, // bounds a malformed/looping chain (48 = the 256-byte space in dwords)
|
||||
|
||||
pub fn next(self: *CapabilityIterator) ?Capability {
|
||||
if (self.cursor == 0 or self.guard >= 48) return null;
|
||||
self.guard += 1;
|
||||
const at = self.config + self.cursor;
|
||||
const id = mmio.readRegister(u8, at + 0);
|
||||
self.cursor = mmio.readRegister(u8, at + 1) & pci_class.capability_pointer_mask;
|
||||
return .{ .id = id, .offset = at };
|
||||
}
|
||||
};
|
||||
|
||||
/// A resolved MSI-X capability from `Function.msix`: `capability` is the absolute
|
||||
/// virtual address of the config-space header, `table` of vector-table entry 0 (in BAR
|
||||
/// space — table writes are MMIO, not config space). Entries reset masked; bring-up
|
||||
/// order is programEntry per vector, unmaskEntry per used vector, `enable`, then
|
||||
/// `Function.setInterruptDisable`.
|
||||
pub const MsiX = struct {
|
||||
capability: usize,
|
||||
table: usize,
|
||||
entry_count: u16,
|
||||
|
||||
/// Write `message` into table entry `entry`, leaving the entry masked (its reset
|
||||
/// state) — the spec requires masking while address/data change. false if `entry`
|
||||
/// is out of range.
|
||||
pub fn programEntry(self: *const MsiX, entry: u16, message: device.Msi) bool {
|
||||
if (entry >= self.entry_count) return false;
|
||||
const at = self.table + @as(usize, entry) * pci_class.msix.entry_size;
|
||||
mmio.writeRegister(u32, at + pci_class.msix.entry_vector_control, pci_class.msix.entry_vector_control_masked);
|
||||
mmio.writeRegister(u32, at + pci_class.msix.entry_address, @truncate(message.address));
|
||||
mmio.writeRegister(u32, at + pci_class.msix.entry_address_high, @intCast(message.address >> 32));
|
||||
mmio.writeRegister(u32, at + pci_class.msix.entry_data, message.data);
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Set the entry's vector-control mask bit — its interrupt is held off (pended in
|
||||
/// the PBA, not lost). false if `entry` is out of range.
|
||||
pub fn maskEntry(self: *const MsiX, entry: u16) bool {
|
||||
return self.writeEntryMask(entry, true);
|
||||
}
|
||||
/// Clear the entry's vector-control mask bit. false if `entry` is out of range.
|
||||
pub fn unmaskEntry(self: *const MsiX, entry: u16) bool {
|
||||
return self.writeEntryMask(entry, false);
|
||||
}
|
||||
fn writeEntryMask(self: *const MsiX, entry: u16, masked: bool) bool {
|
||||
if (entry >= self.entry_count) return false;
|
||||
const at = self.table + @as(usize, entry) * pci_class.msix.entry_size + pci_class.msix.entry_vector_control;
|
||||
const control = mmio.readRegister(u32, at);
|
||||
mmio.writeRegister(u32, at, if (masked)
|
||||
control | pci_class.msix.entry_vector_control_masked
|
||||
else
|
||||
control & ~pci_class.msix.entry_vector_control_masked);
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Set the function-mask control bit: every vector masked regardless of entry bits.
|
||||
pub fn setFunctionMask(self: *const MsiX) void {
|
||||
self.writeControl(pci_class.msix.control_function_mask, true);
|
||||
}
|
||||
/// Clear the function-mask control bit.
|
||||
pub fn clearFunctionMask(self: *const MsiX) void {
|
||||
self.writeControl(pci_class.msix.control_function_mask, false);
|
||||
}
|
||||
/// Set MSI-X Enable. The caller also calls `Function.setInterruptDisable` (INTx off).
|
||||
pub fn enable(self: *const MsiX) void {
|
||||
self.writeControl(pci_class.msix.control_enable, true);
|
||||
}
|
||||
/// Clear MSI-X Enable.
|
||||
pub fn disable(self: *const MsiX) void {
|
||||
self.writeControl(pci_class.msix.control_enable, false);
|
||||
}
|
||||
fn writeControl(self: *const MsiX, bit: u16, set: bool) void {
|
||||
const at = self.capability + pci_class.msix.control;
|
||||
const control = mmio.readRegister(u16, at);
|
||||
mmio.writeRegister(u16, at, if (set) control | bit else control & ~bit);
|
||||
}
|
||||
};
|
||||
|
||||
/// One extended capability. `offset` is the ABSOLUTE virtual address of its header,
|
||||
/// like `Capability.offset`.
|
||||
pub const ExtendedCapability = struct { id: u16, version: u4, offset: usize };
|
||||
|
||||
pub const ExtendedCapabilityIterator = struct {
|
||||
config: usize,
|
||||
cursor: u16 = @intCast(pci_class.extended_capability_start),
|
||||
guard: u32 = 0, // bounds a malformed chain (480 = the 0xF00-byte space / 8-byte minimum spacing)
|
||||
|
||||
pub fn next(self: *ExtendedCapabilityIterator) ?ExtendedCapability {
|
||||
if (self.cursor == 0 or self.guard >= 480) return null;
|
||||
self.guard += 1;
|
||||
const at = self.config + self.cursor;
|
||||
const header = pci_class.ExtendedCapabilityHeader.decode(mmio.readRegister(u32, at));
|
||||
// Id 0 marks an empty list; all-ones is a conventional-PCI function (no
|
||||
// extended space — reads come back as FFs).
|
||||
if (header.id == 0 or header.id == 0xFFFF) return null;
|
||||
// A next pointer below 0x100 would walk into the legacy header; treat it as the
|
||||
// terminator it must be (0 is the normal one). The 0xFFC decode mask already
|
||||
// keeps `config + cursor + 4` inside the 4 KiB page.
|
||||
self.cursor = if (header.next >= pci_class.extended_capability_start) header.next else 0;
|
||||
return .{ .id = header.id, .version = header.version, .offset = at };
|
||||
}
|
||||
};
|
||||
@@ -0,0 +1,343 @@
|
||||
//! The device registry: parse `/system/configuration/devices.csv` into match rules and bind a
|
||||
//! reported device to a driver. This is the data-driven replacement for the
|
||||
//! device manager's three hand-written `switch` tables (`pciDriverForIdentity`,
|
||||
//! `hidDriverFor`, `usbDriverForIdentity`); the registry is now **authoritative**
|
||||
//! — a device that no row matches goes unbound (logged), never guessed.
|
||||
//!
|
||||
//! Pure logic: no hardware access, no syscalls, no allocator. `parse` fills a
|
||||
//! caller-provided `[]Rule` whose string fields (`hid`, `driver`) are slices
|
||||
//! *into the CSV source*, so the source buffer must outlive the rules (the
|
||||
//! manager holds it in a static buffer for the life of the process — zero-copy).
|
||||
//! That keeps this module freestanding and unit-testable with plain `zig test`.
|
||||
//!
|
||||
//! The file format (docs/device-driver-development/device-manager.md, and the
|
||||
//! `/system/configuration/devices.csv` header itself): one rule per line, nine comma-separated
|
||||
//! fields, `#` starts a comment (whole-line or trailing), blank lines ignored.
|
||||
//!
|
||||
//! bus, base, class, prog_if, vendor, device, subsystem, hid, driver
|
||||
//!
|
||||
//! `bus` is `pci`/`usb`/`acpi`; the numeric fields are hex (with or without a
|
||||
//! `0x` prefix); `*` or an empty field is a wildcard (matches anything). For PCI
|
||||
//! the class triple is base/subclass/prog-IF; for USB it is class/subclass/
|
||||
//! protocol with vendor/device the idVendor/idProduct; ACPI matches on `hid`
|
||||
//! (e.g. "PNP0303") with the triple left blank. `driver` is a full ramdisk path.
|
||||
|
||||
const std = @import("std");
|
||||
const csv = @import("csv");
|
||||
|
||||
/// Which bus a rule or a reported device belongs to. `unknown` is what an
|
||||
/// unrecognised `bus` token parses to — such a rule never matches (its bus
|
||||
/// equals no real device's), so a typo fails safe rather than binding wrongly.
|
||||
pub const Bus = enum {
|
||||
pci,
|
||||
usb,
|
||||
acpi,
|
||||
unknown,
|
||||
|
||||
pub fn fromToken(token: []const u8) Bus {
|
||||
if (std.mem.eql(u8, token, "pci")) return .pci;
|
||||
if (std.mem.eql(u8, token, "usb")) return .usb;
|
||||
if (std.mem.eql(u8, token, "acpi")) return .acpi;
|
||||
return .unknown;
|
||||
}
|
||||
};
|
||||
|
||||
/// A reported device's full identity, as the manager assembles it from a
|
||||
/// `child_added`: the bus-native class triple plus the numeric ids the widened
|
||||
/// ABI now carries, or the ACPI `_HID` string. Fields a given bus does not have
|
||||
/// are zero / empty (a PCI function has no `hid`; an ACPI device has no vendor).
|
||||
pub const Identity = struct {
|
||||
bus: Bus,
|
||||
base: u8 = 0,
|
||||
subclass: u8 = 0,
|
||||
prog_if: u8 = 0,
|
||||
vendor: u16 = 0,
|
||||
device: u16 = 0,
|
||||
subsystem: u32 = 0,
|
||||
hid: []const u8 = "",
|
||||
};
|
||||
|
||||
/// One parsed registry row. A `null` field is a wildcard — it matches any value
|
||||
/// and contributes nothing to specificity. String fields point into the CSV
|
||||
/// source that was parsed (see the module doc).
|
||||
pub const Rule = struct {
|
||||
bus: Bus,
|
||||
base: ?u8 = null,
|
||||
subclass: ?u8 = null,
|
||||
prog_if: ?u8 = null,
|
||||
vendor: ?u16 = null,
|
||||
device: ?u16 = null,
|
||||
subsystem: ?u32 = null,
|
||||
hid: ?[]const u8 = null,
|
||||
driver: []const u8,
|
||||
};
|
||||
|
||||
/// Specificity weights: how much each pinned field counts toward "most specific
|
||||
/// wins". Doubling from the coarsest (`base`) so that each level outweighs *all*
|
||||
/// coarser levels combined (1+2+4+8+16 = 31 < 32) — a rule that pins `device`
|
||||
/// always beats any rule that does not, no matter how many coarse fields the
|
||||
/// latter pins. `hid` and `device` share the top tier (the user's "hid and
|
||||
/// device weigh heaviest"); they never co-occur, since `hid` is ACPI-only and
|
||||
/// `device` is a PCI/USB numeric id.
|
||||
const weight_base: u32 = 1;
|
||||
const weight_subclass: u32 = 2;
|
||||
const weight_prog_if: u32 = 4;
|
||||
const weight_vendor: u32 = 8;
|
||||
const weight_subsystem: u32 = 16;
|
||||
const weight_device: u32 = 32;
|
||||
const weight_hid: u32 = 32;
|
||||
|
||||
/// The outcome of `matchDriver`: the winning rule's driver path, its specificity,
|
||||
/// and whether another rule tied it at that specificity. `ambiguous` is a
|
||||
/// registry authoring error (two equally-specific rules claiming one device); the
|
||||
/// manager logs it loudly and binds the first, so a shadowed rule is visible
|
||||
/// rather than silently dropped.
|
||||
pub const Match = struct {
|
||||
driver: []const u8,
|
||||
specificity: u32,
|
||||
ambiguous: bool,
|
||||
};
|
||||
|
||||
/// Whether `rule` matches `id`: same bus, and every pinned (non-wildcard) field
|
||||
/// equal. `hid` compares as a string; the rest as integers.
|
||||
fn matches(rule: Rule, id: Identity) bool {
|
||||
if (rule.bus != id.bus) return false;
|
||||
if (rule.base) |b| if (b != id.base) return false;
|
||||
if (rule.subclass) |s| if (s != id.subclass) return false;
|
||||
if (rule.prog_if) |p| if (p != id.prog_if) return false;
|
||||
if (rule.vendor) |v| if (v != id.vendor) return false;
|
||||
if (rule.device) |d| if (d != id.device) return false;
|
||||
if (rule.subsystem) |s| if (s != id.subsystem) return false;
|
||||
if (rule.hid) |h| if (!std.mem.eql(u8, h, id.hid)) return false;
|
||||
return true;
|
||||
}
|
||||
|
||||
/// The specificity score of a rule — the sum of the weights of its pinned fields.
|
||||
fn specificity(rule: Rule) u32 {
|
||||
var score: u32 = 0;
|
||||
if (rule.base != null) score += weight_base;
|
||||
if (rule.subclass != null) score += weight_subclass;
|
||||
if (rule.prog_if != null) score += weight_prog_if;
|
||||
if (rule.vendor != null) score += weight_vendor;
|
||||
if (rule.device != null) score += weight_device;
|
||||
if (rule.subsystem != null) score += weight_subsystem;
|
||||
if (rule.hid != null) score += weight_hid;
|
||||
return score;
|
||||
}
|
||||
|
||||
/// Bind a reported device to a driver: of every rule that matches `id`, return
|
||||
/// the most specific. `null` when nothing matches (the device goes unbound —
|
||||
/// the authoritative registry does not guess). On an exact specificity tie the
|
||||
/// first such rule in file order wins and `ambiguous` is set.
|
||||
pub fn matchDriver(rules: []const Rule, id: Identity) ?Match {
|
||||
var best: ?Match = null;
|
||||
for (rules) |rule| {
|
||||
if (!matches(rule, id)) continue;
|
||||
const score = specificity(rule);
|
||||
if (best) |current| {
|
||||
if (score > current.specificity) {
|
||||
best = .{ .driver = rule.driver, .specificity = score, .ambiguous = false };
|
||||
} else if (score == current.specificity) {
|
||||
// Two equally-specific rules claim this device — keep the first,
|
||||
// flag the ambiguity for the manager to log.
|
||||
best.?.ambiguous = true;
|
||||
}
|
||||
} else {
|
||||
best = .{ .driver = rule.driver, .specificity = score, .ambiguous = false };
|
||||
}
|
||||
}
|
||||
return best;
|
||||
}
|
||||
|
||||
// --- parsing -----------------------------------------------------------------
|
||||
|
||||
/// What one CSV line parsed to. `malformed` is a non-comment, non-blank line the
|
||||
/// parser could not read (wrong field count, unparsable number, empty driver) —
|
||||
/// the manager counts these and logs, so a broken registry is loud, not silent.
|
||||
const Line = union(enum) {
|
||||
rule: Rule,
|
||||
ignorable, // blank or comment
|
||||
malformed,
|
||||
};
|
||||
|
||||
/// The result of `parse`: how many rules landed in the caller's buffer, and how
|
||||
/// many non-ignorable lines were malformed (for the manager to log). `truncated`
|
||||
/// is set if there were more valid rules than the buffer could hold.
|
||||
pub const ParseResult = struct {
|
||||
count: usize,
|
||||
malformed: usize,
|
||||
truncated: bool,
|
||||
};
|
||||
|
||||
/// Parse one hex field into `T`, honouring `*`/empty as a wildcard (`null`) and
|
||||
/// an optional `0x` prefix. Returns an error only for a genuinely unparsable
|
||||
/// non-wildcard token, so the caller can mark the whole line malformed.
|
||||
fn parseHexField(comptime T: type, field: []const u8) !?T {
|
||||
const token = std.mem.trim(u8, field, " \t");
|
||||
if (token.len == 0 or std.mem.eql(u8, token, "*")) return null;
|
||||
const digits = if (std.mem.startsWith(u8, token, "0x") or std.mem.startsWith(u8, token, "0X"))
|
||||
token[2..]
|
||||
else
|
||||
token;
|
||||
return try std.fmt.parseInt(T, digits, 16);
|
||||
}
|
||||
|
||||
/// Parse a wildcard-or-string field (the `hid` column): `*`/empty → wildcard.
|
||||
fn parseStringField(field: []const u8) ?[]const u8 {
|
||||
const token = std.mem.trim(u8, field, " \t");
|
||||
if (token.len == 0 or std.mem.eql(u8, token, "*")) return null;
|
||||
return token;
|
||||
}
|
||||
|
||||
/// Classify and (if a rule) parse one line. Split out from `parse` so it can be
|
||||
/// unit-tested directly. `line` is the raw line including no newline.
|
||||
fn parseLine(line: []const u8) Line {
|
||||
const body = csv.stripComment(line);
|
||||
if (body.len == 0) return .ignorable;
|
||||
|
||||
// Nine comma-separated fields (csv.fields trims each): bus, base, class,
|
||||
// prog_if, vendor, device, subsystem, hid, driver.
|
||||
var cols: [9][]const u8 = undefined;
|
||||
var count: usize = 0;
|
||||
var it = csv.fields(body);
|
||||
while (it.next()) |field| {
|
||||
if (count >= cols.len) return .malformed; // too many columns
|
||||
cols[count] = field;
|
||||
count += 1;
|
||||
}
|
||||
if (count != cols.len) return .malformed; // too few columns
|
||||
|
||||
const bus = Bus.fromToken(cols[0]);
|
||||
if (bus == .unknown) return .malformed;
|
||||
|
||||
const driver = cols[8];
|
||||
if (driver.len == 0) return .malformed;
|
||||
|
||||
return .{ .rule = .{
|
||||
.bus = bus,
|
||||
.base = parseHexField(u8, cols[1]) catch return .malformed,
|
||||
.subclass = parseHexField(u8, cols[2]) catch return .malformed,
|
||||
.prog_if = parseHexField(u8, cols[3]) catch return .malformed,
|
||||
.vendor = parseHexField(u16, cols[4]) catch return .malformed,
|
||||
.device = parseHexField(u16, cols[5]) catch return .malformed,
|
||||
.subsystem = parseHexField(u32, cols[6]) catch return .malformed,
|
||||
.hid = parseStringField(cols[7]),
|
||||
.driver = driver,
|
||||
} };
|
||||
}
|
||||
|
||||
/// Parse a whole `/system/configuration/devices.csv` into `out_rules`. The string fields of the
|
||||
/// returned rules point into `source`, which must outlive them.
|
||||
pub fn parse(source: []const u8, out_rules: []Rule) ParseResult {
|
||||
var result: ParseResult = .{ .count = 0, .malformed = 0, .truncated = false };
|
||||
var lines = std.mem.splitScalar(u8, source, '\n');
|
||||
while (lines.next()) |line| {
|
||||
switch (parseLine(line)) {
|
||||
.ignorable => {},
|
||||
.malformed => result.malformed += 1,
|
||||
.rule => |rule| {
|
||||
if (result.count >= out_rules.len) {
|
||||
result.truncated = true;
|
||||
continue;
|
||||
}
|
||||
out_rules[result.count] = rule;
|
||||
result.count += 1;
|
||||
},
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
// --- tests -------------------------------------------------------------------
|
||||
|
||||
const testing = std.testing;
|
||||
|
||||
// The worked example from the design: a specific virtio-gpu rule (pins vendor +
|
||||
// device) and a generic display rule (class only) both match the virtio card;
|
||||
// the specific one must win. And a plain VGA adapter still falls to the generic
|
||||
// rule. This is the whole point of widening the ABI to carry vendor/device.
|
||||
test "virtio device rule beats the generic display rule" {
|
||||
const text =
|
||||
\\# bus, base, class, prog_if, vendor, device, subsystem, hid, driver
|
||||
\\pci, 03, 00, 00, *, *, *, *, /system/drivers/display
|
||||
\\pci, 03, 80, *, 1AF4, 1050, *, *, /system/drivers/virtio-gpu
|
||||
;
|
||||
var rules: [8]Rule = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
try testing.expectEqual(@as(usize, 2), parsed.count);
|
||||
try testing.expectEqual(@as(usize, 0), parsed.malformed);
|
||||
|
||||
// The virtio-gpu function: display / other, vendor 1AF4 device 1050.
|
||||
const virtio = matchDriver(rules[0..parsed.count], .{
|
||||
.bus = .pci, .base = 0x03, .subclass = 0x80, .prog_if = 0x00,
|
||||
.vendor = 0x1AF4, .device = 0x1050,
|
||||
}).?;
|
||||
try testing.expect(!virtio.ambiguous);
|
||||
try testing.expectEqualStrings("/system/drivers/virtio-gpu", virtio.driver);
|
||||
|
||||
// A plain VGA adapter (display / VGA) still binds the generic display driver.
|
||||
const vga = matchDriver(rules[0..parsed.count], .{
|
||||
.bus = .pci, .base = 0x03, .subclass = 0x00, .prog_if = 0x00,
|
||||
.vendor = 0x1234, .device = 0x1111,
|
||||
}).?;
|
||||
try testing.expectEqualStrings("/system/drivers/display", vga.driver);
|
||||
}
|
||||
|
||||
test "no matching row leaves the device unbound" {
|
||||
const text = "pci, 0C, 03, 30, *, *, *, *, /system/drivers/usb-xhci-bus\n";
|
||||
var rules: [8]Rule = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
try testing.expectEqual(@as(usize, 1), parsed.count);
|
||||
|
||||
// An AHCI controller (mass storage / SATA / AHCI) has no row — unbound.
|
||||
const unmatched = matchDriver(rules[0..parsed.count], .{
|
||||
.bus = .pci, .base = 0x01, .subclass = 0x06, .prog_if = 0x01,
|
||||
});
|
||||
try testing.expect(unmatched == null);
|
||||
}
|
||||
|
||||
test "acpi rows match on hid" {
|
||||
const text =
|
||||
\\acpi, *, *, *, *, *, *, PNP0303, /system/drivers/ps2-bus
|
||||
\\acpi, *, *, *, *, *, *, PNP0F13, /system/drivers/ps2-bus
|
||||
;
|
||||
var rules: [8]Rule = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
try testing.expectEqual(@as(usize, 2), parsed.count);
|
||||
|
||||
const keyboard = matchDriver(rules[0..parsed.count], .{ .bus = .acpi, .hid = "PNP0303" }).?;
|
||||
try testing.expectEqualStrings("/system/drivers/ps2-bus", keyboard.driver);
|
||||
const nothing = matchDriver(rules[0..parsed.count], .{ .bus = .acpi, .hid = "PNP0A03" });
|
||||
try testing.expect(nothing == null);
|
||||
}
|
||||
|
||||
test "equally specific rules flag ambiguity" {
|
||||
const text =
|
||||
\\pci, 03, 00, 00, *, *, *, *, /system/drivers/display-a
|
||||
\\pci, 03, 00, 00, *, *, *, *, /system/drivers/display-b
|
||||
;
|
||||
var rules: [8]Rule = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
const hit = matchDriver(rules[0..parsed.count], .{
|
||||
.bus = .pci, .base = 0x03, .subclass = 0x00, .prog_if = 0x00,
|
||||
}).?;
|
||||
try testing.expect(hit.ambiguous);
|
||||
try testing.expectEqualStrings("/system/drivers/display-a", hit.driver); // first wins
|
||||
}
|
||||
|
||||
test "comments, blanks, and malformed lines" {
|
||||
const text =
|
||||
\\# a header comment
|
||||
\\
|
||||
\\pci, 0C, 03, 30, *, *, *, *, /system/drivers/usb-xhci-bus # trailing comment
|
||||
\\pci, ZZ, 03, 30, *, *, *, *, /system/drivers/broken
|
||||
\\pci, 03, 00, 00, *, *, *, *,
|
||||
\\bogus-bus, *, *, *, *, *, *, *, /system/drivers/x
|
||||
;
|
||||
var rules: [8]Rule = undefined;
|
||||
const parsed = parse(text, &rules);
|
||||
try testing.expectEqual(@as(usize, 1), parsed.count); // only the xhci row is valid
|
||||
try testing.expectEqual(@as(usize, 3), parsed.malformed); // bad hex, empty driver, bad bus
|
||||
try testing.expectEqualStrings("/system/drivers/usb-xhci-bus", rules[0].driver);
|
||||
try testing.expect(rules[0].hid == null); // trailing comment stripped, hid still wildcard
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user