Compare commits
37
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ac01f627d1 | ||
|
|
f3bc23cb81 | ||
|
|
8d4a7cf240 | ||
|
|
c4f16a5448 | ||
|
|
e4da4e0610 | ||
|
|
9a3238025d | ||
|
|
c96ef87714 | ||
|
|
de9870175f | ||
|
|
be04ebe954 | ||
|
|
62d6a7a150 | ||
|
|
fa8203cdba | ||
|
|
e53d6ebafb | ||
|
|
4476208361 | ||
|
|
c621b649f6 | ||
|
|
d27670ec39 | ||
|
|
ab7594df6e | ||
|
|
d03942b543 | ||
|
|
3f9b6813f7 | ||
|
|
bc2eb67581 | ||
|
|
cb98a9844e | ||
|
|
4701fbd123 | ||
|
|
3b23b11b0e | ||
|
|
902e4a0a9e | ||
|
|
15575960bd | ||
|
|
6f4fdc2789 | ||
|
|
721288c516 | ||
|
|
4194bb6e32 | ||
|
|
fc0b934b7f | ||
|
|
6a687fbc2b | ||
|
|
4e7cbc9792 | ||
|
|
e94adcfc02 | ||
|
|
f477ef7d9f | ||
|
|
9e649178bf | ||
|
|
e376c9e908 | ||
|
|
48b9ed4001 | ||
|
|
081ba1d74e | ||
|
|
203528c8a7 |
@@ -0,0 +1,131 @@
|
|||||||
|
//! The danos build API (docs/build-packages-plan.md): the one shared recipe
|
||||||
|
//! for building a user-space binary. A binary package's build.zig names its
|
||||||
|
//! binary and EXACTLY the modules its source imports — the moral equivalent
|
||||||
|
//! of a C file's include list — and `userBinary` resolves each name from the
|
||||||
|
//! library domain package that exports it. Nothing is pre-wired: an @import
|
||||||
|
//! the package did not declare is a compile error, and a domain none of the
|
||||||
|
//! imports come from never appears in the package's manifest. The only
|
||||||
|
//! implicit dependency is the kernel package, because the shared root shim
|
||||||
|
//! (root.zig, user.ld) lives there and itself reaches start + logging.
|
||||||
|
//!
|
||||||
|
//! Consumers declare this package in their build.zig.zon (as "build-support")
|
||||||
|
//! and @import its build.zig from their own build.zig; nothing is compiled
|
||||||
|
//! from this package itself — it exports build-time functions only.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
pub fn build(b: *std.Build) void {
|
||||||
|
_ = b; // nothing to build: this package exports build-time functions only
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The freestanding x86-64 target every danos binary (kernel and user) is
|
||||||
|
/// built for. SSE2 is part of the x86_64 baseline and UEFI leaves it enabled
|
||||||
|
/// at handoff, so we keep it: disabling it forces soft-float and makes the
|
||||||
|
/// compiler unable to encode the vector ops that std's formatting/runtime
|
||||||
|
/// still emit.
|
||||||
|
pub fn freestandingTarget(b: *std.Build) std.Build.ResolvedTarget {
|
||||||
|
return b.resolveTargetQuery(.{
|
||||||
|
.cpu_arch = .x86_64,
|
||||||
|
.os_tag = .freestanding,
|
||||||
|
.abi = .none,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Resolve one imported module by searching the packages this binary DECLARED
|
||||||
|
/// in its own build.zig.zon — the C include path made literal: an import can
|
||||||
|
/// only be satisfied by a domain the binary claims, and each domain's own
|
||||||
|
/// build.zig (its addModule exports) is the single statement of who owns
|
||||||
|
/// what. There is no name table here to drift.
|
||||||
|
fn moduleFromDeclaredDependencies(b: *std.Build, name: []const u8) *std.Build.Module {
|
||||||
|
for (b.available_deps) |declared| {
|
||||||
|
const dependency = b.dependency(declared[0], .{});
|
||||||
|
if (dependency.builder.modules.get(name)) |module| return module;
|
||||||
|
}
|
||||||
|
@panic(b.fmt(
|
||||||
|
"no declared dependency exports a module named '{s}' — declare the domain that owns it in this package's build.zig.zon",
|
||||||
|
.{name},
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// What `userBinary` needs to know about one user binary.
|
||||||
|
pub const UserBinaryOptions = struct {
|
||||||
|
name: []const u8,
|
||||||
|
/// The program's own source file — it becomes the `program` module the
|
||||||
|
/// root shim imports; a program only defines `pub fn main`.
|
||||||
|
root_source_file: std.Build.LazyPath,
|
||||||
|
/// Exactly the modules the program's source @imports (directly or through
|
||||||
|
/// its same-directory files) — no more, no less. Order is free; sorted
|
||||||
|
/// reads best. An undeclared @import fails the compile; a name no
|
||||||
|
/// declared domain exports fails the build graph, naming the miss.
|
||||||
|
imports: []const []const u8,
|
||||||
|
/// Built multi-threaded (`single_threaded = false`) so real atomics/TLS
|
||||||
|
/// work — required before a binary may call `Thread.spawn`
|
||||||
|
/// (docs/threading.md). Threads are a deliberate per-binary opt-in.
|
||||||
|
threaded: bool = false,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Build one user-space binary the same way for every program (init, the
|
||||||
|
/// services, the drivers): freestanding, ReleaseSmall, `.large` code model
|
||||||
|
/// (the image base is above 4 GiB — smaller models emit 32-bit relocations
|
||||||
|
/// that can't reach), linked with the shared user link script. Pinned to
|
||||||
|
/// LLVM + LLD so the script's PHDRS (segment permissions) are authoritative —
|
||||||
|
/// the kernel's W^X user-ELF loader requires exact perms.
|
||||||
|
///
|
||||||
|
/// The compilation root is not the program's own file but the shared shim
|
||||||
|
/// (the kernel package's root.zig), which supplies the root declarations
|
||||||
|
/// (`main` re-export, panic handler, `_start` pull) so a program only defines
|
||||||
|
/// `pub fn main`. The program's file becomes the `program` module the shim
|
||||||
|
/// imports; reach it through `programModule` to add per-binary non-library
|
||||||
|
/// modules (compile-time options).
|
||||||
|
pub fn userBinary(b: *std.Build, options: UserBinaryOptions) *std.Build.Step.Compile {
|
||||||
|
const kernel = b.dependency("kernel", .{});
|
||||||
|
var imports: std.ArrayListUnmanaged(std.Build.Module.Import) = .empty;
|
||||||
|
for (options.imports) |name| {
|
||||||
|
imports.append(b.allocator, .{
|
||||||
|
.name = name,
|
||||||
|
.module = moduleFromDeclaredDependencies(b, name),
|
||||||
|
}) catch @panic("OOM");
|
||||||
|
}
|
||||||
|
// Settings (target, optimize, code model, ...) live on the root module
|
||||||
|
// only; the program module inherits them.
|
||||||
|
const program_module = b.createModule(.{
|
||||||
|
.root_source_file = options.root_source_file,
|
||||||
|
.imports = imports.items,
|
||||||
|
});
|
||||||
|
const exe = b.addExecutable(.{
|
||||||
|
.name = options.name,
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = kernel.path("root.zig"),
|
||||||
|
.target = freestandingTarget(b),
|
||||||
|
.optimize = .ReleaseSmall,
|
||||||
|
.code_model = .large,
|
||||||
|
.single_threaded = !options.threaded, // a threaded binary needs real atomics/TLS
|
||||||
|
.sanitize_c = .off,
|
||||||
|
.stack_check = false,
|
||||||
|
.stack_protector = false,
|
||||||
|
// The root shim itself imports only start (_start + panic) and
|
||||||
|
// logging (std_options) — straight from the kernel package, so a
|
||||||
|
// program's own import list stays exactly its own.
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "start", .module = kernel.module("start") },
|
||||||
|
.{ .name = "logging", .module = kernel.module("logging") },
|
||||||
|
.{ .name = "program", .module = program_module },
|
||||||
|
},
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
exe.setLinkerScript(kernel.path("user.ld"));
|
||||||
|
exe.entry = .{ .symbol_name = "_start" };
|
||||||
|
exe.image_base = 0x7000_0000_0000;
|
||||||
|
exe.use_llvm = true;
|
||||||
|
exe.use_lld = true;
|
||||||
|
return exe;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The `program` module of a binary built by `userBinary` — the module rooted
|
||||||
|
/// at the program's own source file. Per-binary non-library modules (an
|
||||||
|
/// addOptions build_options) go here, not on the root shim: module imports
|
||||||
|
/// are not transitive, so an import added to the root would be invisible to
|
||||||
|
/// the program's code.
|
||||||
|
pub fn programModule(exe: *std.Build.Step.Compile) *std.Build.Module {
|
||||||
|
return exe.root_module.import_table.get("program").?;
|
||||||
|
}
|
||||||
@@ -0,0 +1,8 @@
|
|||||||
|
.{
|
||||||
|
.name = .build_support,
|
||||||
|
.version = "0.0.0",
|
||||||
|
.fingerprint = 0xad91962994f4be41, // Changing this has security and trust implications.
|
||||||
|
.minimum_zig_version = "0.16.0",
|
||||||
|
.dependencies = .{},
|
||||||
|
.paths = .{""},
|
||||||
|
}
|
||||||
+44
-8
@@ -32,6 +32,49 @@
|
|||||||
// Once all dependencies are fetched, `zig build` no longer requires
|
// Once all dependencies are fetched, `zig build` no longer requires
|
||||||
// internet connectivity.
|
// internet connectivity.
|
||||||
.dependencies = .{
|
.dependencies = .{
|
||||||
|
// The danos build API — the shared user-binary recipe every build file
|
||||||
|
// (root and per-binary packages) consumes (docs/build-packages-plan.md).
|
||||||
|
.@"build-support" = .{ .path = "build-support" },
|
||||||
|
// The library domains, each a package exporting its modules.
|
||||||
|
.kernel = .{ .path = "library/kernel" },
|
||||||
|
.device = .{ .path = "library/device" },
|
||||||
|
.client = .{ .path = "library/client" },
|
||||||
|
.protocol = .{ .path = "library/protocol" },
|
||||||
|
.csv = .{ .path = "library/csv" },
|
||||||
|
.@"xkeyboard-config" = .{ .path = "library/xkeyboard-config" },
|
||||||
|
// Binary packages (phase 2), consumed as artifacts for the boot image.
|
||||||
|
.@"pci-bus" = .{ .path = "system/drivers/pci-bus" },
|
||||||
|
.init = .{ .path = "system/services/init" },
|
||||||
|
.fat = .{ .path = "system/services/fat" },
|
||||||
|
.display = .{ .path = "system/services/display" },
|
||||||
|
.@"display-demo" = .{ .path = "system/services/display-demo" },
|
||||||
|
.@"device-manager" = .{ .path = "system/services/device-manager" },
|
||||||
|
.input = .{ .path = "system/services/input" },
|
||||||
|
.logger = .{ .path = "system/services/logger" },
|
||||||
|
// The discovery pair and the /test fixtures are lazy: only what a
|
||||||
|
// given build actually ships gets its build file loaded and compiled
|
||||||
|
// (-Ddiscovery picks one of the pair; -Dtest-case pulls the fixtures).
|
||||||
|
.acpi = .{ .path = "system/services/acpi", .lazy = true },
|
||||||
|
.fdt = .{ .path = "system/services/fdt", .lazy = true },
|
||||||
|
.@"ps2-bus" = .{ .path = "system/drivers/ps2-bus" },
|
||||||
|
.@"usb-xhci-bus" = .{ .path = "system/drivers/usb-xhci-bus" },
|
||||||
|
.@"usb-hid" = .{ .path = "system/drivers/usb-hid" },
|
||||||
|
.@"usb-storage" = .{ .path = "system/drivers/usb-storage" },
|
||||||
|
.@"virtio-gpu" = .{ .path = "system/drivers/virtio-gpu" },
|
||||||
|
.@"vfs-test" = .{ .path = "test/system/services/vfs-test", .lazy = true },
|
||||||
|
.@"fat-test" = .{ .path = "test/system/services/fat-test", .lazy = true },
|
||||||
|
.@"shared-memory-server" = .{ .path = "test/system/services/shared-memory-server", .lazy = true },
|
||||||
|
.@"shared-memory-client" = .{ .path = "test/system/services/shared-memory-client", .lazy = true },
|
||||||
|
.@"crash-test" = .{ .path = "test/system/services/crash-test", .lazy = true },
|
||||||
|
.@"device-list" = .{ .path = "test/system/services/device-list", .lazy = true },
|
||||||
|
.@"pci-cap-test" = .{ .path = "test/system/services/pci-cap-test", .lazy = true },
|
||||||
|
.@"iommu-fault-test" = .{ .path = "test/system/services/iommu-fault-test", .lazy = true },
|
||||||
|
.@"input-source" = .{ .path = "test/system/services/input-source", .lazy = true },
|
||||||
|
.@"input-test" = .{ .path = "test/system/services/input-test", .lazy = true },
|
||||||
|
.@"args-echo" = .{ .path = "test/system/services/args-echo", .lazy = true },
|
||||||
|
.@"process-test" = .{ .path = "test/system/services/process-test", .lazy = true },
|
||||||
|
.@"thread-test" = .{ .path = "test/system/services/thread-test", .lazy = true },
|
||||||
|
.@"user-memory-test" = .{ .path = "test/system/services/user-memory-test", .lazy = true },
|
||||||
// See `zig fetch --save <url>` for a command-line interface for adding dependencies.
|
// See `zig fetch --save <url>` for a command-line interface for adding dependencies.
|
||||||
//.example = .{
|
//.example = .{
|
||||||
// // When updating this field to a new URL, be sure to delete the corresponding
|
// // When updating this field to a new URL, be sure to delete the corresponding
|
||||||
@@ -70,12 +113,5 @@
|
|||||||
// Paths are relative to the build root. Use the empty string (`""`) to refer to
|
// Paths are relative to the build root. Use the empty string (`""`) to refer to
|
||||||
// the build root itself.
|
// the build root itself.
|
||||||
// A directory listed here means that all files within, recursively, are included.
|
// A directory listed here means that all files within, recursively, are included.
|
||||||
.paths = .{
|
.paths = .{""},
|
||||||
"build.zig",
|
|
||||||
"build.zig.zon",
|
|
||||||
"src",
|
|
||||||
// For example...
|
|
||||||
//"LICENSE",
|
|
||||||
//"README.md",
|
|
||||||
},
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,166 @@
|
|||||||
|
//! Boot-image assembly (docs/build-packages-plan.md, phase 3): everything
|
||||||
|
//! between "here are the built binaries" and "here is a bootable volume".
|
||||||
|
//! The FHS-shaped zig-out install tree, the boot manifest, the boot capsule,
|
||||||
|
//! the FAT32 USB image (+ its serial-enabled twin for the QEMU run steps),
|
||||||
|
//! and the release ISO — with their check steps. The root build.zig decides
|
||||||
|
//! WHAT ships (the bundled list); this file owns HOW it becomes an image.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
/// One user binary and its FHS home on the boot volume (and in zig-out).
|
||||||
|
pub const BundledBinary = struct { path: []const u8, binary: std.Build.LazyPath };
|
||||||
|
|
||||||
|
pub const Options = struct {
|
||||||
|
/// The installed/flashable kernel (serial follows the root -Dserial).
|
||||||
|
kernel: *std.Build.Step.Compile,
|
||||||
|
/// The serial-enabled kernel variant the `run-x86-64` image boots.
|
||||||
|
kernel_serial: *std.Build.Step.Compile,
|
||||||
|
/// The UEFI loader (BOOTX64).
|
||||||
|
efi: *std.Build.Step.Compile,
|
||||||
|
/// Every user binary and data file at its FHS path.
|
||||||
|
bundled: []const BundledBinary,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Wire up the install tree, both FAT32 boot images, the release ISO, and the
|
||||||
|
/// check steps. Returns the serial-enabled FAT image for the QEMU run steps.
|
||||||
|
pub fn addImageSteps(b: *std.Build, options: Options) std.Build.LazyPath {
|
||||||
|
// Everything installs into a FHS-shaped zig-out: it IS the danos filesystem *and*
|
||||||
|
// the boot volume. Each binary lands at its addressed, leaf-collapsed path — the
|
||||||
|
// kernel at zig-out/system/kernel (from system/kernel/kernel.zig), init at
|
||||||
|
// zig-out/system/services/init, and so on (see docs/README.md). The bootloader
|
||||||
|
// then loads these FHS paths off the volume.
|
||||||
|
const kernel_install = b.addInstallArtifact(options.kernel, .{ .dest_dir = .{ .override = .{ .custom = "system" } } });
|
||||||
|
b.getInstallStep().dependOn(&kernel_install.step);
|
||||||
|
|
||||||
|
// UEFI firmware requires the removable-media loader at exactly \EFI\BOOT\BOOTX64.efi,
|
||||||
|
// so that path is fixed by the firmware (it is /boot's EFI stub, conceptually).
|
||||||
|
const efi_install = b.addInstallArtifact(options.efi, .{ .dest_dir = .{ .override = .{ .custom = "EFI/BOOT" } } });
|
||||||
|
b.getInstallStep().dependOn(&efi_install.step);
|
||||||
|
|
||||||
|
// The boot manifest: the FHS path of every bundled binary, one per line. The
|
||||||
|
// EFI loader reads THIS by name and opens each listed path by name — FAT
|
||||||
|
// name lookup is case-insensitive and firmware-portable, unlike directory
|
||||||
|
// ENUMERATION, whose returned names vary by firmware (bare 8.3 entries come
|
||||||
|
// back uppercase on some FAT drivers). The tree walk remains only as the
|
||||||
|
// loader's fallback for hand-assembled sticks without a manifest.
|
||||||
|
var manifest_text: std.ArrayListUnmanaged(u8) = .empty;
|
||||||
|
for (options.bundled) |item| {
|
||||||
|
manifest_text.append(b.allocator, '/') catch @panic("OOM");
|
||||||
|
manifest_text.appendSlice(b.allocator, item.path) catch @panic("OOM");
|
||||||
|
manifest_text.append(b.allocator, '\n') catch @panic("OOM");
|
||||||
|
}
|
||||||
|
const manifest_files = b.addWriteFiles();
|
||||||
|
const manifest_file = manifest_files.add("manifest", manifest_text.items);
|
||||||
|
const manifest_install = b.addInstallFileWithDir(manifest_file, .prefix, "system/manifest");
|
||||||
|
b.getInstallStep().dependOn(&manifest_install.step);
|
||||||
|
|
||||||
|
// The boot capsule: the same bundled list packed into ONE file (v2
|
||||||
|
// initial_ramdisk format), because a single open + sequential read is the
|
||||||
|
// only firmware file I/O shape that is fast everywhere — a per-file tree
|
||||||
|
// walk measured MINUTES on real firmware. The loader tries this first,
|
||||||
|
// then the manifest, then the walk; the running system cannot tell the
|
||||||
|
// difference (it always receives the same in-RAM table). Derived from the
|
||||||
|
// tree in the same build graph, so the two cannot drift.
|
||||||
|
const mk_capsule = b.addSystemCommand(&.{"python3"});
|
||||||
|
mk_capsule.addFileArg(b.path("tools/pack-system-image.py"));
|
||||||
|
const capsule_img = mk_capsule.addOutputFileArg("system.img");
|
||||||
|
for (options.bundled) |item| {
|
||||||
|
mk_capsule.addArg(item.path);
|
||||||
|
mk_capsule.addFileArg(item.binary);
|
||||||
|
}
|
||||||
|
const capsule_install = b.addInstallFile(capsule_img, "boot/system.img");
|
||||||
|
b.getInstallStep().dependOn(&capsule_install.step);
|
||||||
|
|
||||||
|
// Install every bundled binary to its FHS home, so zig-out is a true image of
|
||||||
|
// the filesystem — the same tree make-fat-image.py lays out on the boot volume.
|
||||||
|
for (options.bundled) |item| {
|
||||||
|
const install = b.addInstallFileWithDir(item.binary, .prefix, item.path);
|
||||||
|
b.getInstallStep().dependOn(&install.step);
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- danos-usb.img: the bootable FAT32 USB image ---
|
||||||
|
// Format a real FAT32 image (the in-repo Python builder, no external tools)
|
||||||
|
// holding the EFI stub, the kernel, and the whole /system tree of user
|
||||||
|
// binaries at their FHS paths. QEMU presents this image as a USB mass-storage
|
||||||
|
// device the guest boots from (see run-x86-64 and the test harness), and the
|
||||||
|
// danos fat driver mounts the same image at /volumes/usb.
|
||||||
|
const fat_image = addBootImage(b, options.kernel.getEmittedBin(), options.efi.getEmittedBin(), manifest_file, capsule_img, options.bundled);
|
||||||
|
const fat_image_install = b.addInstallFile(fat_image, "danos-usb.img");
|
||||||
|
b.getInstallStep().dependOn(&fat_image_install.step);
|
||||||
|
|
||||||
|
// The image `run-x86-64` boots: identical to the flashable one but with the
|
||||||
|
// serial log sink compiled in, so a developer always gets the machine-readable
|
||||||
|
// log captured to serial0 — without baking serial into the image users flash.
|
||||||
|
// Built lazily (only when `run-x86-64` is requested), and never installed.
|
||||||
|
const fat_image_serial = addBootImage(b, options.kernel_serial.getEmittedBin(), options.efi.getEmittedBin(), manifest_file, capsule_img, options.bundled);
|
||||||
|
|
||||||
|
// `zig build check-fat-image` — validate the produced image is a real FAT32
|
||||||
|
// with the EFI stub present (the builder's own --verify, no external tools).
|
||||||
|
const check_fat = b.addSystemCommand(&.{"python3"});
|
||||||
|
check_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||||
|
check_fat.addArg("--verify");
|
||||||
|
check_fat.addFileArg(fat_image);
|
||||||
|
const check_fat_step = b.step("check-fat-image", "Verify the FAT32 USB image is valid and bootable");
|
||||||
|
check_fat_step.dependOn(&check_fat.step);
|
||||||
|
|
||||||
|
// --- release-x86-64: danos-x86-64.iso, the flashable release image ---
|
||||||
|
// Wrap the FAT32 boot volume in a hybrid ISO (the in-repo Python builder
|
||||||
|
// again, no xorriso/isohybrid): an ISO9660 whose El Torito EFI boot entry
|
||||||
|
// and MBR ESP partition entry both point at the embedded FAT image. One
|
||||||
|
// file then boots every way release media is consumed — flashed raw to a
|
||||||
|
// USB stick with Etcher or dd, or burned to optical media — while
|
||||||
|
// danos-usb.img stays the raw superfloppy QEMU and the test harness boot.
|
||||||
|
const mk_iso = b.addSystemCommand(&.{"python3"});
|
||||||
|
mk_iso.addFileArg(b.path("tools/make-iso-image.py"));
|
||||||
|
const iso_image = mk_iso.addOutputFileArg("danos-x86-64.iso");
|
||||||
|
mk_iso.addFileArg(fat_image);
|
||||||
|
const iso_install = b.addInstallFile(iso_image, "danos-x86-64.iso");
|
||||||
|
const release_step = b.step("release-x86-64", "Build the flashable x86-64 release ISO (zig-out/danos-x86-64.iso; flash with Etcher or dd)");
|
||||||
|
release_step.dependOn(&iso_install.step);
|
||||||
|
|
||||||
|
// `zig build check-iso-image` — the ISO builder's own --verify (mirroring
|
||||||
|
// check-fat-image): the MBR partition, the El Torito catalog, and the
|
||||||
|
// embedded FAT32 image must all agree.
|
||||||
|
const check_iso = b.addSystemCommand(&.{"python3"});
|
||||||
|
check_iso.addFileArg(b.path("tools/make-iso-image.py"));
|
||||||
|
check_iso.addArg("--verify");
|
||||||
|
check_iso.addFileArg(iso_image);
|
||||||
|
const check_iso_step = b.step("check-iso-image", "Verify the release ISO is a valid hybrid (MBR ESP partition + El Torito EFI entry)");
|
||||||
|
check_iso_step.dependOn(&check_iso.step);
|
||||||
|
|
||||||
|
return fat_image_serial;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Assemble the bootable FAT32 image (the in-repo Python builder) holding the
|
||||||
|
/// EFI stub, the kernel, and every user binary at its FHS path — the volume's
|
||||||
|
/// /system tree IS the system image; the EFI loader walks it at boot and builds
|
||||||
|
/// the in-RAM initial_ramdisk from it. Factored so the serial-enabled
|
||||||
|
/// `run-x86-64` variant can bundle its own serial kernel while sharing the
|
||||||
|
/// loader and user tree (the loader's boot breadcrumbs and init's heartbeat both
|
||||||
|
/// follow the top-level -Dserial). Returns the image's LazyPath.
|
||||||
|
fn addBootImage(
|
||||||
|
b: *std.Build,
|
||||||
|
kernel_bin: std.Build.LazyPath,
|
||||||
|
efi_bin: std.Build.LazyPath,
|
||||||
|
manifest: std.Build.LazyPath,
|
||||||
|
capsule: std.Build.LazyPath,
|
||||||
|
bundled: []const BundledBinary,
|
||||||
|
) std.Build.LazyPath {
|
||||||
|
const mk_fat = b.addSystemCommand(&.{"python3"});
|
||||||
|
mk_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||||
|
const fat_image = mk_fat.addOutputFileArg("danos-usb.img");
|
||||||
|
mk_fat.addArg("64"); // MiB
|
||||||
|
mk_fat.addArg("EFI/BOOT/BOOTX64.efi");
|
||||||
|
mk_fat.addFileArg(efi_bin);
|
||||||
|
mk_fat.addArg("system/kernel");
|
||||||
|
mk_fat.addFileArg(kernel_bin);
|
||||||
|
mk_fat.addArg("system/manifest");
|
||||||
|
mk_fat.addFileArg(manifest);
|
||||||
|
mk_fat.addArg("boot/system.img");
|
||||||
|
mk_fat.addFileArg(capsule);
|
||||||
|
for (bundled) |item| {
|
||||||
|
mk_fat.addArg(item.path);
|
||||||
|
mk_fat.addFileArg(item.binary);
|
||||||
|
}
|
||||||
|
return fat_image;
|
||||||
|
}
|
||||||
+176
@@ -0,0 +1,176 @@
|
|||||||
|
//! The QEMU run steps (docs/build-packages-plan.md, phase 3): `run-x86-64`
|
||||||
|
//! boots the serial-enabled FAT image via UEFI/OVMF; `run-x86-64-gpu` adds a
|
||||||
|
//! virtio-gpu adapter for the native-present display path. OVMF firmware is
|
||||||
|
//! probed across distro/OS layouts (-Dovmf-code / -Dovmf-vars override).
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
/// Wire up the `run-x86-64` and `run-x86-64-gpu` steps around the given
|
||||||
|
/// serial-enabled boot image (the guest boots that self-contained image
|
||||||
|
/// attached as USB storage, not the installed FHS zig-out).
|
||||||
|
pub fn addRunSteps(b: *std.Build, fat_image_serial: std.Build.LazyPath) void {
|
||||||
|
// Firmware lives in different places per OS/distro, so probe the known
|
||||||
|
// layouts (Architecture, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
|
||||||
|
// that exists. Override with -Dovmf-code / -Dovmf-vars if yours is elsewhere.
|
||||||
|
const ovmf_code = b.option(
|
||||||
|
[]const u8,
|
||||||
|
"ovmf-code",
|
||||||
|
"Path to the OVMF_CODE firmware image",
|
||||||
|
) orelse firstExisting(b.graph.io, &.{
|
||||||
|
"/usr/share/edk2/x64/OVMF_CODE.4m.fd", // Architecture
|
||||||
|
"/usr/share/OVMF/OVMF_CODE_4M.fd", // Debian/Ubuntu
|
||||||
|
"/usr/share/OVMF/OVMF_CODE.fd", // older Debian/Ubuntu
|
||||||
|
"/usr/share/edk2-ovmf/x64/OVMF_CODE.fd", // Fedora
|
||||||
|
"/opt/homebrew/share/qemu/edk2-x86_64-code.fd", // macOS Homebrew (Apple Silicon)
|
||||||
|
"/usr/local/share/qemu/edk2-x86_64-code.fd", // macOS Homebrew (Intel)
|
||||||
|
});
|
||||||
|
const ovmf_vars = b.option(
|
||||||
|
[]const u8,
|
||||||
|
"ovmf-vars",
|
||||||
|
"Path to the OVMF_VARS firmware image (a writable copy is made)",
|
||||||
|
) orelse firstExisting(b.graph.io, &.{
|
||||||
|
"/usr/share/edk2/x64/OVMF_VARS.4m.fd", // Architecture
|
||||||
|
"/usr/share/OVMF/OVMF_VARS_4M.fd", // Debian/Ubuntu
|
||||||
|
"/usr/share/OVMF/OVMF_VARS.fd", // older Debian/Ubuntu
|
||||||
|
"/usr/share/edk2-ovmf/x64/OVMF_VARS.fd", // Fedora
|
||||||
|
"/opt/homebrew/share/qemu/edk2-i386-vars.fd", // macOS Homebrew (Apple Silicon)
|
||||||
|
"/usr/local/share/qemu/edk2-i386-vars.fd", // macOS Homebrew (Intel)
|
||||||
|
});
|
||||||
|
|
||||||
|
// The firmware needs to write NVRAM, so give it a writable copy of the vars.
|
||||||
|
const vars_copy = b.addSystemCommand(&.{ "cp", "-f", ovmf_vars });
|
||||||
|
const vars_out = vars_copy.addOutputFileArg("OVMF_VARS.4m.fd");
|
||||||
|
|
||||||
|
// Capture the guest's serial0 (danos's machine-readable log) to the qemu-test
|
||||||
|
// scratch area — a dev/host artifact, kept out of the boot volume we mount.
|
||||||
|
// (/system/logs on the volume belongs to the guest's own logger.) One
|
||||||
|
// timestamped file per run.
|
||||||
|
const log_dir = b.fmt("{s}/qemu-test", .{b.install_path});
|
||||||
|
const make_log_dir = b.addSystemCommand(&.{ "mkdir", "-p", log_dir });
|
||||||
|
|
||||||
|
// --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF ---
|
||||||
|
const run_efi = b.addSystemCommand(&.{
|
||||||
|
"qemu-system-x86_64",
|
||||||
|
"-device",
|
||||||
|
"qemu-xhci,id=xhci",
|
||||||
|
"-device",
|
||||||
|
"usb-mouse,bus=xhci.0",
|
||||||
|
"-device",
|
||||||
|
"usb-kbd,bus=xhci.0",
|
||||||
|
"-machine",
|
||||||
|
"q35",
|
||||||
|
"-m",
|
||||||
|
"128M",
|
||||||
|
"-drive",
|
||||||
|
b.fmt("if=pflash,format=raw,readonly=on,file={s}", .{ovmf_code}),
|
||||||
|
});
|
||||||
|
run_efi.addArg("-drive");
|
||||||
|
run_efi.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
||||||
|
// Boot off the FAT32 USB image: a mass-storage device on the same xHCI bus as
|
||||||
|
// the keyboard and mouse. OVMF finds \EFI\BOOT\BOOTX64.efi on it and boots.
|
||||||
|
// The serial-enabled variant, so serial0 carries the log for this dev boot.
|
||||||
|
run_efi.addArg("-drive");
|
||||||
|
run_efi.addPrefixedFileArg("if=none,id=bootusb,format=raw,file=", fat_image_serial);
|
||||||
|
run_efi.addArgs(&.{
|
||||||
|
"-device",
|
||||||
|
"usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
||||||
|
"-net",
|
||||||
|
"none",
|
||||||
|
// Emulated display advertising 1280x720 as its native (EDID preferred)
|
||||||
|
// resolution, so the kernel's native-resolution switch has something to
|
||||||
|
// find. `-vga none` avoids a second, default adapter.
|
||||||
|
"-vga",
|
||||||
|
"none",
|
||||||
|
"-device",
|
||||||
|
"VGA,edid=on,xres=1280,yres=720",
|
||||||
|
});
|
||||||
|
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
||||||
|
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
|
||||||
|
// We boot the self-contained `fat_image_serial` (added as a file arg above, so
|
||||||
|
// it's already a dependency) — not the installed FHS zig-out — so `run-x86-64`
|
||||||
|
// builds only the serial kernel, never the flashable one. Just make the serial
|
||||||
|
// scratch dir first.
|
||||||
|
run_efi.step.dependOn(&make_log_dir.step);
|
||||||
|
|
||||||
|
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/qemu-test/run-x86-64-serial0-<timestamp>.log");
|
||||||
|
run_efi_step.dependOn(&run_efi.step);
|
||||||
|
|
||||||
|
// --- run-x86-64-gpu: the same boot plus a virtio-gpu adapter ---
|
||||||
|
// The VGA device still supplies the boot (GOP) framebuffer the compositor starts
|
||||||
|
// on; the virtio-gpu function is discovered by the device-manager stack, its
|
||||||
|
// driver announces a shared scanout, and the compositor upgrades off the GOP
|
||||||
|
// floor to fenced, tear-free native presents (docs/display-v2.md).
|
||||||
|
// This is the interactive twin of the `display-native` test case, and 512M
|
||||||
|
// matches it (the whole driver stack + the compositor's surfaces at once).
|
||||||
|
// QEMU shows one head per adapter: pick the virtio-gpu head in the View menu
|
||||||
|
// to watch the native output.
|
||||||
|
const run_gpu = b.addSystemCommand(&.{
|
||||||
|
"qemu-system-x86_64",
|
||||||
|
"-device",
|
||||||
|
"qemu-xhci,id=xhci",
|
||||||
|
"-device",
|
||||||
|
"usb-mouse,bus=xhci.0",
|
||||||
|
"-device",
|
||||||
|
"usb-kbd,bus=xhci.0",
|
||||||
|
"-machine",
|
||||||
|
"q35",
|
||||||
|
"-m",
|
||||||
|
"512M",
|
||||||
|
"-drive",
|
||||||
|
b.fmt("if=pflash,format=raw,readonly=on,file={s}", .{ovmf_code}),
|
||||||
|
});
|
||||||
|
run_gpu.addArg("-drive");
|
||||||
|
run_gpu.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
||||||
|
run_gpu.addArg("-drive");
|
||||||
|
run_gpu.addPrefixedFileArg("if=none,id=bootusb,format=raw,file=", fat_image_serial);
|
||||||
|
run_gpu.addArgs(&.{
|
||||||
|
"-device",
|
||||||
|
"usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
||||||
|
"-net",
|
||||||
|
"none",
|
||||||
|
"-vga",
|
||||||
|
"none",
|
||||||
|
"-device",
|
||||||
|
"VGA,edid=on,xres=1280,yres=720",
|
||||||
|
"-device",
|
||||||
|
"virtio-gpu-pci",
|
||||||
|
});
|
||||||
|
const gpu_serial_log = b.fmt("{s}/run-x86-64-gpu-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
||||||
|
run_gpu.addArgs(&.{ "-serial", b.fmt("file:{s}", .{gpu_serial_log}) });
|
||||||
|
run_gpu.step.dependOn(&make_log_dir.step);
|
||||||
|
|
||||||
|
const run_gpu_step = b.step("run-x86-64-gpu", "Boot in QEMU with a virtio-gpu adapter: the compositor upgrades to fenced (tear-free) native presents; watch the virtio-gpu head in QEMU's View menu");
|
||||||
|
run_gpu_step.dependOn(&run_gpu.step);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Return the first path in `candidates` that exists on the build host, else the
|
||||||
|
/// first candidate as a fallback so a missing-firmware error still names a
|
||||||
|
/// concrete (and, by convention, the primary) path. Used to locate OVMF firmware
|
||||||
|
/// across distro/OS layouts without configuration.
|
||||||
|
fn firstExisting(io: std.Io, candidates: []const []const u8) []const u8 {
|
||||||
|
for (candidates) |path| {
|
||||||
|
std.Io.Dir.accessAbsolute(io, path, .{}) catch continue;
|
||||||
|
return path;
|
||||||
|
}
|
||||||
|
return candidates[0];
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A UTC timestamp like "20260708-153045", for naming a per-run artifact so
|
||||||
|
/// repeated runs don't clobber each other's logs. Resolved when `zig build`
|
||||||
|
/// runs, which is moments before QEMU launches.
|
||||||
|
fn timestamp(b: *std.Build) []const u8 {
|
||||||
|
const ns = std.Io.Clock.now(.real, b.graph.io).nanoseconds;
|
||||||
|
const secs: u64 = @intCast(@divFloor(ns, std.time.ns_per_s));
|
||||||
|
const es = std.time.epoch.EpochSeconds{ .secs = secs };
|
||||||
|
const yd = es.getEpochDay().calculateYearDay();
|
||||||
|
const md = yd.calculateMonthDay();
|
||||||
|
const ds = es.getDaySeconds();
|
||||||
|
return b.fmt("{d:0>4}{d:0>2}{d:0>2}-{d:0>2}{d:0>2}{d:0>2}", .{
|
||||||
|
yd.year,
|
||||||
|
md.month.numeric(),
|
||||||
|
@as(u32, md.day_index) + 1,
|
||||||
|
ds.getHoursIntoDay(),
|
||||||
|
ds.getMinutesIntoHour(),
|
||||||
|
ds.getSecondsIntoMinute(),
|
||||||
|
});
|
||||||
|
}
|
||||||
+16
-3
@@ -208,7 +208,7 @@ the whole reason for the arrangement ([vision.md](vision.md)).
|
|||||||
danos is a **monorepo of sub-projects**. Each service or driver is a directory that is
|
danos is a **monorepo of sub-projects**. Each service or driver is a directory that is
|
||||||
its own Zig module — it can hold as many files as it needs, and other sub-projects
|
its own Zig module — it can hold as many files as it needs, and other sub-projects
|
||||||
reach it *by module name*, never by a path into its files. The source tree deliberately
|
reach it *by module name*, never by a path into its files. The source tree deliberately
|
||||||
**mirrors the runtime FHS** ([danos-file-system-hierarchy-FSH.md](file-system-development/danos-file-system-hierarchy-FSH.md)):
|
**mirrors the runtime file-system hierarchy** ([file-system-hierarchy.md](file-system-development/file-system-hierarchy.md)):
|
||||||
what you see under `system/` in the source is what a running danos represents under
|
what you see under `system/` in the source is what a running danos represents under
|
||||||
`/system`.
|
`/system`.
|
||||||
|
|
||||||
@@ -216,7 +216,7 @@ what you see under `system/` in the source is what a running danos represents un
|
|||||||
name.** `system/services/init/` contains `init.zig` (its root), and produces a binary
|
name.** `system/services/init/` contains `init.zig` (its root), and produces a binary
|
||||||
addressed as **`system/services/init`** — the repeated leaf resolves away:
|
addressed as **`system/services/init`** — the repeated leaf resolves away:
|
||||||
|
|
||||||
| Source (root file) | Addressed as (module / binary / FHS path) |
|
| Source (root file) | Addressed as (module / binary / hierarchy path) |
|
||||||
|----------------------------------------|--------------------------------------------|
|
|----------------------------------------|--------------------------------------------|
|
||||||
| `system/services/init/init.zig` | `system/services/init` → `/system/services/init` |
|
| `system/services/init/init.zig` | `system/services/init` → `/system/services/init` |
|
||||||
| `system/drivers/ps2-bus/ps2-bus.zig` | `system/drivers/ps2-bus` → `/system/drivers/ps2-bus` |
|
| `system/drivers/ps2-bus/ps2-bus.zig` | `system/drivers/ps2-bus` → `/system/drivers/ps2-bus` |
|
||||||
@@ -263,9 +263,20 @@ test/ → /test the test tree: the QEMU harness (qemu_test.py, h
|
|||||||
system/services/ beside the on-image test fixtures — vfs-test/ thread-test/
|
system/services/ beside the on-image test fixtures — vfs-test/ thread-test/
|
||||||
crash-test/ … — whose repo path IS their boot-volume path
|
crash-test/ … — whose repo path IS their boot-volume path
|
||||||
(/test/system/services/<name>)
|
(/test/system/services/<name>)
|
||||||
|
build-support/ the danos build API (build-time only, nothing on the image):
|
||||||
|
the shared user-binary recipe + default-import wiring every
|
||||||
|
build file consumes (docs/build-packages-plan.md)
|
||||||
|
build/ root-build helpers: image assembly (images.zig) + the QEMU
|
||||||
|
run steps (qemu.zig)
|
||||||
tools/ host-side build scripts
|
tools/ host-side build scripts
|
||||||
```
|
```
|
||||||
|
|
||||||
|
**Builds are packages** (docs/build-packages-plan.md): each `library/` domain owns a
|
||||||
|
`build.zig`/`build.zig.zon` exporting its modules (with a standalone `zig build test`),
|
||||||
|
every binary directory is a ~15-line package build, and the root `build.zig`
|
||||||
|
orchestrates — the kernel + loader, what ships, and the aggregate test step — with
|
||||||
|
image assembly in `build/images.zig` and the QEMU run steps in `build/qemu.zig`.
|
||||||
|
|
||||||
**Wire protocols live in `library/protocol/`**, one module per directory
|
**Wire protocols live in `library/protocol/`**, one module per directory
|
||||||
(`library/protocol/vfs/vfs-protocol.zig` is the `vfs-protocol` module), imported by module
|
(`library/protocol/vfs/vfs-protocol.zig` is the `vfs-protocol` module), imported by module
|
||||||
name. A protocol is the seam between a low-level driver and the higher-level service it
|
name. A protocol is the seam between a low-level driver and the higher-level service it
|
||||||
@@ -320,5 +331,7 @@ exception in [coding-standards.md](coding-standards.md) applies to that seam.
|
|||||||
| System services (init, the `fat` filesystem, the device-manager) | `system/services/` |
|
| System services (init, the `fat` filesystem, the device-manager) | `system/services/` |
|
||||||
| Device drivers, one sub-project each (`pci-bus`, `ps2-bus`, `usb-xhci-bus` bus drivers) | `system/drivers/` |
|
| Device drivers, one sub-project each (`pci-bus`, `ps2-bus`, `usb-xhci-bus` bus drivers) | `system/drivers/` |
|
||||||
| On-image test fixtures for the QEMU cases (`vfs-test`, `crash-test`, `thread-test`, …) → `/test/system/services` | `test/system/services/` |
|
| On-image test fixtures for the QEMU cases (`vfs-test`, `crash-test`, `thread-test`, …) → `/test/system/services` | `test/system/services/` |
|
||||||
| Build + `run-x86-64` (QEMU/OVMF) + `release-x86-64` (the flashable ISO) | `build.zig` |
|
| Build orchestration (kernel + loader, what ships, the aggregate test step) | `build.zig` (root; the shared user-binary recipe is `build-support/`, and each `library/` domain + binary package carries its own `build.zig`) |
|
||||||
|
| Image assembly + `release-x86-64` (the flashable ISO) | `build/images.zig` |
|
||||||
|
| `run-x86-64` / `run-x86-64-gpu` (QEMU/OVMF) | `build/qemu.zig` |
|
||||||
| QEMU integration test harness | `test/qemu_test.py` |
|
| QEMU integration test harness | `test/qemu_test.py` |
|
||||||
|
|||||||
@@ -0,0 +1,179 @@
|
|||||||
|
# Plan: packages — hierarchical builds for libraries and binaries
|
||||||
|
|
||||||
|
**Status: complete** (branch `claude/build-packages-plan-174144`). Phase 0
|
||||||
|
(`build-support`), phase 1 (all six library domains), phase 2 (every binary —
|
||||||
|
the pci-bus pilot first, then services, drivers, and test fixtures in waves;
|
||||||
|
multi-binary directories like ps2-bus and usb-hid are one package exporting
|
||||||
|
several artifacts, and the acpi/fdt discovery pair each export an artifact
|
||||||
|
named "discovery" that the root's -Ddiscovery picks between), and phase 3 (the
|
||||||
|
root split into `build/images.zig` + `build/qemu.zig`; the root `build.zig` is
|
||||||
|
~460 lines of orchestration, down from ~1,250). Every phase landed green: unit
|
||||||
|
tests, the QEMU suite at parity with main, boot-image file list unchanged.
|
||||||
|
The `lazyDependency` payoff (What-this-buys #4) is in too: the /test fixtures
|
||||||
|
and the unselected discovery package are lazy — a build loads and compiles
|
||||||
|
only what it ships. And imports are exact: the pre-wired default set is gone;
|
||||||
|
every binary names precisely the modules its source imports and carries only
|
||||||
|
those domains in its manifest (rule 1 below).
|
||||||
|
|
||||||
|
## Why
|
||||||
|
|
||||||
|
`build.zig` was ~1,250 lines, growing by three hand-written stanzas per binary;
|
||||||
|
at a driver per device family that does not scale. More fundamentally: in one
|
||||||
|
monolithic build every binary compiles against library *source*, so a library
|
||||||
|
interface break is silently absorbed by whoever edits everything in one commit —
|
||||||
|
the interface never has to be honest. danos is about isolation; the build should
|
||||||
|
mirror it.
|
||||||
|
|
||||||
|
A **package** here is a build-time unit only — a directory owning a `build.zig`
|
||||||
|
(recipe: what it exports, how to test it) and a `build.zig.zon` (manifest: name
|
||||||
|
+ dependencies). Binaries remain fully static freestanding ELFs; packages change
|
||||||
|
who declares what, not what links to what. Source code is untouched: `@import`
|
||||||
|
uses module names (`"pci"`, `"service"`) exactly as today — only build files
|
||||||
|
know where anything lives.
|
||||||
|
|
||||||
|
## Target shape
|
||||||
|
|
||||||
|
```
|
||||||
|
build-support/ package: the danos build API (userBinary(), defaultImports(), targets)
|
||||||
|
library/kernel/ package "kernel": modules abi, ipc, service, memory, process, logging, time, ... (depends on protocol)
|
||||||
|
library/device/ package "device": modules driver, pci, usb-abi, model, ... (depends on kernel, protocol, csv)
|
||||||
|
library/protocol/ package "protocol": the wire protocols
|
||||||
|
library/client/ package "client" (depends on kernel, protocol)
|
||||||
|
library/csv/ package "csv"
|
||||||
|
library/xkeyboard-config/ package "xkeyboard-config"
|
||||||
|
system/services/<name>/ one package per binary: ~15-line build.zig + zon
|
||||||
|
system/drivers/<name>/ one package per binary
|
||||||
|
build.zig (root) orchestrator: dependency() per binary, image assembly, QEMU, test steps
|
||||||
|
```
|
||||||
|
|
||||||
|
The three shared contracts: `boot-handoff` stays a root module (only the
|
||||||
|
loader↔kernel pair speaks it); `abi` is exported by the kernel package from
|
||||||
|
`../../system/abi.zig` (the source stays with the kernel; userspace's one view
|
||||||
|
of it lives in the package, so every consumer names the same module instance);
|
||||||
|
`device-abi` is exported by device. Reaching outside the package root means the
|
||||||
|
kernel package is valid only as an in-repo path dependency — it could never be
|
||||||
|
fetched by hash — which is fine: path dependencies are the only way any of
|
||||||
|
these packages is consumed.
|
||||||
|
|
||||||
|
Rules:
|
||||||
|
|
||||||
|
- **Imports are exact and per binary.** A binary's build.zig names precisely
|
||||||
|
the modules its source `@import`s — the moral equivalent of a C file's
|
||||||
|
include list — and its zon names only the domains those modules come from
|
||||||
|
(plus `build-support` and `kernel`, which is implicit in every binary: the
|
||||||
|
root shim and user link script live there). Nothing is pre-wired: an
|
||||||
|
undeclared `@import` is a compile error, and build-support resolves each
|
||||||
|
name by searching the packages the zon declares — the domains' own
|
||||||
|
addModule exports are the single statement of who owns what, with no name
|
||||||
|
table anywhere to drift. Availability
|
||||||
|
never meant bloat — Zig only compiles what a program actually imports — but
|
||||||
|
exactness makes the declared interface honest and machine-checked.
|
||||||
|
- **Modules export source, not artifacts** — each consumer compiles libraries
|
||||||
|
with its own flags, so per-binary optimization choices keep working; Zig's
|
||||||
|
cache deduplicates.
|
||||||
|
- **Zon paths are relative and that is accepted.** Binaries sit exactly three
|
||||||
|
levels deep, so the `../../../` prefix is a constant idiom; a library-domain
|
||||||
|
move is a rare, already-breaking event fixed by one sed across manifests, and
|
||||||
|
a stale path fails loudly before anything compiles.
|
||||||
|
- **Cross-cutting build changes live in `build-support` only** — that is the
|
||||||
|
contract that keeps per-binary build files declarative.
|
||||||
|
|
||||||
|
## What this buys
|
||||||
|
|
||||||
|
1. Library interfaces become machine-checked: a consumer can only import what
|
||||||
|
it declared — per binary, down to the single module — and each domain's zon
|
||||||
|
declares what it needs (claim-before-touch, applied to source). A keyboard
|
||||||
|
driver carries `xkeyboard-config` in its manifest; nothing else does.
|
||||||
|
2. Each library domain gets a standalone `zig build test` — runtime-library
|
||||||
|
stability testing in isolation.
|
||||||
|
3. Adding a binary = adding a directory (source + two small files), not editing
|
||||||
|
three places in a 1,250-line file.
|
||||||
|
4. `lazyDependency` lets an image target build only what it ships: the /test
|
||||||
|
fixtures resolve only under -Dtest-case, and only the -Ddiscovery-selected
|
||||||
|
discovery package ever loads.
|
||||||
|
|
||||||
|
## Phases
|
||||||
|
|
||||||
|
Each phase ends green: `zig build test` passes (88/88 QEMU) and the boot
|
||||||
|
image's file list is unchanged. Byte-identical binaries are expected but not
|
||||||
|
required (module reorganization can perturb symbol order); file list is the
|
||||||
|
hard gate.
|
||||||
|
|
||||||
|
**Phase 0 — `build-support`.** Extract `addUserBinary`/`addThreadedUserBinary`,
|
||||||
|
the freestanding target setup, and the default-import wiring into the
|
||||||
|
`build-support` package. Root build consumes it; nothing else moves. This is
|
||||||
|
the cross-cutting-change home, so it lands first.
|
||||||
|
|
||||||
|
**Phase 1 — library domains become packages.** In dependency order: `protocol`
|
||||||
|
and `csv` (the roots) → `kernel` (depends on protocol: file-system speaks
|
||||||
|
vfs-protocol) → `device`, `client`; `xkeyboard-config` stands alone. Each gets
|
||||||
|
build.zig + zon + a standalone test step (client's is empty until its modules
|
||||||
|
grow host tests — kept for uniformity, since the root aggregate depends on
|
||||||
|
every domain's test step). The root build swaps its `createModule` calls for
|
||||||
|
`b.dependency("<domain>").module("<name>")`. **No binary moves in this phase**
|
||||||
|
— the root build is the pilot consumer, which proves the packages without
|
||||||
|
touching 30 binaries.
|
||||||
|
|
||||||
|
**Phase 2 — binaries become packages, in waves.** The template was shaken out
|
||||||
|
by the pci-bus pilot (see Status). Wave A: services (done). Wave B: the
|
||||||
|
remaining drivers (done). Wave C: test fixtures (done). Root build shrank to
|
||||||
|
orchestration per wave. init's `-Dserial` heartbeat flag rides a dependency
|
||||||
|
option; a directory with several binaries (ps2-bus, usb-hid) is one package
|
||||||
|
exporting several artifacts.
|
||||||
|
|
||||||
|
**Phase 3 — root cleanup (done).** What remained of the root build split into
|
||||||
|
`build/images.zig` (the FHS install tree, boot manifest + capsule, FAT32
|
||||||
|
images, release ISO, check steps) and `build/qemu.zig` (the run steps + OVMF
|
||||||
|
probing), imported by a short root `build.zig`.
|
||||||
|
|
||||||
|
**Afterwards** (outside this plan): the intel-uhd-graphics-750 driver is
|
||||||
|
(re)created as a greenfield package. The new-driver checklist's build step
|
||||||
|
(docs/device-driver-development/new-driver-checklist.md, step 2) is already
|
||||||
|
rewritten against the package template.
|
||||||
|
|
||||||
|
## Execution notes (the finished shape)
|
||||||
|
|
||||||
|
- The shared recipe lives in `build-support/build.zig`: `userBinary` (what
|
||||||
|
every binary package calls; each named import resolves by searching the
|
||||||
|
packages the binary's zon declares) and `programModule` (for per-binary
|
||||||
|
addOptions modules). The `start` root shim and `user.ld` are named through the kernel
|
||||||
|
package (Dependency.path).
|
||||||
|
- Adding a binary = adding a directory with source + a ~15-line build.zig +
|
||||||
|
zon (copy any existing binary package, e.g.
|
||||||
|
`system/drivers/pci-bus/build.zig`) listing exactly the modules the source
|
||||||
|
imports and the domains they come from, then one dependency + one bundled
|
||||||
|
entry in the root build.zig and one zon line.
|
||||||
|
- The boot-tree array in the root (search `"etc/init.csv"` or
|
||||||
|
`.getEmittedBin()`) is the image file list — the authoritative comparison
|
||||||
|
target for any future build change.
|
||||||
|
- Package unit tests live in each package's own `test` step; the root
|
||||||
|
aggregate depends on every test-bearing package's step, so `zig build test`
|
||||||
|
at the root still runs everything.
|
||||||
|
|
||||||
|
Verification per phase:
|
||||||
|
|
||||||
|
- Unit tests: `zig build test`.
|
||||||
|
- QEMU integration suite: `python3 test/qemu_test.py` (docs/testing.md; the
|
||||||
|
full suite, all cases must pass).
|
||||||
|
- Image file list: the boot-tree array is the source of truth — snapshot it
|
||||||
|
(paths only) before phase 0 and diff after each phase; `zig build
|
||||||
|
check-fat-image` must also stay green.
|
||||||
|
|
||||||
|
Context a fresh session should read first: this doc, docs/testing.md,
|
||||||
|
docs/coding-standards.md (kebab-case names, no abbreviations), and the
|
||||||
|
`userBinary`/`userBinaryFromImports` bodies in build-support/build.zig. Commit
|
||||||
|
style: no Co-Authored-By trailers.
|
||||||
|
|
||||||
|
## Risks / notes
|
||||||
|
|
||||||
|
- Zig version churn: the package API (`b.dependency`, zon schema) has moved
|
||||||
|
between releases; the work pins against the repo's current Zig and any
|
||||||
|
upgrade lands separately, never mid-phase.
|
||||||
|
- The QEMU size-check tests hardcode source paths (e.g. virtio-gpu protocol
|
||||||
|
struct sizes) — they moved into their binaries' packages with their waves,
|
||||||
|
discharging the carry-along obligation.
|
||||||
|
- Doc updates ride each phase: docs/README.md (repo layout + source map),
|
||||||
|
docs/device-driver-development/new-driver-checklist.md (step 2) and
|
||||||
|
devices-csv.md ("Adding a driver"), and the docs that cite the build recipe
|
||||||
|
(driver-model.md, threading.md, system-requirements.md) reference build
|
||||||
|
shapes that keep changing.
|
||||||
@@ -0,0 +1,205 @@
|
|||||||
|
# The C library compatibility layer
|
||||||
|
|
||||||
|
A design note and milestone plan for **libdanos-c** — the mini C library that lets
|
||||||
|
`zig cc` cross-compile C programs for danos. It is milestone **P0** of
|
||||||
|
[python-on-danos-milestones.md](python-on-danos-milestones.md), expanded here the
|
||||||
|
way [character-devices-and-tty.md](character-devices-and-tty.md) expands P1.
|
||||||
|
CPython is the driving consumer, but the layer is general: any portable C program
|
||||||
|
within its surface should build.
|
||||||
|
|
||||||
|
## What it is — and the three things it is not
|
||||||
|
|
||||||
|
The deliverable is a **sysroot**: a set of C headers plus a static `libdanos-c.a`,
|
||||||
|
handed to `zig cc -target x86_64-freestanding-none` via `-isystem` and linked into
|
||||||
|
every C binary. Three explicit non-goals keep it small:
|
||||||
|
|
||||||
|
- **Not a musl port.** Whole-musl assumes Linux syscall semantics at its bottom
|
||||||
|
(the door the Zig roadmap deferred, twice now). We *lift* musl's pure-computation
|
||||||
|
source files and *write* a danos-native bottom — see the layer split below.
|
||||||
|
- **Not full POSIX — *yet*.** Stage 1's surface is "what CPython's minimal
|
||||||
|
configuration and ordinary portable C need" — roughly 100–150 functions — and
|
||||||
|
at that stage absence is a *feature*: configure scripts probe and adapt, and a
|
||||||
|
linker error is honest. But the end state is a **full C compatibility layer**
|
||||||
|
(see "The road to full coverage" below); the absence table is a schedule of
|
||||||
|
arrivals, not a wall.
|
||||||
|
- **Not a second runtime.** The library is a thin C-ABI re-spelling of the same
|
||||||
|
danos-native surface `runtime` already provides. It contains no policy of its
|
||||||
|
own; when the Zig track's `runtime.os` seam is authored, the libc bottom
|
||||||
|
re-targets it near-mechanically — the fourth appearance of the roadmap's "same
|
||||||
|
surface" symmetry.
|
||||||
|
|
||||||
|
One scoping rule sits above all three — the **size doctrine**: this layer serves
|
||||||
|
**applications only**. The kernel and the system services never link libdanos-c;
|
||||||
|
they stay danos-native Zig over `runtime`, small and static, because leanness is
|
||||||
|
an operating-system property. Applications have their own budget and may be as
|
||||||
|
big as they need to be. The libc is how big software *lands on* danos, never how
|
||||||
|
danos itself is built.
|
||||||
|
|
||||||
|
## The layer split: lift the mathematics, write the plumbing
|
||||||
|
|
||||||
|
The realization that makes 100–150 functions tractable: a libc is two very
|
||||||
|
different kinds of code, and the hard kind is portable.
|
||||||
|
|
||||||
|
| Layer | Contents | Source |
|
||||||
|
|-------|----------|--------|
|
||||||
|
| **Pure computation** | `string.h`/`memcpy` family, all of libm, `strtod`/`dtoa`, `strtol`, `qsort`, `ctype` tables, `gmtime` calendar math, the `printf`/`scanf` engines, `setjmp` (a dozen instructions of x86-64 asm) | **Lift from musl**, vendored under `library/c/third-party/musl/` (MIT; files compile standalone) |
|
||||||
|
| **OS plumbing** | fds (`open`/`read`/`write`/`close`/`lseek`/`stat`/`getcwd`/`chdir`/`isatty`), `mmap`/`munmap`, clocks, `exit`, `getenv`, `getentropy` | **Write in Zig**, exporting C ABI over the `runtime` syscall + VFS client surface |
|
||||||
|
| **The middle** | `malloc` over danos `mmap` (simple free-list; CPython's arenas sit above), `FILE*` buffering, `errno` | **Write in Zig** (small, danos-shaped) |
|
||||||
|
| **Entry** | `crt0`: the existing danos entry shim ([sysv.md](os-development/sysv.md)) bridged to C `main(argc, argv, envp)`, `environ` initialised, `exit` flushing stdio | **Write** |
|
||||||
|
|
||||||
|
Two liftings deserve their own line because getting them wrong is silent
|
||||||
|
corruption rather than a linker error:
|
||||||
|
|
||||||
|
- **`strtod`/float formatting.** Python's float `repr` guarantees shortest
|
||||||
|
round-trip; that property lives entirely in these routines. musl's are correct;
|
||||||
|
an improvised one would be subtly wrong for years. Lift, never write.
|
||||||
|
- **The stdio engines.** musl's `vfprintf`/`vfscanf` are self-contained around
|
||||||
|
its `FILE` abstraction (function-pointer read/write slots), so the whole
|
||||||
|
formatted-I/O engine lifts too — we implement only the fd-backed slots
|
||||||
|
(`__stdio_write`-shaped) and the buffering glue.
|
||||||
|
|
||||||
|
## Header policy
|
||||||
|
|
||||||
|
Hand-write the headers as danos's own minimal set rather than importing musl's
|
||||||
|
(musl's are entangled with Linux ABI details), borrowing declarations freely.
|
||||||
|
Freestanding compiler headers (`stdint.h`, `stddef.h`, `stdarg.h`, `stdbool.h`,
|
||||||
|
`float.h`, `limits.h`) come from clang via `zig cc` — do not duplicate them.
|
||||||
|
`errno.h` values are the danos errno enum re-spelled with POSIX names; there is no
|
||||||
|
Linux numbering to be compatible with, so the enum is the truth.
|
||||||
|
|
||||||
|
Deliberate absences, and their planned arrivals — this table is the
|
||||||
|
compatibility matrix, and "the road to full coverage" below is the schedule
|
||||||
|
that empties it:
|
||||||
|
|
||||||
|
| Absent | Arrives with |
|
||||||
|
|--------|--------------|
|
||||||
|
| `pthread.h` | the post-P5 pthread subset over `thread_spawn`/futex — but see the risk below |
|
||||||
|
| real `signal.h` (beyond no-op `signal()`/`raise` stubs) | M17 signals-over-IPC in the libc |
|
||||||
|
| `dlfcn.h` | [dynamic-libraries.md](dynamic-libraries.md) D1 |
|
||||||
|
| `fork`/`exec*`/`wait*` | P5 exposes danos spawn as `posix_spawn`; `fork` itself never (see below) |
|
||||||
|
| `socket.h` | a future networking track |
|
||||||
|
| locale beyond `"C"` | stage 3 evaluation (CPython is UTF-8-mode happy without it) |
|
||||||
|
| pipes (`pipe()`) | P5 process-control cluster |
|
||||||
|
|
||||||
|
## The road to full coverage
|
||||||
|
|
||||||
|
The layer grows in three stages; only stage 1 is a current milestone (P0), but
|
||||||
|
the stages exist so stage-1 decisions never have to be unmade:
|
||||||
|
|
||||||
|
- **Stage 1 — CPython-minimal** (P0, the slicing below): ~100–150 functions,
|
||||||
|
static-only, absences honest.
|
||||||
|
- **Stage 2 — the danos-complete layer**: the full hosted C11 standard library,
|
||||||
|
plus every POSIX facility danos semantics support, landing as its enabling
|
||||||
|
milestone lands — pipes and `posix_spawn` at P5, real signals at M17, the
|
||||||
|
pthread subset after P5, `dlfcn.h` at
|
||||||
|
[dynamic-libraries](dynamic-libraries.md) D1, sockets with networking. Stage 2
|
||||||
|
is not one milestone but the standing rule that **every system capability
|
||||||
|
gets its C spelling when it ships**, so the matrix above drains as the OS
|
||||||
|
grows.
|
||||||
|
- **Stage 3 — ecosystem grade**: the point where "portable C program" generally
|
||||||
|
means "builds on danos" (autotools-style probing included). Reaching it is
|
||||||
|
mostly stage 2 compounding, plus the long tail (locale, wide-char,
|
||||||
|
`fnmatch`/`glob`/`regex` — the last three lift from musl like the rest). At
|
||||||
|
this stage, re-evaluate hand-grown-vs-musl-port once with real data; the
|
||||||
|
standing recommendation remains danos-native — musl's bottom assumes Linux
|
||||||
|
syscall semantics, and by stage 3 the danos bottom exists and is tested —
|
||||||
|
with musl continuing as the quarry for computation code.
|
||||||
|
|
||||||
|
Two boundaries are permanent and worth stating at every stage: **`fork` never
|
||||||
|
comes** — danos is a spawn-shaped OS, and `fork`'s address-space-duplication
|
||||||
|
semantics are hostile to everything from capabilities to threads; software that
|
||||||
|
hard-requires `fork` (not `posix_spawn`) stays off the platform. And the
|
||||||
|
**public ABI stays the vDSO + IPC protocols** — a full libc is a compatibility
|
||||||
|
*layer*, not a second stable system ABI.
|
||||||
|
|
||||||
|
## Milestone slicing
|
||||||
|
|
||||||
|
1. **sysroot-skeleton** — layout under `library/c/` (a build package:
|
||||||
|
`include/`, Zig sources, vendored musl subtree); `crt0`; string/mem +
|
||||||
|
`ctype` lifted; a `build.zig` step making C binaries first-class targets.
|
||||||
|
*Test:* a C program using only computation links and runs in QEMU
|
||||||
|
(`c-hello` printing via a raw `write` extern to `debug_write`).
|
||||||
|
2. **fd-plumbing** — `errno`; open/read/write/close/lseek/stat/unlink/mkdir/
|
||||||
|
rename over the `runtime` VFS client; `getcwd`/`chdir`/`getenv`/
|
||||||
|
`getentropy` arriving as P1 lands them (stubbed truthfully until then:
|
||||||
|
`getenv` empty, `getentropy` `ENOSYS`). *Test:* QEMU `c-file-io` — create,
|
||||||
|
write, reopen, read back, stat size + mtime through FAT.
|
||||||
|
3. **malloc** — free-list allocator over danos `mmap`; `calloc`/`realloc`/
|
||||||
|
`free`; alignment guarantees documented. *Test:* host + QEMU allocator
|
||||||
|
torture (interleaved sizes, realloc growth, alignment asserts).
|
||||||
|
4. **stdio** — `FILE*`, buffering modes, the lifted printf/scanf engines wired
|
||||||
|
to the fd slots; `snprintf` family; stdin/stdout/stderr over fd 0/1/2.
|
||||||
|
*Test:* host round-trip suite for format engines (especially `%.17g`
|
||||||
|
float round-trip); QEMU `c-stdio` cooked-line echo once P1's console exists.
|
||||||
|
5. **mathematics-and-time** — libm lifted wholesale; `strtod`/`strtol`;
|
||||||
|
`clock_gettime` (monotonic + realtime over `clock`/`wall_clock`);
|
||||||
|
`gmtime`/`mktime`/`strftime` (UTC only — no timezone database);
|
||||||
|
`setjmp`/`longjmp`; `qsort`/`bsearch`; `abort`/`assert`. *Test:* host
|
||||||
|
`strtod`/`dtoa` vectors against known-hard cases; QEMU `c-time` sanity
|
||||||
|
against the wall clock.
|
||||||
|
|
||||||
|
Slices 1, 3, 4-host, and 5-host have **no dependency on P1** and can start
|
||||||
|
immediately; slice 2 and the QEMU halves interleave with P1 as it lands.
|
||||||
|
|
||||||
|
**Exit for the layer as a whole** (= P0's exit): `c-hello` and `c-file-io` green
|
||||||
|
in the QEMU suite, and the host-side computation tests green — at which point P2
|
||||||
|
(CPython configure) becomes the layer's real integration test.
|
||||||
|
|
||||||
|
## Testing strategy: two targets, on purpose
|
||||||
|
|
||||||
|
The computation layer is target-independent, so it is unit-tested **on the host**
|
||||||
|
(built for the host triple, compared against the host libc's answers —
|
||||||
|
thousands of cheap oracle checks for `strtod`, `printf`, libm edge cases). The
|
||||||
|
plumbing layer only means anything **on danos**, so it is tested in the QEMU
|
||||||
|
suite like every other subsystem. Keeping the split explicit stops the slow-QEMU
|
||||||
|
suite from absorbing tests that a host `zig test` runs in milliseconds.
|
||||||
|
|
||||||
|
## Risks and gotchas
|
||||||
|
|
||||||
|
- **CPython's configure may insist on pthreads.** WASI-class targets build
|
||||||
|
threadless, but verify this *first* in P2 bring-up; the fallback is a
|
||||||
|
truthfully-single-threaded `pthread.h` stub set (create returns `EAGAIN`,
|
||||||
|
mutexes are no-ops — valid when only one thread can exist). Decide from
|
||||||
|
evidence, not assumption.
|
||||||
|
- **`long double` is x87 80-bit on x86-64.** musl's libm handles it, but keep
|
||||||
|
CPython away from it (`configure` uses `double` throughout by default);
|
||||||
|
don't hand-write anything touching x87.
|
||||||
|
- **errno is a contract, not a convention.** The Zig plumbing must map every
|
||||||
|
`runtime` error to a POSIX name consistently — CPython turns errno into
|
||||||
|
exception types (`FileNotFoundError` is `ENOENT`). One table, tested.
|
||||||
|
- **`malloc` alignment**: 16-byte minimum on x86-64 (SSE spills in
|
||||||
|
compiled C). The free-list must guarantee it from day one; retrofitting
|
||||||
|
alignment bugs out of an allocator is misery.
|
||||||
|
- **Vendoring discipline.** The musl subtree is lift-only — never edited in
|
||||||
|
place (patches live beside it if ever needed), pinned to one musl release,
|
||||||
|
with the file list documented so a version bump is a re-copy, not an
|
||||||
|
archaeology dig.
|
||||||
|
- **stdio buffering vs. crashes.** Buffered stdout + a crashing program eats
|
||||||
|
output — the classic debugging trap. `stderr` stays unbuffered (per C
|
||||||
|
standard) and `exit`/`abort` flush; document that `_exit` does not.
|
||||||
|
|
||||||
|
## Decisions needing sign-off
|
||||||
|
|
||||||
|
- **Lift-from-musl for all pure computation** (vendored, pinned, unedited) rather
|
||||||
|
than writing or porting whole-musl.
|
||||||
|
- **Hand-written danos-native headers**; danos errno values are the numbering.
|
||||||
|
- **`library/c/` as a build package** producing both the sysroot and the
|
||||||
|
first-class C-binary build step.
|
||||||
|
- The **deliberate-absence table** as the living compatibility matrix, drained
|
||||||
|
by the three-stage road above — with exactly one permanent "never": `fork`.
|
||||||
|
- **Full coverage as the end state** (stage 3), reached by the standing rule
|
||||||
|
that every system capability ships with its C spelling — not by a musl port.
|
||||||
|
|
||||||
|
## Related
|
||||||
|
|
||||||
|
- [python-on-danos-milestones.md](python-on-danos-milestones.md) — this is P0.
|
||||||
|
- [dynamic-libraries.md](dynamic-libraries.md) — ships in this sysroot
|
||||||
|
(`dlfcn.h` + the loader) once its D1 lands.
|
||||||
|
- [python-on-danos.md](python-on-danos.md) — the design note that scoped the
|
||||||
|
layer.
|
||||||
|
- [character-devices-and-tty.md](character-devices-and-tty.md) — P1; supplies
|
||||||
|
the console that makes stdio interactive.
|
||||||
|
- [zig-self-hosting.md](zig-self-hosting.md) — the `runtime.os` seam the
|
||||||
|
plumbing layer will re-target when it exists.
|
||||||
|
- [os-development/sysv.md](os-development/sysv.md) — the entry stack `crt0`
|
||||||
|
bridges.
|
||||||
@@ -0,0 +1,174 @@
|
|||||||
|
# Character devices, the console, and the tty question
|
||||||
|
|
||||||
|
A design note for the **stream** half of the device world. danos has block devices
|
||||||
|
(the USB storage service behind the FAT mount) but no character devices — and three
|
||||||
|
tracks now need them at once: the terminal application, Zig self-hosting Phase 1
|
||||||
|
("wire fd 0/1/2 to a console byte stream"), and [Python on danos](python-on-danos.md)
|
||||||
|
Phase 1. This note settles what a character device *is* on danos before any of those
|
||||||
|
tracks build one.
|
||||||
|
|
||||||
|
## The Unix picture, briefly
|
||||||
|
|
||||||
|
Unix splits devices in two: **block devices** are seekable arrays of fixed-size
|
||||||
|
sectors (disks); **character devices** are unseekable byte streams (keyboards,
|
||||||
|
serial ports, terminals, `/dev/null`, entropy). A **tty** is the canonical
|
||||||
|
character device — a byte stream plus a *line discipline* (echo, line buffering,
|
||||||
|
erase handling, Ctrl-C-to-signal) that lives in the kernel. A **pty** is a pair of
|
||||||
|
character devices (master/slave) that exists so a *userspace* program — a terminal
|
||||||
|
emulator — can impersonate terminal hardware to the kernel's in-kernel line
|
||||||
|
discipline.
|
||||||
|
|
||||||
|
The identification asked for and confirmed: yes, tty and pty are character
|
||||||
|
devices in this taxonomy.
|
||||||
|
|
||||||
|
## The realization that shapes everything: danos already has the mechanism
|
||||||
|
|
||||||
|
A Unix character device is an in-kernel dispatch table: major/minor numbers route
|
||||||
|
`read()`/`write()` to a driver. danos already has exactly that dispatch — the VFS:
|
||||||
|
`fs_resolve` routes a path to a mounted backend service, and `Operation.mount`
|
||||||
|
attaches a backend *endpoint* at a prefix. What is missing is not a device model;
|
||||||
|
it is **one node kind with stream semantics**. And the protocol already reserved
|
||||||
|
it: `NodeKind.character_device = 2` sits unimplemented in
|
||||||
|
[vfs-protocol.zig](../library/protocol/vfs/vfs-protocol.zig), exactly like
|
||||||
|
`symbolic_link`.
|
||||||
|
|
||||||
|
So the design is small:
|
||||||
|
|
||||||
|
**A character device on danos is a VFS node, served by an ordinary service over
|
||||||
|
the existing VFS wire protocol, whose read/write have stream semantics.**
|
||||||
|
|
||||||
|
No device numbers, no `/dev` special casing, no new syscalls, no new protocol —
|
||||||
|
a service is reachable at a path, clients open it with `runtime.fs` like any
|
||||||
|
file, and the node kind says what it is. (Since the protocol namespace landed
|
||||||
|
in design, that path is `/protocol/console` — a protocol node, see
|
||||||
|
[os-development/protocol-namespace.md](os-development/protocol-namespace.md) —
|
||||||
|
rather than a mounted device file; the stream semantics below are unchanged.)
|
||||||
|
|
||||||
|
### Stream semantics (the actual contract change)
|
||||||
|
|
||||||
|
For a node whose kind is `character_device`:
|
||||||
|
|
||||||
|
- **`offset` is ignored** on read and write; there is no seek position. (`lseek`,
|
||||||
|
when the C layer exists, returns `ESPIPE`.)
|
||||||
|
- **Reads block** until at least one byte is available, then return what is there —
|
||||||
|
**short reads are normal**, not EOF. A zero-length read reply means the stream
|
||||||
|
is closed (hangup), not end-of-file-at-size.
|
||||||
|
- **`FileStatus.size` is 0** and means nothing; `mtime` may be 0.
|
||||||
|
- Writes may be short if the service's buffer is full; the client loops as it
|
||||||
|
already must for the 256-byte message cap.
|
||||||
|
|
||||||
|
This is a semantics note on existing operations, not a wire change — the `Request`
|
||||||
|
and `Reply` structs are untouched. The one true protocol addition is a **`control`
|
||||||
|
operation** (appended to `Operation`, values stable): a typed request the stream's
|
||||||
|
service interprets. Deliberately *not* an `ioctl` grab-bag — the control payloads
|
||||||
|
are enumerated per protocol, starting with the terminal set below.
|
||||||
|
|
||||||
|
## The first character device is a pseudo-device
|
||||||
|
|
||||||
|
The first device is deliberately **not hardware**: an in-memory **loopback** — a
|
||||||
|
byte queue served over the stream contract, where bytes written to one end are
|
||||||
|
read from the other. It is the reference implementation of the semantics above
|
||||||
|
(blocking reads, short reads, hangup on close, the `control` round-trip), it
|
||||||
|
tests deterministically with no QEMU serial scripting, and it keeps hardware off
|
||||||
|
the critical path entirely. `null` and `zero` come along nearly for free as
|
||||||
|
degenerate cases. This is a decision, not a convenience: the dead-COM1 boot bug
|
||||||
|
on real hardware already proved serial cannot be assumed present or alive, so
|
||||||
|
**nothing in this milestone writes to COM1**. (A serial-backed stream node can
|
||||||
|
exist *later* as one more optional backend for headless debugging; it is on
|
||||||
|
nobody's critical path.)
|
||||||
|
|
||||||
|
The loopback is also not throwaway — it is the seed of P5's `pipe()`, which is
|
||||||
|
the same object with two fds.
|
||||||
|
|
||||||
|
## The console service
|
||||||
|
|
||||||
|
A `console` service owns the line discipline — **in userspace**, where a
|
||||||
|
microkernel wants it, not in the kernel as Unix has it:
|
||||||
|
|
||||||
|
- **The discipline is a pure library first**: bytes and key events in, bytes
|
||||||
|
out, no I/O of its own — developed and host-tested against in-memory buffers,
|
||||||
|
then shared verbatim between the console and the future terminal application.
|
||||||
|
- **Input**: subscribes to keyboard `InputEvent` IPC (the structured events that
|
||||||
|
exist today) and cooks them into bytes. Cooked mode is the default: echo, line
|
||||||
|
buffering, backspace/erase, so a line is delivered on Enter. Raw mode delivers
|
||||||
|
bytes as they come (the REPL's line editor and any full-screen program need it).
|
||||||
|
- **Output is a pluggable sink**, and the stream contract is independent of it:
|
||||||
|
the bring-up sink is in-memory (readable back by tests, mirrored to the boot
|
||||||
|
log), and the real one is the framebuffer text renderer when the display
|
||||||
|
track's font work lands.
|
||||||
|
- **Control set** (the `control` payloads): mode raw/cooked, echo on/off, and
|
||||||
|
window-size query — the minimal termios. Ctrl-C-to-signal joins when M17
|
||||||
|
signals-over-IPC lands; until then Ctrl-C is just a byte.
|
||||||
|
- Mounts itself at `/device/console` as a `character_device` node.
|
||||||
|
|
||||||
|
**fd 0/1/2** then stop being special: spawn hands the child three open handles
|
||||||
|
(console by default; anything else if the parent chooses), and `runtime`'s fd
|
||||||
|
table maps 0/1/2 to them. `isatty` is simply "does `status` say
|
||||||
|
`character_device`" — no side channel needed.
|
||||||
|
|
||||||
|
## The pty answer: there is no pty
|
||||||
|
|
||||||
|
The pty exists in Unix *because the line discipline is in the kernel* — userspace
|
||||||
|
terminal emulators need a kernel gadget to impersonate hardware. On danos the
|
||||||
|
terminal emulator is already a userspace server, so the pair collapses:
|
||||||
|
|
||||||
|
**The graphical terminal application serves the VFS stream protocol itself and
|
||||||
|
hands its own endpoints to the children it spawns as their fd 0/1/2.**
|
||||||
|
|
||||||
|
The terminal *is* the console service for its children — same protocol, same
|
||||||
|
control set, same line discipline code (shared as a library with the boot
|
||||||
|
console). No master/slave device pair, no `/dev/pts`, no new kernel object. When
|
||||||
|
CPython arrives, the libc's `isatty`/read/write see a character device and are
|
||||||
|
none the wiser; when xonsh eventually wants job control, that lands as control
|
||||||
|
messages + M17 signals, still with no pty object.
|
||||||
|
|
||||||
|
What this costs: programs that *specifically* manipulate Unix ptys
|
||||||
|
(`os.openpty()`, `pexpect`-style tools) have no direct equivalent — the danos
|
||||||
|
answer is "spawn the child yourself with your own stream endpoints," which is the
|
||||||
|
same capability with less machinery. Accepted.
|
||||||
|
|
||||||
|
## Milestone slicing
|
||||||
|
|
||||||
|
1. **pseudo-devices** — VFS honors `character_device` semantics end to end;
|
||||||
|
`Operation.control` added; the in-memory **loopback** (plus `null`/`zero`)
|
||||||
|
as the first device. QEMU test: one client writes, another reads — open,
|
||||||
|
offsetless read/write, blocking read, short read, hangup on close, control
|
||||||
|
round-trip. No hardware anywhere.
|
||||||
|
2. **console-service** — the line-discipline library (host-tested, pure) plus
|
||||||
|
the console composing keyboard `InputEvent`s with an in-memory output sink;
|
||||||
|
mounted at `/device/console`. QEMU test injects key events and reads cooked
|
||||||
|
lines and raw bytes back through the sink.
|
||||||
|
3. **fd-inheritance** — spawn passes 0/1/2 handles; `runtime` fd table; `isatty`
|
||||||
|
via `status`; existing binaries' stdout migrates from `debug_write` to fd 1
|
||||||
|
(the logger keeps its own path).
|
||||||
|
4. **terminal-as-server** — deferred to the terminal application milestone
|
||||||
|
(Python track P3): the terminal reuses the discipline library and serves its
|
||||||
|
children directly.
|
||||||
|
|
||||||
|
Steps 1–3 are exactly the shared seam that Zig self-hosting Phase 1 and Python
|
||||||
|
Phase 1 both list; neither track repeats them.
|
||||||
|
|
||||||
|
## Decisions needing sign-off
|
||||||
|
|
||||||
|
- **No pty object; the terminal serves its children directly** (the section
|
||||||
|
above) — the load-bearing simplification.
|
||||||
|
- **`control` as an enumerated, typed operation** rather than an ioctl-style
|
||||||
|
opaque pass-through.
|
||||||
|
- **Line discipline in userspace services** (console + terminal, shared library),
|
||||||
|
never in the kernel.
|
||||||
|
|
||||||
|
## Related
|
||||||
|
|
||||||
|
- [python-on-danos.md](python-on-danos.md) — consumes this as its Phase 1.
|
||||||
|
- [zig-self-hosting.md](zig-self-hosting.md) — ditto ("stdio as fds").
|
||||||
|
- [file-system-development/vfs-protocol.md](file-system-development/vfs-protocol.md) —
|
||||||
|
the wire protocol this note extends.
|
||||||
|
- [file-system-development/file-system-hierarchy.md](file-system-development/file-system-hierarchy.md)
|
||||||
|
— the tree the console surfaces in.
|
||||||
|
- [os-development/protocol-namespace.md](os-development/protocol-namespace.md) —
|
||||||
|
supersedes this note's device-node naming: the console lands as a protocol
|
||||||
|
(`/protocol/console`, a protocol node), not a `/dev`-style device file. The
|
||||||
|
stream semantics designed here (line discipline, cooked/raw modes) carry over
|
||||||
|
unchanged.
|
||||||
|
- [device-driver-development/input.md](device-driver-development/input.md) — the
|
||||||
|
`InputEvent` stream the console cooks.
|
||||||
@@ -91,6 +91,9 @@ mechanism), replacing first-come-first-served `device_claim` with policy. Identi
|
|||||||
`child_added` is per-bus: PCI children carry the class triple (`pci_class`, as the
|
`child_added` is per-bus: PCI children carry the class triple (`pci_class`, as the
|
||||||
xHCI match already uses); USB children carry the (class, subclass, protocol) triple
|
xHCI match already uses); USB children carry the (class, subclass, protocol) triple
|
||||||
from usb-ids.zig — each bus's native language, decoded by the shared ids modules.
|
from usb-ids.zig — each bus's native language, decoded by the shared ids modules.
|
||||||
|
(Since the registry landed, `child_added` also carries a `bus` discriminator and
|
||||||
|
the numeric `vendor`/`device`/`subsystem` ids the finer match levels need —
|
||||||
|
see [/etc/devices.csv](devices-csv.md).)
|
||||||
|
|
||||||
## Supervision and restart
|
## Supervision and restart
|
||||||
|
|
||||||
@@ -169,9 +172,15 @@ published exit events, signals + `process`). On top of those:
|
|||||||
- **Manager death**: drivers survive the manager; the restarted manager re-learns
|
- **Manager death**: drivers survive the manager; the restarted manager re-learns
|
||||||
the world (above). Checkpointing driver state with the manager is deferred until
|
the world (above). Checkpointing driver state with the manager is deferred until
|
||||||
something demonstrates the need.
|
something demonstrates the need.
|
||||||
- **Matching stays code until the third bus.** `driverFor`/`pciDriverFor` were
|
- **Matching is a registry, not code (resolved 2026-07-26).** `driverFor`/
|
||||||
honest at two bus types; the third was expected to trigger the manifest (a driver
|
`pciDriverFor` were honest at two bus types; the third (USB) was matched in code
|
||||||
declares what it binds: a PCI class triple, a USB class triple, an ACPI `_HID`).
|
too, and then the switch tables started to hurt — they keyed PCI matches on the
|
||||||
(Since then: the third bus — USB — arrived and is matched in code too. Today's
|
class triple alone, so a virtio-gpu could only be matched as a generic display
|
||||||
matchers are `pciDriverForIdentity`, `hidDriverFor`, and `usbDriverForIdentity`;
|
function and the driver had to re-confirm its `1AF4:1050` identity from config
|
||||||
the manifest waits until code matching actually hurts.)
|
space after being spawned. The manifest the earlier note anticipated landed as a
|
||||||
|
human-readable registry: **[/etc/devices.csv](devices-csv.md)**, parsed by the
|
||||||
|
pure `device-registry` module and read by the manager at boot. A row binds a
|
||||||
|
driver to a device by any of base / subclass / prog-IF / vendor / device /
|
||||||
|
subsystem / `_HID`, most-specific match winning; it is authoritative (no
|
||||||
|
compiled-in fallback — an unmatched device is logged, never guessed).
|
||||||
|
`pciDriverForIdentity`, `hidDriverFor`, and `usbDriverForIdentity` are gone.
|
||||||
|
|||||||
@@ -0,0 +1,110 @@
|
|||||||
|
# /etc/devices.csv — the device registry
|
||||||
|
|
||||||
|
**Status: built (2026-07-26).** The device manager reads `/etc/devices.csv` at
|
||||||
|
boot and binds every device a bus driver reports to the driver the registry
|
||||||
|
names. It replaces the three hand-written `switch` tables that used to live in
|
||||||
|
the manager (`pciDriverForIdentity`, `hidDriverFor`, `usbDriverForIdentity`) —
|
||||||
|
the "manifest" [device-manager.md](device-manager.md) anticipated once code
|
||||||
|
matching started to hurt. The parser and matcher are the pure, unit-tested
|
||||||
|
`device-registry` module (`library/device/registry/device-registry.zig`).
|
||||||
|
|
||||||
|
## Why a registry
|
||||||
|
|
||||||
|
The switch tables keyed PCI matches on the 24-bit class/subclass/prog-IF triple
|
||||||
|
alone. That is too coarse: a virtio-gpu is just "display / other" by class, so it
|
||||||
|
could only be *class-matched* and the driver had to re-confirm its real
|
||||||
|
`1AF4:1050` identity from config space **after** the manager had already spawned
|
||||||
|
it. The registry lets a rule bind on the full identity — down to vendor, device,
|
||||||
|
and subsystem — so the manager makes the precise decision itself, and the driver
|
||||||
|
comes up already knowing it is the right one.
|
||||||
|
|
||||||
|
It is also **data, not code**: teaching the system new hardware is a line in a
|
||||||
|
file, not an edit-and-recompile of the manager. And it is **greppable** — one
|
||||||
|
place to read "what binds what," the same idea as Linux's `modules.alias`.
|
||||||
|
|
||||||
|
## The file
|
||||||
|
|
||||||
|
One rule per line, nine comma-separated fields; `#` starts a comment (whole-line
|
||||||
|
or trailing); blank lines are ignored. Whitespace around a field is trimmed, so
|
||||||
|
columns may be padded for readability.
|
||||||
|
|
||||||
|
```
|
||||||
|
# bus base class prog_if vendor device subsystem hid driver
|
||||||
|
pci, 0C, 03, 30, *, *, *, *, /system/drivers/usb-xhci-bus
|
||||||
|
pci, 03, 00, 00, *, *, *, *, /system/drivers/display
|
||||||
|
pci, 03, 80, *, 1AF4, 1050, *, *, /system/drivers/virtio-gpu
|
||||||
|
usb, 03, 01, 01, *, *, *, *, /system/drivers/usb-hid-keyboard
|
||||||
|
acpi, *, *, *, *, *, *, PNP0303, /system/drivers/ps2-bus
|
||||||
|
```
|
||||||
|
|
||||||
|
| Field | Meaning | Notes |
|
||||||
|
|---|---|---|
|
||||||
|
| `bus` | `pci` \| `usb` \| `acpi` | which bus reported the device; picks the namespace for the id columns |
|
||||||
|
| `base` | PCI base class / USB class | hex |
|
||||||
|
| `class` | PCI subclass / USB subclass | hex |
|
||||||
|
| `prog_if` | PCI prog-IF / USB protocol | hex |
|
||||||
|
| `vendor` | PCI vendor / USB idVendor | hex |
|
||||||
|
| `device` | PCI device / USB idProduct | hex |
|
||||||
|
| `subsystem` | PCI subsystem, `(ssvid<<16)\|ssid` | hex; blank for usb/acpi |
|
||||||
|
| `hid` | ACPI `_HID` (e.g. `PNP0303`) | blank for pci/usb |
|
||||||
|
| `driver` | full ramdisk path to spawn | e.g. `/system/drivers/virtio-gpu` |
|
||||||
|
|
||||||
|
`*` or an empty field is a **wildcard** — it matches anything and adds nothing to
|
||||||
|
a rule's specificity.
|
||||||
|
|
||||||
|
## Levels of detection: most-specific-wins
|
||||||
|
|
||||||
|
Several rows may match one device. The manager picks the **most specific** — the
|
||||||
|
one that pins the finest-grained fields. Specificity weights double from the
|
||||||
|
coarsest level so each outweighs all coarser levels combined:
|
||||||
|
|
||||||
|
```
|
||||||
|
base(1) < class(2) < prog_if(4) < vendor(8) < subsystem(16) < device(32) ≈ hid(32)
|
||||||
|
```
|
||||||
|
|
||||||
|
So the generic `pci, 03, 00, 00, …/display` rule and the precise
|
||||||
|
`pci, 03, 80, *, 1AF4, 1050, …/virtio-gpu` rule coexist: the virtio card
|
||||||
|
(vendor 1AF4, device 1050) takes the specific rule; a plain VGA adapter still
|
||||||
|
falls to the generic one. Two rules that match a device with the *same*
|
||||||
|
specificity are a registry authoring error — the manager logs it loudly and binds
|
||||||
|
the first, so the shadowed rule is visible rather than silently dropped.
|
||||||
|
|
||||||
|
## Authoritative — no code fallback
|
||||||
|
|
||||||
|
There is no compiled-in default table behind the registry. A device that no row
|
||||||
|
matches goes **unbound** and is logged; the manager never guesses. A missing or
|
||||||
|
empty `/etc/devices.csv` therefore means nothing matches — which is loud at boot,
|
||||||
|
not a silent half-working system.
|
||||||
|
|
||||||
|
## How the manager reads it
|
||||||
|
|
||||||
|
`/etc/devices.csv` is bundled into the initial ramdisk (`build.zig`'s `bundled`
|
||||||
|
list). The kernel serves the initrd's `/etc` tree directly — the `fat` service is
|
||||||
|
spawned *after* the device manager and is irrelevant to `/etc` — so the manager
|
||||||
|
reads the file with a plain `fs.open("/etc/devices.csv")` + `read`, with no
|
||||||
|
filesystem service running and no boot-ordering dependency. It parses the bytes
|
||||||
|
once in `initialise`, before any bus driver can report a device to match.
|
||||||
|
|
||||||
|
## Feeding the matcher: the widened report
|
||||||
|
|
||||||
|
Finer-grained matching needs identity the old ABI threw away. Two things carry it
|
||||||
|
now: `child_added` (and `DeviceDescriptor`) grew `vendor` / `device` /
|
||||||
|
`subsystem` fields, filled by the PCI bus driver from config space (offsets
|
||||||
|
0x00 and 0x2C); and each bus driver states its `bus` in the report (a `BusKind`),
|
||||||
|
so the manager reads a PCI class triple and a USB class triple — the same 24 bits
|
||||||
|
in different namespaces — against the right `bus` column.
|
||||||
|
|
||||||
|
## Adding a driver
|
||||||
|
|
||||||
|
(The step-by-step walkthrough with a worked example is
|
||||||
|
[new-driver-checklist.md](new-driver-checklist.md).)
|
||||||
|
|
||||||
|
1. Create `system/drivers/<name>/` with the driver source plus a ~15-line
|
||||||
|
package `build.zig` + `build.zig.zon` (copy an existing driver package,
|
||||||
|
e.g. `system/drivers/pci-bus/`; per-driver extras go through
|
||||||
|
`build_support.programModule`). Then bundle it at `/system/drivers/<name>`:
|
||||||
|
one dependency + one bundled entry in the root `build.zig`, one line in the
|
||||||
|
root `build.zig.zon`.
|
||||||
|
2. Add a row to `etc/devices.csv` naming the identity it binds and its full path.
|
||||||
|
|
||||||
|
No device-manager change is required — the registry is the seam.
|
||||||
@@ -20,9 +20,10 @@ Read [display.md](display.md) first for the *why*; this is the *what* and the *o
|
|||||||
|
|
||||||
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym abbreviations in
|
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym abbreviations in
|
||||||
full, kebab-case file names, no `Co-Authored-By` trailers on commits. New user binaries
|
full, kebab-case file names, no `Co-Authored-By` trailers on commits. New user binaries
|
||||||
go through `addUserBinary` in [build.zig](../../build.zig) and get packed into the
|
go through build-support's shared user-binary recipe and get packed into the
|
||||||
initial-ramdisk; protocols are `b.addModule("…-protocol", …)` and imported into the
|
initial-ramdisk; protocols are modules exported by the `library/protocol` package.
|
||||||
`runtime` module.
|
(This section predates the build-packages split; see
|
||||||
|
[build-packages-plan.md](../build-packages-plan.md) for the current build shape.)
|
||||||
|
|
||||||
## How to verify along the way
|
## How to verify along the way
|
||||||
|
|
||||||
|
|||||||
@@ -18,9 +18,11 @@ lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run,
|
|||||||
|
|
||||||
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym abbreviations,
|
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym abbreviations,
|
||||||
kebab-case file names, no `Co-Authored-By` trailers. New user binaries go through
|
kebab-case file names, no `Co-Authored-By` trailers. New user binaries go through
|
||||||
`addUserBinary` and get packed into the initial-ramdisk; protocols are
|
build-support's shared user-binary recipe and get packed into the initial-ramdisk;
|
||||||
`b.addModule("…-protocol", …)` imported into `runtime`; new syscalls extend
|
protocols are modules exported by the `library/protocol` package; new syscalls extend
|
||||||
[abi.zig](../../system/abi.zig) `SystemCall` + a `library/runtime` wrapper.
|
[abi.zig](../../system/abi.zig) `SystemCall` + a `library/kernel` wrapper.
|
||||||
|
(This section predates the build-packages split; see
|
||||||
|
[build-packages-plan.md](../build-packages-plan.md) for the current build shape.)
|
||||||
|
|
||||||
## How to verify along the way
|
## How to verify along the way
|
||||||
|
|
||||||
|
|||||||
@@ -29,7 +29,7 @@ which one you're holding decides what you can do.
|
|||||||
|
|
||||||
- **The PCI class-0x03 device is the raw controller** — BARs, config space, registers,
|
- **The PCI class-0x03 device is the raw controller** — BARs, config space, registers,
|
||||||
IO ports. It is what you actually *own* after boot. On QEMU's emulated adapter
|
IO ports. It is what you actually *own* after boot. On QEMU's emulated adapter
|
||||||
([`-device VGA,edid=on`](../../build.zig), the Bochs VBE/DISPI model) the `base` GOP handed
|
([`-device VGA,edid=on`](../../build/qemu.zig), the Bochs VBE/DISPI model) the `base` GOP handed
|
||||||
you *is* that device's linear-framebuffer BAR — the same physical memory, seen through
|
you *is* that device's linear-framebuffer BAR — the same physical memory, seen through
|
||||||
a different door. On a real discrete GPU, GOP's `base` is an aperture inside the GPU's
|
a different door. On a real discrete GPU, GOP's `base` is an aperture inside the GPU's
|
||||||
VRAM BAR. danos already decodes this device
|
VRAM BAR. danos already decodes this device
|
||||||
|
|||||||
@@ -138,13 +138,17 @@ a higher-level service (block ↔ filesystem, a scanout driver ↔ the composito
|
|||||||
private wire to its *hardware* — virtio-gpu's command set — is not that; it stays a
|
private wire to its *hardware* — virtio-gpu's command set — is not that; it stays a
|
||||||
driver-private file, like the virtio-pci transport beside it.
|
driver-private file, like the virtio-pci transport beside it.
|
||||||
|
|
||||||
The build side of this has since landed: [`addUserBinary`](build.zig) injects the
|
The build side of this has since landed: every binary owns a package whose
|
||||||
default modules — the library/kernel concern modules (`ipc`, `memory`, `process`, `time`,
|
~15-line `build.zig` names EXACTLY the modules its source imports — the moral
|
||||||
`logging`, `file-system`, `thread`, `service`), the device/service clients (`driver`,
|
equivalent of a C file's include list — and the shared recipe in
|
||||||
`block`, `display`, `input`), plus `mmio`, `xkeyboard-config`, `acpi-ids` — into every user
|
[`build-support/build.zig`](../../build-support/build.zig) (`userBinary`)
|
||||||
binary, and per-binary extras — protocol modules, bus logic — are added with
|
resolves each name from the library domain that exports it (kernel's concern
|
||||||
`programModule(exe).addImport(...)`. That's the *entire* mechanism — Zig modules
|
modules, the device driver libraries, the service clients, the protocols). An
|
||||||
already give you everything else.
|
undeclared `@import` is a compile error, and a domain none of the imports come
|
||||||
|
from never appears in the binary's manifest — a keyboard driver declares
|
||||||
|
`xkeyboard-config`; nothing else does (see
|
||||||
|
[build-packages-plan.md](../build-packages-plan.md)). That's the *entire*
|
||||||
|
mechanism — Zig modules already give you everything else.
|
||||||
|
|
||||||
The discipline that makes this work: **a class driver must not import a bus's *hardware*
|
The discipline that makes this work: **a class driver must not import a bus's *hardware*
|
||||||
logic module.** `usb-hid` imports `usb` (the transfer client) and `input-protocol`, never
|
logic module.** `usb-hid` imports `usb` (the transfer client) and `input-protocol`, never
|
||||||
|
|||||||
@@ -1,34 +1,70 @@
|
|||||||
# IPC: message-passing channels
|
# IPC: the kernel-ipc transport
|
||||||
|
|
||||||
Inter-process communication is the **backbone of a microkernel**. Once drivers and
|
Inter-process communication is the **backbone of a microkernel**. Once drivers and
|
||||||
services run isolated in their own address spaces ([vision](../vision.md)), they can't
|
services run isolated in their own address spaces ([vision](../vision.md)), they can't
|
||||||
just call each other — a request becomes a **message**. In a microkernel, whatever
|
just call each other — a request becomes bytes on a wire. In a microkernel, whatever
|
||||||
was a function call across a monolithic kernel is IPC, so it's a first-class
|
was a function call across a monolithic kernel is IPC, so it's a first-class
|
||||||
concern, not an afterthought.
|
concern, not an afterthought.
|
||||||
|
|
||||||
There are two layers, built a milestone apart:
|
This document describes **one transport** — the bottom layer (L0) of the
|
||||||
|
communication stack defined in
|
||||||
|
[communication.md](../os-development/communication.md), which owns the model
|
||||||
|
and the vocabulary (*protocol*, *channel*, *packet*, *signal*, *endpoint*).
|
||||||
|
kernel-ipc is the **first** transport, not the only possible one: in
|
||||||
|
buffer-plus-doorbell terms it is a kernel-owned mailbox with the scheduler as
|
||||||
|
the doorbell. Its distinguishing properties, which the layers above may rely
|
||||||
|
on where they say so:
|
||||||
|
|
||||||
- **`system/kernel/ipc.zig`** — a bounded blocking channel between *kernel threads*,
|
- **Rendezvous.** A call is a synchronous meeting, copied sender-page to
|
||||||
described below. The primitive, and where the blocking discipline was worked out.
|
receiver-page — natural backpressure, no queue to size.
|
||||||
- **`system/kernel/ipc-synchronous.zig`** — synchronous call/reply between *processes*, across
|
- **Capability carriage.** The *only* transport that can move a handle
|
||||||
address spaces. What user-space servers and drivers actually talk over. It's the
|
between processes. Channels are therefore always established over
|
||||||
second half of this document.
|
kernel-ipc, and it remains every channel's control path even when bulk
|
||||||
|
data is negotiated onto a fatter transport (a shared-memory ring).
|
||||||
|
- **Verified source.** Every delivery carries the kernel-stamped badge — the
|
||||||
|
identity the channel layer attaches to received packets.
|
||||||
|
- **Bounded packets.** 256 bytes call/reply, 64 pushed — the floor every
|
||||||
|
protocol may assume on any transport.
|
||||||
|
|
||||||
## The channel
|
Three properties keep the networking analogy honest — kernel-ipc is
|
||||||
|
networking-*shaped*, not TCP:
|
||||||
|
|
||||||
The first form is a **bounded blocking channel** (`system/kernel/ipc.zig`): a fixed-size
|
- **Channels over it are RPC-shaped, not streams.** Packets, call/reply,
|
||||||
ring buffer of messages with a producer/consumer rendezvous, built on the
|
datagram pushes — closer to UDP plus RPC than to a byte stream. Ordering
|
||||||
scheduler's [wait queues](../os-development/scheduling.md).
|
exists per exchange (a reply answers its call), not across a channel.
|
||||||
|
- **Possession is the connection.** There is no handshake state in the
|
||||||
|
kernel: holding the capability *is* having the channel. A provider's one
|
||||||
|
endpoint terminates every client's channel at once, demultiplexed by badge
|
||||||
|
— like every client sharing the server's listening socket, with
|
||||||
|
per-connection state living in the provider, keyed by badge. A *private*
|
||||||
|
channel (a dedicated endpoint pair) is built when wanted: that is exactly
|
||||||
|
what `subscribe` does.
|
||||||
|
- **Packets never fragment.** If it doesn't fit in a packet, it isn't a
|
||||||
|
packet: bulk data lives in shared memory and a packet (or signal) is the
|
||||||
|
doorbell. The display path already works this way.
|
||||||
|
|
||||||
|
The rest of this document is the implementation, bottom-up: the kernel-thread
|
||||||
|
queue the blocking discipline was worked out on, then endpoints — this
|
||||||
|
transport's termination points.
|
||||||
|
|
||||||
|
## The kernel-thread queue
|
||||||
|
|
||||||
|
The first form is a **bounded blocking queue** (`system/kernel/ipc.zig`): a
|
||||||
|
fixed-size ring buffer of messages with a producer/consumer rendezvous, built
|
||||||
|
on the scheduler's [wait queues](../os-development/scheduling.md). (Its type
|
||||||
|
is still named `Channel(T, capacity)` — it predates the vocabulary above, and
|
||||||
|
is a *queue between kernel threads in one address space*, not a channel in
|
||||||
|
the model's sense; a rename can ride a later flag-day.)
|
||||||
|
|
||||||
`Channel(T, capacity)` is generic over the message type and buffer size. It holds a
|
`Channel(T, capacity)` is generic over the message type and buffer size. It holds a
|
||||||
ring buffer, a count, and two wait queues:
|
ring buffer, a count, and two wait queues:
|
||||||
|
|
||||||
- **`send(msg)`** — if the channel is full, block on the *not-full* queue; otherwise
|
- **`send(msg)`** — if the queue is full, block on the *not-full* queue; otherwise
|
||||||
write the message, bump the count, and wake a waiting receiver.
|
write the message, bump the count, and wake a waiting receiver.
|
||||||
- **`receive()`** — if the channel is empty, block on the *not-empty* queue; otherwise
|
- **`receive()`** — if the queue is empty, block on the *not-empty* queue; otherwise
|
||||||
take a message, drop the count, and wake a waiting sender.
|
take a message, drop the count, and wake a waiting sender.
|
||||||
|
|
||||||
Neither side busy-waits: a full channel parks the sender, an empty one parks the
|
Neither side busy-waits: a full queue parks the sender, an empty one parks the
|
||||||
receiver, and each operation wakes the other side when it makes progress possible.
|
receiver, and each operation wakes the other side when it makes progress possible.
|
||||||
|
|
||||||
Two details make it correct:
|
Two details make it correct:
|
||||||
@@ -45,47 +81,52 @@ Two details make it correct:
|
|||||||
CPU. `waitLocked` / `wakeLocked` are the variants that assume the caller already
|
CPU. `waitLocked` / `wakeLocked` are the variants that assume the caller already
|
||||||
holds that critical section.
|
holds that critical section.
|
||||||
|
|
||||||
## Verifying it
|
### Verifying it
|
||||||
|
|
||||||
The `ipc` test (see [testing.md](../testing.md)) runs a producer and a consumer passing
|
The `ipc` test (see [testing.md](../testing.md)) runs a producer and a consumer passing
|
||||||
**100 messages through a 4-slot channel**. The small buffer means the channel goes
|
**100 messages through a 4-slot queue**. The small buffer means the queue goes
|
||||||
full and empty over and over, so both the blocking-send and blocking-receive paths are
|
full and empty over and over, so both the blocking-send and blocking-receive paths are
|
||||||
exercised heavily. The messages arrive intact and in order (their sum is the
|
exercised heavily. The messages arrive intact and in order (their sum is the
|
||||||
expected `5050`), and neither task busy-waits — they block and wake each other.
|
expected `5050`), and neither task busy-waits — they block and wake each other.
|
||||||
|
|
||||||
## Endpoints: call/reply across address spaces
|
## Endpoints: the termination points
|
||||||
|
|
||||||
A channel connects two kernel threads sharing one address space. Real servers are
|
A queue connects two kernel threads sharing one address space. Real providers are
|
||||||
*processes*, so the payload has to cross an address-space boundary. That's
|
*processes*, so a packet has to cross an address-space boundary. That's
|
||||||
`system/kernel/ipc-synchronous.zig`, and its shape is L4's: a synchronous **rendezvous** at an
|
`system/kernel/ipc-synchronous.zig`, and its shape is L4's: a synchronous **rendezvous** at an
|
||||||
`Endpoint`, with the message copied directly from the sender's pages to the receiver's
|
`Endpoint`, with the packet copied directly from the sender's pages to the receiver's
|
||||||
(`copyAcross` walks both sets of page tables through the physmap — no CR3 switch, no
|
(`copyAcross` walks both sets of page tables through the physmap — no CR3 switch, no
|
||||||
bounce buffer).
|
bounce buffer).
|
||||||
|
|
||||||
Two syscalls carry it:
|
Two syscalls carry the request/reply exchange:
|
||||||
|
|
||||||
- **`ipc_call(h, msg, reply)`** — copy `msg` to the server, block until it replies.
|
- **`ipc_call(h, msg, reply)`** — copy the request packet to the provider, block
|
||||||
|
until the reply packet comes back.
|
||||||
- **`ipc_reply_wait(h, reply, recv)`** — reply to the client you're still holding (if
|
- **`ipc_reply_wait(h, reply, recv)`** — reply to the client you're still holding (if
|
||||||
any), then block for the next request. One syscall, because a server's steady state
|
any), then block for the next request. One syscall, because a provider's steady state
|
||||||
is *always* "finish the last one, wait for the next".
|
is *always* "finish the last one, wait for the next".
|
||||||
|
|
||||||
An endpoint is reached by **handle** — a small integer index into the process's handle
|
An endpoint is reached by **handle** — a small integer index into the process's handle
|
||||||
table (`Task.handles`), exactly like a file descriptor, and just as unforgeable. The
|
table (`Task.handles`), exactly like a file descriptor, and just as unforgeable.
|
||||||
bootstrap problem (how do you get the first handle?) is solved by a tiny name registry:
|
The provider never learns the client's identity beyond the **badge** delivered
|
||||||
a server calls `ipc_register(service_id, h)` under a well-known small integer, and a
|
alongside each packet: the caller's task id, stamped by the kernel —
|
||||||
client calls `ipc_lookup(service_id)`.
|
unforgeable source addressing, a property a network's source field lacks.
|
||||||
|
|
||||||
The server never learns the client's identity beyond a **badge**, delivered alongside
|
The bootstrap problem — how a channel is first established — is the subject of
|
||||||
the message: the caller's task id.
|
[protocol-namespace.md](../os-development/protocol-namespace.md): a protocol is
|
||||||
|
resolved by name and the channel arrives as a capability. (The mechanism this
|
||||||
|
replaces, `ipc_register`/`ipc_lookup` under compile-time `ServiceId` integers,
|
||||||
|
is retired by that design.)
|
||||||
|
|
||||||
### Interrupts are messages too
|
### Interrupts are signals
|
||||||
|
|
||||||
`notifyFromIsr` posts an *asynchronous* notification to an endpoint — no payload, no
|
`notifyFromIsr` posts an *asynchronous* signal to an endpoint — no payload, no
|
||||||
reply owed — and wakes whoever is blocked in `reply_wait`. Its badge has the top bit
|
reply owed — and wakes whoever is blocked in `reply_wait`. Its badge has the top bit
|
||||||
set (`notify_badge_bit`), which is how a driver's single event loop distinguishes "a
|
set (`notify_badge_bit`), which is how a driver's single event loop distinguishes "a
|
||||||
client wants something" from "the hardware wants something". Notifications sit in a
|
client wants something" from "the hardware wants something". Signals sit in a
|
||||||
small coalescing ring on the endpoint, so an interrupt taken while the driver was busy
|
small coalescing ring on the endpoint, so an interrupt taken while the driver was busy
|
||||||
elsewhere is not lost.
|
elsewhere is not lost — coalesced, never dropped, which is exactly a signal's
|
||||||
|
contract (the *count* may collapse; the *fact* may not).
|
||||||
|
|
||||||
This is what makes a user-space driver possible at all, and it's the subject of
|
This is what makes a user-space driver possible at all, and it's the subject of
|
||||||
[drivers.md](drivers.md).
|
[drivers.md](drivers.md).
|
||||||
@@ -93,40 +134,44 @@ This is what makes a user-space driver possible at all, and it's the subject of
|
|||||||
## What's next (partly done since)
|
## What's next (partly done since)
|
||||||
|
|
||||||
- **Priority inheritance** through IPC — still open: a high-priority client
|
- **Priority inheritance** through IPC — still open: a high-priority client
|
||||||
blocked on a low-priority server suffers unbounded priority inversion.
|
blocked on a low-priority provider suffers unbounded priority inversion.
|
||||||
- **Handle transfer.** *Landed as cap-passing (M13)*: `ipc_call` and
|
- **Handle transfer.** *Landed as cap-passing (M13)*: `ipc_call` and
|
||||||
`ipc_reply_wait` carry an optional capability alongside the bytes (`send_cap`),
|
`ipc_reply_wait` carry an optional capability alongside the bytes (`send_cap`),
|
||||||
copying an endpoint or shared-memory handle into the peer's table. First user:
|
copying an endpoint or shared-memory handle into the peer's table — the
|
||||||
[input](input.md) subscribers register by handing over their own endpoint, and
|
mechanism by which channels are established and private channels built. First
|
||||||
class drivers get a private channel to one device.
|
user: [input](input.md) subscribers register by handing over their own
|
||||||
|
endpoint, and class drivers get a private channel to one device.
|
||||||
- **Asynchronous / buffered send** for the cases where a rendezvous is the wrong
|
- **Asynchronous / buffered send** for the cases where a rendezvous is the wrong
|
||||||
shape (logging, notifications between servers). *Landed as `ipc_send`* — a
|
shape (logging, event fan-out). *Landed as `ipc_send`* — a
|
||||||
non-blocking post to an endpoint's bounded payload queue, delivered through
|
non-blocking post of an event packet (≤ 64 bytes) to an endpoint's bounded
|
||||||
`reply_wait` as a buffered message (badge bit `notify_message_bit`). Built for, and
|
queue, delivered through `reply_wait` (badge bit `notify_message_bit`). Built
|
||||||
first used by, the [input service](input.md)'s keyboard-event broadcast, where a
|
for, and first used by, the [input service](input.md)'s keyboard-event
|
||||||
synchronous push would let one dead subscriber hang the fan-out. A full queue drops
|
broadcast, where a synchronous push would let one dead subscriber hang the
|
||||||
the oldest (discrete messages, not a coalescing level like the notification ring).
|
fan-out. A full queue drops the oldest — event packets are droppable by
|
||||||
- **A bounded reply** — half landed. The copy is still 256 bytes
|
design ([protocol-namespace.md](../os-development/protocol-namespace.md)'s
|
||||||
(`MESSAGE_MAXIMUM`) under the big kernel lock, but bulk transfer got its shared
|
wiring section states the rule).
|
||||||
|
- **A bounded reply** — half landed. The copy is still one packet
|
||||||
|
(256 bytes) under the big kernel lock, but bulk transfer got its shared
|
||||||
pages: `shared_memory_create`/`map`/`physical`, the region handle delegated as
|
pages: `shared_memory_create`/`map`/`physical`, the region handle delegated as
|
||||||
a capability (above). virtio-gpu's scanout surface is the first user
|
a capability (above) — the packets-never-fragment rule in practice.
|
||||||
|
virtio-gpu's scanout surface is the first user
|
||||||
([display-v2.md](display-v2.md)).
|
([display-v2.md](display-v2.md)).
|
||||||
|
|
||||||
## Lifecycle conventions over IPC (M17)
|
## Lifecycle conventions over IPC (M17)
|
||||||
|
|
||||||
Three conventions from [process-lifecycle.md](../os-development/process-lifecycle.md) ride the
|
Three conventions from [process-lifecycle.md](../os-development/process-lifecycle.md) ride the
|
||||||
notification mechanism:
|
signal mechanism:
|
||||||
|
|
||||||
- **Signals** arrive as notifications on the endpoint a process nominated with
|
- **Process signals** arrive as endpoint signals on the endpoint a process
|
||||||
`signal_bind` (`process.bindSignals`): badge = the signal bit plus the
|
nominated with `signal_bind` (`process.bindSignals`): badge = the signal bit
|
||||||
coalesced pending mask (`process.signalsFrom` decodes). Statements,
|
plus the coalesced pending mask (`process.signalsFrom` decodes). Statements,
|
||||||
never questions; no payload, no reply.
|
never questions; no payload, no reply.
|
||||||
- **One-shot timers** (`timer_bind`, `time.timerOnce`) land as a
|
- **One-shot timers** (`timer_bind`, `time.timerOnce`) land as a
|
||||||
timer-bit notification — the timed wait: a service arms a deadline and keeps
|
timer-bit signal — the timed wait: a service arms a deadline and keeps
|
||||||
serving, instead of blocking in sleep.
|
serving, instead of blocking in sleep.
|
||||||
- **The universal ping**: a **zero-length request is the liveness probe**,
|
- **The universal ping**: a **zero-length request is the liveness probe**,
|
||||||
answered with a zero-length reply by the service harness itself
|
answered with a zero-length reply by the service harness itself
|
||||||
(`service.run`). No protocol's requests start at length zero, so the
|
(`service.run`). No protocol's requests start at length zero, so the
|
||||||
encoding cannot collide, and a wedged service simply fails to answer — which
|
encoding cannot collide, and a wedged service simply fails to answer — which
|
||||||
is the diagnosis. Deep health ("can I reach my hardware?") stays a per-service
|
is the diagnosis. Deep health ("can I reach my hardware?") stays a per-service
|
||||||
protocol message.
|
protocol packet.
|
||||||
|
|||||||
@@ -0,0 +1,218 @@
|
|||||||
|
# New driver: the minimum steps
|
||||||
|
|
||||||
|
The shortest path from "a device shows up in the boot log" to "my process is
|
||||||
|
running with its registers mapped". This is the checklist; the reasoning behind
|
||||||
|
every step lives in [Writing a driver](drivers.md), the matching rules in
|
||||||
|
[devices.csv](devices-csv.md), and interrupts in
|
||||||
|
[device interrupts](device-interrupts.md).
|
||||||
|
|
||||||
|
Worked example throughout: the Intel UHD 750 iGPU, which the boot log reports as
|
||||||
|
|
||||||
|
```
|
||||||
|
pci-bus: 0:2.0 bus=pci base=03 class=00 prog_if=00 vendor=8086 device=4C8A ...
|
||||||
|
```
|
||||||
|
|
||||||
|
## 1. Create the source file
|
||||||
|
|
||||||
|
`system/drivers/<name>/<name>.zig` — kebab-case, abbreviations spelled out
|
||||||
|
([coding standards](../coding-standards.md)). The directory name, the binary
|
||||||
|
name, and the `devices.csv` driver path must all agree; a mismatch fails
|
||||||
|
silently (the device-manager logs the spawn failure, nothing else happens).
|
||||||
|
|
||||||
|
The complete minimal driver — claims its device, logs every resource, maps the
|
||||||
|
register window, then sleeps in the harness loop:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
//! /system/drivers/intel-uhd-graphics-750 — spawned by the device manager with
|
||||||
|
//! the device-tree id as argv[1]; claims that device and no other.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const device = @import("driver");
|
||||||
|
const ipc = @import("ipc");
|
||||||
|
const memory = @import("memory");
|
||||||
|
const process = @import("process");
|
||||||
|
const service = @import("service");
|
||||||
|
|
||||||
|
/// No protocol yet: the kernel's IPC ceiling (MESSAGE_MAXIMUM) sizes the buffers.
|
||||||
|
const message_maximum = 256;
|
||||||
|
|
||||||
|
var controller_id: u64 = 0;
|
||||||
|
var register_base: usize = 0;
|
||||||
|
|
||||||
|
fn initialise(endpoint: ipc.Handle) bool {
|
||||||
|
_ = endpoint; // needed later, for irq binding and timers
|
||||||
|
|
||||||
|
if (!device.claim(controller_id)) {
|
||||||
|
std.log.err("unable to claim device {d}", .{controller_id});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Fetch our own descriptor back for the device's resources.
|
||||||
|
const buffer = memory.allocator().alloc(device.DeviceDescriptor, 64) catch return false;
|
||||||
|
defer memory.allocator().free(buffer);
|
||||||
|
const total = device.enumerate(buffer);
|
||||||
|
const descriptor = for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||||
|
if (d.id == controller_id) break d;
|
||||||
|
} else {
|
||||||
|
std.log.err("device {d} not in the device tree", .{controller_id});
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
|
||||||
|
// Log every resource BEFORE choosing one (see step 5).
|
||||||
|
var register_index: u64 = 0;
|
||||||
|
for (descriptor.resources[0..@intCast(descriptor.resource_count)], 0..) |resource, index| {
|
||||||
|
std.log.info("resource {d}: kind={d} start=0x{x} len=0x{x}", .{
|
||||||
|
index, resource.kind, resource.start, resource.len,
|
||||||
|
});
|
||||||
|
// The 16 MiB window is GTTMMADR, the register BAR (this device also has
|
||||||
|
// a 256 MiB memory BAR, GMADR — "first memory resource" would be wrong).
|
||||||
|
if (resource.kind == @intFromEnum(device.ResourceKind.memory) and
|
||||||
|
resource.len == 16 * 1024 * 1024) register_index = index;
|
||||||
|
}
|
||||||
|
if (register_index == 0) {
|
||||||
|
std.log.err("register BAR not found", .{});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
register_base = device.mmioMap(controller_id, register_index) orelse {
|
||||||
|
std.log.err("mmio_map failed", .{});
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
std.log.info("registers mapped at 0x{x}", .{register_base});
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||||
|
_ = message;
|
||||||
|
_ = reply;
|
||||||
|
_ = sender;
|
||||||
|
_ = capability;
|
||||||
|
return 0; // no protocol yet; the zero-length ping is answered by the harness
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn main(init: process.Init) void {
|
||||||
|
const argument = init.arguments.get(1) orelse {
|
||||||
|
std.log.err("missing device id (argv[1])", .{});
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
controller_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||||
|
std.log.err("malformed device id '{s}'", .{argument});
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
service.run(message_maximum, .{
|
||||||
|
.init = initialise,
|
||||||
|
.on_message = onMessage,
|
||||||
|
// .on_notification only once an IRQ or timer is bound
|
||||||
|
});
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
`claim` is the capability gate: MMIO mapping, DMA grants, and IRQ binding all
|
||||||
|
require it, and it pins the IOMMU domain to this process
|
||||||
|
([drivers.md — claim before touch](drivers.md#the-capability-claim-before-touch)).
|
||||||
|
|
||||||
|
## 2. Create the build package and register it in the root build
|
||||||
|
|
||||||
|
The driver directory is its own build package
|
||||||
|
([build-packages-plan.md](../build-packages-plan.md)): a ~15-line `build.zig`
|
||||||
|
plus a `build.zig.zon` beside the source. Copy both from an existing driver —
|
||||||
|
`system/drivers/pci-bus/` is the template — and adjust the name, root source
|
||||||
|
file, and the import list. The list names EXACTLY the modules the driver's
|
||||||
|
source `@import`s (the moral equivalent of its include list; an undeclared
|
||||||
|
import is a compile error):
|
||||||
|
|
||||||
|
```zig
|
||||||
|
pub fn build(b: *std.Build) void {
|
||||||
|
const exe = build_support.userBinary(b, .{
|
||||||
|
.name = "intel-uhd-graphics-750",
|
||||||
|
.root_source_file = b.path("intel-uhd-graphics-750.zig"),
|
||||||
|
.imports = &.{ "driver", "ipc", "memory", "process", "service" },
|
||||||
|
});
|
||||||
|
b.installArtifact(exe);
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
The zon declares `build-support`, `kernel` (implicit in every binary: the root
|
||||||
|
shim lives there), and the homes of the listed imports — for the minimal
|
||||||
|
driver above that is kernel alone plus `device` (for `driver`); add
|
||||||
|
`protocol`, `client`, ... only when an import comes from them (again, copy
|
||||||
|
pci-bus's zon and adjust). For the `.fingerprint` field, leave the copied
|
||||||
|
value in place and `zig build` will reject it and suggest the fresh one to
|
||||||
|
paste.
|
||||||
|
|
||||||
|
Then three one-liners in the root build register the package: the dependency
|
||||||
|
and a row in the boot-tree array in `build.zig` (search for
|
||||||
|
`virtio_gpu_package` to land in the right places),
|
||||||
|
|
||||||
|
```zig
|
||||||
|
const intel_uhd_graphics_750_exe = b.dependency("intel-uhd-graphics-750", .{}).artifact("intel-uhd-graphics-750");
|
||||||
|
```
|
||||||
|
|
||||||
|
```zig
|
||||||
|
.{ .path = "system/drivers/intel-uhd-graphics-750", .binary = intel_uhd_graphics_750_exe.getEmittedBin() },
|
||||||
|
```
|
||||||
|
|
||||||
|
and the path entry in the root `build.zig.zon`:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
.@"intel-uhd-graphics-750" = .{ .path = "system/drivers/intel-uhd-graphics-750" },
|
||||||
|
```
|
||||||
|
|
||||||
|
Without the boot-tree row the binary never reaches the image and the
|
||||||
|
device-manager has nothing to spawn. (The package also builds standalone:
|
||||||
|
`cd system/drivers/intel-uhd-graphics-750 && zig build`.)
|
||||||
|
|
||||||
|
## 3. Add the match rule to `etc/devices.csv`
|
||||||
|
|
||||||
|
One row: bus, class triplet, vendor/device, driver path. **Copy the class
|
||||||
|
triplet from the pci-bus boot log line, not from another row** — for the iGPU
|
||||||
|
above the correct rule is
|
||||||
|
|
||||||
|
```
|
||||||
|
pci, 03, 00, 00, 8086, 4C8A, *, *, /system/drivers/intel-uhd-graphics-750
|
||||||
|
```
|
||||||
|
|
||||||
|
Field-by-field rules and the most-specific-wins policy: [devices.csv](devices-csv.md).
|
||||||
|
The registry is authoritative: an unmatched device is logged unbound, never
|
||||||
|
guessed — so a wrong nibble here means the driver simply never starts.
|
||||||
|
|
||||||
|
## 4. First contact: read, predict, verify
|
||||||
|
|
||||||
|
Before writing any register, read one whose value you can predict from state
|
||||||
|
the firmware already programmed (for a display controller: the pipe source
|
||||||
|
size of the live mode). Registers are volatile loads at `register_base +
|
||||||
|
offset`, where `offset` is what the device's manual lists:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
fn read32(offset: usize) u32 {
|
||||||
|
return @as(*volatile u32, @ptrFromInt(register_base + offset)).*;
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
A matching read proves the whole chain — CSV match, spawn, claim, BAR choice,
|
||||||
|
mapping — with zero risk to the hardware.
|
||||||
|
|
||||||
|
## 5. Verify the plumbing
|
||||||
|
|
||||||
|
- `zig build test` still passes.
|
||||||
|
- On the image: `/var/log/<boot-stamp>/system/services/device-manager.log`
|
||||||
|
shows `spawned <name> for device <N>`, and
|
||||||
|
`/var/log/<boot-stamp>/system/drivers/<name>.log` holds the resource list and
|
||||||
|
your first read.
|
||||||
|
- If the driver did not spawn, diagnose in this order: binary on the image
|
||||||
|
(step 2) → CSV row matches the log line exactly (step 3) → path identical in
|
||||||
|
both (step 1).
|
||||||
|
|
||||||
|
## Later, when the device needs them
|
||||||
|
|
||||||
|
- **Interrupts**: MSI/MSI-X via the `pci` module, delivered as notifications to
|
||||||
|
`on_notification` — see [device interrupts](device-interrupts.md) and the
|
||||||
|
xHCI driver's `setupMsi` (QEMU trap documented there: enable MSI-X before
|
||||||
|
unmasking the device's own interrupt-enable bit).
|
||||||
|
- **DMA**: grant-backed buffers, bounded by the IOMMU domain established at
|
||||||
|
claim time ([driver model](driver-model.md)).
|
||||||
|
- **Children**: a bus driver publishes what it finds via `device_register`
|
||||||
|
([drivers.md — publishing children](drivers.md#publishing-children-device_register)).
|
||||||
|
- **A protocol**: replace `message_maximum` with the protocol's own maximum and
|
||||||
|
dispatch on the operation word in `onMessage` — every service under
|
||||||
|
`system/services/` is an example.
|
||||||
@@ -0,0 +1,124 @@
|
|||||||
|
# Dynamic libraries on danos
|
||||||
|
|
||||||
|
A design note and milestone plan for shared objects: building them, loading them
|
||||||
|
with `dlopen`, and — the part that needs kernel work — actually *sharing* them
|
||||||
|
between processes. Directional, post-P5 of
|
||||||
|
[python-on-danos-milestones.md](python-on-danos-milestones.md); nothing on the
|
||||||
|
CPython bring-up path depends on it.
|
||||||
|
|
||||||
|
## Reconciling the earlier "rejected"
|
||||||
|
|
||||||
|
Dynamic libraries were evaluated once before and rejected — but as an answer to a
|
||||||
|
*different question*: whether they could claw back ReleaseSafe's measured ~2×
|
||||||
|
code size. They cannot (the safety checks inline at every call site; no library
|
||||||
|
scheme dedups them), and that verdict stands for that question. The reasons to
|
||||||
|
build them now are the ones that investigation never weighed:
|
||||||
|
|
||||||
|
- **`ctypes` and runtime FFI** — Python calling into a danos library without
|
||||||
|
rebuilding the interpreter. This is the piece that makes Python prototyping
|
||||||
|
self-serve: drop a `.so` on the image, `ctypes.CDLL` it, iterate.
|
||||||
|
- **Loadable CPython extension modules** — today every C extension means
|
||||||
|
relinking the interpreter (`Modules/Setup`); with `dlopen`, an extension is a
|
||||||
|
file.
|
||||||
|
- **One interpreter image, many Python services** — a statically-linked CPython
|
||||||
|
is tens of megabytes *per process*. A shared `libpython` mapped read-only once
|
||||||
|
(milestone D3 below) makes Python services cheap enough to be the default way
|
||||||
|
to prototype one.
|
||||||
|
- **Plugin-shaped applications** — the UI toolkit and the terminal will want
|
||||||
|
them eventually.
|
||||||
|
|
||||||
|
The scoping that dissolves the apparent contradiction is the **size doctrine**:
|
||||||
|
leanness is an *operating-system* property — the kernel and system services stay
|
||||||
|
small and statically linked, and none of them ever link the loader — while
|
||||||
|
*applications* have their own budget and may be big. Dynamic libraries are an
|
||||||
|
**application-layer facility**, full stop.
|
||||||
|
|
||||||
|
What also does **not** change: the public ABI stays the vDSO + the IPC
|
||||||
|
protocols. Shared objects are artifacts *within* one system image, versioned by
|
||||||
|
the build — not a new stable ABI surface for the OS.
|
||||||
|
|
||||||
|
## Design
|
||||||
|
|
||||||
|
- **Format and codegen are free.** ELF shared objects with position-independent
|
||||||
|
code; `zig cc -fPIC -shared` against the [libdanos-c](c-library-compatibility.md)
|
||||||
|
sysroot already emits them. The work is entirely on the loading side.
|
||||||
|
- **The loader lives in userspace, inside the libc.** `dlopen` reads the `.so`
|
||||||
|
through the VFS, maps its segments, applies relocations, resolves symbols
|
||||||
|
against the process and the `DT_NEEDED` dependency graph, runs constructors,
|
||||||
|
returns a handle. No kernel loader changes in v1 — segments land in anonymous
|
||||||
|
`mmap` as private copies.
|
||||||
|
- **Bind-now, always.** All relocations resolved at `dlopen` time
|
||||||
|
(`RTLD_NOW` semantics only). Lazy PLT binding buys startup latency danos does
|
||||||
|
not care about, at the price of a writable GOT dance and a much subtler
|
||||||
|
loader. Not worth it; keep it out permanently.
|
||||||
|
- **W^X from day one.** Map, relocate, then flip text pages read-execute —
|
||||||
|
which requires memory-protection change (`mprotect`-shaped) in the danos
|
||||||
|
`mmap` surface if it is not already there. No page is ever writable and
|
||||||
|
executable at once.
|
||||||
|
- **TLS in shared objects is deferred.** Thread-local storage models
|
||||||
|
(initial-exec vs. general-dynamic) are the deep end of every dynamic linker.
|
||||||
|
v1 refuses a `.so` with a TLS segment; revisit alongside the post-P5 pthread
|
||||||
|
subset, which is when it could matter.
|
||||||
|
- **Executables stay static until D4.** v1 is "a static binary that can
|
||||||
|
`dlopen`" — no `PT_INTERP`, no program interpreter, no dynamically-linked
|
||||||
|
`main` binaries. That keeps process startup untouched.
|
||||||
|
|
||||||
|
## Milestones
|
||||||
|
|
||||||
|
1. **D1 — dlopen in-process.** The `.so` build target; the loader in libdanos-c:
|
||||||
|
map, relocate (`RELATIVE`/`GLOB_DAT`/`JUMP_SLOT`), resolve, constructors;
|
||||||
|
`dlopen`/`dlsym`/`dlerror`/`dlclose`; private anonymous mappings; no TLS.
|
||||||
|
*Test:* QEMU `dlopen-hello` — load a `.so`, call a symbol, unload, reload.
|
||||||
|
2. **D2 — the FFI payoff.** `DT_NEEDED` dependency graphs; a **libffi port**
|
||||||
|
(x86-64 SysV assembly is upstream; the port is its closure-allocation paths,
|
||||||
|
which must respect W^X); CPython's `ctypes` enabled; extension modules
|
||||||
|
loadable from file. *Test:* QEMU — a Python script `ctypes.CDLL`s a danos
|
||||||
|
`.so` and round-trips a call; a `.so` extension module imports.
|
||||||
|
3. **D3 — actual sharing (the kernel milestone).** Shared read-only file-backed
|
||||||
|
mappings — a page-cache-shaped facility so N processes mapping `libpython`
|
||||||
|
hold one physical copy. This is the memory-win milestone and the only one
|
||||||
|
touching the kernel; design it with the existing shm machinery in view
|
||||||
|
(the shared-fate walks already locked the relevant paths). *Test:* N Python
|
||||||
|
services up; measure physical pages against N× the static baseline.
|
||||||
|
4. **D4 — dynamically-linked executables** (optional, evaluate after D3):
|
||||||
|
`PT_INTERP`, a danos program interpreter, and the spawn path teaching the
|
||||||
|
loader about it. Only worth it if the image-size or update story demands it.
|
||||||
|
|
||||||
|
## Risks and gotchas
|
||||||
|
|
||||||
|
- **Scope creep is the failure mode.** Every dynamic linker grows toward glibc.
|
||||||
|
The fences: bind-now only, no lazy binding ever, no TLS until pthreads demand
|
||||||
|
it, no dlopen-from-memory, no versioned symbols. Each fence removed is a
|
||||||
|
design discussion, not a patch.
|
||||||
|
- **Code loading is a security event.** `dlopen` turns file bytes into executable
|
||||||
|
code, so W^X discipline is table stakes and *what may be dlopened* is a
|
||||||
|
capability question — the natural danos answer is that loadability follows VFS
|
||||||
|
readability of the `.so`, and services' images are supervised like any other
|
||||||
|
artifact. Revisit explicitly at D3 when mappings become shared.
|
||||||
|
- **`dlclose` is where loaders go to die.** Constructors/destructors,
|
||||||
|
dangling function pointers, re-open identity. Keep v1 semantics honest and
|
||||||
|
simple: `dlclose` runs destructors and unmaps; holding pointers past it is
|
||||||
|
undefined; no reference-counted deferral cleverness.
|
||||||
|
- **The ReleaseSafe fact still applies to `.so`s** — a ReleaseSafe shared object
|
||||||
|
carries its inlined checks like any static code; D3's sharing saves *copies*,
|
||||||
|
not check overhead. Size expectations should be set accordingly.
|
||||||
|
|
||||||
|
## Decisions needing sign-off
|
||||||
|
|
||||||
|
- Dynamic libraries join the roadmap at all (this note exists because the
|
||||||
|
earlier size-motivated rejection was re-opened for ABI/sharing reasons).
|
||||||
|
- **Bind-now only; no lazy binding, permanently.**
|
||||||
|
- **Loader in userspace libc; kernel involvement only at D3** (shared read-only
|
||||||
|
mappings).
|
||||||
|
- **Static executables until D4**, and D4 only on demonstrated need.
|
||||||
|
|
||||||
|
## Related
|
||||||
|
|
||||||
|
- [c-library-compatibility.md](c-library-compatibility.md) — the sysroot the
|
||||||
|
loader ships in; its absence table gains `dlfcn.h` at D1.
|
||||||
|
- [python-on-danos.md](python-on-danos.md) — the `ctypes` story this unlocks.
|
||||||
|
- [python-on-danos-milestones.md](python-on-danos-milestones.md) — sequencing;
|
||||||
|
this work is post-P5.
|
||||||
|
- [os-development/memory-map.md](os-development/memory-map.md) /
|
||||||
|
[os-development/paging.md](os-development/paging.md) — where W^X and shared
|
||||||
|
mappings land.
|
||||||
@@ -1,128 +0,0 @@
|
|||||||
# DanOS Filesystem Hierarchy Standard (DFHS)
|
|
||||||
|
|
||||||
Most modern Unix and Unix-like operating systems follow the FHS. DanOS has its own FHS structure which extends the unix FHS. Root path resolution is provided by the kernel-resident VFS root (`fs_resolve`, `system/kernel/vfs.zig`); mounted filesystem servers serve the subtrees they own.
|
|
||||||
|
|
||||||
## Directory structure
|
|
||||||
|
|
||||||
| Path | Description |
|
|
||||||
|------------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
|
||||||
| / | Primary hierarchy root and root directory of the entire file system hierarchy. |
|
|
||||||
| /bin | Essential command binaries that need to be available in single-user mode, including to bring up the system or repair it, for all users (e.g., cat, ls, cp). |
|
|
||||||
| /boot | Boot loader files (e.g., EFI, initial-ramdisk.img ). |
|
|
||||||
| /dev | POSIX Device files (e.g., /dev/null, /dev/disk0, /dev/tty, /dev/random). |
|
|
||||||
| /etc | Host-specific system-wide configuration files. |
|
|
||||||
| /home | Users' home directories, containing saved files, personal settings, etc. |
|
|
||||||
| /lib | Libraries essential for the binaries in /bin and /sbin. eg realtime, system, ipc etc. |
|
|
||||||
| /sbin | Essential system binaries (e.g init) |
|
|
||||||
| /srv | Site-specific data served by this system, such as data and scripts for web servers, data offered by FTP servers, and repositories for version control systems |
|
|
||||||
| /system | DanOS operating system files (similar idea to C:\Windows). A true representation of danos — its layout mirrors the source tree, so `/system` is what danos *is*. |
|
|
||||||
| /system/devices | danos virtual device tree e.g. similar to /sys on linux but with danos device tree conventions (the structures in the devices module) |
|
|
||||||
| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/pci-bus, /system/drivers/ps2-bus) |
|
|
||||||
| /system/services | system-service binaries — init, the FAT server, and other user-mode servers (e.g. /system/services/init, /system/services/fat) |
|
|
||||||
| /system/kernel | the kernel image |
|
|
||||||
| /test | Test fixtures for the QEMU integration suite. Read-only and initrd-backed like /system, and its layout likewise mirrors the source tree (the repo's test/ directory). Present on development and test images; a volume without it still boots. |
|
|
||||||
| /test/system/services | test-fixture binaries (e.g. /test/system/services/vfs-test, /test/system/services/thread-test) — the same path in the repo source tree and on the boot volume |
|
|
||||||
| /tmp | Directory for temporary files (see also /var/tmp). Often not preserved between system reboots and may be severely size-restricted. |
|
|
||||||
| /usr | Secondary hierarchy for read-only user data; contains the majority of (multi-)user utilities and applications. Should be shareable and read-only. |
|
|
||||||
| /var | Variable files: files whose content is expected to continually change during normal operation of the system, such as logs, spool files, and temporary e-mail files. |
|
|
||||||
|
|
||||||
## File types
|
|
||||||
|
|
||||||
POSIX specifies the long format of the ls command to represent the Unix file type as the first letter for an entry.
|
|
||||||
|
|
||||||
| type | symbol | Description |
|
|
||||||
|-------------------|--------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
|
||||||
| regular | - | An ordinary file holding an uninterpreted byte stream. Reads and writes are positional, and the file grows on demand (e.g., a binary in /bin, a config file in /etc). |
|
|
||||||
| directory | d | A container mapping names to other files. It may only be modified through directory operations, never written to directly. |
|
|
||||||
| symbolic link | l | A file whose contents are a path that is resolved in its place. The target need not exist, and may cross mount points. |
|
|
||||||
| FIFO special | p | A named pipe: an in-order byte stream between processes, where writers block until a reader opens the other end. |
|
|
||||||
| block special | b | A device node addressed in fixed-size blocks with the kernel free to buffer and reorder access (e.g., /dev/disk0). |
|
|
||||||
| character special | c | A device node addressed as an unbuffered byte stream, delivered to the driver in order (e.g., /dev/tty, /dev/null). |
|
|
||||||
| socket | s | A named endpoint for bidirectional message-passing between processes, bound to a path rather than an address. |
|
|
||||||
|
|
||||||
## /dev
|
|
||||||
|
|
||||||
`/dev` holds the names through which processes reach devices. It is deliberately not
|
|
||||||
the device tree: the tree — every node discovered by ACPI or PCI enumeration, with its
|
|
||||||
resources and its parent — lives under [/system/devices](#directory-structure) and is
|
|
||||||
addressed by device id. `/dev` is the much smaller set of devices that have a driver
|
|
||||||
willing to serve them, addressed by name.
|
|
||||||
|
|
||||||
A device node is not a file the VFS can read. The bytes live in a driver process
|
|
||||||
([drivers.md](../device-driver-development/drivers.md)), so opening a `/dev` name has to resolve to that driver's
|
|
||||||
IPC endpoint, and subsequent reads and writes are calls against it. Resolve-to-endpoint
|
|
||||||
is exactly what the kernel's `fs_resolve` already does for any mounted backend, and
|
|
||||||
`FileStatus.kind` is the field that marks a device node; **what is not implemented today
|
|
||||||
is `/dev` itself** — no service mounts it. (The flat eight-node ramfs this section once
|
|
||||||
described is retired: the kernel-resident VFS root in `system/kernel/vfs.zig` serves a
|
|
||||||
read-only initrd mount per top-level tree — `/system`, and `/test` on images that carry
|
|
||||||
the fixtures — with real directories and node kinds, and filesystem
|
|
||||||
backends such as the FAT server mount the rest.) The three sections below describe the
|
|
||||||
intended shape, and are honest about which parts the kernel can already support.
|
|
||||||
|
|
||||||
### Character devices
|
|
||||||
|
|
||||||
A character device is a byte stream with no addressable position: bytes are delivered
|
|
||||||
to the driver in the order written, and a read consumes what is there. Terminals,
|
|
||||||
serial lines, keyboards and mice are all of this shape. These are the natural first
|
|
||||||
device nodes in danos, because a character driver needs nothing the kernel doesn't
|
|
||||||
already provide — it claims its device, maps its registers with `mmio_map`, and blocks
|
|
||||||
on `replyWait` for either an interrupt or a client request. `system/drivers/ps2-bus/ps2-bus.zig`
|
|
||||||
is already that program, minus the file-node client half.
|
|
||||||
|
|
||||||
The obstacle was never the file type; it is which hardware a ring-3 driver can reach.
|
|
||||||
Direct `in`/`out` from user space is still a #GP (no TSS I/O bitmap, IOPL never raised),
|
|
||||||
but a driver no longer needs it: **`io_read`/`io_write`** grant port access the same way
|
|
||||||
`mmio_map` grants memory — gated by `device_claim` and the device's discovered `io_port`
|
|
||||||
resource. So the 16550 UART at `0x3F8` and the PS/2 controller at `0x60`/`0x64` (and thus
|
|
||||||
`/dev/ttyS0` and a keyboard node) are now writable as ordinary ring-3 drivers; the
|
|
||||||
low-rate legacy hardware that needs port I/O is fine with a syscall per access. A
|
|
||||||
memory-mapped device such as the framebuffer, needing no port I/O at all, remains the
|
|
||||||
easiest first entry.
|
|
||||||
|
|
||||||
### Block devices
|
|
||||||
|
|
||||||
A block device is addressed in fixed-size blocks and, unlike a character device, the
|
|
||||||
layer above is free to buffer, reorder, coalesce and retry requests against it. Disks
|
|
||||||
and other persistent storage are the whole population of this class.
|
|
||||||
|
|
||||||
A block driver is now **writable, but not yet memory-safe.** Every storage controller
|
|
||||||
worth naming is a bus master: it is programmed by handing it the physical address of a
|
|
||||||
descriptor ring and left to read and write memory on its own. That ring is exactly what
|
|
||||||
**`dma_alloc`** now provides — physically contiguous, pinned, uncacheable, with its
|
|
||||||
physical address disclosed — and **`/lib/device/mmio`**'s barriers order the descriptor writes
|
|
||||||
against the doorbell, and **`msi_bind`** delivers completions. So an AHCI or NVMe driver
|
|
||||||
can be written today (the M14/M15 work in [driver-model.md](../device-driver-development/driver-model.md); the earlier
|
|
||||||
"cannot host a block driver at all" is no longer true).
|
|
||||||
|
|
||||||
What is *not* yet true is that it is safe. A device programmed with an arbitrary physical
|
|
||||||
address writes to arbitrary physical memory, and page tables do not sit between a device
|
|
||||||
and RAM — an IOMMU does. The IOMMU is now *detected* (M16), but no translation domains
|
|
||||||
are programmed, so granting a DMA-capable device to a driver process is still equivalent
|
|
||||||
to granting ring 0. Until per-device domains confine a driver's DMA to the buffers it
|
|
||||||
`dma_alloc`'d, a block driver works but forfeits the isolation that motivates user-space
|
|
||||||
drivers — enforcement is the next step, and lands with that first driver. A ramdisk over
|
|
||||||
the initial ramdisk remains the one block-shaped thing that needs no driver process at all.
|
|
||||||
|
|
||||||
### Pseudo-devices
|
|
||||||
|
|
||||||
A pseudo-device has the interface of a device and no hardware behind it: `/dev/null`
|
|
||||||
discarding writes and reading as end-of-file, `/dev/zero` reading as an endless run of
|
|
||||||
zero bytes, `/dev/full` failing writes with `ENOSPC`, `/dev/random` and `/dev/urandom`
|
|
||||||
yielding unpredictable bytes.
|
|
||||||
|
|
||||||
These are the only `/dev` entries danos can implement immediately, and they are the
|
|
||||||
sensible place to start, because they are exactly the entries that need no driver
|
|
||||||
process, no `device_claim`, no MMIO grant and no interrupt. A future pseudo-device
|
|
||||||
service would answer them out of its own address space — `null` and `zero` are a few
|
|
||||||
lines each in its `read` and `write` handlers — and mount itself at `/dev` the way the
|
|
||||||
FAT server mounts `/mnt/usb`. The two pieces of structure every later device node
|
|
||||||
depends on (and that the flat ramfs of the time lacked) exist now: directories, so that
|
|
||||||
`/dev/null` is a path rather than a name; and a populated `FileStatus.kind`, so that a
|
|
||||||
caller can tell a character device from a regular file.
|
|
||||||
|
|
||||||
`/dev/random` is the one that is not free. It needs an entropy source, and the honest
|
|
||||||
options on this kernel are `RDRAND`/`RDSEED` where CPUID advertises them, and the HPET
|
|
||||||
counter's low bits as a poor fallback. Neither is a seeded CSPRNG, and a `/dev/random`
|
|
||||||
that is merely unpredictable-looking is worse than none — nothing should be keyed from
|
|
||||||
it until it is a real one.
|
|
||||||
@@ -0,0 +1,90 @@
|
|||||||
|
# The danos file-system hierarchy
|
||||||
|
|
||||||
|
danos is not unix, and its tree does not follow the unix FHS. Paths are the
|
||||||
|
system's universal namespace — files, the device inventory, and protocol
|
||||||
|
endpoints all live in one tree — but what a path *yields* differs by subtree:
|
||||||
|
bytes, facts, or a connection. Root path resolution is provided by the
|
||||||
|
kernel-resident VFS root (`fs_resolve`, `system/kernel/vfs.zig`); mounted
|
||||||
|
backends serve the subtrees they own.
|
||||||
|
|
||||||
|
Naming follows the codebase conventions: kebab-case, full words, no
|
||||||
|
abbreviations. Every top-level name says what its subtree *is*.
|
||||||
|
|
||||||
|
## The tree
|
||||||
|
|
||||||
|
| Path | What it is |
|
||||||
|
|-------------------------|---------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||||
|
| `/` | The root of the one namespace. |
|
||||||
|
| `/applications` | Installed applications, one directory per application — the directory is the identity, the same rule as source sub-projects. *(Planned; empty today.)* |
|
||||||
|
| `/protocol` | The contract namespace: one protocol node per contract, grouped into directories by domain (`/protocol/display`, `/protocol/networking/ip`). Synthetic — no bytes; opening a name yields a connection to the current provider. See [protocol-namespace.md](../os-development/protocol-namespace.md). |
|
||||||
|
| `/system` | The operating system — what danos *is*. Its program subtrees mirror the source tree exactly. |
|
||||||
|
| `/system/kernel` | The kernel image. |
|
||||||
|
| `/system/drivers` | Driver binaries, one per sub-project (`/system/drivers/pci-bus`, `/system/drivers/ps2-bus`). |
|
||||||
|
| `/system/services` | System-service binaries (`/system/services/init`, `/system/services/fat`). |
|
||||||
|
| `/system/devices` | The device inventory: every node hardware discovery found, with its resources and parent — the structures of the devices module, as a browsable virtual tree. Informational only; you *read about* hardware here and *talk to* it through `/protocol`. *(Planned; served by device-manager.)* |
|
||||||
|
| `/system/configuration` | Machine configuration (`init.csv`, `devices.csv`). Writable, served from the boot volume. |
|
||||||
|
| `/system/logs` | Per-boot logs: `/system/logs/<boot-stamp>/<binary-path>.log`. Writable, served from the boot volume. |
|
||||||
|
| `/test` | Test fixtures for the QEMU integration suite. Read-only and initrd-backed like the program subtrees of `/system`, mirroring the repo's `test/` directory. Present on development and test images; a volume without it still boots. |
|
||||||
|
| `/volumes` | Attached storage volumes, one directory per volume (`/volumes/usb`). A volume's own tree appears beneath its name. |
|
||||||
|
|
||||||
|
Read-only and writable halves of `/system`: the program subtrees (`kernel`,
|
||||||
|
`drivers`, `services`) and the future `devices` are immutable at runtime —
|
||||||
|
initrd-backed or synthetic — while `configuration` and `logs` are mutable
|
||||||
|
machine state served by the boot-volume FAT backend. The kernel's
|
||||||
|
reserved-prefix rule (no mount may shadow `/system`, `/test`, or `/protocol`)
|
||||||
|
needs a carve-out for exactly these two writable subtrees; that lands with the
|
||||||
|
path migration below.
|
||||||
|
|
||||||
|
Deliberately not defined yet: a temporary-files location and per-application
|
||||||
|
mutable storage. Both belong to the `/applications` design and will be
|
||||||
|
specified there, not guessed at here.
|
||||||
|
|
||||||
|
## Node kinds
|
||||||
|
|
||||||
|
What a path resolves to. These fill `FileStatus.kind` and
|
||||||
|
`DirectoryEntry.kind` in the [vfs protocol](vfs-protocol.md)
|
||||||
|
(`library/protocol/vfs/vfs-protocol.zig`); enum values are append-only.
|
||||||
|
|
||||||
|
| Kind | Meaning |
|
||||||
|
|--------------------|-------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||||
|
| `regular` | An ordinary file: an uninterpreted byte stream, positional reads and writes, grows on demand. |
|
||||||
|
| `directory` | A container mapping names to nodes; modified only through directory operations. |
|
||||||
|
| `character_device` | A node whose read/write have **stream semantics**: unseekable, reads block until bytes exist, size is meaningless. The console and every tty-shaped node ([character-devices-and-tty.md](../character-devices-and-tty.md)); what a POSIX layer's `isatty` detects. |
|
||||||
|
| `block_device` | A node addressed in fixed-size sectors — a raw volume. Reserved: recognized, nothing serves one yet. |
|
||||||
|
| `symbolic_link` | Reserved: a recognized value, not implemented by any backend. |
|
||||||
|
| `fifo` | Reserved for the future pipe object (wanted by the POSIX compatibility layer); not implemented. |
|
||||||
|
| `protocol` | A node naming a contract: `open` yields an IPC connection (an endpoint capability) instead of a file id — the kind of every leaf under `/protocol`. *(Being added; see protocol-namespace.md.)* |
|
||||||
|
|
||||||
|
Note the layering: `protocol` says what *opening the name* does (you get a
|
||||||
|
conversation); `character_device`/`block_device` say what *read and write
|
||||||
|
mean* on a node a provider serves you. The two compose — `/protocol/console`
|
||||||
|
is a protocol node in the registry, and the node opened over that connection
|
||||||
|
reports `character_device`, which is what gives it stream semantics. Only
|
||||||
|
`socket` is retired (its value stays reserved for wire stability): a named
|
||||||
|
rendezvous point is exactly what a protocol node is.
|
||||||
|
|
||||||
|
## What is deliberately absent
|
||||||
|
|
||||||
|
There is no `/bin`, `/boot`, `/dev`, `/etc`, `/home`, `/lib`, `/mnt`, `/sbin`,
|
||||||
|
`/srv`, `/tmp`, `/usr`, or `/var`. These encode unix history — the
|
||||||
|
binary/library split of small disks, configuration-as-scattered-text, devices
|
||||||
|
as magic files — that danos does not carry. A POSIX compatibility layer (the
|
||||||
|
Python track's mini-libc) may *present* whichever of these its programs
|
||||||
|
expect, mapped onto the real tree; the tree itself stays danos-native.
|
||||||
|
|
||||||
|
## Migration
|
||||||
|
|
||||||
|
The tree above is the specification; some code still writes the unix paths it
|
||||||
|
replaced. The flag-day converting them:
|
||||||
|
|
||||||
|
| Today (in code) | Becomes | Where |
|
||||||
|
|------------------------------------------|-------------------------------------------|-----------------------------------------------------------------|
|
||||||
|
| `/etc/init.csv` | `/system/configuration/init.csv` | `system/services/init/init.zig` |
|
||||||
|
| `/etc/devices.csv` | `/system/configuration/devices.csv` | `system/services/device-manager/device-manager.zig` |
|
||||||
|
| `/var/log/...` | `/system/logs/...` | `system/services/logger/logger.zig`, the FAT server's `/var` mount |
|
||||||
|
| `/mnt/usb` | `/volumes/usb` | `system/services/fat/fat.zig`, the fat/vfs tests |
|
||||||
|
| `ServiceId` lookup | resolve + open under `/protocol` | every service and client; [protocol-namespace.md](../os-development/protocol-namespace.md) |
|
||||||
|
|
||||||
|
The boot-image builder and the on-volume directory layout move in the same
|
||||||
|
change, so a freshly written image and the paths the services expect never
|
||||||
|
disagree.
|
||||||
@@ -149,8 +149,8 @@ Bitwise OR in `Request.flags`, meaningful for `open` only:
|
|||||||
|
|
||||||
## NodeKind
|
## NodeKind
|
||||||
|
|
||||||
Aligned to the FSH file-type table
|
Aligned to the node-kind table in the file-system hierarchy
|
||||||
(docs/danos-file-system-hierarchy-FSH.md):
|
(docs/file-system-development/file-system-hierarchy.md):
|
||||||
|
|
||||||
| value | kind |
|
| value | kind |
|
||||||
|------:|------|
|
|------:|------|
|
||||||
@@ -163,7 +163,12 @@ Aligned to the FSH file-type table
|
|||||||
| 6 | socket |
|
| 6 | socket |
|
||||||
|
|
||||||
Clients should map unknown values to *regular* rather than reject — the
|
Clients should map unknown values to *regular* rather than reject — the
|
||||||
table can grow.
|
table can grow. Kind 6 (`socket`) keeps its wire value but is retired from
|
||||||
|
the design — a named rendezvous point is exactly what a `protocol` node is,
|
||||||
|
planned as value 7 with the protocol namespace
|
||||||
|
(docs/os-development/protocol-namespace.md). `character_device` (stream
|
||||||
|
semantics — the tty/console shape) and `block_device` (raw sector-addressed
|
||||||
|
volumes, reserved) remain part of the design.
|
||||||
|
|
||||||
## Lifetimes and trust
|
## Lifetimes and trust
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,120 @@
|
|||||||
|
# Communication: the four layers
|
||||||
|
|
||||||
|
*Design, agreed 2026-07-31. The model document — the vocabulary and layering
|
||||||
|
every other communication document speaks.*
|
||||||
|
|
||||||
|
danos separates **what is said** from **how the bytes move**, so that the
|
||||||
|
mechanism is replaceable. The shape is a network stack's, cut into four
|
||||||
|
layers; a program only ever touches the top two.
|
||||||
|
|
||||||
|
```
|
||||||
|
L3 namespace /protocol/... names establishment points protocol-namespace.md
|
||||||
|
L2 protocol the language: packet schemas, verbs, targets the envelope, library/protocol/*
|
||||||
|
L1 channel two ends exchanging packets and signals the client library's Channel
|
||||||
|
L0 transport a buffer + a doorbell: moves the bytes ipc.md (kernel-ipc), later shm-ring, …
|
||||||
|
```
|
||||||
|
|
||||||
|
## Vocabulary
|
||||||
|
|
||||||
|
| Term | Meaning |
|
||||||
|
|---|---|
|
||||||
|
| **protocol** | The language: which packets exist, what their fields mean, which verbs a provider answers. Defined transport-independently in a `library/protocol/*` module. |
|
||||||
|
| **channel** | An open conversation between two processes, speaking one protocol. Established by opening a `/protocol/...` name; both ends can send and receive. |
|
||||||
|
| **packet** | The unit a protocol transmits: a bounded, atomic header+payload. Never fragmented — if it doesn't fit, it isn't a packet; bulk data rides shared memory with a packet as the doorbell. |
|
||||||
|
| **signal** | A payload-less poke below the packet layer: "something happened, come look." Coalescing — the count may collapse, the fact may not. |
|
||||||
|
| **transport** | What moves the bytes of one channel: a buffer plus a doorbell. Chosen (and upgradable) at establishment, invisible above L1. |
|
||||||
|
| **endpoint** | A termination point where a transport delivers. The kernel-ipc transport's endpoint is its kernel mailbox object. |
|
||||||
|
|
||||||
|
## Addressing: parties by channel, objects by target
|
||||||
|
|
||||||
|
There are no network-style addresses in a packet. The two questions addresses
|
||||||
|
answer are answered at different layers:
|
||||||
|
|
||||||
|
- **Who am I talking to?** The **channel**, decided once at establishment.
|
||||||
|
Opening `/protocol/input` yields a channel; every packet sent on it goes to
|
||||||
|
the peer. Nothing to route per-packet — like TCP, where no HTTP request
|
||||||
|
carries the server's IP.
|
||||||
|
- **Who sent this?** Attached to every received packet **by the channel
|
||||||
|
layer**, from identity the transport can verify — under kernel-ipc, the
|
||||||
|
kernel-stamped badge. The sender never writes a source field, which is what
|
||||||
|
makes source unforgeable (the property a network's spoofable source header
|
||||||
|
lacks).
|
||||||
|
- **Which of your things?** The packet's **`target`** field: *object*
|
||||||
|
addressing within the already-chosen peer — the vfs protocol's node id, the
|
||||||
|
display protocol's layer id, a block volume. `target = 0` addresses the
|
||||||
|
provider itself; a protocol without objects never uses it.
|
||||||
|
|
||||||
|
`target` is how instance multiplicity stays out of the namespace. Ten USB
|
||||||
|
sticks and the namespace still holds exactly one name, `/protocol/block`: a
|
||||||
|
channel to the provider, `enumerate` lists the current volumes as targets, a
|
||||||
|
`targets_changed` signal announces hotplug, and a read names its volume in
|
||||||
|
`target`. The unix `/dev/sda`,`/dev/sdb` problem is dissolved, not renamed.
|
||||||
|
|
||||||
|
If a future transport genuinely routes between machines, *it* carries real
|
||||||
|
source/destination addressing internally at L0 — the way IP runs under TCP —
|
||||||
|
and none of it surfaces into the packet header. Protocols stay ignorant of
|
||||||
|
distance.
|
||||||
|
|
||||||
|
## The transport (L0): a buffer and a doorbell
|
||||||
|
|
||||||
|
Strip any transport to its skeleton and the same two parts remain:
|
||||||
|
|
||||||
|
| Transport | Buffer | Doorbell | Status |
|
||||||
|
|---|---|---|---|
|
||||||
|
| **kernel-ipc** | kernel-owned mailbox (the `Endpoint`) | the scheduler (rendezvous wake) | the first transport — [ipc.md](../device-driver-development/ipc.md) |
|
||||||
|
| **shm-ring** | user-owned shared-memory ring | a signal | exists ad hoc (display bulk); to be formalized — the unlock for the 256-byte ceiling |
|
||||||
|
| network | NIC queue | an interrupt | someday, when danos networks |
|
||||||
|
|
||||||
|
Transports differ in their **properties**, which the channel layer exposes and
|
||||||
|
the protocol layer may depend on:
|
||||||
|
|
||||||
|
- **packet ceiling** — kernel-ipc: 256 bytes request/reply, 64 pushed. An
|
||||||
|
shm-ring's ceiling is its slot size. Kernel-ipc's 256 is the *floor* every
|
||||||
|
protocol may assume everywhere.
|
||||||
|
- **synchrony** — kernel-ipc's call is a rendezvous: natural backpressure, no
|
||||||
|
queue to size. An asynchronous transport buffers, so a channel over one
|
||||||
|
needs explicit flow control. Backpressure is a *transport property*, not a
|
||||||
|
channel guarantee — protocols that rely on it say so.
|
||||||
|
- **droppability** — pushed event packets may drop when a ring fills;
|
||||||
|
request/reply may not.
|
||||||
|
- **capability carriage** — **only kernel-ipc can move a capability.**
|
||||||
|
Handles are kernel objects; a user-space ring cannot transfer one. So
|
||||||
|
kernel-ipc is always the *establishment and control* transport — channels
|
||||||
|
are born on it, capabilities ride it — even when a channel's data is
|
||||||
|
negotiated onto something fatter.
|
||||||
|
|
||||||
|
That negotiation is the upgrade path: a channel starts on kernel-ipc; the
|
||||||
|
protocol's handshake may then delegate a shared-memory region (as a
|
||||||
|
capability, over kernel-ipc) and move its bulk traffic there. The display
|
||||||
|
path already does exactly this by hand; formalizing it in the channel layer
|
||||||
|
makes it every protocol's option.
|
||||||
|
|
||||||
|
## The channel (L1)
|
||||||
|
|
||||||
|
A channel has two ends, and **the ends are peers**: each may send packets,
|
||||||
|
each may receive, each may signal. Request/reply is a *pattern* over the
|
||||||
|
channel — a send with a correlated receive, which the kernel-ipc transport
|
||||||
|
happens to accelerate as a single rendezvous — not the definition of it. The
|
||||||
|
event stream (subscribe, then pushes) and the change signal (poke, then
|
||||||
|
re-read) are the other two patterns; all three are catalogued in
|
||||||
|
[protocol-namespace.md](protocol-namespace.md)'s wiring section.
|
||||||
|
|
||||||
|
The channel layer's obligations: deliver packets whole, attach the verified
|
||||||
|
source to every receive, expose the transport's properties, and hide the
|
||||||
|
transport's mechanics. The client library's `Channel` type is this layer made
|
||||||
|
concrete — a program holds channels that speak protocols and never touches a
|
||||||
|
raw handle.
|
||||||
|
|
||||||
|
## The protocol (L2) and the namespace (L3)
|
||||||
|
|
||||||
|
A protocol defines its packets through the envelope — every packet begins
|
||||||
|
`{operation, target}`, reserved verbs (`describe`, `enumerate`, `subscribe`,
|
||||||
|
`unsubscribe`) mean the same thing in every protocol, and `Define` checks
|
||||||
|
every packet against the transport floor at compile time. The full treatment,
|
||||||
|
including how names are granted, resolved, and restricted per process, is
|
||||||
|
[protocol-namespace.md](protocol-namespace.md).
|
||||||
|
|
||||||
|
Establishment points are named by contract — `/protocol/display`, never
|
||||||
|
`/protocol/ipc-1` — because the name must outlive the mechanism: a
|
||||||
|
transport named in the namespace could never be swapped, which would defeat
|
||||||
|
this document's premise.
|
||||||
@@ -24,8 +24,9 @@ EFI/BOOT/BOOTX64.efi <- the "removable media" default for x86-64
|
|||||||
```
|
```
|
||||||
|
|
||||||
The boot volume is **FHS-shaped** (see the repository-layout note in
|
The boot volume is **FHS-shaped** (see the repository-layout note in
|
||||||
[README.md](../README.md)): `build.zig` installs `boot/efi.zig` (built for the `uefi`
|
[README.md](../README.md)): the root `build.zig` compiles `boot/efi.zig` (built
|
||||||
target) at `EFI/BOOT/BOOTX64.efi` — the one path UEFI firmware fixes — and lays
|
for the `uefi` target) and `build/images.zig` places it at
|
||||||
|
`EFI/BOOT/BOOTX64.efi` — the one path UEFI firmware fixes — and lays
|
||||||
the rest out by FHS path: the kernel at `system/kernel`, init at
|
the rest out by FHS path: the kernel at `system/kernel`, init at
|
||||||
`system/services/init`, the pre-packed boot capsule at `boot/system.img`
|
`system/services/init`, the pre-packed boot capsule at `boot/system.img`
|
||||||
([system-image.md](system-image.md)).
|
([system-image.md](system-image.md)).
|
||||||
|
|||||||
@@ -0,0 +1,472 @@
|
|||||||
|
# The protocol namespace
|
||||||
|
|
||||||
|
*Design, agreed 2026-07-31. Supersedes the `ServiceId` registry. Not yet implemented —
|
||||||
|
the migration plan at the end is the work list.*
|
||||||
|
|
||||||
|
How a program finds, connects to, and is restricted from the things it talks to.
|
||||||
|
Three ideas, kept deliberately separate:
|
||||||
|
|
||||||
|
1. **Naming** — a path under `/protocol` names a *contract*, not a service.
|
||||||
|
2. **Access** — resolving that path yields an endpoint *capability*; what a process
|
||||||
|
cannot resolve, it cannot reach.
|
||||||
|
3. **Transport** — unchanged: packets over channels, moved by whichever
|
||||||
|
transport the channel rides (kernel-ipc first).
|
||||||
|
This document is layers **L3** (the namespace) and **L2** (the protocol
|
||||||
|
and its envelope) of the communication stack;
|
||||||
|
[communication.md](communication.md) owns the model and the vocabulary
|
||||||
|
(*protocol* the language, *channel* the conversation, *packet* the
|
||||||
|
transmitted unit, *signal* the payload-less poke, *transport* the
|
||||||
|
replaceable mechanism), and
|
||||||
|
[ipc.md](../device-driver-development/ipc.md) is the first transport.
|
||||||
|
|
||||||
|
## Why ServiceId has to go
|
||||||
|
|
||||||
|
Today a service calls `ipc_register(service_id, endpoint)` and a client calls
|
||||||
|
`ipc_lookup(service_id)`, where `ServiceId` is a compile-time enum in `abi.zig`
|
||||||
|
backed by a flat 16-slot table in the kernel. Three defects, in rising order:
|
||||||
|
|
||||||
|
- **Static.** The id space is baked into the ABI at compile time. A third-party
|
||||||
|
program can never introduce a service; the one place danos is *less* dynamic
|
||||||
|
than its own design.
|
||||||
|
- **Ungated.** `ipc_register` is callable by any process and *replaces* an
|
||||||
|
existing registration. Any process can hijack `.fat` or `.display` and
|
||||||
|
impersonate it. `ipc_lookup` is equally ambient.
|
||||||
|
- **Unrestrictable.** Because lookup is a syscall available to everyone, there is
|
||||||
|
no point at which "this process may not talk to the display" can be enforced.
|
||||||
|
Any future file-access restriction would be bypassable by speaking to the FAT
|
||||||
|
server directly.
|
||||||
|
|
||||||
|
## Naming: contracts, not services
|
||||||
|
|
||||||
|
`/protocol/<name>` names a protocol — the contract a conversation follows — and
|
||||||
|
resolving it connects you to whatever process currently provides that contract.
|
||||||
|
The client never cared *which* binary answers; it cares that its messages are
|
||||||
|
understood. Naming the contract makes that explicit, and buys:
|
||||||
|
|
||||||
|
- **Swappable providers.** Replace the display server; `/protocol/display`
|
||||||
|
routes to the new one; clients notice nothing.
|
||||||
|
- **Test fakes.** Spawn a program whose namespace wires `/protocol/display` to a
|
||||||
|
mock. The name promises the protocol; the mock speaks it.
|
||||||
|
- **One vocabulary.** The names mirror `library/protocol/`: a program imports
|
||||||
|
the `display-protocol` module, then opens `/protocol/display`. What you
|
||||||
|
compiled against and what you ask the namespace for are the same word.
|
||||||
|
|
||||||
|
A leaf names one contract — kebab-case, full words, matching the
|
||||||
|
`library/protocol/` module that defines its wire format — and related
|
||||||
|
contracts group into directories: `/protocol/networking/ip`,
|
||||||
|
`/protocol/networking/bluetooth`. Directories organize *contracts only*;
|
||||||
|
they never encode addressing (see below), so a directory appears because a
|
||||||
|
domain has several contracts, never because hardware multiplied. The module
|
||||||
|
tree mirrors the namespace (`library/protocol/networking/ip` ↔
|
||||||
|
`/protocol/networking/ip`), and registrar grants scope naturally to subtrees
|
||||||
|
— an application installed at `/applications/foo` can be granted
|
||||||
|
`/protocol/applications/foo/...` and nothing above it. `/protocol` is
|
||||||
|
top level, beside `/system` and `/applications`, because the boundary it names
|
||||||
|
is spoken on both sides: applications talk to protocols as much as the OS does
|
||||||
|
(see [file-system-hierarchy.md](../file-system-development/file-system-hierarchy.md)).
|
||||||
|
|
||||||
|
**Addressing lives inside the protocol, never in the path.** Which volume, which
|
||||||
|
layer, which input device — that is a destination field in the messages, the way
|
||||||
|
TCP carries a destination address, and the way danos protocols already work (the
|
||||||
|
display protocol multiplexes layer ids; the vfs protocol addresses node ids).
|
||||||
|
The namespace answers exactly one question — *may this process speak this
|
||||||
|
protocol at all* — so `/protocol/block` is one name no matter how many disks are
|
||||||
|
attached. The source address is never in the message either: it is the IPC
|
||||||
|
badge, stamped by the kernel per message, unforgeable — a property TCP's source
|
||||||
|
address does not have.
|
||||||
|
|
||||||
|
`/system/devices` (the device inventory) stays purely informational: facts for
|
||||||
|
diagnosis, never a routing mechanism. Unix conflated the two in `/dev`; danos
|
||||||
|
does not. You *read about* hardware in `/system/devices`; you *talk to* it
|
||||||
|
through `/protocol`.
|
||||||
|
|
||||||
|
## Resolution: a protocol node in the VFS
|
||||||
|
|
||||||
|
The kernel VFS router already does the hard part: `fs_resolve` matches a mount
|
||||||
|
prefix and installs the backend's endpoint capability in the caller's handle
|
||||||
|
table. The registry is just a backend mounted at `/protocol` — ring 3, like FAT.
|
||||||
|
Connecting is a normal vfs-protocol `open` with one twist in the reply:
|
||||||
|
|
||||||
|
```
|
||||||
|
client kernel router registry backend
|
||||||
|
│ fs_resolve("/protocol/display") │
|
||||||
|
│──────────────────────────▶│ prefix match: /protocol │
|
||||||
|
│◀── registry endpoint ─────│ (capability installed) │
|
||||||
|
│ vfs open("display") ──────────────────────────────────────▶│
|
||||||
|
│◀───────────────── Reply + capability = provider endpoint ──│
|
||||||
|
│ ipc_call(provider, display-protocol messages...) │
|
||||||
|
```
|
||||||
|
|
||||||
|
Both capability moves use machinery the kernel already has: request-direction
|
||||||
|
and reply-direction `send_cap` on `call`/`replyWait`. The vfs protocol needs two
|
||||||
|
additions, both append-only:
|
||||||
|
|
||||||
|
- `NodeKind.protocol` — a node that names a contract; its `open` establishes
|
||||||
|
a **channel** (delivered as an endpoint capability) instead of returning a
|
||||||
|
file id. The node is the protocol, the channel is the conversation, and the
|
||||||
|
addressing inside the packets decides where within the provider each one
|
||||||
|
lands. `readdir` over `/protocol` lists protocol nodes like any others, so
|
||||||
|
the tree stays browsable for diagnosis.
|
||||||
|
- The convention that an `open` reply may carry a capability. File backends
|
||||||
|
(FAT) never use it; synthetic backends (the registry, later the device
|
||||||
|
inventory) do.
|
||||||
|
|
||||||
|
The path lookup happens once, at connect time. The hot path — `ipc_call` on the
|
||||||
|
cached endpoint — is untouched. A provider crash turns the cached endpoint dead
|
||||||
|
(`-EPEER`), and the client's recovery is to re-resolve: the restart story falls
|
||||||
|
out of the naming layer for free.
|
||||||
|
|
||||||
|
## Registration: the registrar, held by init
|
||||||
|
|
||||||
|
The registry backend is **init**. It is already PID 1, already spawns every
|
||||||
|
service from its manifest, and already holds the supervision link to each — it
|
||||||
|
is the process that *knows* which binary is which. (If init grows
|
||||||
|
uncomfortable, the same design lifts into a dedicated registry service that
|
||||||
|
init spawns first and delegates to; nothing below changes.)
|
||||||
|
|
||||||
|
- **Binding.** A service creates its endpoint and sends the registry a `bind`
|
||||||
|
request with the protocol name as payload and the endpoint attached as the
|
||||||
|
call's capability.
|
||||||
|
- **Authorization.** Init's manifest gains a column: the protocols each spawned
|
||||||
|
binary may bind. A `bind` from any process not granted that name is refused
|
||||||
|
(`-EPERM`) — the badge identifies the caller, the supervision records map
|
||||||
|
badge to binary. This is the registrar authority; it never leaves init.
|
||||||
|
- **Collision is an error.** A name already bound refuses a second bind — never
|
||||||
|
last-writer-wins. When a provider dies, init (its supervisor) unbinds its
|
||||||
|
names; the restarted instance binds again.
|
||||||
|
- **Provenance.** The registry records name → task id → binary path, so a
|
||||||
|
diagnostic listing answers "who serves this?" at a glance:
|
||||||
|
|
||||||
|
```
|
||||||
|
/protocol/display pid 12 /system/services/display
|
||||||
|
/protocol/input pid 7 /system/services/input
|
||||||
|
```
|
||||||
|
|
||||||
|
`ipc_register` and `ipc_lookup` retire; the `ServiceId` enum leaves `abi.zig`.
|
||||||
|
The kernel keeps one residual rule: `/protocol` becomes a reserved prefix like
|
||||||
|
`/system` — `fs_mount` refuses to shadow it, and init's boot-time mount is the
|
||||||
|
only one it will ever hold. (Full gating of `fs_mount` is a separate item on
|
||||||
|
the security track; the reserved prefix closes the hole for this namespace
|
||||||
|
without waiting for it.)
|
||||||
|
|
||||||
|
## Restriction: per-process namespaces, not ACLs
|
||||||
|
|
||||||
|
danos has no users and no principals, deliberately. Restriction is therefore
|
||||||
|
**delegation**: what a process may open is decided by whoever spawned it, and
|
||||||
|
enforcement is absence — a protocol you cannot resolve does not exist for you.
|
||||||
|
"Permission denied" and "not found" are the same answer, which is the same
|
||||||
|
discipline the device layer already follows: the claim is the capability; here,
|
||||||
|
the resolvable name is the capability.
|
||||||
|
|
||||||
|
Two stages, deliberately ordered so the useful half lands first:
|
||||||
|
|
||||||
|
**Stage one — the registry filters by badge.** Init is both the spawner and the
|
||||||
|
registry, so its manifest already knows which binary may *open* which protocols
|
||||||
|
(a second manifest column, beside the bind grants). An `open` from a process
|
||||||
|
whose binary is not granted that protocol is refused. No new kernel mechanism
|
||||||
|
at all; the display driver's view can be narrowed to nothing, a future
|
||||||
|
downloaded application's to `display` and `input`, today.
|
||||||
|
|
||||||
|
**Stage two — spawn passes the namespace.** `spawn` gains an initial
|
||||||
|
capability: the child's connection to *its* registry view, chosen by the
|
||||||
|
spawner. A newly spawned process starts with an empty handle table and this one
|
||||||
|
handle — its world is whatever its parent wired in. This removes the last
|
||||||
|
ambient reach (`fs_resolve` finding `/protocol` globally), lets any supervisor
|
||||||
|
— not just init — narrow or fake a child's view (an application launcher
|
||||||
|
granting an app only what its manifest declares; a test harness substituting
|
||||||
|
every provider), and composes down the supervision tree. Stage one's manifest
|
||||||
|
column becomes the *content* of the view init builds, so nothing is thrown
|
||||||
|
away.
|
||||||
|
|
||||||
|
### A worked example: the microphone prompt
|
||||||
|
|
||||||
|
The scenario stage two exists for: an application opens
|
||||||
|
`/protocol/audio-input`, and the user should be asked. The supervisor is an
|
||||||
|
ordinary user process — an application launcher — and the flow needs no new
|
||||||
|
security concepts:
|
||||||
|
|
||||||
|
1. The launcher spawned the app with a namespace channel that terminates at
|
||||||
|
**the launcher itself**. The app's whole world is a conversation with its
|
||||||
|
supervisor.
|
||||||
|
2. The app's `open("audio-input")` packet lands in the launcher,
|
||||||
|
badge-stamped. The launcher spawned the app, so badge → binary path
|
||||||
|
(`/applications/foo`) is its own supervision record — "remember my choice"
|
||||||
|
needs no identity system.
|
||||||
|
3. Grant unknown → the launcher parks the request and shows a prompt (it is a
|
||||||
|
user process with display access; init never does UI). Blocking an open on
|
||||||
|
a human is architecturally fine: opens are connect-time, never hot-path.
|
||||||
|
4. **Yes** → the launcher opens `/protocol/audio-input` in *its own*
|
||||||
|
namespace and attaches the resulting channel to the parked reply. The app
|
||||||
|
cannot tell a prompt happened — a consented open is indistinguishable from
|
||||||
|
a direct one, merely slower.
|
||||||
|
5. **No** → refuse the open, indistinguishable from "no such protocol" — or
|
||||||
|
hand the app a **fake**: a silence-generating provider. The test-fake
|
||||||
|
mechanism doubles as a privacy feature.
|
||||||
|
|
||||||
|
The capability discipline holds throughout: the launcher can only grant what
|
||||||
|
it holds — if init never gave the launcher `audio-input`, no prompt can
|
||||||
|
conjure it. Consent is delegation flowing down the supervision tree, never a
|
||||||
|
global ACL edit. And the provider still sees the app's badge on every packet,
|
||||||
|
so a coarser second check at the audio service remains possible.
|
||||||
|
|
||||||
|
Two mechanical requirements this scenario pins on stage two:
|
||||||
|
|
||||||
|
- **Parked replies.** A prompt takes seconds, and the service loop holds one
|
||||||
|
outstanding reply today — the launcher must park request A, keep serving B
|
||||||
|
and C, and reply to A later (by badge). The kernel already tracks owed
|
||||||
|
replies (that is how death delivers `-EPEER`); multiple parked replies is
|
||||||
|
the extension, in the harness and, if needed, the kernel.
|
||||||
|
- **Granted channels are dedicated, hence revocable.** Once the app holds a
|
||||||
|
channel capability, nobody reaches into its handle table — so a
|
||||||
|
prompt-granted channel must be one that can be *killed*: a dedicated
|
||||||
|
endpoint pair (or per-client session at the provider) whose death turns
|
||||||
|
the app's capability into `-EPEER`. Revoking microphone access is then
|
||||||
|
killing that channel, using machinery that already exists.
|
||||||
|
|
||||||
|
One adjacent problem, named and deferred: **trusted UI**. The prompt is only
|
||||||
|
meaningful if the app cannot draw a convincing fake or overlay the real one —
|
||||||
|
a display-layer question (a reserved surface for the supervisor chain), owned
|
||||||
|
by the display track, not this one.
|
||||||
|
|
||||||
|
Fine-grained restriction *within* a protocol (this process may use volume A but
|
||||||
|
not volume B) is not the namespace's job. The capability-shaped answer, when it
|
||||||
|
is needed: the supervisor pre-opens a connection scoped to one target and passes
|
||||||
|
that connection to the child, which never opens `/protocol/block` at all.
|
||||||
|
Delegation again, not ACLs.
|
||||||
|
|
||||||
|
## The envelope: one addressing scheme for every protocol
|
||||||
|
|
||||||
|
Every protocol module today hand-rolls its `Request`/`Reply` with an
|
||||||
|
`operation` first field. That convention becomes a library, so addressing is
|
||||||
|
uniform and the rules are enforced by construction rather than by review. New
|
||||||
|
module: **`library/protocol/envelope`** (the one protocol-layer module that is
|
||||||
|
not itself a protocol).
|
||||||
|
|
||||||
|
```zig
|
||||||
|
/// Every packet a danos protocol transmits begins with this header.
|
||||||
|
pub const Header = extern struct {
|
||||||
|
operation: u32, // the verb; values 0..15 are reserved universal verbs
|
||||||
|
_padding: u32 = 0,
|
||||||
|
/// Object addressing, never party addressing: which of the peer's
|
||||||
|
/// objects this packet operates on — a volume, layer, node, device.
|
||||||
|
/// 0 addresses the provider itself. Parties are addressed by the
|
||||||
|
/// channel; the protocol defines target's meaning; the field's place
|
||||||
|
/// and width are universal.
|
||||||
|
target: u64 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Reserved verbs, answered by every provider.
|
||||||
|
pub const operation_describe: u32 = 0; // -> protocol name, version, target kinds
|
||||||
|
pub const operation_enumerate: u32 = 1; // -> the current targets, one per reply page
|
||||||
|
pub const operation_subscribe: u32 = 2; // capability = the subscriber's endpoint
|
||||||
|
pub const operation_unsubscribe: u32 = 3;
|
||||||
|
pub const first_protocol_operation: u32 = 16;
|
||||||
|
|
||||||
|
/// Every reply begins with this.
|
||||||
|
pub const Status = extern struct {
|
||||||
|
status: i32, // 0 or a negative errno
|
||||||
|
_padding: u32 = 0,
|
||||||
|
len: u32 = 0, // payload bytes following the header
|
||||||
|
_padding2: u32 = 0,
|
||||||
|
};
|
||||||
|
```
|
||||||
|
|
||||||
|
A protocol is then *defined through* the envelope, not beside it:
|
||||||
|
|
||||||
|
```zig
|
||||||
|
pub const Protocol = envelope.Define(.{
|
||||||
|
.name = "display",
|
||||||
|
.version = 1,
|
||||||
|
.operations = &.{
|
||||||
|
.{ .name = "configure_layer", .request = ConfigureLayer, .reply = void },
|
||||||
|
.{ .name = "blit", .request = Blit, .reply = void },
|
||||||
|
...
|
||||||
|
},
|
||||||
|
});
|
||||||
|
```
|
||||||
|
|
||||||
|
`Define` is comptime and is where the enforcement lives:
|
||||||
|
|
||||||
|
- verbs are numbered automatically from `first_protocol_operation`, so no
|
||||||
|
protocol can collide with the reserved range;
|
||||||
|
- every packet is size-checked at compile time against the kernel-ipc floor
|
||||||
|
— `packet_maximum` (256) for request/reply, `post_maximum` (64) for event
|
||||||
|
packets. Ceilings are transport properties
|
||||||
|
([communication.md](communication.md)); the floor is what every protocol
|
||||||
|
may assume on any transport. The errors that today surface as runtime
|
||||||
|
truncation become compile errors, and packets-never-fragment is enforced
|
||||||
|
at the source;
|
||||||
|
- the generated type carries encode/decode helpers and a provider-side dispatch
|
||||||
|
table, so a provider answers `describe` automatically and unknown operations
|
||||||
|
with `-ENOSYS` uniformly;
|
||||||
|
- the service harness (`library/kernel/service.zig`) accepts the generated
|
||||||
|
dispatch type, which is what makes the envelope *enforced*: a protocol that
|
||||||
|
bypasses `Define` does not plug into the harness.
|
||||||
|
|
||||||
|
Universal conventions that ride on the reserved verbs:
|
||||||
|
|
||||||
|
- **`describe`** is the version handshake. Version lives in the handshake, not
|
||||||
|
in every message — the 256-byte budget is too small to spend per call.
|
||||||
|
- **`enumerate`** is how multi-target protocols expose their targets, and the
|
||||||
|
standard `targets_changed` notification (a notify bit) tells subscribers to
|
||||||
|
re-enumerate — arrival and removal of volumes, layers, devices all take the
|
||||||
|
same shape. Hotplug fits the notification ring far better than a filesystem
|
||||||
|
tree ever did.
|
||||||
|
- **Source is the badge.** No protocol defines a "sender" field; the kernel's
|
||||||
|
per-message badge is the only source identity, and providers key per-client
|
||||||
|
state on it.
|
||||||
|
|
||||||
|
### Paths resolve once; integers do the work
|
||||||
|
|
||||||
|
A rule the envelope makes official: **a path appears in a conversation at most
|
||||||
|
once — at resolve or open — and everything after it addresses integers.** The
|
||||||
|
namespace resolves `/protocol/display` to an endpoint; a backend's `open`
|
||||||
|
resolves a path payload to a node id; from then on every packet carries the
|
||||||
|
integer in `target`. Integers compare in one instruction and fit the fixed
|
||||||
|
header, and the 256-byte message budget never re-carries path strings on the
|
||||||
|
hot path. This is already the system's shape — vfs node ids, display layer ids
|
||||||
|
— and the envelope pins it as the required shape for every protocol.
|
||||||
|
|
||||||
|
Two integer identities, not to be confused:
|
||||||
|
|
||||||
|
- **An open handle** — what vfs `open` returns today: transient, meaningful
|
||||||
|
only within one client's session with one provider, swept when the client
|
||||||
|
exits. Cheap, and all a protocol usually needs. Handles must be **scoped per
|
||||||
|
client** — validated against the badge, or drawn from a per-client id
|
||||||
|
namespace. (Today the FAT server's node ids are guessable small integers
|
||||||
|
honoured across clients; that hole closes with this rule.)
|
||||||
|
- **A persistent node identity** — a unix inode number, stable across opens
|
||||||
|
and renames. danos deliberately does not promise this, because FAT cannot
|
||||||
|
deliver it: a FAT file's identity is its directory entry, and rename or
|
||||||
|
truncation moves every candidate anchor. If a future filesystem or a cache
|
||||||
|
layer needs stable identity, that is the backend's promise to make, never
|
||||||
|
the protocol's assumption.
|
||||||
|
|
||||||
|
The five existing protocol modules (`vfs`, `display`, `input`, `power`,
|
||||||
|
`block`, plus `scanout`, `usb-transfer`, `device-manager`) rebase onto the
|
||||||
|
envelope during the migration flag-day. `input-protocol`'s subscribe/publish
|
||||||
|
split and `vfs-protocol`'s node addressing both map cleanly (`node` and layer
|
||||||
|
ids become `target`).
|
||||||
|
|
||||||
|
## Wiring: how conversations flow
|
||||||
|
|
||||||
|
The patterns below are channel-layer (L1) shapes; the delivery mechanics are
|
||||||
|
the kernel-ipc transport's, described here because it is the transport every
|
||||||
|
channel starts on. Kernel-ipc provides exactly three delivery shapes, and
|
||||||
|
every one is unicast. An endpoint is a mailbox owned by one process — its
|
||||||
|
creator receives; anyone holding its capability sends into it. That direction
|
||||||
|
never reverses:
|
||||||
|
|
||||||
|
1. **Synchronous call** — request/reply. The kernel parks the caller and
|
||||||
|
`replyWait` delivers the reply straight back, so the provider answers
|
||||||
|
without holding any capability to the client. Badge-stamped, blocking, and
|
||||||
|
the *only* shape that carries capabilities (in the request, and in the
|
||||||
|
reply — which is how a reverse path is bootstrapped).
|
||||||
|
2. **Asynchronous send** — an event packet pushed into the receiver's post
|
||||||
|
ring, at most `post_maximum` (64) bytes, no reply owed, never blocks the
|
||||||
|
sender. Strictly one-way: to be pushed to, you must first hand the pusher
|
||||||
|
your endpoint.
|
||||||
|
3. **Signals** — payload-less notification bits, below the packet layer,
|
||||||
|
coalescing: "something changed, come look."
|
||||||
|
|
||||||
|
A bidirectional link is therefore always **a pair of endpoints**, one per
|
||||||
|
direction, each delivered by cap-passing. Three conversation patterns are
|
||||||
|
built from these, and the envelope names all three:
|
||||||
|
|
||||||
|
- **Request/response** — the synchronous call. The default, and the only
|
||||||
|
place capabilities move.
|
||||||
|
- **Event stream** — `subscribe` (a synchronous call whose attached
|
||||||
|
capability is the subscriber's own endpoint), after which the provider
|
||||||
|
pushes events asynchronously; `unsubscribe` or subscriber exit ends it.
|
||||||
|
Listened-to, not blocked-on.
|
||||||
|
- **Change signal** — a signal plus re-read: `targets_changed` →
|
||||||
|
`enumerate`. For state whose truth lives with the provider.
|
||||||
|
|
||||||
|
**Broadcast is a provider pattern, never a kernel primitive.** The kernel
|
||||||
|
does not know subscriber sets — a service does. The input service is the
|
||||||
|
model: sources *publish* (a unicast call to the service), the service
|
||||||
|
*broadcasts* (a fan-out loop of asynchronous sends over its subscriber list,
|
||||||
|
so one dead subscriber can never stall the rest). One fan-out point per event
|
||||||
|
domain, owned by the service that defines the event.
|
||||||
|
|
||||||
|
The harness owns the machinery: the subscriber table, the dead-subscriber
|
||||||
|
sweep (via process-exit notifications), and the fan-out loop — all written by
|
||||||
|
hand in `input.zig` today, lifted into the service harness so every protocol
|
||||||
|
gets identical semantics. `Define` declares a protocol's events (`.events`),
|
||||||
|
and each event type is checked against `post_maximum` at compile time,
|
||||||
|
generalizing the assert `input-protocol` already carries.
|
||||||
|
|
||||||
|
**Event packets are droppable.** A slow subscriber's ring fills, and the
|
||||||
|
provider must not block on it — so an event stream is a hint or a coalescing
|
||||||
|
signal, never a ledger. Anything that must not be lost is either re-readable
|
||||||
|
state (the change-signal pattern) or bulk data in shared memory with a
|
||||||
|
packet as the doorbell, which is how the display path already works — the
|
||||||
|
packets-never-fragment rule and this one are the same rule seen from two
|
||||||
|
sides.
|
||||||
|
|
||||||
|
**Source direction (open point).** Today event sources are *clients*: an
|
||||||
|
input driver resolves `/protocol/input` and delivers each event as a
|
||||||
|
synchronous `publish` call — one capability, obtained by resolution, covers
|
||||||
|
everything, and the badge tells the service exactly who each event came from.
|
||||||
|
The inversion — the service subscribing to each driver — would require every
|
||||||
|
driver to be individually discoverable and its endpoint ferried to the
|
||||||
|
service, machinery whose payoff (the service choosing its sources) the
|
||||||
|
namespace already provides more cheaply: only a process granted open on
|
||||||
|
`/protocol/input` can publish into it. Sources stay clients for now;
|
||||||
|
revisited at restriction stage two, when a supervisor can wire capabilities
|
||||||
|
at spawn time.
|
||||||
|
|
||||||
|
## What this deliberately does not solve
|
||||||
|
|
||||||
|
The wider security track, for which this namespace is the foundation, not the
|
||||||
|
whole:
|
||||||
|
|
||||||
|
- **File access restriction** — the point of the exercise. The same stage-two
|
||||||
|
namespace mechanism extends from protocol names to file paths: the spawner
|
||||||
|
decides which subtrees resolve. Designed separately once this lands.
|
||||||
|
- `fs_mount` gating beyond the reserved prefixes; `system_spawn` gating;
|
||||||
|
`klog_read` being world-readable; backends checking the badge on per-node
|
||||||
|
operations (the FAT server honours node ids across clients today).
|
||||||
|
- Kernel hardening items already noted in-tree: SMEP/SMAP and SYSRET
|
||||||
|
canonical-RIP, now designed in [smep-smap.md](smep-smap.md).
|
||||||
|
- Pipes/FIFOs for the POSIX layer — a byte-stream object *beside* message IPC,
|
||||||
|
wanted by the Python track, unrelated to naming.
|
||||||
|
- **Trusted UI** — a permission prompt an application cannot fake or overlay
|
||||||
|
(see the microphone example). A display-track concern: the supervisor chain
|
||||||
|
needs a reserved surface.
|
||||||
|
|
||||||
|
## Migration plan
|
||||||
|
|
||||||
|
Flag-day per phase, in the style of the DMA-capability conversion — no
|
||||||
|
dual-stack periods, the QEMU suite green at each phase boundary.
|
||||||
|
|
||||||
|
**P1 — mechanics, no behavior change.** The `envelope` module with its comptime
|
||||||
|
`Define`, unit tests; `NodeKind.protocol` and the open-reply-capability
|
||||||
|
convention in `vfs-protocol`; existing protocols untouched.
|
||||||
|
|
||||||
|
**P2 — the registry.** Init serves `/protocol` (bind with manifest
|
||||||
|
authorization, collision refusal, unbind on provider death, provenance);
|
||||||
|
kernel reserves the `/protocol` prefix; every service converts from
|
||||||
|
`ipc_register` to `bind`, every client from `ipc_lookup` to resolve-and-open;
|
||||||
|
`ServiceId`, `ipc_register`, `ipc_lookup` deleted. Tests: unauthorized bind
|
||||||
|
refused, collision refused, provider restart re-binds and a client re-resolves.
|
||||||
|
|
||||||
|
**P3 — restriction, stage one.** The open-grant column in init's manifest;
|
||||||
|
registry refuses ungranted opens. Test: a fixture process denied a protocol its
|
||||||
|
neighbour is granted.
|
||||||
|
|
||||||
|
**P4 — protocol rebase.** Existing protocol modules re-expressed through
|
||||||
|
`Define`; providers move onto the generated dispatch; `describe`/`enumerate`
|
||||||
|
answered everywhere; the conformance test fixture exercises the reserved verbs
|
||||||
|
against every registered provider.
|
||||||
|
|
||||||
|
**P5 — restriction, stage two.** Spawn's initial capability; namespace views
|
||||||
|
built by the spawner; ambient resolution of `/protocol` retired. Includes the
|
||||||
|
two requirements the microphone example pins: **parked replies** (a
|
||||||
|
supervisor parks an open, keeps serving, replies later by badge) and
|
||||||
|
**dedicated, killable granted channels** (revocation = channel death →
|
||||||
|
`-EPEER`). Scoped separately — it touches `spawn`, the loader contract, and
|
||||||
|
every supervisor — and lands together with the file-path half of namespacing.
|
||||||
|
|
||||||
|
The unix-path migration ([file-system-hierarchy.md](../file-system-development/file-system-hierarchy.md#migration))
|
||||||
|
is independent of P1–P5 and can land before or after.
|
||||||
@@ -0,0 +1,143 @@
|
|||||||
|
# SMEP and SMAP — supervisor-mode hardening
|
||||||
|
|
||||||
|
*Design, 2026-07-31. Not yet implemented. Companion to
|
||||||
|
[protocol-namespace.md](protocol-namespace.md) on the security track — this is
|
||||||
|
the hardware half; that is the namespace half.*
|
||||||
|
|
||||||
|
Two CR4 bits that make the CPU refuse the two things a kernel should never do
|
||||||
|
with user memory:
|
||||||
|
|
||||||
|
- **SMEP** (Supervisor Mode Execution Prevention, CR4 bit 20): instruction
|
||||||
|
fetch in ring 0 from a page whose U/S bit says *user* → #PF. Kills the
|
||||||
|
classic ret2usr exploit shape — a kernel bug that redirects control flow
|
||||||
|
can no longer land in attacker-prepared user code.
|
||||||
|
- **SMAP** (Supervisor Mode Access Prevention, CR4 bit 21): data read/write
|
||||||
|
in ring 0 to a user page → #PF, unless `EFLAGS.AC` is set. `stac`/`clac`
|
||||||
|
open and close deliberate access windows; danos's design needs no windows
|
||||||
|
at all (below).
|
||||||
|
|
||||||
|
Detection is CPUID leaf 7, subleaf 0, EBX bit 7 (SMEP) and bit 20 (SMAP).
|
||||||
|
Both bits are per-core state: the BSP and every AP must set them.
|
||||||
|
|
||||||
|
## Why, in danos terms
|
||||||
|
|
||||||
|
Every syscall argument is an attacker-controlled integer, and several take
|
||||||
|
pointers. A kernel bug that dereferences a crafted pointer reads, writes, or
|
||||||
|
executes memory of the attacker's choosing — the exact bug class the
|
||||||
|
isolation tracks exist to prevent. SMEP/SMAP turn that class from "silent
|
||||||
|
compromise" into "immediate, attributable #PF with a kernel RIP in the log."
|
||||||
|
|
||||||
|
The second benefit matters as much as the first: **SMAP is a permanent
|
||||||
|
tripwire.** Once it is on, any *future* syscall that touches user memory
|
||||||
|
directly — instead of going through the checked copy layer — faults the
|
||||||
|
first time the QEMU suite runs it. The discipline stops depending on review.
|
||||||
|
|
||||||
|
## Where danos already stands
|
||||||
|
|
||||||
|
The design is closer than it looks, because the IPC layer was built right:
|
||||||
|
|
||||||
|
- **The copy layer is already SMAP-proof.** `copyAcross` and `copyFromUser`
|
||||||
|
(`system/kernel/ipc-synchronous.zig:305,333`) never dereference a user
|
||||||
|
virtual address: they walk the page tables and move bytes through the
|
||||||
|
physmap — kernel mappings throughout. SMAP cannot object.
|
||||||
|
- **Syscall entry already clears AC.** `SFMASK = 0x4_0700` clears IF, TF,
|
||||||
|
DF, **AC** on every `syscall`
|
||||||
|
(`system/kernel/architecture/x86_64/per-cpu.zig:76`). The syscall path is
|
||||||
|
SMAP-clean from day one.
|
||||||
|
- **The interrupt path is not.** Hardware does *not* clear AC on IDT
|
||||||
|
delivery, and ring 3 can set AC with `popfq` — so a hostile process could
|
||||||
|
take an interrupt with AC=1 and have the handler run with SMAP suspended.
|
||||||
|
`isr_common` (`system/kernel/architecture/x86_64/isr.s:366`) needs a
|
||||||
|
`clac` beside its `swapgs`.
|
||||||
|
- **CR4 today:** the BSP inherits firmware CR4 (no kernel write anywhere);
|
||||||
|
APs set PAE/OSFXSR/OSXMMEXCPT in `trampoline.s:62-68`. Neither path sets
|
||||||
|
SMEP/SMAP yet, and both must.
|
||||||
|
- **The stragglers.** Nine syscalls still dereference user pointers raw
|
||||||
|
after a bounds check — every one is a SMAP #PF waiting to happen, and
|
||||||
|
every one is *already* a latent kernel fault today (an unmapped-but-in-
|
||||||
|
range user page oopses the kernel instead of failing the call). The
|
||||||
|
verified sweep of `system/kernel/process.zig` (2026-07-31; a
|
||||||
|
whole-kernel `@ptrFromInt` audit found no user-address dereference
|
||||||
|
outside this file):
|
||||||
|
|
||||||
|
| Syscall | Raw access | Direction |
|
||||||
|
|---|---|---|
|
||||||
|
| `system_spawn` | name + argument blob (`:972`, `:980`) | read |
|
||||||
|
| `fs_resolve` | path in (`:1780`), result out (`:1797`) | read + write |
|
||||||
|
| `fs_mount` | prefix + rewrite strings (`:1864`, `:1865`) | read |
|
||||||
|
| `fs_unmount` | prefix string (`:1883`) | read |
|
||||||
|
| `fs_node` | read buffer out (`:1820`) | write |
|
||||||
|
| `debug_write` | message bytes (`:1700`; read twice — memcpy `:1710` and `log.append` `:1717`) | read |
|
||||||
|
| `klog_read` | log bytes out (`:1741`) | write |
|
||||||
|
| `klog_status` | status struct out (`:1758`) | write |
|
||||||
|
| `process_enumerate` | descriptor array out (`:1132`) | write |
|
||||||
|
| `device_enumerate` | descriptor array out (`:388`) | write |
|
||||||
|
|
||||||
|
For the write-direction rows the `@ptrFromInt` is in process.zig but the
|
||||||
|
stores happen in callees (`scheduler.enumerate`
|
||||||
|
`system/kernel/scheduler.zig:1209`, `devices_broker.enumerate`
|
||||||
|
`devices-broker.zig:136`, `log.readAt` `log.zig:209`, the vfs node calls
|
||||||
|
`vfs.zig:257/269/289`) — converting them means bounce buffers plus
|
||||||
|
`copyToUser` around those calls, not just editing the process.zig lines.
|
||||||
|
(Some paths already do it right — the futex word and the device-register
|
||||||
|
descriptor go through `copyFromUser` (`:1087`, `:924`). The write
|
||||||
|
direction has no public helper yet, but the mechanism exists:
|
||||||
|
`copyAcross` with a kernel source is exactly how IPC replies reach user
|
||||||
|
buffers, so `copyToUser` is a mechanical mirror.)
|
||||||
|
|
||||||
|
- **One known gap inside the copy layer itself:** the walk checks presence,
|
||||||
|
not the leaf U/S and writable bits (`ipc-synchronous.zig:20-22` flags
|
||||||
|
this). Today that is nearly moot — the user half contains only mappings
|
||||||
|
the kernel itself created for that process — but it must close before
|
||||||
|
shared or copy-on-write mappings exist, and closing it is part of making
|
||||||
|
the copy layer the single trusted door.
|
||||||
|
|
||||||
|
## The plan
|
||||||
|
|
||||||
|
**H1 — copy discipline (the real work).** A `user-memory` kernel module:
|
||||||
|
`copyFromUser` / `copyToUser` (the missing write direction) via the physmap
|
||||||
|
walk, with U/S and writable leaf checks closing the in-tree TODO. Convert
|
||||||
|
the nine stragglers. This fixes the latent unmapped-page kernel fault on
|
||||||
|
its own — it is worth doing even if SMEP/SMAP never shipped. QEMU suite
|
||||||
|
green; no behavior change visible to correct programs.
|
||||||
|
|
||||||
|
**H2 — SMEP.** A leaf-7 feature probe (the kernel has per-leaf `cpuid`
|
||||||
|
helpers in `apic.zig` to generalize); set CR4.SMEP during per-CPU bring-up
|
||||||
|
on BSP and APs — prefer the Zig-side per-CPU init over the trampoline
|
||||||
|
assembly, so one code path covers every core and the trampoline stays
|
||||||
|
minimal. Audit first that ring 0 never executes user-mapped pages: kernel
|
||||||
|
text lives in the kernel half, `jump_to_user` is kernel code, and the AP
|
||||||
|
trampoline page is kernel-mapped — expected clean, verify before flipping.
|
||||||
|
|
||||||
|
**H3 — SMAP.** Add `clac` at `isr_common` entry. `clac` is #UD on CPUs
|
||||||
|
without SMAP, so the instruction is a 3-byte NOP in the image, patched to
|
||||||
|
`clac` at boot when CPUID advertises SMAP (one-time patch beats a
|
||||||
|
conditional branch in the hottest path in the kernel). Then set CR4.SMAP in
|
||||||
|
the same per-CPU init. From this point the whole QEMU suite doubles as the
|
||||||
|
enforcement test: any missed raw dereference is a vector-14 with a kernel
|
||||||
|
RIP and a user CR2 — loud and attributable.
|
||||||
|
|
||||||
|
**H4 — keep it honest.** A line in the coding standards: kernel code
|
||||||
|
touches user memory only through `user-memory`; there is no `stac` anywhere
|
||||||
|
in the tree, and a PR that adds one is wrong by definition. SMAP enforces
|
||||||
|
the rule mechanically at test time.
|
||||||
|
|
||||||
|
Feature-gating follows the timekeeping rule (work on any VM, real Intel,
|
||||||
|
real AMD): both bits are probed, absence is logged and tolerated — like the
|
||||||
|
IOMMU's fail-open, the machine still boots, just unhardened. QEMU: TCG
|
||||||
|
implements both; KVM inherits the host (Intel Ivy Bridge+ for SMEP,
|
||||||
|
Broadwell+ for SMAP; AMD Zen+ for both). The test images should run with
|
||||||
|
`-cpu max` so the suite always exercises the enabled paths.
|
||||||
|
|
||||||
|
## Adjacent, deliberately separate
|
||||||
|
|
||||||
|
- **SYSRET canonical-RIP hardening** (`isr.s:192-194` documents it): a
|
||||||
|
non-canonical return RIP makes `sysretq` #GP *in ring 0* on Intel. Same
|
||||||
|
hardening bucket, independent fix (validate RCX before `sysretq`, fall
|
||||||
|
back to `iretq`), should ride the same branch as H2/H3 but is not
|
||||||
|
SMEP/SMAP.
|
||||||
|
- **KPTI / Meltdown-class leaks are out of scope.** SMEP/SMAP police
|
||||||
|
architectural accesses, not speculative ones. danos runs one kernel
|
||||||
|
mapping in every address space and accepts that on affected hardware;
|
||||||
|
revisit only if the threat model ever includes hostile native code on
|
||||||
|
shared machines.
|
||||||
@@ -11,7 +11,7 @@ sequential pass and hands the bytes to the kernel unmodified.
|
|||||||
|
|
||||||
The capsule is a *performance artifact*, not a source of truth. The boot
|
The capsule is a *performance artifact*, not a source of truth. The boot
|
||||||
volume's `/system` and `/test` file trees remain the canonical layout (see
|
volume's `/system` and `/test` file trees remain the canonical layout (see
|
||||||
[danos-file-system-hierarchy-FSH.md](../file-system-development/danos-file-system-hierarchy-FSH.md));
|
[file-system-hierarchy.md](../file-system-development/file-system-hierarchy.md));
|
||||||
the capsule is a pre-baked snapshot of the same binaries, derived from the same
|
the capsule is a pre-baked snapshot of the same binaries, derived from the same
|
||||||
build graph, so the running system is identical whether the loader read the
|
build graph, so the running system is identical whether the loader read the
|
||||||
capsule or walked the tree.
|
capsule or walked the tree.
|
||||||
@@ -36,11 +36,11 @@ so it need be no fancier. Little-endian throughout:
|
|||||||
|
|
||||||
```
|
```
|
||||||
Header magic: u32 = "DNR2" (0x32524E44), count: u32
|
Header magic: u32 = "DNR2" (0x32524E44), count: u32
|
||||||
Entry × count name: [64]u8 (NUL-padded FHS path), offset: u64, len: u64
|
Entry × count name: [64]u8 (NUL-padded hierarchy path), offset: u64, len: u64
|
||||||
blobs... each entry's file bytes, at its offset within the image
|
blobs... each entry's file bytes, at its offset within the image
|
||||||
```
|
```
|
||||||
|
|
||||||
- **Names are full FHS paths** (`/system/services/init`), not basenames — that
|
- **Names are full hierarchy paths** (`/system/services/init`), not basenames — that
|
||||||
is what "v2" means. The 64-byte capacity matches `abi.maximum_process_name`,
|
is what "v2" means. The 64-byte capacity matches `abi.maximum_process_name`,
|
||||||
so a task named after its binary path is never truncated. Paths longer than
|
so a task named after its binary path is never truncated. Paths longer than
|
||||||
63 bytes are a build error (`pack-system-image.py` rejects them).
|
63 bytes are a build error (`pack-system-image.py` rejects them).
|
||||||
@@ -54,14 +54,14 @@ blobs... each entry's file bytes, at its offset within the image
|
|||||||
|
|
||||||
## How it is built
|
## How it is built
|
||||||
|
|
||||||
`build.zig` maintains one `bundled` list — every user binary and its FHS home.
|
`build.zig` maintains one `bundled` list — every user binary and its hierarchy home.
|
||||||
Three artifacts are derived from that same list, in the same build graph, so
|
Three artifacts are derived from that same list, in the same build graph, so
|
||||||
they cannot drift apart:
|
they cannot drift apart:
|
||||||
|
|
||||||
1. **The tree**: each binary installed at its FHS path (`zig-out/system/...`
|
1. **The tree**: each binary installed at its hierarchy path (`zig-out/system/...`
|
||||||
and `zig-out/test/...`, mirrored onto the FAT boot volume by
|
and `zig-out/test/...`, mirrored onto the FAT boot volume by
|
||||||
`tools/make-fat-image.py`).
|
`tools/make-fat-image.py`).
|
||||||
2. **The manifest** (`system/manifest`): the FHS path of every bundled binary,
|
2. **The manifest** (`system/manifest`): the hierarchy path of every bundled binary,
|
||||||
one per line — the loader's per-file fallback input.
|
one per line — the loader's per-file fallback input.
|
||||||
3. **The capsule**: `tools/pack-system-image.py` packs the same binaries into
|
3. **The capsule**: `tools/pack-system-image.py` packs the same binaries into
|
||||||
the v2 container, installed at `zig-out/boot/system.img` and placed on the
|
the v2 container, installed at `zig-out/boot/system.img` and placed on the
|
||||||
@@ -103,7 +103,7 @@ the kernel (`kernel.zig`) then publishes the same bytes twice, to two
|
|||||||
consumers:
|
consumers:
|
||||||
|
|
||||||
- **The process layer** (`process.zig`): `system_spawn` looks binaries up in
|
- **The process layer** (`process.zig`): `system_spawn` looks binaries up in
|
||||||
the ramdisk via `Reader.find` — exact FHS path, or unique basename for
|
the ramdisk via `Reader.find` — exact hierarchy path, or unique basename for
|
||||||
pre-path callers — and loads them as fresh ring-3 processes. The stored path
|
pre-path callers — and loads them as fresh ring-3 processes. The stored path
|
||||||
becomes the task's name.
|
becomes the task's name.
|
||||||
- **The VFS root** (`vfs.zig`, `setInitialRamdisk`): the image is mounted as
|
- **The VFS root** (`vfs.zig`, `setInitialRamdisk`): the image is mounted as
|
||||||
@@ -111,7 +111,7 @@ consumers:
|
|||||||
paths, so `/system` and, when the fixtures are bundled, `/test`. Directory
|
paths, so `/system` and, when the fixtures are bundled, `/test`. Directory
|
||||||
nodes are derived from the entry paths (the unique parents), so the trees
|
nodes are derived from the entry paths (the unique parents), so the trees
|
||||||
are listable and their files readable over the normal VFS protocol — the
|
are listable and their files readable over the normal VFS protocol — the
|
||||||
FHS boot tree every process sees comes straight out of the capsule bytes.
|
boot tree every process sees comes straight out of the capsule bytes.
|
||||||
|
|
||||||
The image is never copied after the handoff and never mutated: the initrd is
|
The image is never copied after the handoff and never mutated: the initrd is
|
||||||
immutable, which is what makes the VFS's node serving lock-free.
|
immutable, which is what makes the VFS's node serving lock-free.
|
||||||
|
|||||||
@@ -22,11 +22,12 @@ lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run,
|
|||||||
## Conventions
|
## Conventions
|
||||||
|
|
||||||
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym abbreviations,
|
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym abbreviations,
|
||||||
kebab-case file names, no `Co-Authored-By` trailers. New user binaries go through
|
kebab-case file names, no `Co-Authored-By` trailers. New user binaries are
|
||||||
`addUserBinary` (with the new `threaded` flag where a binary spawns threads) and get
|
packages whose build.zig calls `build_support.userBinary` (with `.threaded =
|
||||||
packed into the initial-ramdisk; new syscalls extend [abi.zig](../../system/abi.zig)
|
true` where a binary spawns threads) and get packed into the initial-ramdisk;
|
||||||
`SystemCall` + a `library/runtime` wrapper; test services live beside the code they
|
new syscalls extend [abi.zig](../../system/abi.zig) `SystemCall` + a
|
||||||
exercise and register a `ServiceId` if they must be looked up.
|
`library/kernel` wrapper; test services live beside the code they exercise and
|
||||||
|
register a `ServiceId` if they must be looked up.
|
||||||
|
|
||||||
## How to verify along the way
|
## How to verify along the way
|
||||||
|
|
||||||
|
|||||||
@@ -61,9 +61,10 @@ runtime — rebuilt in lockstep — knows the mapping.
|
|||||||
backend would either bake danos syscall numbers into std (breaking ABI privacy and
|
backend would either bake danos syscall numbers into std (breaking ABI privacy and
|
||||||
renumbering) or fork std to route back through the runtime — a permanent rebase
|
renumbering) or fork std to route back through the runtime — a permanent rebase
|
||||||
cost that buys nothing the native type doesn't.
|
cost that buys nothing the native type doesn't.
|
||||||
2. **Our user binaries are built `single_threaded = true`** ([build.zig](../../build.zig)
|
2. **Our user binaries are built `single_threaded = true`** (the shared recipe in
|
||||||
`addUserBinary`), which compiles threading out entirely and makes atomics and TLS
|
[build-support/build.zig](../../build-support/build.zig)), which compiles threading
|
||||||
single-threaded. Threads need this flipped per binary regardless.
|
out entirely and makes atomics and TLS single-threaded. Threads need this flipped
|
||||||
|
per binary regardless.
|
||||||
|
|
||||||
So we take the *shape* of `std.Thread`, not the *type*. The cost of replicating the
|
So we take the *shape* of `std.Thread`, not the *type*. The cost of replicating the
|
||||||
surface (spawn/join/Mutex/Condition) is small; the cost of the std type is the ABI
|
surface (spawn/join/Mutex/Condition) is small; the cost of the std type is the ABI
|
||||||
@@ -236,9 +237,10 @@ see the intro). Two scoped pieces, as built:
|
|||||||
|
|
||||||
### Build: multi-threaded codegen, opt-in
|
### Build: multi-threaded codegen, opt-in
|
||||||
|
|
||||||
A binary opts in by being added with `addThreadedUserBinary` — as `addUserBinary`,
|
A binary opts in with `.threaded = true` in its package's
|
||||||
but the shared implementation builds it `single_threaded = false` — so atomics and
|
`build_support.userBinary` call — the shared recipe in build-support then builds it
|
||||||
(later) TLS are real. Threads and atomics are unsound in a `single_threaded` image,
|
`single_threaded = false` — so atomics
|
||||||
|
and (later) TLS are real. Threads and atomics are unsound in a `single_threaded` image,
|
||||||
so a binary must opt in **before** it may call `Thread.spawn`. Everyone else
|
so a binary must opt in **before** it may call `Thread.spawn`. Everyone else
|
||||||
stays single-threaded and lean.
|
stays single-threaded and lean.
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,192 @@
|
|||||||
|
# Python on danos: the milestone plan
|
||||||
|
|
||||||
|
The execution plan for [python-on-danos.md](python-on-danos.md). That note holds
|
||||||
|
the *why* and the design decisions; this one slices the work into milestones with
|
||||||
|
concrete deliverables, tests, and exit criteria. Milestones are numbered **P0–P5**
|
||||||
|
(track-local — the global M-series stays with the driver/lifecycle tracks).
|
||||||
|
|
||||||
|
Dependencies at a glance:
|
||||||
|
|
||||||
|
```
|
||||||
|
P0 toolchain + mini-libc ──┐
|
||||||
|
P1 streams + console + seam ─┴─→ P2 CPython minimal ─→ P3 terminal + REPL
|
||||||
|
│ │
|
||||||
|
└─→ P4 danos module │
|
||||||
|
+ Python service│
|
||||||
|
P5 process control + shell ←─────────────────────────────────┘
|
||||||
|
```
|
||||||
|
|
||||||
|
P0 and P1 are independent of each other and can proceed in parallel. P1 is shared
|
||||||
|
work — it is also Zig self-hosting Phase 1 and the first three slices of
|
||||||
|
[character-devices-and-tty.md](character-devices-and-tty.md).
|
||||||
|
|
||||||
|
## P0 — Toolchain + the C library compatibility layer
|
||||||
|
|
||||||
|
**Goal:** a C hello-world, cross-compiled on the host with `zig cc`, runs on danos.
|
||||||
|
|
||||||
|
Design and slicing live in
|
||||||
|
[c-library-compatibility.md](c-library-compatibility.md): the **libdanos-c**
|
||||||
|
sysroot (hand-written danos-native headers + `libc.a`) as a `library/c/` build
|
||||||
|
package — pure computation (string, libm, `strtod`, the printf/scanf engines)
|
||||||
|
lifted from a vendored, pinned musl subtree; the OS plumbing written in Zig over
|
||||||
|
the `runtime` surface (re-targeting `runtime.os` when the Zig track authors it);
|
||||||
|
`malloc` over danos `mmap`; a `crt0` bridging the danos entry shim to C `main`.
|
||||||
|
Driven by `zig cc -target x86_64-freestanding-none -isystem` (the triple becomes
|
||||||
|
`x86_64-danos` if the Zig fork lands first; nothing else changes).
|
||||||
|
|
||||||
|
Its five slices (sysroot-skeleton, fd-plumbing, malloc, stdio,
|
||||||
|
mathematics-and-time) carry their own tests — host-side oracle suites for the
|
||||||
|
computation layer, QEMU cases (`c-hello`, `c-file-io`, `c-stdio`, `c-time`) for
|
||||||
|
the plumbing.
|
||||||
|
|
||||||
|
**Exit:** `c-hello` and `c-file-io` green in the QEMU suite; host computation
|
||||||
|
tests green.
|
||||||
|
|
||||||
|
## P1 — Stream nodes, console, and the seam pieces
|
||||||
|
|
||||||
|
**Goal:** the shared Phase-1 surface exists: byte-stream stdio, cwd, environment,
|
||||||
|
entropy. Design and slicing live in
|
||||||
|
[character-devices-and-tty.md](character-devices-and-tty.md); this milestone is
|
||||||
|
its slices 1–3 plus three small seam pieces:
|
||||||
|
|
||||||
|
- **cwd/chdir** — per-process current directory used by path resolution (the
|
||||||
|
kernel already anchors a VFS root per `fs_resolve`; the cwd is the same idea,
|
||||||
|
process-scoped, with `getcwd`/`chdir` exposed through `runtime` and the libc).
|
||||||
|
- **Environment** — spawn carries an environment block; the SysV entry stack's
|
||||||
|
`envp` slot ([sysv.md](os-development/sysv.md)) stops being empty; `getenv`
|
||||||
|
reads it. An empty block stays valid.
|
||||||
|
- **Entropy** — a kernel `entropy` syscall (RDSEED/RDRAND with a jitter fallback,
|
||||||
|
mirroring the TSC-reliability posture of not trusting one CPU feature blindly);
|
||||||
|
the libc exposes `getentropy`.
|
||||||
|
|
||||||
|
- **Tests.** QEMU: the character-device tests from the tty note (offsetless
|
||||||
|
read/write, blocking read, cooked/raw control round-trip), plus `cwd-basics`
|
||||||
|
(chdir + relative open), `env-roundtrip` (spawn with env, child reads it),
|
||||||
|
`entropy-sane` (nonzero, changing, correct length).
|
||||||
|
|
||||||
|
**Exit:** a C program reads a cooked line from fd 0 and echoes it to fd 1 —
|
||||||
|
injected key events in, bytes read back through the console's in-memory sink,
|
||||||
|
all under QEMU with no hardware involved — and `getcwd`/`getenv`/`getentropy`
|
||||||
|
return real answers.
|
||||||
|
|
||||||
|
## P2 — CPython, minimal configuration
|
||||||
|
|
||||||
|
**Goal:** `python -c 'print(2**100)'` runs on danos under QEMU.
|
||||||
|
|
||||||
|
- Pin **CPython 3.13.x**; vendor as `third-party/cpython/` or fetch via the build
|
||||||
|
(decide with the build-packages conventions).
|
||||||
|
- Host build-Python of the same version (`--with-build-python`).
|
||||||
|
- `config.site` cache for the cross answers; `config.sub` patch so
|
||||||
|
`x86_64-unknown-danos` parses; a small `configure`/`pyconfig` patch set kept as
|
||||||
|
rebasable diffs, WASI-style.
|
||||||
|
- `--disable-shared`; static `Modules/Setup`: `posix errno _io _codecs _weakref
|
||||||
|
time math _stat _collections itertools _functools _locale _sre` plus what the
|
||||||
|
interpreter core insists on; threadless build (WASI precedent).
|
||||||
|
- `Lib/` on the FAT image under the hierarchy (e.g. `/system/python/lib`);
|
||||||
|
`PYTHONHOME` set accordingly; `.pyc` written with **checked-hash
|
||||||
|
invalidation** (FAT's 2-second mtime granularity makes mtime-based validation
|
||||||
|
lie during fast edit-run cycles).
|
||||||
|
- `PYTHONHASHSEED` pinned only if P1's entropy slipped — otherwise real
|
||||||
|
hash randomization from day one.
|
||||||
|
- **Tests.** QEMU: `python-expr` (the exit criterion), `python-file` (run a
|
||||||
|
script from FAT, write a file, read it back), then a curated slice of CPython's
|
||||||
|
own suite (`test_int`, `test_float`, `test_io`, `test_dict`) as a
|
||||||
|
longer-running target — the suite is the porting harness.
|
||||||
|
|
||||||
|
**Exit:** the four QEMU cases green; the CPython test slice green or with a
|
||||||
|
short, documented skip list.
|
||||||
|
|
||||||
|
## P3 — Terminal + REPL: the first real application
|
||||||
|
|
||||||
|
**Goal:** an interactive `python` REPL in a graphical danos terminal — the
|
||||||
|
milestone demo for the OS.
|
||||||
|
|
||||||
|
- Depends on the display track's font rendering (its stated next step) — until
|
||||||
|
that lands, the REPL is exercised end-to-end through the pseudo-device
|
||||||
|
harness from P1+P2 (scripted input in, output read back), so P2's exit is
|
||||||
|
never blocked on graphics; the graphical terminal is the *interactive* debut.
|
||||||
|
- The terminal application: draws with the UI toolkit / display client, consumes
|
||||||
|
keyboard `InputEvent`s, and — per the tty note's load-bearing decision —
|
||||||
|
**serves the VFS stream protocol itself** to its children, reusing the console's
|
||||||
|
line-discipline library. Spawns `python` with its endpoints as fd 0/1/2.
|
||||||
|
- Raw mode + the control set give the REPL line editing; window-size control
|
||||||
|
gives it wrapping.
|
||||||
|
- **Tests.** QEMU: scripted terminal session (inject key events, assert rendered
|
||||||
|
or captured output). Real-hardware smoke on the Intel box joins the existing
|
||||||
|
checklist.
|
||||||
|
|
||||||
|
**Exit:** typing `2+2` into the terminal on the QEMU GPU target prints `4`.
|
||||||
|
|
||||||
|
## P4 — The `danos` extension module + a Python service
|
||||||
|
|
||||||
|
**Goal:** Python can speak danos: IPC, capabilities, spawn.
|
||||||
|
|
||||||
|
- The `danos` module, **written in Zig against `Python.h`**, statically linked
|
||||||
|
via `Modules/Setup`: endpoints (create/send/receive), capability passing,
|
||||||
|
spawn + exit-notification, and the service bootstrap (announce, supervision
|
||||||
|
handshake) — the same surface Zig services use, re-exposed.
|
||||||
|
- UI-toolkit bindings as a second module once the toolkit's API settles.
|
||||||
|
- Prototype **one real service in Python** — policy-shaped, not data-plane
|
||||||
|
(candidates: hot-plug policy, a settings service) — speaking an existing wire
|
||||||
|
protocol, supervised by the device manager like any service.
|
||||||
|
- **Tests.** QEMU: `python-ipc-echo` (Python service echoes over an endpoint, a
|
||||||
|
Zig client asserts), plus the prototype service's own protocol test.
|
||||||
|
|
||||||
|
**Exit:** a Python process runs as a supervised danos service exchanging IPC
|
||||||
|
with Zig peers.
|
||||||
|
|
||||||
|
## P5 — Process control, then the shell
|
||||||
|
|
||||||
|
**Goal:** danos can spawn arbitrary programs with arguments and pipes; a small
|
||||||
|
Python shell uses it.
|
||||||
|
|
||||||
|
The kernel/VFS cluster a shell forces (any shell, any language):
|
||||||
|
|
||||||
|
- **exec-of-path** — spawn an arbitrary VFS path, not a named ramdisk binary;
|
||||||
|
- **argv/envp** — carried through spawn onto the child's entry stack (env from
|
||||||
|
P1, argv new);
|
||||||
|
- **numeric exit status** — extend the exit record beyond the categorical
|
||||||
|
`ExitReason` (the gotcha the Zig roadmap flagged: `WEXITSTATUS` must be real);
|
||||||
|
- **fd inheritance + pipes** — a kernel or service pipe (a character device by
|
||||||
|
the tty note's definition) and spawn-time fd mapping.
|
||||||
|
|
||||||
|
Then, in order: `subprocess` enabled in CPython (maps onto spawn + the
|
||||||
|
exit-notification endpoint — no fork, Windows-style); a **small Python shell** (a
|
||||||
|
few hundred lines over `subprocess` + the console: prompt, argv parsing, pipes,
|
||||||
|
cwd) as the forcing function that reveals what job control actually needs.
|
||||||
|
|
||||||
|
**Explicitly deferred past P5:** the pthread subset over `thread_spawn`/futex,
|
||||||
|
signals-in-libc via M17, termios job control (Ctrl-C to foreground child), and
|
||||||
|
**xonsh** — which wants all three and is the arc's endpoint, not a milestone.
|
||||||
|
|
||||||
|
**Tests.** QEMU: `spawn-argv-exit` (child echoes argv, exits 42, parent sees
|
||||||
|
42), `pipe-through` (parent → child → parent), `python-subprocess`, and a
|
||||||
|
scripted shell session.
|
||||||
|
|
||||||
|
**Exit:** the Python shell runs `program | program` typed at the terminal and
|
||||||
|
reports the exit status.
|
||||||
|
|
||||||
|
## Post-P5 outlook
|
||||||
|
|
||||||
|
Two tracks continue past this plan, each with its own design doc rather than a
|
||||||
|
P-number here:
|
||||||
|
|
||||||
|
- **Dynamic libraries** ([dynamic-libraries.md](dynamic-libraries.md), D1–D4) —
|
||||||
|
an application-layer facility (the OS stays static and lean): `dlopen` in the
|
||||||
|
libc, then libffi + `ctypes` + loadable extension modules, then shared
|
||||||
|
read-only mappings so N Python services hold one physical `libpython`.
|
||||||
|
- **The full C compatibility layer**
|
||||||
|
([c-library-compatibility.md](c-library-compatibility.md), stages 2–3) — the
|
||||||
|
standing rule that every system capability ships with its C spelling, draining
|
||||||
|
the absence table toward "portable C builds on danos"; `fork` is the one
|
||||||
|
permanent exception.
|
||||||
|
|
||||||
|
## Related
|
||||||
|
|
||||||
|
- [python-on-danos.md](python-on-danos.md) — the design note this executes.
|
||||||
|
- [c-library-compatibility.md](c-library-compatibility.md) — P0's design.
|
||||||
|
- [character-devices-and-tty.md](character-devices-and-tty.md) — P1's design.
|
||||||
|
- [zig-self-hosting.md](zig-self-hosting.md) — shares P1; its fork makes P0's
|
||||||
|
triple prettier but gates nothing here.
|
||||||
|
- [os-development/process-management.md](os-development/process-management.md) —
|
||||||
|
the spawn/exit surface P5 extends.
|
||||||
@@ -0,0 +1,261 @@
|
|||||||
|
# Python on danos: the CPython milestone
|
||||||
|
|
||||||
|
A design note (not built yet) on bringing **CPython** to danos, compiled with the Zig
|
||||||
|
toolchain (`zig cc`). Like [zig-self-hosting.md](zig-self-hosting.md), it is
|
||||||
|
forward-looking: it sets a direction and the decisions that follow from it.
|
||||||
|
|
||||||
|
## Why Python, and why now
|
||||||
|
|
||||||
|
The Zig self-hosting road is gated on a compiler fork and a long std-library seam.
|
||||||
|
Python is the **stop-gap that removes the wait**: a working CPython gives danos a way
|
||||||
|
to write programs — services, tools, application prototypes — *without* the Zig
|
||||||
|
compiler being self-hosted, and it brings the pure-Python package ecosystem along as
|
||||||
|
a bonus. The intended division of labour:
|
||||||
|
|
||||||
|
- **Zig** — the kernel, drivers, and anything on a data plane (interrupt paths,
|
||||||
|
DMA rings, block I/O). Unchanged.
|
||||||
|
- **Python** — the control plane and the prototyping surface: services that are
|
||||||
|
event loops over IPC, policy logic that changes often, application experiments,
|
||||||
|
and eventually the shell.
|
||||||
|
|
||||||
|
Python is also the scripting language for the terminal-and-shell arc: the first
|
||||||
|
real danos application is planned as a terminal, a terminal wants a shell, a shell
|
||||||
|
wants a scripting language — and [xonsh](https://xon.sh) (a shell written in
|
||||||
|
Python) marks where that road can end.
|
||||||
|
|
||||||
|
### Non-goals
|
||||||
|
|
||||||
|
- **No drivers in Python.** Interrupt handling, ring management, and DMA stay in
|
||||||
|
Zig. Python may *supervise and configure* drivers; it does not sit in their hot
|
||||||
|
paths (interpreter overhead and garbage-collection pauses in an interrupt path
|
||||||
|
are disqualifying).
|
||||||
|
- **No dynamic loading during bring-up, no `pip`.** The whole arc here ships
|
||||||
|
statically linked. Dynamic libraries are a real *later* milestone
|
||||||
|
([dynamic-libraries.md](dynamic-libraries.md)) — an application-layer
|
||||||
|
facility that unlocks `ctypes` and loadable extension modules; the operating
|
||||||
|
system itself stays static and lean regardless (the size doctrine below).
|
||||||
|
`pip` stays out either way until a networking track exists.
|
||||||
|
- **No fork.** `os.fork` will not exist. This costs almost nothing (see "The
|
||||||
|
spawn model fits").
|
||||||
|
|
||||||
|
## The realization that shapes everything: the compiler is not the obstacle
|
||||||
|
|
||||||
|
`zig cc` is a full Clang-based C cross-compiler, and CPython is portable C with
|
||||||
|
official precedent for stranger targets than danos — the WASI port is upstream
|
||||||
|
tier-2, and it runs **without fork, without dynamic loading, and without working
|
||||||
|
threads**. Every "CPython can't possibly run there" objection has already been
|
||||||
|
answered upstream by a target *more* constrained than danos.
|
||||||
|
|
||||||
|
What CPython actually needs is a **C environment**: headers and a `libc.a`. danos
|
||||||
|
has neither — and that is the whole project. In the language of the Zig roadmap's
|
||||||
|
three doors, this is the **door-2-shaped work** (the deferred "musl door"), not the
|
||||||
|
`std.os.danos` seam: CPython never touches Zig's std.
|
||||||
|
|
||||||
|
### The same surface, a third time
|
||||||
|
|
||||||
|
The Zig roadmap observed that door 1 (`std.os.danos`) and door 2 (a libc) implement
|
||||||
|
the *same* ~30 danos-facing operations at different layers. CPython consumes that
|
||||||
|
identical surface through C spellings. So nothing here is throwaway: the
|
||||||
|
danos-native operations backing `runtime.os` are the same ones the libc bottoms out
|
||||||
|
in, and the gaps this track must close (stdio byte streams, cwd, environment,
|
||||||
|
entropy) are **exactly the Phase-1 gaps the Zig roadmap already lists**. The two
|
||||||
|
tracks share a road until Python forks off at "build the libc."
|
||||||
|
|
||||||
|
## Where danos stands: coverage vs. the gaps
|
||||||
|
|
||||||
|
Judged against the minimal CPython configuration (static, WASI-like):
|
||||||
|
|
||||||
|
| CPython need | danos today | Gap |
|
||||||
|
|--------------|-------------|-----|
|
||||||
|
| open/read/write/close/lseek, readdir | VFS + FAT via `runtime.fs` | none — wrap in C |
|
||||||
|
| mkdir / unlink / rename / truncate | done (self-hosting Phase 2) | none |
|
||||||
|
| stat with mtime | done (`wall_clock` + FAT mtime) | none |
|
||||||
|
| mmap/munmap (object allocator) | native syscalls | none |
|
||||||
|
| monotonic + wall clock | `clock` + `wall_clock` syscalls | none |
|
||||||
|
| a place for `Lib/` | FAT boot image | none — better than WASI has it |
|
||||||
|
| fork / exec | not needed (subprocess disabled at first) | — |
|
||||||
|
| dynamic loading | not needed (static extension modules) | — |
|
||||||
|
| getcwd / chdir | — | **missing** (shared with Zig Phase 1) |
|
||||||
|
| environment variables | `Init` has no env | **missing** (can start empty) |
|
||||||
|
| entropy | — | **missing** (hash seed; `PYTHONHASHSEED` pins it meanwhile) |
|
||||||
|
| byte-stream stdin/stdout (fd 0/1/2) | `debug_write` out; structured `InputEvent` in | **missing** (shared with Zig Phase 1; the REPL needs it) |
|
||||||
|
| signals | — | stubs suffice (WASI precedent); M17 signals-over-IPC maps on later |
|
||||||
|
| threads | native `thread_spawn`/futex | build threadless first; a pthread subset later (xonsh needs it) |
|
||||||
|
|
||||||
|
The clustering repeats the Zig roadmap's: **files, memory, and time are done; the
|
||||||
|
work is the C packaging plus the small seam pieces** (tty bytes, cwd, env, entropy).
|
||||||
|
|
||||||
|
## The libc decision: hand-rolled in Zig, computation lifted from musl
|
||||||
|
|
||||||
|
Two viable shapes were considered:
|
||||||
|
|
||||||
|
| Option | What it is | Verdict |
|
||||||
|
|--------|-----------|---------|
|
||||||
|
| **Mini-libc in Zig** | C-ABI-exporting Zig library over `runtime.os`/`runtime.fs`, shipped as headers + `libc.a`. | **Take this.** Reuses the danos-native surface directly; no Linux assumptions to fight. |
|
||||||
|
| **Port musl** | Full musl with a danos syscall backend. | Defer, again. musl assumes Linux syscall semantics in places; heavier than the need. |
|
||||||
|
|
||||||
|
The trick that makes the mini-libc tractable: musl's `string/`, `math/` (libm —
|
||||||
|
CPython needs essentially all of it), and number-conversion layers are **pure
|
||||||
|
computation with no syscalls**. Lift those wholesale (MIT-licensed, designed to
|
||||||
|
compile standalone) and hand-write only:
|
||||||
|
|
||||||
|
- the OS-facing bottom: fds, `mmap`, clocks, `exit`, `getcwd` — thin C-ABI wrappers
|
||||||
|
over `runtime.os`;
|
||||||
|
- a `FILE*` stdio layer (buffered, over the fd layer);
|
||||||
|
- `malloc` over danos `mmap` (a simple allocator is fine; CPython does its own
|
||||||
|
small-object arena management above it);
|
||||||
|
- the headers (`stdio.h`, `stdlib.h`, `string.h`, `math.h`, `errno.h`, …).
|
||||||
|
|
||||||
|
Estimate: **100–150 functions**, of which the hard 40% (libm, string, printf/strtod
|
||||||
|
cores) are lifted, not written. Correctness hot spots are `strtod`/`dtoa` (Python's
|
||||||
|
float repr round-trips through them) — another reason to lift musl's, not improvise.
|
||||||
|
|
||||||
|
## C interop: static extension modules, not ctypes
|
||||||
|
|
||||||
|
"Python can interface with C libraries" is true on danos with one important
|
||||||
|
correction: **`ctypes` does not work at first** — it is built on `dlopen` + libffi,
|
||||||
|
both of which arrive only with the [dynamic-libraries](dynamic-libraries.md)
|
||||||
|
milestone (D2). Until then the interop story is the other, older one:
|
||||||
|
|
||||||
|
- **Extension modules statically linked into the interpreter** via CPython's
|
||||||
|
`Modules/Setup` mechanism (the standard route for embedded/static builds).
|
||||||
|
- **Zig speaks C ABI natively**, so danos extension modules are written in Zig
|
||||||
|
against `Python.h` — no C required. Two modules are planned from the start:
|
||||||
|
- **`danos`** — the system module: endpoints, send/receive, capability passing,
|
||||||
|
spawn, exit notification. This is what makes a Python *service* possible: an
|
||||||
|
event loop over IPC, speaking the same wire protocols as Zig services.
|
||||||
|
- **UI toolkit bindings** — the in-progress danos UI toolkit exposed to Python,
|
||||||
|
so application prototypes drive real windows.
|
||||||
|
|
||||||
|
The package story follows: **pure-Python packages work** (unpack into
|
||||||
|
`Lib/site-packages` on the FAT image); packages with C extensions must be
|
||||||
|
cross-compiled and baked into the interpreter — a curated set chosen per image,
|
||||||
|
not `pip install`. That is the honest shape of the stop-gap.
|
||||||
|
|
||||||
|
## The roadmap
|
||||||
|
|
||||||
|
### Phase 0 — Toolchain + libc bring-up
|
||||||
|
|
||||||
|
`zig cc -target x86_64-freestanding-none` plus `-isystem` the danos headers and the
|
||||||
|
mini-libc archive. No compiler fork required — this track deliberately avoids the
|
||||||
|
Zig roadmap's Phase-0 gate (if the fork lands first, the triple becomes a clean
|
||||||
|
`x86_64-danos`; nothing else changes). Exit criterion: a **hello-world C program**
|
||||||
|
compiles on the host and runs on danos, printing via the libc's `write`.
|
||||||
|
|
||||||
|
### Phase 1 — The shared seam pieces
|
||||||
|
|
||||||
|
The same list as Zig self-hosting Phase 1, closed once for both tracks:
|
||||||
|
|
||||||
|
- fd 0/1/2 as console **byte** streams (output exists as `debug_write`; input is a
|
||||||
|
new small thing — cooked line input first, raw mode when the REPL wants editing);
|
||||||
|
- `getcwd`/`chdir`;
|
||||||
|
- environment variables (an empty block is a valid start);
|
||||||
|
- an entropy syscall or service (until then, builds pin `PYTHONHASHSEED`).
|
||||||
|
|
||||||
|
### Phase 2 — Cross-compile CPython, minimal configuration
|
||||||
|
|
||||||
|
Pin one CPython release (3.13 — strongest WASI-era cross-compile support). The
|
||||||
|
mechanics are well-trodden upstream since 3.11:
|
||||||
|
|
||||||
|
- a same-version **build-Python on the host** (`--with-build-python`);
|
||||||
|
- a `config.site` cache answering what configure cannot probe cross
|
||||||
|
(`ac_cv_file__dev_ptmx=no` and friends);
|
||||||
|
- a `config.sub` patch so `x86_64-unknown-danos` parses;
|
||||||
|
- `--disable-shared`, static `Modules/Setup` with a minimal module set
|
||||||
|
(`posix`, `errno`, `_io`, `_codecs`, `time`, `math`, …);
|
||||||
|
- `Lib/` shipped on the FAT image; `PYTHONHOME` pointed at it.
|
||||||
|
|
||||||
|
Exit criterion: `python -c 'print(2**100)'` runs on danos under QEMU.
|
||||||
|
|
||||||
|
### Phase 3 — Terminal + REPL: the first real application
|
||||||
|
|
||||||
|
Depends on the display track's font rendering (already its stated next step) and
|
||||||
|
Phase 1's tty. A terminal emulator drawing a `python` REPL is the milestone demo:
|
||||||
|
interactive, self-evidently real, and it needs **zero** process-control machinery.
|
||||||
|
|
||||||
|
### Phase 4 — The `danos` module and Python services
|
||||||
|
|
||||||
|
Write the `danos` extension module and the UI-toolkit bindings; prototype one real
|
||||||
|
service in Python (a policy-shaped one — e.g. hot-plug policy or a settings
|
||||||
|
service) speaking the existing IPC protocols. This is the payoff phase for
|
||||||
|
"prototyping a service or application."
|
||||||
|
|
||||||
|
### Phase 5 — Process control, then the shell
|
||||||
|
|
||||||
|
The shell — any shell, in any language — forces the surface danos has deferred so
|
||||||
|
far: **exec-of-path, argv/envp passing, numeric exit status (`WEXITSTATUS`, not the
|
||||||
|
categorical `ExitReason`), fd inheritance, and pipes.** That is a kernel/VFS
|
||||||
|
milestone cluster of its own. Then, in order:
|
||||||
|
|
||||||
|
1. `subprocess` enabled in CPython (maps onto danos spawn — see below);
|
||||||
|
2. a **small Python shell** (a few hundred lines over `subprocess` + line input, no
|
||||||
|
job control) — the forcing function that reveals which process-control pieces
|
||||||
|
actually matter;
|
||||||
|
3. **explicitly deferred:** a pthread subset over `thread_spawn`/futex
|
||||||
|
(create/join/mutex/condition/thread-locals), signals via M17 signals-over-IPC,
|
||||||
|
termios job control — and then **xonsh**, which wants all three.
|
||||||
|
|
||||||
|
### The spawn model fits
|
||||||
|
|
||||||
|
One genuinely good alignment: **CPython does not need fork.** `subprocess` maps
|
||||||
|
cleanly onto a posix_spawn-style model — exactly what danos has — and the existing
|
||||||
|
exit-notification-via-endpoint is a *better* fit for `Popen.wait` than Unix's
|
||||||
|
`wait` semantics. `os.fork` simply won't exist, as on Windows, and almost nothing
|
||||||
|
in practice cares.
|
||||||
|
|
||||||
|
## Risks and gotchas
|
||||||
|
|
||||||
|
- **Binary size — and the size doctrine that makes it acceptable.** danos's
|
||||||
|
leanness mandate applies to the **operating system**: the kernel and the system
|
||||||
|
services stay small (the kernel is measured in kilobytes, not megabytes), and
|
||||||
|
nothing in this track changes that — Python never enters the OS layer. An
|
||||||
|
**application** budget is different: a statically-linked CPython with its
|
||||||
|
module set will be tens of megabytes in ReleaseSafe (the measured ~2×
|
||||||
|
safety-check factor compounds it), and that is *allowed* — applications live
|
||||||
|
on the FAT image, not in the kernel's world. It still shapes the image, and it
|
||||||
|
means every Python service shares one interpreter binary + per-service
|
||||||
|
scripts, so the spawn model needs **argv** before "run this .py" works at all.
|
||||||
|
- **FAT mtime granularity is 2 seconds.** CPython's `.pyc` cache validation is
|
||||||
|
mtime-based by default; a rapid edit-run cycle can see stale bytecode. Use
|
||||||
|
hash-based `.pyc` invalidation (PEP 552, `--invalidation-mode checked-hash` at
|
||||||
|
freeze time) or accept the quirk during bring-up.
|
||||||
|
- **FAT name lookups are case-insensitive.** Long file names preserve case but
|
||||||
|
match insensitively — the same world Python inhabits on Windows/macOS, so
|
||||||
|
importlib copes, but two modules differing only by case cannot coexist on the
|
||||||
|
image.
|
||||||
|
- **`strtod`/float repr correctness.** Python's float round-tripping is exacting;
|
||||||
|
lift musl's conversions rather than writing them, and run CPython's float tests
|
||||||
|
early.
|
||||||
|
- **Threadless build is load-bearing, initially.** Like WASI, the first builds have
|
||||||
|
no working `threading`. The escape hatch is real (danos has native threads and
|
||||||
|
futexes; a pthread subset is Phase-5 work) but keep the configuration honestly
|
||||||
|
single-threaded until then.
|
||||||
|
- **The test suite is the porting harness.** CPython ships its own conformance
|
||||||
|
suite; getting `test_builtin`, `test_int`, `test_float`, `test_io` running on
|
||||||
|
danos early converts "it seems to work" into a checklist. Budget image space for
|
||||||
|
the test `Lib/` tree during bring-up.
|
||||||
|
- **Entropy before exposure.** `PYTHONHASHSEED=0` is fine for bring-up and wrong
|
||||||
|
forever; hash randomization exists because attacker-controlled dict keys are a
|
||||||
|
denial-of-service vector. Land the entropy source before any Python service
|
||||||
|
parses external input.
|
||||||
|
|
||||||
|
## Related
|
||||||
|
|
||||||
|
- [python-on-danos-milestones.md](python-on-danos-milestones.md) — the execution
|
||||||
|
plan (P0–P5) for this note.
|
||||||
|
- [c-library-compatibility.md](c-library-compatibility.md) — the mini-libc
|
||||||
|
(libdanos-c) design behind Phase 0.
|
||||||
|
- [character-devices-and-tty.md](character-devices-and-tty.md) — the stream-node /
|
||||||
|
console / no-pty design behind Phase 1.
|
||||||
|
- [zig-self-hosting.md](zig-self-hosting.md) — the sibling track; shares Phase 1,
|
||||||
|
diverges at the libc.
|
||||||
|
- [os-development/syscall.md](os-development/syscall.md) — the kernel ABI the
|
||||||
|
mini-libc bottoms out in.
|
||||||
|
- [os-development/vdso.md](os-development/vdso.md) — the public ABI boundary the
|
||||||
|
`danos` extension module wraps.
|
||||||
|
- [os-development/sysv.md](os-development/sysv.md) — the entry stack (argv/envp)
|
||||||
|
the spawn-argv work extends.
|
||||||
|
- [device-driver-development/ipc.md](device-driver-development/ipc.md) — the IPC
|
||||||
|
surface Python services speak.
|
||||||
|
- [file-system-development/file-system-hierarchy.md](file-system-development/file-system-hierarchy.md)
|
||||||
|
— where `Lib/` and `site-packages` land on the image.
|
||||||
@@ -0,0 +1,394 @@
|
|||||||
|
# Security track execution plan: paths, protocol namespace, SMEP/SMAP
|
||||||
|
|
||||||
|
The design is settled in
|
||||||
|
[communication.md](os-development/communication.md),
|
||||||
|
[protocol-namespace.md](os-development/protocol-namespace.md),
|
||||||
|
[file-system-hierarchy.md](file-system-development/file-system-hierarchy.md),
|
||||||
|
and [smep-smap.md](os-development/smep-smap.md). This file is the build order
|
||||||
|
— one phase at a time, each phase green before the next starts. Delete or
|
||||||
|
archive this file when the last milestone lands.
|
||||||
|
|
||||||
|
**Context a fresh session should read first:** the four design docs above,
|
||||||
|
then this plan's *Settled decisions* section — those decisions came out of a
|
||||||
|
full-code grounding pass (2026-07-31) and must not be re-derived or reopened.
|
||||||
|
|
||||||
|
**Definition of green, every phase:** `zig build` clean, `zig build test`
|
||||||
|
clean, `python3 test/qemu_test.py` passes (existing scenarios plus the
|
||||||
|
phase's new ones — record the suite count in the checkbox), and the relevant
|
||||||
|
design doc's status/known-gap lines updated in the same commit. Commit per
|
||||||
|
green phase, style `area: lower-case declarative summary`, **no co-author
|
||||||
|
trailers**. On a suite failure, read
|
||||||
|
`zig-out/qemu-test/<case>-failed-serial.log` before changing anything.
|
||||||
|
|
||||||
|
**Workflow:** work in a dedicated git worktree on feature branches cut from
|
||||||
|
`main` (one branch per milestone group as marked below); when a group's
|
||||||
|
phases are all green, merge to `main` and push. The loop marks a phase `[x]`
|
||||||
|
in the same commit that lands it.
|
||||||
|
|
||||||
|
**Numbering note:** milestones use the design docs' own names (PM, H1–H3,
|
||||||
|
HS, P1–P4) — the M-number sequence is left alone (M19–M22 are reserved by
|
||||||
|
the logging/USB-lifecycle track).
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
- [x] **Phase 0** — baseline: suite green on `main` (106/106, 2026-07-31; `zig build` + `zig build test` clean at 9a32380), plan committed
|
||||||
|
- [x] **PM** — path-migration flag-day (`/etc`→`/system/configuration`, `/var/log`→`/system/logs`, `/mnt/usb`→`/volumes/usb`; vfs carve-out for the two writable `/system` subtrees, FAT's `/var` mount split in two; suite 106/106)
|
||||||
|
- [x] **H1** — the `user-memory` module; nine stragglers converted; leaf U/S+W checks (plus physmap-coverage confirmation, so an `mmio_map`'d buffer cannot fault ring 0 — this also closes the same hazard on the IPC path; `fs_resolve`'s out-capacity bound made overflow-safe; suite 107/107)
|
||||||
|
- [x] **merge** group 1 → main, push (f3bc23c, 2026-07-31)
|
||||||
|
- [ ] **P1** — envelope module + `Define`; vfs `NodeKind.protocol` + open-reply-capability; client `Channel`
|
||||||
|
- [ ] **P2** — registry in init; `/protocol` reserved; ServiceId flag-day (11 binds, 17 lookups)
|
||||||
|
- [ ] **P3** — open grants: `protocol.csv` enforcement, denial test
|
||||||
|
- [ ] **merge** group 2 → main, push
|
||||||
|
- [ ] **P4a** — clean protocols rebased onto `Define` (vfs, block, display, scanout, input)
|
||||||
|
- [ ] **P4b** — misfit protocols rebased (device-manager, power, usb-transfer)
|
||||||
|
- [ ] **P4c** — harness subscriber lift + badge-scoped per-client integers
|
||||||
|
- [ ] **merge** group 3 → main, push
|
||||||
|
- [ ] **H2** — SMEP on every core
|
||||||
|
- [ ] **HS** — SYSRET canonical-RIP guard
|
||||||
|
- [ ] **H3** — SMAP + boot-patched `clac`; `-cpu max` in the harness; negative tests
|
||||||
|
- [ ] **merge** group 4 → main, push
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Settled decisions (grounding pass, 2026-07-31 — do not reopen)
|
||||||
|
|
||||||
|
These resolve every open wrinkle the code inventory surfaced. Where one
|
||||||
|
amends a design doc, the amendment lands in the same commit as the phase
|
||||||
|
that implements it.
|
||||||
|
|
||||||
|
1. **Every packet — request, reply, and event — begins with the envelope
|
||||||
|
`Header`, exactly as the design says; the header is FOLDED, never
|
||||||
|
stacked.** It absorbs each protocol's existing operation/id fields
|
||||||
|
rather than sitting on top of them, so the two apparent 64-byte-limit
|
||||||
|
offenders fit: `ChildAdded` re-lays to 60 bytes (its packed operation
|
||||||
|
byte and `device_id` become `Header.operation`/`.target`);
|
||||||
|
`InterruptReport` puts `device_token` in `Header.target` and trims
|
||||||
|
inline data 48 → 40 bytes (largest real report today is 8). A
|
||||||
|
headerless-events variant was considered and REJECTED (2026-07-31): it
|
||||||
|
re-invents per-protocol mini-headers and breaks uniform tooling. No
|
||||||
|
design-doc amendment; `Define`'s event check stays ≤ 64 *including*
|
||||||
|
the header.
|
||||||
|
2. **Bind/open authorization is chain-attested identity: the
|
||||||
|
kernel-stamped binary name PLUS the supervision chain**, both read from
|
||||||
|
the kernel's process records (`ProcessDescriptor` carries `name` and
|
||||||
|
`supervisor`; init walks the chain with `process_enumerate` — no new
|
||||||
|
protocol). A grant row names the binary *and* the supervisor expected
|
||||||
|
in its chain, so a malicious process re-spawning a granted binary
|
||||||
|
(ungated `spawn`, hostile argv — the confused deputy) is refused: its
|
||||||
|
chain roots at the attacker, not at init or device-manager. Name alone
|
||||||
|
is NOT sufficient — that was considered and rejected (2026-07-31).
|
||||||
|
Pure delegation (device-manager forwarding driver binds as
|
||||||
|
capabilities — "option B") is deliberately deferred to P5, whose
|
||||||
|
spawner-wired namespaces subsume it. Amends protocol-namespace.md's
|
||||||
|
"Authorization" bullet in P2.
|
||||||
|
3. **Grants live in a new manifest, `/system/configuration/protocol.csv`**
|
||||||
|
(rows: `binary-path, supervisor, bind|open, protocol-name`, where
|
||||||
|
`supervisor` is the binary expected in the caller's supervision chain —
|
||||||
|
`init` for init's own children, `kernel` for harness-spawned fixtures),
|
||||||
|
not in extra init.csv columns — today every post-path init.csv field is
|
||||||
|
argv, and overloading that is ambiguous. init parses both files.
|
||||||
|
4. **Test fixtures bind under `/protocol/test/...`**, granted to any
|
||||||
|
binary whose path starts `/test/` — the subtree-scoping rule from the
|
||||||
|
design doc, dogfooded. `shared_memory_test` (the borrowed-ServiceId
|
||||||
|
hack) becomes `/protocol/test/shared-memory`; process-test's child gets
|
||||||
|
`/protocol/test/process`.
|
||||||
|
5. **Rebind after provider death:** a `bind` hitting an existing binding
|
||||||
|
succeeds only if the current owner process is dead (init checks
|
||||||
|
liveness); otherwise `-EBUSY`. Init also unbinds in `restartChild`
|
||||||
|
before respawning its own children. This preserves collision-refusal
|
||||||
|
while making restart work for providers init does not supervise.
|
||||||
|
6. **Cross-thread service access** (the display mouse-listener's
|
||||||
|
per-thread self-lookup, `display.zig:512`): threads resolve and open
|
||||||
|
`/protocol/<name>` like any client — once, at thread startup. No
|
||||||
|
special mechanism.
|
||||||
|
7. **The envelope module is `library/protocol/envelope/envelope.zig`**
|
||||||
|
(module name `envelope`) — the one protocol-package module not ending
|
||||||
|
in `-protocol`, because it is not a protocol. Wired as a new
|
||||||
|
`addModule` row in `library/protocol/build.zig` with its host tests in
|
||||||
|
that package's test step.
|
||||||
|
8. **The QEMU harness gains `-cpu max`** (in `qemu_args`,
|
||||||
|
`test/qemu_test.py:66-83`) so TCG exposes SMEP/SMAP — without it the
|
||||||
|
enabled paths never execute in CI. Landed in H2 so the flag soaks
|
||||||
|
before H3 depends on it.
|
||||||
|
9. **Scenario fixtures that need the registry are init-driven.** Kernel
|
||||||
|
test cases that today spawn providers directly (shared-memory,
|
||||||
|
process-test) either spawn init first or move to init.csv-driven
|
||||||
|
scenario boots — resolved per-case in P2 with the suite as the
|
||||||
|
arbiter.
|
||||||
|
10. **The capsule-staleness caveat is documented, not fixed.** On-volume
|
||||||
|
edits to `/system/configuration/*.csv` do not reach the initrd copy
|
||||||
|
the loader boots (capsule shadows tree). Same drift exists today with
|
||||||
|
`/etc`; PM adds the note to file-system-hierarchy.md and moves on.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## PM — path-migration flag-day
|
||||||
|
|
||||||
|
One commit, everything moves together. The authoritative site inventory is
|
||||||
|
the grounding pass; the checklist order:
|
||||||
|
|
||||||
|
1. Move repo `etc/` → `configuration/` sources; fix the three CSVs'
|
||||||
|
self-referencing headers (`etc/init.csv:1,12`, `etc/devices.csv:1`,
|
||||||
|
`etc/init-diagnose.csv:1`).
|
||||||
|
2. `build.zig:309-311`: bundled entries `etc/...` →
|
||||||
|
`system/configuration/...` (this alone re-shapes the image, manifest,
|
||||||
|
and capsule — `tools/make-fat-image.py` and the EFI loader need
|
||||||
|
nothing; the tree-walk fallback even starts picking the CSVs up, a
|
||||||
|
bonus fix).
|
||||||
|
3. `system/kernel/vfs.zig` `mountBackend` (`:332-340`): allow exactly
|
||||||
|
`/system/configuration` and `/system/logs` as backend prefixes beneath
|
||||||
|
the initrd `/system` mount; keep refusing everything else under
|
||||||
|
`/system` and `/test`.
|
||||||
|
4. `system/services/fat/fat.zig`: `mount_point` → `/volumes/usb` (`:25`);
|
||||||
|
replace the `/var` mount (`:155`) with two `mountRewritten` calls for
|
||||||
|
`/system/configuration` and `/system/logs`; update the mount log lines
|
||||||
|
(the harness matches them).
|
||||||
|
5. `system/services/init/init.zig:76` and
|
||||||
|
`system/services/device-manager/device-manager.zig:48`: open the new
|
||||||
|
CSV paths; update the message strings (`init.zig:77,92`,
|
||||||
|
`device-manager.zig:49,61-63,454`).
|
||||||
|
6. `system/services/logger/logger.zig:44`: `base = "/system/logs"`
|
||||||
|
(buffers derive from `base.len` comptime — nothing else changes).
|
||||||
|
7. `system/kernel/tests.zig:2808-2810`: exclude `/system/configuration/`
|
||||||
|
from the spawn-everything sweep (the CSVs are not programs).
|
||||||
|
8. Tests: `fat-test.zig` and `vfs-test.zig` `/mnt/usb` literals →
|
||||||
|
`/volumes/usb`; harness regexes `test/qemu_test.py:175,211,632,717`.
|
||||||
|
9. Comment sweep (init, device-manager, logger, fat, engine, vfs, abi,
|
||||||
|
file-system, csv, device, protocol/device-manager, drivers, acpi,
|
||||||
|
build.zig — full list in the grounding inventory); delete vestigial
|
||||||
|
repo `var/`.
|
||||||
|
|
||||||
|
**Test:** no new case — the existing 106 are the test, since fat/logger/
|
||||||
|
init/device-manager scenarios all assert the new paths through their
|
||||||
|
regexes. Suite stays 106.
|
||||||
|
|
||||||
|
## H1 — user-memory copy discipline
|
||||||
|
|
||||||
|
New kernel module `system/kernel/user-memory.zig`:
|
||||||
|
|
||||||
|
- `copyFromUser` moves from ipc-synchronous.zig (which re-exports or
|
||||||
|
imports it); new `copyToUser(user_as, user_va, source) bool` — the
|
||||||
|
mechanical mirror (kernel-source `copyAcross` already does this for IPC
|
||||||
|
replies at `ipc-synchronous.zig:431,460`).
|
||||||
|
- The page walk gains leaf U/S and writable checks: `paging.translateIn`
|
||||||
|
(`architecture/x86_64/paging.zig:513-525`) tests only `present` today —
|
||||||
|
add a flags-accumulating variant (2 MiB leaves included); reads require
|
||||||
|
U/S, writes require U/S+W. Closes the TODO at
|
||||||
|
`ipc-synchronous.zig:20-22`.
|
||||||
|
- Convert the nine stragglers (table in smep-smap.md). Read direction is
|
||||||
|
local to `process.zig`; the write direction restructures callees with
|
||||||
|
kernel bounce buffers: `scheduler.enumerate` (`scheduler.zig:1209`),
|
||||||
|
`devices_broker.enumerate` (`devices-broker.zig:136`), `log.readAt`
|
||||||
|
(`log.zig:209`), and the `fs_node` flows through
|
||||||
|
`vfs.nodeRead/nodeStatus/nodeReaddir` (`vfs.zig:257/269/289`).
|
||||||
|
|
||||||
|
**Test:** kernel unit coverage in `system/kernel/tests.zig` for
|
||||||
|
`copyToUser` bounds/permission refusals; one new QEMU case `user-memory` —
|
||||||
|
a fixture passes an unmapped-but-in-range buffer to `klog_read`,
|
||||||
|
`process_enumerate`, and `fs_resolve` and asserts `-EFAULT` returns with
|
||||||
|
the system still alive (today each would oops the kernel). Suite 107.
|
||||||
|
|
||||||
|
## P1 — envelope, vfs additions, Channel
|
||||||
|
|
||||||
|
- `library/protocol/envelope/envelope.zig`: `Header` {operation:u32, pad,
|
||||||
|
target:u64}, `Status`, reserved verbs (describe=0, enumerate=1,
|
||||||
|
subscribe=2, unsubscribe=3, protocol verbs from 16), `packet_maximum`
|
||||||
|
= 256 / `post_maximum` = 64 (the floor constants protocols compile
|
||||||
|
against — nothing exports them today), and comptime
|
||||||
|
`Define(.{name, version, operations, events})` generating request/reply
|
||||||
|
types, encode/decode, a provider dispatch table (automatic `describe`,
|
||||||
|
`-ENOSYS` for unknown verbs), and compile-time size checks:
|
||||||
|
request/reply ≤ 256, each `.events` entry ≤ 64 *including* its Header
|
||||||
|
(decision 1). Host unit tests in the protocol package's test step.
|
||||||
|
- `library/protocol/vfs/vfs-protocol.zig`: `NodeKind.protocol = 7`; the
|
||||||
|
open-reply-may-carry-capability convention documented in the module.
|
||||||
|
Rewrite the value-pinning unit test (`:108-117`) to pin the *new*
|
||||||
|
stable values.
|
||||||
|
- `library/kernel/file-system.zig` + a new `Channel` type in
|
||||||
|
`library/kernel` (or `library/client`): `open("/protocol/<name>")` →
|
||||||
|
resolve, vfs open, receive the reply capability → a `Channel` wrapping
|
||||||
|
the handle with `call`/typed helpers. Nothing uses it yet — P2 converts
|
||||||
|
the world.
|
||||||
|
- Docs: vfs-protocol.md's NodeKind table gains value 7 (no
|
||||||
|
protocol-namespace.md amendment — decision 1 conforms to it as written).
|
||||||
|
|
||||||
|
**Test:** host unit tests only (envelope round-trips, size-check compile
|
||||||
|
errors via `error` tests, Channel plumbing against a mock). Suite stays
|
||||||
|
107.
|
||||||
|
|
||||||
|
## P2 — the registry; ServiceId flag-day
|
||||||
|
|
||||||
|
The single biggest phase; one branch, may be several commits, green at the
|
||||||
|
end of each.
|
||||||
|
|
||||||
|
- **init as registry backend** (`system/services/init/init.zig`): a second
|
||||||
|
endpoint (the supervision endpoint's reply-empty loop is unsuitable for
|
||||||
|
a vfs backend); serve vfs `open`/`readdir` over `/protocol` plus the
|
||||||
|
`bind` operation (name payload + capability). Mount `/protocol` before
|
||||||
|
spawning children. Parse `/system/configuration/protocol.csv`
|
||||||
|
(decision 3). Authorization by chain-attested identity (decision 2):
|
||||||
|
badge → kernel process records → binary name **and** supervision chain
|
||||||
|
(walk `supervisor` links) checked against the grant row's expected
|
||||||
|
supervisor. Unbind on child death in `restartChild`; dead-owner rebind
|
||||||
|
rule (decision 5).
|
||||||
|
Provenance: readdir/diagnostics show name → pid → binary path.
|
||||||
|
- **Kernel:** reserve `/protocol` — `mountBackend` refuses mounts at or
|
||||||
|
under it once bound, `installMount`'s remount-replace path refuses it,
|
||||||
|
and `fs_unmount` refuses it (`vfs.zig:164-181,332-351`,
|
||||||
|
`process.zig:1879-1889`). First mount wins (init is PID 1).
|
||||||
|
- **Harness:** `library/kernel/service.zig` `Callbacks.service:
|
||||||
|
?abi.ServiceId` becomes a protocol name; the register call (`:49-51`)
|
||||||
|
becomes bind-with-retry via the registry.
|
||||||
|
- **Flag-day conversion** — all 11 registration sites and 17 lookup sites
|
||||||
|
from the grounding inventory: providers (input:123, ps2-bus:223,
|
||||||
|
device-manager:569, acpi:193, usb-xhci-bus:676, usb-storage:205,
|
||||||
|
fat:307, display:699, virtio-gpu:550, shared-memory-server:43,
|
||||||
|
process-test:130 → `/protocol/test/...` per decision 4); clients
|
||||||
|
(input-client:53, display-client:28, driver.zig:173, usb.zig:139,
|
||||||
|
block.zig:72+87, ps2-bus keyboard:35 + mouse:34, virtio-gpu:478,
|
||||||
|
display:314+512 (decision 6), acpi:212, init:218+245 — init
|
||||||
|
short-circuits its own registry, shared-memory-client:22,
|
||||||
|
process-test:85, device-list:22, crash-test:32). Retry loops keep their
|
||||||
|
cadence, wrapping resolve+open instead of lookup.
|
||||||
|
- **Delete:** `abi.zig:36-37` (syscall ids — leave holes),
|
||||||
|
`abi.zig:287-303` (enum), `process.zig:223-224,314-343`,
|
||||||
|
`ipc-synchronous.zig:41-43,646-664` and the registry sweep in
|
||||||
|
`:121-140`; the wrappers `library/kernel/ipc.zig:33-35,47-50`; comment
|
||||||
|
sweep (irq.zig:50, tests.zig:3744, vdso.md's syscall table, the docs
|
||||||
|
list in the inventory).
|
||||||
|
- Kernel-spawned test scenarios made init-driven where they need the
|
||||||
|
registry (decision 9).
|
||||||
|
|
||||||
|
**Test:** new QEMU case `protocol-registry`: a fixture asserts (a) bind of
|
||||||
|
an ungranted name → `-EPERM`, (b) bind collision with a live owner →
|
||||||
|
`-EBUSY`, (c) provider kill → re-resolve reaches the restarted instance.
|
||||||
|
Every existing scenario doubles as conversion proof. Suite 108.
|
||||||
|
|
||||||
|
## P3 — open grants (restriction stage one)
|
||||||
|
|
||||||
|
- `protocol.csv` `open` rows enforced in the registry's `open` handler,
|
||||||
|
same name-based identity as bind. Default rows grant what today's
|
||||||
|
clients need (from the P2 conversion table); a deliberate hole for the
|
||||||
|
test fixture.
|
||||||
|
- Docs: protocol-namespace.md stage-one section gets its "landed" line.
|
||||||
|
|
||||||
|
**Test:** new QEMU case `protocol-denied`: a fixture granted
|
||||||
|
`/protocol/test/shared-memory` but not `/protocol/display` asserts open of
|
||||||
|
the first succeeds and the second fails identically to not-found. Suite
|
||||||
|
109.
|
||||||
|
|
||||||
|
## P4a — clean protocols onto Define
|
||||||
|
|
||||||
|
vfs, block, display, scanout, input — the modules whose shapes map
|
||||||
|
directly (grounding inventory §1,3,4,6,8):
|
||||||
|
|
||||||
|
- vfs: `node` → `target`; `Reply.node` (open's result) moves to reply
|
||||||
|
payload — `library/kernel/file-system.zig` decoders change; readdir
|
||||||
|
stays a protocol verb.
|
||||||
|
- block: pure renumber; `attach`'s DMA cap rides the call as today.
|
||||||
|
- display: the overloaded 40-byte `Request` becomes per-operation structs
|
||||||
|
(attach_scanout's field abuse dies); `layer` → `target`; blit payload
|
||||||
|
grows to 224 bytes.
|
||||||
|
- scanout: renumber; drop its bogus `message_maximum=64` (sync floor is
|
||||||
|
256); fix virtio-gpu's hard-coded `service.run(256, …)` to the
|
||||||
|
generated constant.
|
||||||
|
- input: subscribe merges into reserved subscribe; publish renumbers;
|
||||||
|
the event re-lays onto the Header folded (operation = event kind,
|
||||||
|
target = 0; 16 + 28-byte payload = 44 ≤ 64); **input moves onto the
|
||||||
|
service harness** (it is the last hand-rolled loop, no ping/terminate
|
||||||
|
compliance today).
|
||||||
|
|
||||||
|
**Test:** new QEMU case `protocol-conformance`: a fixture opens every
|
||||||
|
registered protocol and asserts `describe` answers (name, version) and an
|
||||||
|
unknown verb returns `-ENOSYS`. Existing input/display/fat scenarios prove
|
||||||
|
the rebase. Suite 110.
|
||||||
|
|
||||||
|
## P4b — misfit protocols onto Define
|
||||||
|
|
||||||
|
device-manager, power, usb-transfer (inventory §2,5,7 — the u8-operation
|
||||||
|
re-layouts and raw-offset readers):
|
||||||
|
|
||||||
|
- device-manager: u8 operations → Header; its enumerate=4/subscribe=5
|
||||||
|
merge into the reserved verbs; `ChildAdded` splits its dual role —
|
||||||
|
request struct and event, both Header-first (folded to 60 B ≤ 64);
|
||||||
|
`ChildRemoved`'s (parent, bus_address) addressing stays payload.
|
||||||
|
- power: u8 operations → Header; subscribe merges; **init's raw
|
||||||
|
byte-offset event parsing (`init.zig:171-173`) and acpi's
|
||||||
|
`message[0]` dispatch (`acpi.zig:435-467`) are rewritten against the
|
||||||
|
generated types** — the two silent-breakage sites, called out so the
|
||||||
|
loop treats them as first-class conversions, not collateral.
|
||||||
|
- usb-transfer: `device_token` → `target` (already layout-identical);
|
||||||
|
`InterruptReport` re-lays onto the Header (`device_token` → `target`,
|
||||||
|
inline data trimmed 48 → 40 — largest real report is 8); control/bulk
|
||||||
|
budgets re-verified by `Define` (Status absorbs `actual_length`).
|
||||||
|
|
||||||
|
**Test:** existing scenarios are the proof (device hot-add, power button,
|
||||||
|
USB storage/HID all exercise these wires); the conformance case now covers
|
||||||
|
three more providers. Suite 110.
|
||||||
|
|
||||||
|
## P4c — harness subscriber lift + badge scoping
|
||||||
|
|
||||||
|
- `library/kernel/service.zig` grows the subscriber table, exit-
|
||||||
|
notification sweep, and fan-out loop declared via `Define(.events)`;
|
||||||
|
input (:33-116), acpi (:67-68,393-406), and device-manager (:155-166)
|
||||||
|
delete their hand-rolled variants. One sweep idiom: exit notifications
|
||||||
|
(fat's pattern), replacing input's process-list polling and acpi's
|
||||||
|
none-at-all.
|
||||||
|
- Badge-scoped per-client integers (the guessable-id holes): fat node ids
|
||||||
|
gain owner checks on every operation (`fat.zig:72-76`), xhci device
|
||||||
|
tokens validate sender and sweep on exit (`usb-xhci-bus.zig:66-88,479`),
|
||||||
|
display layers gain an owner field.
|
||||||
|
|
||||||
|
**Test:** extend the fat scenario: a second fixture guesses the first's
|
||||||
|
node id and asserts refusal; kernel-side unit test for the harness sweep.
|
||||||
|
Suite 111.
|
||||||
|
|
||||||
|
## H2 — SMEP
|
||||||
|
|
||||||
|
- Generalize the cpuid helper (`apic.zig:351-365`, private, subleaf-0) to
|
||||||
|
a shared probe; gate on `cpuid(0).eax >= 7`.
|
||||||
|
- Set CR4 bit 20 in `per-cpu.zig:initSystemCall` (or a sibling called
|
||||||
|
from both `cpu.zig:148` and `smp.zig:181` — the one path both BSP and
|
||||||
|
every AP already execute). Log enabled/absent (fail-open, IOMMU style).
|
||||||
|
- Harness: add `-cpu max` to `qemu_args` (decision 8).
|
||||||
|
|
||||||
|
**Test:** new QEMU case `fault-smep` mirroring the `fault-*` injector
|
||||||
|
pattern (`tests.zig:3906-3938`): ring-0 call through a pointer into a
|
||||||
|
user-mapped page; expect `page fault (vector 14)` + `error code : 0x11` +
|
||||||
|
kernel-half IP, machine reports the exception (deliberate-exception cases
|
||||||
|
put the text in `expect`, per `qemu_test.py:189`). Suite 112.
|
||||||
|
|
||||||
|
## HS — SYSRET canonical-RIP guard
|
||||||
|
|
||||||
|
- `isr.s` syscall exit (`:256`): validate RCX canonicality before
|
||||||
|
`sysretq`; non-canonical → `iretq` fallback (or kill), per the hazard
|
||||||
|
note at `isr.s:192-194`.
|
||||||
|
|
||||||
|
**Test:** kernel unit case driving a thread whose return RIP is forged
|
||||||
|
non-canonical via the syscall path if constructible cheaply; otherwise the
|
||||||
|
review-level proof plus the existing fault cases regression. Suite 112.
|
||||||
|
|
||||||
|
## H3 — SMAP
|
||||||
|
|
||||||
|
- `clac` patch site at `isr_common` (`isr.s:367`, before the CPL test —
|
||||||
|
ring-0 nesting inherits AC too): assemble a 3-byte NOP, patch to `clac`
|
||||||
|
at boot through the physmap (the `process.zig:1990-1995` /
|
||||||
|
`smp.zig:79-111` precedent), BSP-only before AP bring-up.
|
||||||
|
- Set CR4 bit 21 in the same per-CPU init as SMEP.
|
||||||
|
- Coding standards: kernel code touches user memory only through
|
||||||
|
`user-memory`; no `stac` anywhere, ever.
|
||||||
|
|
||||||
|
**Test:** new QEMU case `fault-smap`: ring-0 deliberate read of a mapped
|
||||||
|
user page; expect vector 14 + `error code : 0x1` + kernel IP. And the
|
||||||
|
whole suite becomes the tripwire — any missed straggler now fails loudly.
|
||||||
|
Suite 113.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
**Explicitly out of scope** (own tracks, after this plan): P5 restriction
|
||||||
|
stage two (spawn's initial capability, namespace views, parked replies,
|
||||||
|
dedicated killable channels — needs a design session on the spawn
|
||||||
|
contract), file-path namespacing, trusted UI (display track), pipes/FIFOs
|
||||||
|
(Python track), `/applications` and its storage, `fs_mount`/`spawn`/
|
||||||
|
`klog_read` gating beyond the `/protocol` reserved prefix, KPTI, IPC
|
||||||
|
priority inheritance.
|
||||||
@@ -95,9 +95,9 @@ hypervisor configured for UEFI firmware and an xHCI USB controller.
|
|||||||
|
|
||||||
| Requirement | Detail | Source |
|
| Requirement | Detail | Source |
|
||||||
|---|---|---|
|
|---|---|---|
|
||||||
| **x86-64, 64-bit only** | Kernel and loader are built exclusively for `x86_64`; the loader rejects any non-x86-64 kernel ELF (`error.WrongArchitecture`). | `build.zig:481`, `boot/efi.zig:622` |
|
| **x86-64, 64-bit only** | Kernel and loader are built exclusively for `x86_64`; the loader rejects any non-x86-64 kernel ELF (`error.WrongArchitecture`). | `build-support/build.zig` (`freestandingTarget`), `boot/efi.zig:622` |
|
||||||
| **Long mode + PAE + NX** | AP trampoline sets `CR4.PAE`, `EFER.LME`, `EFER.NXE`; NX is used in kernel page-table entries. | `system/kernel/architecture/x86_64/trampoline.s:62` |
|
| **Long mode + PAE + NX** | AP trampoline sets `CR4.PAE`, `EFER.LME`, `EFER.NXE`; NX is used in kernel page-table entries. | `system/kernel/architecture/x86_64/trampoline.s:62` |
|
||||||
| **SSE / SSE2** | Baseline: the compiler emits SSE for ordinary struct copies. Trampoline enables `CR4.OSFXSR` + `OSXMMEXCPT` and clears `CR0.EM`. | `build.zig:477`, `trampoline.s:62` |
|
| **SSE / SSE2** | Baseline: the compiler emits SSE for ordinary struct copies. Trampoline enables `CR4.OSFXSR` + `OSXMMEXCPT` and clears `CR0.EM`. | `build-support/build.zig` (`freestandingTarget`), `trampoline.s:62` |
|
||||||
| **`syscall` / `sysret`** | Primary user↔kernel entry path. `EFER.SCE` enabled; `STAR`/`LSTAR`/`SFMASK` programmed per core. (`int 0x80` exists as a parallel gate.) | `architecture/x86_64/per-cpu.zig:71`, `isr.s:196` |
|
| **`syscall` / `sysret`** | Primary user↔kernel entry path. `EFER.SCE` enabled; `STAR`/`LSTAR`/`SFMASK` programmed per core. (`int 0x80` exists as a parallel gate.) | `architecture/x86_64/per-cpu.zig:71`, `isr.s:196` |
|
||||||
| **Local APIC (xAPIC)** | LAPIC accessed via MMIO at `0xFEE00000`. LAPIC ID read as a `u8` — classic xAPIC. **x2APIC is not supported** (no MSR path). | `apic.zig:67`, `apic.zig:646` |
|
| **Local APIC (xAPIC)** | LAPIC accessed via MMIO at `0xFEE00000`. LAPIC ID read as a `u8` — classic xAPIC. **x2APIC is not supported** (no MSR path). | `apic.zig:67`, `apic.zig:646` |
|
||||||
| **CPUID + RDTSC** | CPUID leaf `0x15` for TSC frequency; RDTSC is the monotonic clock. | `apic.zig:333`, `apic.zig:113` |
|
| **CPUID + RDTSC** | CPUID leaf `0x15` for TSC frequency; RDTSC is the monotonic clock. | `apic.zig:333`, `apic.zig:113` |
|
||||||
@@ -108,7 +108,7 @@ hypervisor configured for UEFI firmware and an xHCI USB controller.
|
|||||||
- **UEFI only.** A custom UEFI application loader is installed to
|
- **UEFI only.** A custom UEFI application loader is installed to
|
||||||
`\EFI\BOOT\BOOTX64.efi`. There is **no BIOS, multiboot, or limine** path. The
|
`\EFI\BOOT\BOOTX64.efi`. There is **no BIOS, multiboot, or limine** path. The
|
||||||
loader tolerates UEFI Class-3 machines with no legacy PIC/PIT.
|
loader tolerates UEFI Class-3 machines with no legacy PIC/PIT.
|
||||||
(`build.zig:246`, `boot/efi.zig`)
|
(`build/images.zig` — the EFI/BOOT install — and `boot/efi.zig`)
|
||||||
- **ACPI is the hardware-discovery mechanism.** The RSDP is taken from the UEFI
|
- **ACPI is the hardware-discovery mechanism.** The RSDP is taken from the UEFI
|
||||||
configuration table (ACPI 2.0 GUID preferred, 1.0 fallback). Without a valid
|
configuration table (ACPI 2.0 GUID preferred, 1.0 fallback). Without a valid
|
||||||
RSDP there is **no device discovery** — no SMP, no IOAPIC routing, no PCI/USB.
|
RSDP there is **no device discovery** — no SMP, no IOAPIC routing, no PCI/USB.
|
||||||
|
|||||||
+3
-1
@@ -12,7 +12,9 @@ There are two layers:
|
|||||||
`system/abi.zig`, `library/device/model/device-abi.zig`) now spans ~26 modules:
|
`system/abi.zig`, `library/device/model/device-abi.zig`) now spans ~26 modules:
|
||||||
protocol and on-wire definitions (VFS, USB, virtio-gpu), the FAT engine, the
|
protocol and on-wire definitions (VFS, USB, virtio-gpu), the FAT engine, the
|
||||||
display compositor, PS/2 and HID decoding, the kernel log ring, and the
|
display compositor, PS/2 and HID decoding, the kernel log ring, and the
|
||||||
runtime's `time`/`thread` — the full list is the test step in `build.zig`.
|
runtime's `time`/`thread` — the list is distributed across the library-domain
|
||||||
|
and binary packages' own `test` steps, which the root `zig build test`
|
||||||
|
aggregates (docs/build-packages-plan.md).
|
||||||
These compile for the host and run natively.
|
These compile for the host and run natively.
|
||||||
- **QEMU integration tests** (`python3 test/qemu_test.py`) — boot the real kernel
|
- **QEMU integration tests** (`python3 test/qemu_test.py`) — boot the real kernel
|
||||||
and check its behaviour. This is the interesting part.
|
and check its behaviour. This is the interesting part.
|
||||||
|
|||||||
@@ -347,7 +347,7 @@ Two current decisions fall out of this roadmap:
|
|||||||
- [syscall.md](os-development/syscall.md) — the kernel↔runtime ABI `runtime.os` is built on.
|
- [syscall.md](os-development/syscall.md) — the kernel↔runtime ABI `runtime.os` is built on.
|
||||||
- [sysv.md](os-development/sysv.md) — the entry stack (`argc/argv/envp/auxv`) danos already constructs.
|
- [sysv.md](os-development/sysv.md) — the entry stack (`argc/argv/envp/auxv`) danos already constructs.
|
||||||
- [ipc.md](device-driver-development/ipc.md) — the IPC the VFS/FAT operations travel over.
|
- [ipc.md](device-driver-development/ipc.md) — the IPC the VFS/FAT operations travel over.
|
||||||
- [danos-file-system-hierarchy-FSH.md](file-system-development/danos-file-system-hierarchy-FSH.md) — the
|
- [file-system-hierarchy.md](file-system-development/file-system-hierarchy.md) — the
|
||||||
filesystem layout the file surface serves.
|
filesystem layout the file surface serves.
|
||||||
- [coding-standards.md](coding-standards.md) — danos naming (why the compat spellings
|
- [coding-standards.md](coding-standards.md) — danos naming (why the compat spellings
|
||||||
are confined, and now retired).
|
are confined, and now retired).
|
||||||
|
|||||||
@@ -0,0 +1,38 @@
|
|||||||
|
//! The "client" library domain (library/client): userspace-service clients —
|
||||||
|
//! they talk to services over IPC, not to the kernel. Client modules end in
|
||||||
|
//! `-client` the way wire protocols end in `-protocol`, so a service, its
|
||||||
|
//! protocol, and its client never share a name (`display` the service,
|
||||||
|
//! `display-protocol` the wire contract, `display-client` a program's view).
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
pub fn build(b: *std.Build) void {
|
||||||
|
const kernel = b.dependency("kernel", .{});
|
||||||
|
const protocol = b.dependency("protocol", .{});
|
||||||
|
|
||||||
|
const ipc = kernel.module("ipc");
|
||||||
|
const time = kernel.module("time");
|
||||||
|
|
||||||
|
_ = b.addModule("display-client", .{
|
||||||
|
.root_source_file = b.path("display/display-client.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "ipc", .module = ipc },
|
||||||
|
.{ .name = "time", .module = time },
|
||||||
|
.{ .name = "display-protocol", .module = protocol.module("display-protocol") },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
_ = b.addModule("input-client", .{
|
||||||
|
.root_source_file = b.path("input/input-client.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "ipc", .module = ipc },
|
||||||
|
.{ .name = "time", .module = time },
|
||||||
|
.{ .name = "input-protocol", .module = protocol.module("input-protocol") },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
// Standalone `zig build test`, kept for uniformity across the domains (the
|
||||||
|
// root aggregate depends on every domain's test step). The clients have no
|
||||||
|
// host-runnable unit tests yet — they are thin IPC conversation wrappers —
|
||||||
|
// so the step is empty until one grows some.
|
||||||
|
_ = b.step("test", "Run the client unit tests (none yet)");
|
||||||
|
}
|
||||||
@@ -0,0 +1,13 @@
|
|||||||
|
.{
|
||||||
|
.name = .client,
|
||||||
|
.version = "0.0.0",
|
||||||
|
.fingerprint = 0xc74404553e73d4ff, // Changing this has security and trust implications.
|
||||||
|
.minimum_zig_version = "0.16.0",
|
||||||
|
.dependencies = .{
|
||||||
|
// The clients converse over ipc with time-bounded waits.
|
||||||
|
.kernel = .{ .path = "../kernel" },
|
||||||
|
// Each client speaks its service's wire protocol.
|
||||||
|
.protocol = .{ .path = "../protocol" },
|
||||||
|
},
|
||||||
|
.paths = .{""},
|
||||||
|
}
|
||||||
@@ -0,0 +1,20 @@
|
|||||||
|
//! The "csv" library domain: shared CSV helpers (comment stripping, field
|
||||||
|
//! iteration) for the /system/configuration/*.csv config files — the device registry and the
|
||||||
|
//! init service list both parse them.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
pub fn build(b: *std.Build) void {
|
||||||
|
_ = b.addModule("csv", .{ .root_source_file = b.path("csv.zig") });
|
||||||
|
|
||||||
|
// Standalone `zig build test` for this domain alone; the root build keeps
|
||||||
|
// its aggregate test step.
|
||||||
|
const test_step = b.step("test", "Run the csv unit tests");
|
||||||
|
const csv_tests = b.addTest(.{
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path("csv.zig"),
|
||||||
|
.target = b.resolveTargetQuery(.{}),
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
test_step.dependOn(&b.addRunArtifact(csv_tests).step);
|
||||||
|
}
|
||||||
@@ -0,0 +1,8 @@
|
|||||||
|
.{
|
||||||
|
.name = .csv,
|
||||||
|
.version = "0.0.0",
|
||||||
|
.fingerprint = 0x8a4525791f4e5b6, // Changing this has security and trust implications.
|
||||||
|
.minimum_zig_version = "0.16.0",
|
||||||
|
.dependencies = .{},
|
||||||
|
.paths = .{""},
|
||||||
|
}
|
||||||
@@ -0,0 +1,56 @@
|
|||||||
|
//! Minimal CSV helpers shared by the `/system/configuration/*.csv` config files — the device
|
||||||
|
//! registry (`/system/configuration/devices.csv`) and the init service list (`/system/configuration/init.csv`).
|
||||||
|
//! Freestanding, no allocator: returned fields are slices into the source line,
|
||||||
|
//! so the source must outlive them. `#` starts a comment (whole-line or trailing);
|
||||||
|
//! whitespace around a field is trimmed, so columns may be padded for alignment.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
/// Strip a trailing `#` comment and surrounding whitespace from one raw line.
|
||||||
|
/// A blank or comment-only line returns "" (length 0) — the caller's skip signal.
|
||||||
|
pub fn stripComment(raw: []const u8) []const u8 {
|
||||||
|
const body = if (std.mem.indexOfScalar(u8, raw, '#')) |hash| raw[0..hash] else raw;
|
||||||
|
return std.mem.trim(u8, body, " \t\r\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Iterate the comma-separated fields of a line body, each trimmed of spaces and
|
||||||
|
/// tabs. Build it from a `stripComment`ed body.
|
||||||
|
pub const Fields = struct {
|
||||||
|
inner: std.mem.SplitIterator(u8, .scalar),
|
||||||
|
|
||||||
|
/// The next field, trimmed, or null when the row is exhausted.
|
||||||
|
pub fn next(self: *Fields) ?[]const u8 {
|
||||||
|
const field = self.inner.next() orelse return null;
|
||||||
|
return std.mem.trim(u8, field, " \t");
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
pub fn fields(body: []const u8) Fields {
|
||||||
|
return .{ .inner = std.mem.splitScalar(u8, body, ',') };
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- tests -------------------------------------------------------------------
|
||||||
|
|
||||||
|
const testing = std.testing;
|
||||||
|
|
||||||
|
test "stripComment trims and drops comments" {
|
||||||
|
try testing.expectEqualStrings("a, b", stripComment(" a, b # trailing\r\n"));
|
||||||
|
try testing.expectEqualStrings("", stripComment(" # whole-line comment"));
|
||||||
|
try testing.expectEqualStrings("", stripComment(" \t "));
|
||||||
|
try testing.expectEqualStrings("x", stripComment("x"));
|
||||||
|
}
|
||||||
|
|
||||||
|
test "fields splits and trims each column" {
|
||||||
|
var it = fields(stripComment("pci, 03 , 80 , /system/drivers/x # note"));
|
||||||
|
try testing.expectEqualStrings("pci", it.next().?);
|
||||||
|
try testing.expectEqualStrings("03", it.next().?);
|
||||||
|
try testing.expectEqualStrings("80", it.next().?);
|
||||||
|
try testing.expectEqualStrings("/system/drivers/x", it.next().?);
|
||||||
|
try testing.expect(it.next() == null);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "a single field yields one column then null" {
|
||||||
|
var it = fields(stripComment("/system/services/input"));
|
||||||
|
try testing.expectEqualStrings("/system/services/input", it.next().?);
|
||||||
|
try testing.expect(it.next() == null);
|
||||||
|
}
|
||||||
@@ -28,6 +28,18 @@ pub const Device = struct {
|
|||||||
return .{ .block_size = result.block_size, .block_count = result.block_count };
|
return .{ .block_size = result.block_size, .block_count = result.block_count };
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Hand the block server a DMA-region capability (`handle` — from a `shareable`
|
||||||
|
/// dma_alloc) so it forwards it to the controller and the buffer's physical
|
||||||
|
/// addresses become reachable by the device. Call once per buffer before naming it
|
||||||
|
/// in `read`/`write`. Harmless success when no IOMMU is enforcing.
|
||||||
|
pub fn attach(self: Device, handle: ipc.Handle) bool {
|
||||||
|
var request = block_protocol.Request{ .operation = @intFromEnum(block_protocol.Operation.attach), .lba = 0, .count = 0, .physical = 0 };
|
||||||
|
var reply: [block_protocol.reply_size]u8 = undefined;
|
||||||
|
const result = ipc.callCap(self.endpoint, std.mem.asBytes(&request), &reply, handle) catch return false;
|
||||||
|
if (result.len < block_protocol.reply_size) return false;
|
||||||
|
return std.mem.bytesToValue(block_protocol.Reply, reply[0..block_protocol.reply_size]).status == 0;
|
||||||
|
}
|
||||||
|
|
||||||
/// Read `count` blocks starting at `lba` into the DMA buffer at `physical`.
|
/// Read `count` blocks starting at `lba` into the DMA buffer at `physical`.
|
||||||
pub fn read(self: Device, lba: u64, count: u32, physical: u64) bool {
|
pub fn read(self: Device, lba: u64, count: u32, physical: u64) bool {
|
||||||
return self.transfer(.read, lba, count, physical);
|
return self.transfer(.read, lba, count, physical);
|
||||||
|
|||||||
@@ -0,0 +1,137 @@
|
|||||||
|
//! The "device" library domain (library/device): what a driver author imports.
|
||||||
|
//! The flat reference data (device-abi, pci-class, acpi-ids, usb-abi, usb-ids),
|
||||||
|
//! typed MMIO access, the driver-side client libraries (driver, pci, usb,
|
||||||
|
//! block), the AML interpreter, and the data-driven device registry.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
pub fn build(b: *std.Build) void {
|
||||||
|
const kernel = b.dependency("kernel", .{});
|
||||||
|
const protocol = b.dependency("protocol", .{});
|
||||||
|
const csv = b.dependency("csv", .{});
|
||||||
|
|
||||||
|
const abi = kernel.module("abi");
|
||||||
|
const system_call = kernel.module("system-call");
|
||||||
|
const ipc = kernel.module("ipc");
|
||||||
|
const time = kernel.module("time");
|
||||||
|
|
||||||
|
// The devices sub-project's public interface (the flat wire types),
|
||||||
|
// importable by user space, unlike the kernel-internal device model it
|
||||||
|
// also feeds (system/kernel/device-model.zig).
|
||||||
|
const device_abi = b.addModule("device-abi", .{
|
||||||
|
.root_source_file = b.path("model/device-abi.zig"),
|
||||||
|
});
|
||||||
|
// PCI class-code decoding (class/subclass/prog-IF -> names). Pure reference
|
||||||
|
// data, shared by kernel discovery and any user-space PCI tool.
|
||||||
|
const pci_class = b.addModule("pci-class", .{
|
||||||
|
.root_source_file = b.path("pci/pci-class.zig"),
|
||||||
|
});
|
||||||
|
// ACPI/PnP hardware-ID (_HID) names — the flat analog of pci-class.
|
||||||
|
_ = b.addModule("acpi-ids", .{
|
||||||
|
.root_source_file = b.path("acpi/acpi-ids.zig"),
|
||||||
|
});
|
||||||
|
// The AML interpreter, a build module so the ring-3 acpi service can run
|
||||||
|
// the same parser the kernel does (docs/discovery.md). Pure Zig, no kernel
|
||||||
|
// imports — one source, two builds.
|
||||||
|
_ = b.addModule("aml", .{
|
||||||
|
.root_source_file = b.path("acpi/aml/aml.zig"),
|
||||||
|
});
|
||||||
|
// The USB device-framework wire ABI (chapter-9 set-up packets, standard +
|
||||||
|
// class requests, descriptors) and the USB class-code taxonomy.
|
||||||
|
const usb_abi = b.addModule("usb-abi", .{
|
||||||
|
.root_source_file = b.path("usb/usb-abi.zig"),
|
||||||
|
});
|
||||||
|
const usb_ids = b.addModule("usb-ids", .{
|
||||||
|
.root_source_file = b.path("usb/usb-ids.zig"),
|
||||||
|
});
|
||||||
|
// Typed volatile MMIO register access + memory-ordering barriers, for
|
||||||
|
// drivers on top of an mmio_map grant. Depends only on `builtin`.
|
||||||
|
const mmio = b.addModule("mmio", .{
|
||||||
|
.root_source_file = b.path("mmio/mmio.zig"),
|
||||||
|
});
|
||||||
|
// The driver author's interface: device access + the device-manager hello
|
||||||
|
// handshake, folded together.
|
||||||
|
const driver = b.addModule("driver", .{
|
||||||
|
.root_source_file = b.path("driver/driver.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "abi", .module = abi },
|
||||||
|
.{ .name = "device-abi", .module = device_abi },
|
||||||
|
.{ .name = "system-call", .module = system_call },
|
||||||
|
.{ .name = "ipc", .module = ipc },
|
||||||
|
.{ .name = "time", .module = time },
|
||||||
|
.{ .name = "device-manager-protocol", .module = protocol.module("device-manager-protocol") },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
// A device driver's view of its claimed PCI function: config-space header
|
||||||
|
// fields, BAR decode + map, capability walks (legacy + extended), MSI/MSI-X
|
||||||
|
// programming, power state, and function-level reset — the generic PCI
|
||||||
|
// mechanics every leaf PCI driver used to re-derive inline.
|
||||||
|
_ = b.addModule("pci", .{
|
||||||
|
.root_source_file = b.path("pci/pci.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "driver", .module = driver },
|
||||||
|
.{ .name = "mmio", .module = mmio },
|
||||||
|
.{ .name = "pci-class", .module = pci_class },
|
||||||
|
.{ .name = "time", .module = time },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
// The USB class-driver transfer client: open a device on the xHCI bus and
|
||||||
|
// drive it (control / interrupt / bulk). Re-exports usb-abi / usb-ids as
|
||||||
|
// usb.abi / usb.ids for a single USB import.
|
||||||
|
_ = b.addModule("usb", .{
|
||||||
|
.root_source_file = b.path("usb/usb.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "ipc", .module = ipc },
|
||||||
|
.{ .name = "time", .module = time },
|
||||||
|
.{ .name = "usb-transfer-protocol", .module = protocol.module("usb-transfer-protocol") },
|
||||||
|
.{ .name = "usb-abi", .module = usb_abi },
|
||||||
|
.{ .name = "usb-ids", .module = usb_ids },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
// The block-device client — a device type, so it lives here.
|
||||||
|
_ = b.addModule("block", .{
|
||||||
|
.root_source_file = b.path("block/block.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "ipc", .module = ipc },
|
||||||
|
.{ .name = "time", .module = time },
|
||||||
|
.{ .name = "block-protocol", .module = protocol.module("block-protocol") },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
// The device registry: parse /system/configuration/devices.csv into match rules and bind a
|
||||||
|
// reported device to a driver. Pure logic (no hardware, no syscalls), so it
|
||||||
|
// unit-tests on the host; the device manager imports it.
|
||||||
|
_ = b.addModule("device-registry", .{
|
||||||
|
.root_source_file = b.path("registry/device-registry.zig"),
|
||||||
|
.imports = &.{.{ .name = "csv", .module = csv.module("csv") }},
|
||||||
|
});
|
||||||
|
|
||||||
|
// Standalone `zig build test` for this domain alone; the root build keeps
|
||||||
|
// its aggregate test step.
|
||||||
|
const test_step = b.step("test", "Run the device library unit tests");
|
||||||
|
for ([_][]const u8{
|
||||||
|
"model/device-abi.zig", // wire-type sizes
|
||||||
|
"pci/pci-class.zig", // class/subclass/prog-IF name decoding
|
||||||
|
"acpi/acpi-ids.zig", // _HID name decoding
|
||||||
|
"acpi/aml/aml.zig", // AML parse + interpret, incl. Notify dispatch
|
||||||
|
"usb/usb-abi.zig", // wire sizes + bit packings + set-up packet encodings
|
||||||
|
"usb/usb-ids.zig", // class/subclass/protocol code assignments
|
||||||
|
"mmio/mmio.zig", // barriers assemble + registers round-trip
|
||||||
|
}) |root| {
|
||||||
|
const device_tests = b.addTest(.{
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path(root),
|
||||||
|
.target = b.resolveTargetQuery(.{}),
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
test_step.dependOn(&b.addRunArtifact(device_tests).step);
|
||||||
|
}
|
||||||
|
// The registry needs its csv import wired, so it doesn't fit the loop.
|
||||||
|
const registry_tests = b.addTest(.{
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path("registry/device-registry.zig"),
|
||||||
|
.target = b.resolveTargetQuery(.{}),
|
||||||
|
.imports = &.{.{ .name = "csv", .module = csv.module("csv") }},
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
test_step.dependOn(&b.addRunArtifact(registry_tests).step);
|
||||||
|
}
|
||||||
@@ -0,0 +1,15 @@
|
|||||||
|
.{
|
||||||
|
.name = .device,
|
||||||
|
.version = "0.0.0",
|
||||||
|
.fingerprint = 0x92fb68eace23a4f, // Changing this has security and trust implications.
|
||||||
|
.minimum_zig_version = "0.16.0",
|
||||||
|
.dependencies = .{
|
||||||
|
// driver/block/usb/pci build on the kernel library's concern modules.
|
||||||
|
.kernel = .{ .path = "../kernel" },
|
||||||
|
// driver speaks device-manager-protocol; block/usb their transfer protocols.
|
||||||
|
.protocol = .{ .path = "../protocol" },
|
||||||
|
// device-registry parses /system/configuration/devices.csv with the shared csv helpers.
|
||||||
|
.csv = .{ .path = "../csv" },
|
||||||
|
},
|
||||||
|
.paths = .{""},
|
||||||
|
}
|
||||||
@@ -99,6 +99,27 @@ pub fn msiBind(device_id: u64, endpoint: usize) ?Msi {
|
|||||||
return .{ .address = rax, .data = @intCast(rdx) };
|
return .{ .address = rax, .data = @intCast(rdx) };
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Map a delegated DMA-region (or shared-memory) capability into a claimed device's
|
||||||
|
/// IOMMU domain, so the device may DMA to that buffer. The caller must own `device_id`
|
||||||
|
/// and hold `handle` (received over IPC or from its own `dma.alloc(.. | shareable)`).
|
||||||
|
/// Idempotent. Returns true on success (and trivially when no IOMMU is present).
|
||||||
|
pub fn dmaBind(device_id: u64, handle: usize) bool {
|
||||||
|
return !failed(sc.systemCall2(.dma_bind, device_id, handle));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Unmap a previously `dmaBind`'d buffer from the device's domain.
|
||||||
|
pub fn dmaUnbind(device_id: u64, handle: usize) bool {
|
||||||
|
return !failed(sc.systemCall2(.dma_unbind, device_id, handle));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Drain and log any pending IOMMU translation faults, returning the count seen. A
|
||||||
|
/// diagnostic: a driver that suspects its device attempted an out-of-domain DMA (or a
|
||||||
|
/// test proving enforcement) forces the hardware's fault records to the log now. Returns
|
||||||
|
/// 0 when no IOMMU is present.
|
||||||
|
pub fn iommuFaultDrain() usize {
|
||||||
|
return sc.systemCall0(.iommu_fault_drain);
|
||||||
|
}
|
||||||
|
|
||||||
/// Read `width` bytes (1, 2, or 4) from a port in a claimed device's `io_port`
|
/// Read `width` bytes (1, 2, or 4) from a port in a claimed device's `io_port`
|
||||||
/// resource, at byte `offset` within it. Ring 3 has no direct `in`/`out`, so a legacy
|
/// resource, at byte `offset` within it. Ring 3 has no direct `in`/`out`, so a legacy
|
||||||
/// driver (PS/2, 16550 UART) reaches its ports through this claim-gated call — each
|
/// driver (PS/2, 16550 UART) reaches its ports through this claim-gated call — each
|
||||||
|
|||||||
@@ -125,6 +125,15 @@ pub const DeviceDescriptor = extern struct {
|
|||||||
// `pci_device` is (an xHCI controller, an AHCI controller) — decode the triple into
|
// `pci_device` is (an xHCI controller, an AHCI controller) — decode the triple into
|
||||||
// names with the pci-class module.
|
// names with the pci-class module.
|
||||||
pci_class: u64,
|
pci_class: u64,
|
||||||
|
// Numeric identity beyond the class triple, mirrored in the bus report's
|
||||||
|
// ChildAdded so /system/configuration/devices.csv can bind on it: `vendor`/`device` are the PCI
|
||||||
|
// vendor/device (or USB idVendor/idProduct), `subsystem` is the PCI subsystem id
|
||||||
|
// packed `(subsystem_vendor << 16) | subsystem_device`. Zero where the bus has no
|
||||||
|
// such concept. Defaulted so existing descriptor literals keep compiling and lay
|
||||||
|
// out identically until they choose to set them.
|
||||||
|
vendor: u16 = 0,
|
||||||
|
device: u16 = 0,
|
||||||
|
subsystem: u32 = 0,
|
||||||
hid_len: u64,
|
hid_len: u64,
|
||||||
resource_count: u64,
|
resource_count: u64,
|
||||||
hid: [8]u8,
|
hid: [8]u8,
|
||||||
|
|||||||
@@ -52,16 +52,134 @@ pub const config_vendor_id: usize = 0x00;
|
|||||||
pub const config_device_id: usize = 0x02;
|
pub const config_device_id: usize = 0x02;
|
||||||
pub const config_command: usize = 0x04;
|
pub const config_command: usize = 0x04;
|
||||||
pub const config_status: usize = 0x06;
|
pub const config_status: usize = 0x06;
|
||||||
pub const config_capabilities_pointer: usize = 0x34;
|
pub const config_revision_id: usize = 0x08;
|
||||||
|
pub const config_class_code: usize = 0x09; // 3 bytes: prog-IF 0x09, subclass 0x0A, base class 0x0B
|
||||||
pub const config_bar0: usize = 0x10; // BAR0; BAR n is at config_bar0 + n*4
|
pub const config_bar0: usize = 0x10; // BAR0; BAR n is at config_bar0 + n*4
|
||||||
|
pub const config_subsystem_vendor_id: usize = 0x2C;
|
||||||
|
pub const config_subsystem_id: usize = 0x2E;
|
||||||
|
pub const config_expansion_rom: usize = 0x30;
|
||||||
|
pub const config_capabilities_pointer: usize = 0x34;
|
||||||
|
pub const config_interrupt_line: usize = 0x3C;
|
||||||
|
pub const config_interrupt_pin: usize = 0x3D; // 0 = none, 1..4 = INTA..INTD
|
||||||
|
|
||||||
/// Command register: Memory-Space enable (bit 1) | Bus-Master enable (bit 2).
|
/// Command register bits.
|
||||||
pub const command_memory_and_bus_master: u16 = 0x06;
|
pub const command_io_space: u16 = 0x0001; // bit 0: I/O-space decode enable
|
||||||
|
pub const command_memory_space: u16 = 0x0002; // bit 1: memory-space decode enable
|
||||||
|
pub const command_bus_master: u16 = 0x0004; // bit 2: bus-master (DMA) enable
|
||||||
|
pub const command_interrupt_disable: u16 = 0x0400; // bit 10: suppress legacy INTx (MSI/MSI-X unaffected)
|
||||||
|
/// The pair a bus-mastering driver enables together: decode my BARs, let me DMA.
|
||||||
|
pub const command_memory_and_bus_master: u16 = command_memory_space | command_bus_master;
|
||||||
|
|
||||||
|
/// Status register bit 3: legacy INTx is asserted (upstream of the command bit-10 gate).
|
||||||
|
pub const status_interrupt: u16 = 0x0008;
|
||||||
/// Status register bit 4: a capability list is present at config_capabilities_pointer.
|
/// Status register bit 4: a capability list is present at config_capabilities_pointer.
|
||||||
pub const status_capabilities_list: u16 = 0x10;
|
pub const status_capabilities_list: u16 = 0x10;
|
||||||
/// Capability pointers are dword-aligned; the low two bits are reserved.
|
/// Capability pointers are dword-aligned; the low two bits are reserved.
|
||||||
pub const capability_pointer_mask: u8 = 0xFC;
|
pub const capability_pointer_mask: u8 = 0xFC;
|
||||||
|
|
||||||
|
/// Capability IDs — the first byte of each entry in the legacy capability list.
|
||||||
|
/// Non-exhaustive: hardware may report IDs not named here.
|
||||||
|
pub const CapabilityId = enum(u8) {
|
||||||
|
power_management = 0x01,
|
||||||
|
msi = 0x05,
|
||||||
|
vendor_specific = 0x09,
|
||||||
|
pci_express = 0x10,
|
||||||
|
msix = 0x11,
|
||||||
|
_,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// MSI capability (id 0x05) register layout. Offsets are relative to the capability
|
||||||
|
/// header; whether the address is one or two dwords (and therefore where the data word
|
||||||
|
/// sits) depends on `control_64bit_capable`.
|
||||||
|
pub const msi = struct {
|
||||||
|
pub const control: usize = 0x02; // u16 Message Control
|
||||||
|
pub const control_enable: u16 = 0x0001;
|
||||||
|
pub const control_multiple_message_capable_mask: u16 = 0x000E; // bits 3:1, log2(vectors requested)
|
||||||
|
pub const control_multiple_message_enable_mask: u16 = 0x0070; // bits 6:4, log2(vectors granted)
|
||||||
|
pub const control_64bit_capable: u16 = 0x0080; // bit 7: address is 64-bit (layout shifts)
|
||||||
|
pub const control_per_vector_masking: u16 = 0x0100; // bit 8
|
||||||
|
pub const address: usize = 0x04; // u32 low address dword (both layouts)
|
||||||
|
pub const address_high: usize = 0x08; // u32, present only when 64-bit capable
|
||||||
|
pub const data_32: usize = 0x08; // u16 message data, 32-bit layout
|
||||||
|
pub const data_64: usize = 0x0C; // u16 message data, 64-bit layout
|
||||||
|
pub const mask_bits_32: usize = 0x0C; // u32, only with per-vector masking
|
||||||
|
pub const mask_bits_64: usize = 0x10;
|
||||||
|
};
|
||||||
|
|
||||||
|
/// MSI-X capability (id 0x11) register layout, plus the 16-byte vector table entry that
|
||||||
|
/// lives in BAR space (not configuration space) at the decoded (BIR, offset).
|
||||||
|
pub const msix = struct {
|
||||||
|
pub const control: usize = 0x02; // u16 Message Control
|
||||||
|
pub const control_table_size_mask: u16 = 0x07FF; // bits 10:0, encoded as N-1
|
||||||
|
pub const control_function_mask: u16 = 0x4000; // bit 14: mask every vector
|
||||||
|
pub const control_enable: u16 = 0x8000; // bit 15
|
||||||
|
pub const table_offset_word: usize = 0x04; // u32: BIR in bits 2:0, table offset in bits 31:3
|
||||||
|
pub const pba_offset_word: usize = 0x08; // u32: same encoding, pending-bit array
|
||||||
|
pub const bir_mask: u32 = 0x0000_0007;
|
||||||
|
pub const offset_mask: u32 = 0xFFFF_FFF8;
|
||||||
|
pub const entry_size: usize = 16; // table entry stride; offsets within an entry:
|
||||||
|
pub const entry_address: usize = 0x0; // u32 low
|
||||||
|
pub const entry_address_high: usize = 0x4; // u32 high
|
||||||
|
pub const entry_data: usize = 0x8; // u32
|
||||||
|
pub const entry_vector_control: usize = 0xC; // u32
|
||||||
|
pub const entry_vector_control_masked: u32 = 0x1; // bit 0; entries reset to masked
|
||||||
|
|
||||||
|
/// Where the table (or pending-bit array) lives, decoded from its offset/BIR dword.
|
||||||
|
pub const TableLocation = struct { bar: u8, offset: u32 };
|
||||||
|
pub fn tableLocation(word: u32) TableLocation {
|
||||||
|
return .{ .bar = @intCast(word & bir_mask), .offset = word & offset_mask };
|
||||||
|
}
|
||||||
|
/// Number of table entries (the control field encodes N-1).
|
||||||
|
pub fn tableSize(control_value: u16) u16 {
|
||||||
|
return (control_value & control_table_size_mask) + 1;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Power-management capability (id 0x01) register layout.
|
||||||
|
pub const power_management = struct {
|
||||||
|
pub const capabilities: usize = 0x02; // u16 PMC (read-only: version, D-state support)
|
||||||
|
pub const control_status: usize = 0x04; // u16 PMCSR
|
||||||
|
pub const control_status_power_state_mask: u16 = 0x0003; // bits 1:0
|
||||||
|
pub const power_state_d0: u16 = 0x0;
|
||||||
|
pub const power_state_d3_hot: u16 = 0x3;
|
||||||
|
pub const control_status_pme_enable: u16 = 0x0100; // bit 8: plain RW — preserve on writes
|
||||||
|
pub const control_status_pme_status: u16 = 0x8000; // bit 15: RW1C — write 0 or you clear it
|
||||||
|
};
|
||||||
|
|
||||||
|
/// PCI Express capability (id 0x10) register layout — the slice function-level reset
|
||||||
|
/// needs; the full capability is much larger.
|
||||||
|
pub const pci_express = struct {
|
||||||
|
pub const capabilities: usize = 0x02; // u16 PCIe Capabilities register
|
||||||
|
pub const device_capabilities: usize = 0x04; // u32
|
||||||
|
pub const device_capabilities_flr: u32 = 1 << 28; // Function Level Reset supported
|
||||||
|
pub const device_control: usize = 0x08; // u16
|
||||||
|
pub const device_control_initiate_flr: u16 = 1 << 15;
|
||||||
|
pub const device_status: usize = 0x0A; // u16
|
||||||
|
pub const device_status_transactions_pending: u16 = 1 << 5;
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Extended (PCI Express) capabilities start here in the 4 KiB configuration space; a
|
||||||
|
/// conventional-PCI function has nothing there (the space reads as all-ones).
|
||||||
|
pub const extended_capability_start: usize = 0x100;
|
||||||
|
/// Extended-capability next pointers are dword-aligned within the 4 KiB space.
|
||||||
|
pub const extended_capability_pointer_mask: u16 = 0xFFC;
|
||||||
|
|
||||||
|
/// The 32-bit header at the start of each extended capability: ID in bits 15:0,
|
||||||
|
/// version in 19:16, next offset in 31:20 (0 = end of list).
|
||||||
|
pub const ExtendedCapabilityHeader = struct {
|
||||||
|
id: u16,
|
||||||
|
version: u4,
|
||||||
|
next: u16,
|
||||||
|
|
||||||
|
pub fn decode(word: u32) ExtendedCapabilityHeader {
|
||||||
|
return .{
|
||||||
|
.id = @truncate(word),
|
||||||
|
.version = @truncate(word >> 16),
|
||||||
|
.next = @intCast((word >> 20) & extended_capability_pointer_mask),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
/// BAR bit layout: bit 0 selects I/O (1) vs memory (0) space; for a memory BAR, bits 2:1
|
/// BAR bit layout: bit 0 selects I/O (1) vs memory (0) space; for a memory BAR, bits 2:1
|
||||||
/// give the type (00 = 32-bit, 10 = 64-bit spanning the next BAR), and the base address is
|
/// give the type (00 = 32-bit, 10 = 64-bit spanning the next BAR), and the base address is
|
||||||
/// the dword with the low 4 flag bits masked off.
|
/// the dword with the low 4 flag bits masked off.
|
||||||
@@ -595,3 +713,37 @@ test "named parts pack to the raw triple" {
|
|||||||
};
|
};
|
||||||
try std.testing.expectEqual(@as(u24, 0x0C_03_30), xhci.pack());
|
try std.testing.expectEqual(@as(u24, 0x0C_03_30), xhci.pack());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
test "MSI-X table word decodes to BIR and offset" {
|
||||||
|
const eq = std.testing.expectEqual;
|
||||||
|
// BIR 3, table at 0x2000 within that BAR.
|
||||||
|
try eq(msix.TableLocation{ .bar = 3, .offset = 0x2000 }, msix.tableLocation(0x0000_2003));
|
||||||
|
// BIR 0, offset 0 — the degenerate-but-common "table at BAR start" case.
|
||||||
|
try eq(msix.TableLocation{ .bar = 0, .offset = 0 }, msix.tableLocation(0));
|
||||||
|
// Table size encodes N-1 in bits 10:0; enable/function-mask bits must not leak in.
|
||||||
|
try eq(@as(u16, 11), msix.tableSize(msix.control_enable | 0x000A));
|
||||||
|
try eq(@as(u16, 1), msix.tableSize(0));
|
||||||
|
try eq(@as(u16, 2048), msix.tableSize(msix.control_table_size_mask));
|
||||||
|
}
|
||||||
|
|
||||||
|
test "extended capability header unpacks id, version, next" {
|
||||||
|
const eq = std.testing.expectEqual;
|
||||||
|
// AER (id 0x0001), version 1, next capability at 0x140.
|
||||||
|
const aer = ExtendedCapabilityHeader.decode(0x1401_0001);
|
||||||
|
try eq(@as(u16, 0x0001), aer.id);
|
||||||
|
try eq(@as(u4, 1), aer.version);
|
||||||
|
try eq(@as(u16, 0x140), aer.next);
|
||||||
|
// A zero header is the "nothing here" terminator.
|
||||||
|
const none = ExtendedCapabilityHeader.decode(0);
|
||||||
|
try eq(@as(u16, 0), none.id);
|
||||||
|
try eq(@as(u16, 0), none.next);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "command bits and capability ids compose" {
|
||||||
|
const eq = std.testing.expectEqual;
|
||||||
|
try eq(command_memory_space | command_bus_master, command_memory_and_bus_master);
|
||||||
|
try eq(@as(u8, 0x05), @intFromEnum(CapabilityId.msi));
|
||||||
|
try eq(@as(u8, 0x11), @intFromEnum(CapabilityId.msix));
|
||||||
|
try eq(@as(u8, 0x01), @intFromEnum(CapabilityId.power_management));
|
||||||
|
try eq(@as(u8, 0x10), @intFromEnum(CapabilityId.pci_express));
|
||||||
|
}
|
||||||
|
|||||||
+283
-7
@@ -1,7 +1,8 @@
|
|||||||
//! library/device/pci/pci.zig — a device driver's view of the ONE PCI function it has
|
//! library/device/pci/pci.zig — a device driver's view of the ONE PCI function it has
|
||||||
//! claimed. Config space is mapped as resource 0; this gives header-field accessors, BAR
|
//! claimed. Config space is mapped as resource 0 (a full 4 KiB ECAM page); this gives
|
||||||
//! decode + map, and a capability-list iterator, so a driver never re-derives the
|
//! header-field accessors, BAR decode + map, capability walks (legacy and extended),
|
||||||
//! config-space layout by hand.
|
//! MSI/MSI-X programming, power-state handling, and function-level reset, so a driver
|
||||||
|
//! never re-derives the config-space layout by hand.
|
||||||
//!
|
//!
|
||||||
//! This is the *device-owned* view: read my own function's live config, map my own BARs.
|
//! This is the *device-owned* view: read my own function's live config, map my own BARs.
|
||||||
//! The bus enumerator's view — probing arbitrary, not-yet-claimed functions and sizing
|
//! The bus enumerator's view — probing arbitrary, not-yet-claimed functions and sizing
|
||||||
@@ -13,6 +14,18 @@ const std = @import("std");
|
|||||||
const mmio = @import("mmio");
|
const mmio = @import("mmio");
|
||||||
const pci_class = @import("pci-class");
|
const pci_class = @import("pci-class");
|
||||||
const device = @import("driver");
|
const device = @import("driver");
|
||||||
|
const time = @import("time");
|
||||||
|
|
||||||
|
/// Spec recovery time after a D3hot -> D0 transition.
|
||||||
|
const d0_recovery_millis: u64 = 10;
|
||||||
|
/// How long to wait for in-flight transactions to drain before a function-level reset
|
||||||
|
/// (then reset anyway — resetting a stuck function is the point of FLR).
|
||||||
|
const flr_pending_timeout_millis: u64 = 100;
|
||||||
|
/// The spec's maximum FLR completion time.
|
||||||
|
const flr_settle_millis: u64 = 100;
|
||||||
|
/// How long to wait for the function to become readable again after an FLR.
|
||||||
|
const flr_ready_timeout_millis: u64 = 1000;
|
||||||
|
const flr_poll_interval_millis: u64 = 10;
|
||||||
|
|
||||||
/// A claimed PCI function whose configuration space is mapped (resource 0). `descriptor`
|
/// A claimed PCI function whose configuration space is mapped (resource 0). `descriptor`
|
||||||
/// must outlive the Function — the driver's `device.enumerate` buffer does, for the whole
|
/// must outlive the Function — the driver's `device.enumerate` buffer does, for the whole
|
||||||
@@ -42,12 +55,61 @@ pub const Function = struct {
|
|||||||
pub fn status(self: *const Function) u16 {
|
pub fn status(self: *const Function) u16 {
|
||||||
return mmio.readRegister(u16, self.config + pci_class.config_status);
|
return mmio.readRegister(u16, self.config + pci_class.config_status);
|
||||||
}
|
}
|
||||||
|
pub fn revisionId(self: *const Function) u8 {
|
||||||
|
return mmio.readRegister(u8, self.config + pci_class.config_revision_id);
|
||||||
|
}
|
||||||
|
/// Subsystem vendor ID (config 0x2C) — with `subsystemId`, the standard key for
|
||||||
|
/// board-level quirk matching.
|
||||||
|
pub fn subsystemVendorId(self: *const Function) u16 {
|
||||||
|
return mmio.readRegister(u16, self.config + pci_class.config_subsystem_vendor_id);
|
||||||
|
}
|
||||||
|
pub fn subsystemId(self: *const Function) u16 {
|
||||||
|
return mmio.readRegister(u16, self.config + pci_class.config_subsystem_id);
|
||||||
|
}
|
||||||
|
/// Interrupt pin (config 0x3D): 0 = none, 1..4 = INTA..INTD.
|
||||||
|
pub fn interruptPin(self: *const Function) u8 {
|
||||||
|
return mmio.readRegister(u8, self.config + pci_class.config_interrupt_pin);
|
||||||
|
}
|
||||||
|
/// The live class-code triple (config 0x09..0x0B), same shape discovery records.
|
||||||
|
pub fn classCode(self: *const Function) pci_class.ClassCode {
|
||||||
|
return .{
|
||||||
|
.prog_if = mmio.readRegister(u8, self.config + pci_class.config_class_code),
|
||||||
|
.subclass = mmio.readRegister(u8, self.config + pci_class.config_class_code + 1),
|
||||||
|
.base = mmio.readRegister(u8, self.config + pci_class.config_class_code + 2),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
/// Set Memory-Space + Bus-Master enable in the command register. Firmware often leaves
|
fn commandSetBits(self: *const Function, bits: u16) void {
|
||||||
/// a secondary display's decode off; a bus-mastering device must enable both.
|
|
||||||
pub fn enableMemoryAndBusMaster(self: *const Function) void {
|
|
||||||
const at = self.config + pci_class.config_command;
|
const at = self.config + pci_class.config_command;
|
||||||
mmio.writeRegister(u16, at, mmio.readRegister(u16, at) | pci_class.command_memory_and_bus_master);
|
mmio.writeRegister(u16, at, mmio.readRegister(u16, at) | bits);
|
||||||
|
}
|
||||||
|
fn commandClearBits(self: *const Function, bits: u16) void {
|
||||||
|
const at = self.config + pci_class.config_command;
|
||||||
|
mmio.writeRegister(u16, at, mmio.readRegister(u16, at) & ~bits);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Set Memory-Space + Bus-Master Enable in the command register. Firmware only enables
|
||||||
|
/// memory decode on devices it used at boot; any other device has dead BARs until its
|
||||||
|
/// driver sets it. Bus mastering is separately required for the device to do DMA.
|
||||||
|
pub fn enableMemoryAndBusMaster(self: *const Function) void {
|
||||||
|
self.commandSetBits(pci_class.command_memory_and_bus_master);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Clear Bus-Master Enable — stop the device initiating DMA. The quiesce half of a
|
||||||
|
/// driver's shutdown (or a supervisor restart): after this the device can no longer
|
||||||
|
/// write memory the process is about to stop owning.
|
||||||
|
pub fn disableBusMaster(self: *const Function) void {
|
||||||
|
self.commandClearBits(pci_class.command_bus_master);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Set command bit 10: suppress legacy INTx assertion. MSI/MSI-X are unaffected —
|
||||||
|
/// set this when enabling either, so the device cannot also raise the shared pin.
|
||||||
|
pub fn setInterruptDisable(self: *const Function) void {
|
||||||
|
self.commandSetBits(pci_class.command_interrupt_disable);
|
||||||
|
}
|
||||||
|
/// Clear command bit 10, re-allowing legacy INTx assertion.
|
||||||
|
pub fn clearInterruptDisable(self: *const Function) void {
|
||||||
|
self.commandClearBits(pci_class.command_interrupt_disable);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Decode BAR `bar` (0..5) and map it: read the BAR register, reject I/O-space BARs,
|
/// Decode BAR `bar` (0..5) and map it: read the BAR register, reject I/O-space BARs,
|
||||||
@@ -86,6 +148,129 @@ pub const Function = struct {
|
|||||||
0;
|
0;
|
||||||
return .{ .config = self.config, .cursor = first };
|
return .{ .config = self.config, .cursor = first };
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// First capability with `id`, or null.
|
||||||
|
pub fn findCapability(self: *const Function, id: pci_class.CapabilityId) ?Capability {
|
||||||
|
var walk = self.capabilities();
|
||||||
|
while (walk.next()) |capability| {
|
||||||
|
if (capability.id == @intFromEnum(id)) return capability;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Program the MSI capability with the kernel's `msi_bind` result and enable it —
|
||||||
|
/// one vector (multiple-message-enable 0, matching the kernel's single-vector
|
||||||
|
/// grant), INTx suppressed. false if the function has no MSI capability.
|
||||||
|
pub fn programMsi(self: *const Function, message: device.Msi) bool {
|
||||||
|
const cap = self.findCapability(.msi) orelse return false;
|
||||||
|
const control_at = cap.offset + pci_class.msi.control;
|
||||||
|
const control = mmio.readRegister(u16, control_at);
|
||||||
|
// Program the registers while the capability is disabled.
|
||||||
|
mmio.writeRegister(u16, control_at, control & ~pci_class.msi.control_enable);
|
||||||
|
mmio.writeRegister(u32, cap.offset + pci_class.msi.address, @truncate(message.address));
|
||||||
|
const data_offset = if (control & pci_class.msi.control_64bit_capable != 0) offset: {
|
||||||
|
mmio.writeRegister(u32, cap.offset + pci_class.msi.address_high, @intCast(message.address >> 32));
|
||||||
|
break :offset pci_class.msi.data_64;
|
||||||
|
} else pci_class.msi.data_32;
|
||||||
|
// Message data is a 16-bit register in both layouts.
|
||||||
|
mmio.writeRegister(u16, cap.offset + data_offset, @truncate(message.data));
|
||||||
|
mmio.writeRegister(u16, control_at, (control & ~pci_class.msi.control_multiple_message_enable_mask) | pci_class.msi.control_enable);
|
||||||
|
self.setInterruptDisable();
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Clear the MSI enable bit. No-op if the function has no MSI capability.
|
||||||
|
pub fn disableMsi(self: *const Function) void {
|
||||||
|
const cap = self.findCapability(.msi) orelse return;
|
||||||
|
const control_at = cap.offset + pci_class.msi.control;
|
||||||
|
mmio.writeRegister(u16, control_at, mmio.readRegister(u16, control_at) & ~pci_class.msi.control_enable);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The function's MSI-X capability with its vector table mapped: the table's BIR is
|
||||||
|
/// resolved through `mapBar` (a free cache hit when it is a BAR the driver already
|
||||||
|
/// mapped). null if the capability is absent or the table's BAR cannot be mapped.
|
||||||
|
pub fn msix(self: *Function) ?MsiX {
|
||||||
|
const cap = self.findCapability(.msix) orelse return null;
|
||||||
|
const control = mmio.readRegister(u16, cap.offset + pci_class.msix.control);
|
||||||
|
const word = mmio.readRegister(u32, cap.offset + pci_class.msix.table_offset_word);
|
||||||
|
const location = pci_class.msix.tableLocation(word);
|
||||||
|
const bar_base = self.mapBar(location.bar) orelse return null;
|
||||||
|
return .{
|
||||||
|
.capability = cap.offset,
|
||||||
|
.table = bar_base + location.offset,
|
||||||
|
.entry_count = pci_class.msix.tableSize(control),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Bring the function to D0. Firmware can leave a non-boot device in D3hot, where
|
||||||
|
/// its BARs and MSI registers do not decode; call this before touching either. No
|
||||||
|
/// power-management capability means the function is always at D0: nothing to do.
|
||||||
|
/// Preserves PME-Enable and never clears the write-1-to-clear PME-Status bit.
|
||||||
|
pub fn ensurePowerStateD0(self: *const Function) void {
|
||||||
|
const cap = self.findCapability(.power_management) orelse return;
|
||||||
|
const at = cap.offset + pci_class.power_management.control_status;
|
||||||
|
const pmcsr = mmio.readRegister(u16, at);
|
||||||
|
if (pmcsr & pci_class.power_management.control_status_power_state_mask == pci_class.power_management.power_state_d0) return;
|
||||||
|
// PME-Status is RW1C: echoing a read 1 back would clear it, so write it as 0.
|
||||||
|
mmio.writeRegister(u16, at, (pmcsr & ~pci_class.power_management.control_status_power_state_mask & ~pci_class.power_management.control_status_pme_status) | pci_class.power_management.power_state_d0);
|
||||||
|
time.sleepMillis(d0_recovery_millis);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Function Level Reset via the PCI Express capability: return the hardware to a
|
||||||
|
/// known state (a supervisor re-claiming a device after its driver died, or a driver
|
||||||
|
/// recovering a wedged function). The six BAR dwords are saved and restored — FLR
|
||||||
|
/// clears them, and the bus enumerator's assignment must survive for the descriptor
|
||||||
|
/// correlation and `mapBar` cache to stay valid. Everything else is reset: command
|
||||||
|
/// enables and MSI/MSI-X programming are gone, so the caller re-runs its whole
|
||||||
|
/// bring-up afterwards. false if the function has no PCI Express capability, does
|
||||||
|
/// not advertise FLR (conventional-PCI Advanced Features FLR is a possible
|
||||||
|
/// follow-up), or never became readable again. Blocks for at least 100 ms.
|
||||||
|
pub fn functionLevelReset(self: *const Function) bool {
|
||||||
|
const cap = self.findCapability(.pci_express) orelse return false;
|
||||||
|
const device_capabilities = mmio.readRegister(u32, cap.offset + pci_class.pci_express.device_capabilities);
|
||||||
|
if (device_capabilities & pci_class.pci_express.device_capabilities_flr == 0) return false;
|
||||||
|
|
||||||
|
// Stop new DMA, then give in-flight transactions a bounded chance to drain —
|
||||||
|
// and reset anyway on timeout, since resetting a stuck function is the point.
|
||||||
|
self.disableBusMaster();
|
||||||
|
var waited: u64 = 0;
|
||||||
|
while (mmio.readRegister(u16, cap.offset + pci_class.pci_express.device_status) & pci_class.pci_express.device_status_transactions_pending != 0) {
|
||||||
|
if (waited >= flr_pending_timeout_millis) break;
|
||||||
|
time.sleepMillis(flr_poll_interval_millis);
|
||||||
|
waited += flr_poll_interval_millis;
|
||||||
|
}
|
||||||
|
|
||||||
|
var bars: [6]u32 = undefined;
|
||||||
|
for (&bars, 0..) |*bar, index| bar.* = mmio.readRegister(u32, self.config + pci_class.config_bar0 + index * 4);
|
||||||
|
|
||||||
|
const control_at = cap.offset + pci_class.pci_express.device_control;
|
||||||
|
mmio.writeRegister(u16, control_at, mmio.readRegister(u16, control_at) | pci_class.pci_express.device_control_initiate_flr);
|
||||||
|
time.sleepMillis(flr_settle_millis);
|
||||||
|
|
||||||
|
waited = 0;
|
||||||
|
while (self.vendorId() == 0xFFFF) {
|
||||||
|
if (waited >= flr_ready_timeout_millis) return false;
|
||||||
|
time.sleepMillis(flr_poll_interval_millis);
|
||||||
|
waited += flr_poll_interval_millis;
|
||||||
|
}
|
||||||
|
for (bars, 0..) |bar, index| mmio.writeRegister(u32, self.config + pci_class.config_bar0 + index * 4, bar);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Iterate the extended (PCI Express) capability list at 0x100.. in the 4 KiB ECAM
|
||||||
|
/// page. Empty on a conventional-PCI function (the space reads as all-ones).
|
||||||
|
pub fn extendedCapabilities(self: *const Function) ExtendedCapabilityIterator {
|
||||||
|
return .{ .config = self.config };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// First extended capability with `id`, or null.
|
||||||
|
pub fn findExtendedCapability(self: *const Function, id: u16) ?ExtendedCapability {
|
||||||
|
var walk = self.extendedCapabilities();
|
||||||
|
while (walk.next()) |capability| {
|
||||||
|
if (capability.id == id) return capability;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
/// One capability header. `offset` is the ABSOLUTE virtual address of the header, so the
|
/// One capability header. `offset` is the ABSOLUTE virtual address of the header, so the
|
||||||
@@ -106,3 +291,94 @@ pub const CapabilityIterator = struct {
|
|||||||
return .{ .id = id, .offset = at };
|
return .{ .id = id, .offset = at };
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/// A resolved MSI-X capability from `Function.msix`: `capability` is the absolute
|
||||||
|
/// virtual address of the config-space header, `table` of vector-table entry 0 (in BAR
|
||||||
|
/// space — table writes are MMIO, not config space). Entries reset masked; bring-up
|
||||||
|
/// order is programEntry per vector, unmaskEntry per used vector, `enable`, then
|
||||||
|
/// `Function.setInterruptDisable`.
|
||||||
|
pub const MsiX = struct {
|
||||||
|
capability: usize,
|
||||||
|
table: usize,
|
||||||
|
entry_count: u16,
|
||||||
|
|
||||||
|
/// Write `message` into table entry `entry`, leaving the entry masked (its reset
|
||||||
|
/// state) — the spec requires masking while address/data change. false if `entry`
|
||||||
|
/// is out of range.
|
||||||
|
pub fn programEntry(self: *const MsiX, entry: u16, message: device.Msi) bool {
|
||||||
|
if (entry >= self.entry_count) return false;
|
||||||
|
const at = self.table + @as(usize, entry) * pci_class.msix.entry_size;
|
||||||
|
mmio.writeRegister(u32, at + pci_class.msix.entry_vector_control, pci_class.msix.entry_vector_control_masked);
|
||||||
|
mmio.writeRegister(u32, at + pci_class.msix.entry_address, @truncate(message.address));
|
||||||
|
mmio.writeRegister(u32, at + pci_class.msix.entry_address_high, @intCast(message.address >> 32));
|
||||||
|
mmio.writeRegister(u32, at + pci_class.msix.entry_data, message.data);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Set the entry's vector-control mask bit — its interrupt is held off (pended in
|
||||||
|
/// the PBA, not lost). false if `entry` is out of range.
|
||||||
|
pub fn maskEntry(self: *const MsiX, entry: u16) bool {
|
||||||
|
return self.writeEntryMask(entry, true);
|
||||||
|
}
|
||||||
|
/// Clear the entry's vector-control mask bit. false if `entry` is out of range.
|
||||||
|
pub fn unmaskEntry(self: *const MsiX, entry: u16) bool {
|
||||||
|
return self.writeEntryMask(entry, false);
|
||||||
|
}
|
||||||
|
fn writeEntryMask(self: *const MsiX, entry: u16, masked: bool) bool {
|
||||||
|
if (entry >= self.entry_count) return false;
|
||||||
|
const at = self.table + @as(usize, entry) * pci_class.msix.entry_size + pci_class.msix.entry_vector_control;
|
||||||
|
const control = mmio.readRegister(u32, at);
|
||||||
|
mmio.writeRegister(u32, at, if (masked)
|
||||||
|
control | pci_class.msix.entry_vector_control_masked
|
||||||
|
else
|
||||||
|
control & ~pci_class.msix.entry_vector_control_masked);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Set the function-mask control bit: every vector masked regardless of entry bits.
|
||||||
|
pub fn setFunctionMask(self: *const MsiX) void {
|
||||||
|
self.writeControl(pci_class.msix.control_function_mask, true);
|
||||||
|
}
|
||||||
|
/// Clear the function-mask control bit.
|
||||||
|
pub fn clearFunctionMask(self: *const MsiX) void {
|
||||||
|
self.writeControl(pci_class.msix.control_function_mask, false);
|
||||||
|
}
|
||||||
|
/// Set MSI-X Enable. The caller also calls `Function.setInterruptDisable` (INTx off).
|
||||||
|
pub fn enable(self: *const MsiX) void {
|
||||||
|
self.writeControl(pci_class.msix.control_enable, true);
|
||||||
|
}
|
||||||
|
/// Clear MSI-X Enable.
|
||||||
|
pub fn disable(self: *const MsiX) void {
|
||||||
|
self.writeControl(pci_class.msix.control_enable, false);
|
||||||
|
}
|
||||||
|
fn writeControl(self: *const MsiX, bit: u16, set: bool) void {
|
||||||
|
const at = self.capability + pci_class.msix.control;
|
||||||
|
const control = mmio.readRegister(u16, at);
|
||||||
|
mmio.writeRegister(u16, at, if (set) control | bit else control & ~bit);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One extended capability. `offset` is the ABSOLUTE virtual address of its header,
|
||||||
|
/// like `Capability.offset`.
|
||||||
|
pub const ExtendedCapability = struct { id: u16, version: u4, offset: usize };
|
||||||
|
|
||||||
|
pub const ExtendedCapabilityIterator = struct {
|
||||||
|
config: usize,
|
||||||
|
cursor: u16 = @intCast(pci_class.extended_capability_start),
|
||||||
|
guard: u32 = 0, // bounds a malformed chain (480 = the 0xF00-byte space / 8-byte minimum spacing)
|
||||||
|
|
||||||
|
pub fn next(self: *ExtendedCapabilityIterator) ?ExtendedCapability {
|
||||||
|
if (self.cursor == 0 or self.guard >= 480) return null;
|
||||||
|
self.guard += 1;
|
||||||
|
const at = self.config + self.cursor;
|
||||||
|
const header = pci_class.ExtendedCapabilityHeader.decode(mmio.readRegister(u32, at));
|
||||||
|
// Id 0 marks an empty list; all-ones is a conventional-PCI function (no
|
||||||
|
// extended space — reads come back as FFs).
|
||||||
|
if (header.id == 0 or header.id == 0xFFFF) return null;
|
||||||
|
// A next pointer below 0x100 would walk into the legacy header; treat it as the
|
||||||
|
// terminator it must be (0 is the normal one). The 0xFFC decode mask already
|
||||||
|
// keeps `config + cursor + 4` inside the 4 KiB page.
|
||||||
|
self.cursor = if (header.next >= pci_class.extended_capability_start) header.next else 0;
|
||||||
|
return .{ .id = header.id, .version = header.version, .offset = at };
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|||||||
@@ -0,0 +1,343 @@
|
|||||||
|
//! The device registry: parse `/system/configuration/devices.csv` into match rules and bind a
|
||||||
|
//! reported device to a driver. This is the data-driven replacement for the
|
||||||
|
//! device manager's three hand-written `switch` tables (`pciDriverForIdentity`,
|
||||||
|
//! `hidDriverFor`, `usbDriverForIdentity`); the registry is now **authoritative**
|
||||||
|
//! — a device that no row matches goes unbound (logged), never guessed.
|
||||||
|
//!
|
||||||
|
//! Pure logic: no hardware access, no syscalls, no allocator. `parse` fills a
|
||||||
|
//! caller-provided `[]Rule` whose string fields (`hid`, `driver`) are slices
|
||||||
|
//! *into the CSV source*, so the source buffer must outlive the rules (the
|
||||||
|
//! manager holds it in a static buffer for the life of the process — zero-copy).
|
||||||
|
//! That keeps this module freestanding and unit-testable with plain `zig test`.
|
||||||
|
//!
|
||||||
|
//! The file format (docs/device-driver-development/device-manager.md, and the
|
||||||
|
//! `/system/configuration/devices.csv` header itself): one rule per line, nine comma-separated
|
||||||
|
//! fields, `#` starts a comment (whole-line or trailing), blank lines ignored.
|
||||||
|
//!
|
||||||
|
//! bus, base, class, prog_if, vendor, device, subsystem, hid, driver
|
||||||
|
//!
|
||||||
|
//! `bus` is `pci`/`usb`/`acpi`; the numeric fields are hex (with or without a
|
||||||
|
//! `0x` prefix); `*` or an empty field is a wildcard (matches anything). For PCI
|
||||||
|
//! the class triple is base/subclass/prog-IF; for USB it is class/subclass/
|
||||||
|
//! protocol with vendor/device the idVendor/idProduct; ACPI matches on `hid`
|
||||||
|
//! (e.g. "PNP0303") with the triple left blank. `driver` is a full ramdisk path.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const csv = @import("csv");
|
||||||
|
|
||||||
|
/// Which bus a rule or a reported device belongs to. `unknown` is what an
|
||||||
|
/// unrecognised `bus` token parses to — such a rule never matches (its bus
|
||||||
|
/// equals no real device's), so a typo fails safe rather than binding wrongly.
|
||||||
|
pub const Bus = enum {
|
||||||
|
pci,
|
||||||
|
usb,
|
||||||
|
acpi,
|
||||||
|
unknown,
|
||||||
|
|
||||||
|
pub fn fromToken(token: []const u8) Bus {
|
||||||
|
if (std.mem.eql(u8, token, "pci")) return .pci;
|
||||||
|
if (std.mem.eql(u8, token, "usb")) return .usb;
|
||||||
|
if (std.mem.eql(u8, token, "acpi")) return .acpi;
|
||||||
|
return .unknown;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
/// A reported device's full identity, as the manager assembles it from a
|
||||||
|
/// `child_added`: the bus-native class triple plus the numeric ids the widened
|
||||||
|
/// ABI now carries, or the ACPI `_HID` string. Fields a given bus does not have
|
||||||
|
/// are zero / empty (a PCI function has no `hid`; an ACPI device has no vendor).
|
||||||
|
pub const Identity = struct {
|
||||||
|
bus: Bus,
|
||||||
|
base: u8 = 0,
|
||||||
|
subclass: u8 = 0,
|
||||||
|
prog_if: u8 = 0,
|
||||||
|
vendor: u16 = 0,
|
||||||
|
device: u16 = 0,
|
||||||
|
subsystem: u32 = 0,
|
||||||
|
hid: []const u8 = "",
|
||||||
|
};
|
||||||
|
|
||||||
|
/// One parsed registry row. A `null` field is a wildcard — it matches any value
|
||||||
|
/// and contributes nothing to specificity. String fields point into the CSV
|
||||||
|
/// source that was parsed (see the module doc).
|
||||||
|
pub const Rule = struct {
|
||||||
|
bus: Bus,
|
||||||
|
base: ?u8 = null,
|
||||||
|
subclass: ?u8 = null,
|
||||||
|
prog_if: ?u8 = null,
|
||||||
|
vendor: ?u16 = null,
|
||||||
|
device: ?u16 = null,
|
||||||
|
subsystem: ?u32 = null,
|
||||||
|
hid: ?[]const u8 = null,
|
||||||
|
driver: []const u8,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Specificity weights: how much each pinned field counts toward "most specific
|
||||||
|
/// wins". Doubling from the coarsest (`base`) so that each level outweighs *all*
|
||||||
|
/// coarser levels combined (1+2+4+8+16 = 31 < 32) — a rule that pins `device`
|
||||||
|
/// always beats any rule that does not, no matter how many coarse fields the
|
||||||
|
/// latter pins. `hid` and `device` share the top tier (the user's "hid and
|
||||||
|
/// device weigh heaviest"); they never co-occur, since `hid` is ACPI-only and
|
||||||
|
/// `device` is a PCI/USB numeric id.
|
||||||
|
const weight_base: u32 = 1;
|
||||||
|
const weight_subclass: u32 = 2;
|
||||||
|
const weight_prog_if: u32 = 4;
|
||||||
|
const weight_vendor: u32 = 8;
|
||||||
|
const weight_subsystem: u32 = 16;
|
||||||
|
const weight_device: u32 = 32;
|
||||||
|
const weight_hid: u32 = 32;
|
||||||
|
|
||||||
|
/// The outcome of `matchDriver`: the winning rule's driver path, its specificity,
|
||||||
|
/// and whether another rule tied it at that specificity. `ambiguous` is a
|
||||||
|
/// registry authoring error (two equally-specific rules claiming one device); the
|
||||||
|
/// manager logs it loudly and binds the first, so a shadowed rule is visible
|
||||||
|
/// rather than silently dropped.
|
||||||
|
pub const Match = struct {
|
||||||
|
driver: []const u8,
|
||||||
|
specificity: u32,
|
||||||
|
ambiguous: bool,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Whether `rule` matches `id`: same bus, and every pinned (non-wildcard) field
|
||||||
|
/// equal. `hid` compares as a string; the rest as integers.
|
||||||
|
fn matches(rule: Rule, id: Identity) bool {
|
||||||
|
if (rule.bus != id.bus) return false;
|
||||||
|
if (rule.base) |b| if (b != id.base) return false;
|
||||||
|
if (rule.subclass) |s| if (s != id.subclass) return false;
|
||||||
|
if (rule.prog_if) |p| if (p != id.prog_if) return false;
|
||||||
|
if (rule.vendor) |v| if (v != id.vendor) return false;
|
||||||
|
if (rule.device) |d| if (d != id.device) return false;
|
||||||
|
if (rule.subsystem) |s| if (s != id.subsystem) return false;
|
||||||
|
if (rule.hid) |h| if (!std.mem.eql(u8, h, id.hid)) return false;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The specificity score of a rule — the sum of the weights of its pinned fields.
|
||||||
|
fn specificity(rule: Rule) u32 {
|
||||||
|
var score: u32 = 0;
|
||||||
|
if (rule.base != null) score += weight_base;
|
||||||
|
if (rule.subclass != null) score += weight_subclass;
|
||||||
|
if (rule.prog_if != null) score += weight_prog_if;
|
||||||
|
if (rule.vendor != null) score += weight_vendor;
|
||||||
|
if (rule.device != null) score += weight_device;
|
||||||
|
if (rule.subsystem != null) score += weight_subsystem;
|
||||||
|
if (rule.hid != null) score += weight_hid;
|
||||||
|
return score;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Bind a reported device to a driver: of every rule that matches `id`, return
|
||||||
|
/// the most specific. `null` when nothing matches (the device goes unbound —
|
||||||
|
/// the authoritative registry does not guess). On an exact specificity tie the
|
||||||
|
/// first such rule in file order wins and `ambiguous` is set.
|
||||||
|
pub fn matchDriver(rules: []const Rule, id: Identity) ?Match {
|
||||||
|
var best: ?Match = null;
|
||||||
|
for (rules) |rule| {
|
||||||
|
if (!matches(rule, id)) continue;
|
||||||
|
const score = specificity(rule);
|
||||||
|
if (best) |current| {
|
||||||
|
if (score > current.specificity) {
|
||||||
|
best = .{ .driver = rule.driver, .specificity = score, .ambiguous = false };
|
||||||
|
} else if (score == current.specificity) {
|
||||||
|
// Two equally-specific rules claim this device — keep the first,
|
||||||
|
// flag the ambiguity for the manager to log.
|
||||||
|
best.?.ambiguous = true;
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
best = .{ .driver = rule.driver, .specificity = score, .ambiguous = false };
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return best;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- parsing -----------------------------------------------------------------
|
||||||
|
|
||||||
|
/// What one CSV line parsed to. `malformed` is a non-comment, non-blank line the
|
||||||
|
/// parser could not read (wrong field count, unparsable number, empty driver) —
|
||||||
|
/// the manager counts these and logs, so a broken registry is loud, not silent.
|
||||||
|
const Line = union(enum) {
|
||||||
|
rule: Rule,
|
||||||
|
ignorable, // blank or comment
|
||||||
|
malformed,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The result of `parse`: how many rules landed in the caller's buffer, and how
|
||||||
|
/// many non-ignorable lines were malformed (for the manager to log). `truncated`
|
||||||
|
/// is set if there were more valid rules than the buffer could hold.
|
||||||
|
pub const ParseResult = struct {
|
||||||
|
count: usize,
|
||||||
|
malformed: usize,
|
||||||
|
truncated: bool,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// Parse one hex field into `T`, honouring `*`/empty as a wildcard (`null`) and
|
||||||
|
/// an optional `0x` prefix. Returns an error only for a genuinely unparsable
|
||||||
|
/// non-wildcard token, so the caller can mark the whole line malformed.
|
||||||
|
fn parseHexField(comptime T: type, field: []const u8) !?T {
|
||||||
|
const token = std.mem.trim(u8, field, " \t");
|
||||||
|
if (token.len == 0 or std.mem.eql(u8, token, "*")) return null;
|
||||||
|
const digits = if (std.mem.startsWith(u8, token, "0x") or std.mem.startsWith(u8, token, "0X"))
|
||||||
|
token[2..]
|
||||||
|
else
|
||||||
|
token;
|
||||||
|
return try std.fmt.parseInt(T, digits, 16);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Parse a wildcard-or-string field (the `hid` column): `*`/empty → wildcard.
|
||||||
|
fn parseStringField(field: []const u8) ?[]const u8 {
|
||||||
|
const token = std.mem.trim(u8, field, " \t");
|
||||||
|
if (token.len == 0 or std.mem.eql(u8, token, "*")) return null;
|
||||||
|
return token;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Classify and (if a rule) parse one line. Split out from `parse` so it can be
|
||||||
|
/// unit-tested directly. `line` is the raw line including no newline.
|
||||||
|
fn parseLine(line: []const u8) Line {
|
||||||
|
const body = csv.stripComment(line);
|
||||||
|
if (body.len == 0) return .ignorable;
|
||||||
|
|
||||||
|
// Nine comma-separated fields (csv.fields trims each): bus, base, class,
|
||||||
|
// prog_if, vendor, device, subsystem, hid, driver.
|
||||||
|
var cols: [9][]const u8 = undefined;
|
||||||
|
var count: usize = 0;
|
||||||
|
var it = csv.fields(body);
|
||||||
|
while (it.next()) |field| {
|
||||||
|
if (count >= cols.len) return .malformed; // too many columns
|
||||||
|
cols[count] = field;
|
||||||
|
count += 1;
|
||||||
|
}
|
||||||
|
if (count != cols.len) return .malformed; // too few columns
|
||||||
|
|
||||||
|
const bus = Bus.fromToken(cols[0]);
|
||||||
|
if (bus == .unknown) return .malformed;
|
||||||
|
|
||||||
|
const driver = cols[8];
|
||||||
|
if (driver.len == 0) return .malformed;
|
||||||
|
|
||||||
|
return .{ .rule = .{
|
||||||
|
.bus = bus,
|
||||||
|
.base = parseHexField(u8, cols[1]) catch return .malformed,
|
||||||
|
.subclass = parseHexField(u8, cols[2]) catch return .malformed,
|
||||||
|
.prog_if = parseHexField(u8, cols[3]) catch return .malformed,
|
||||||
|
.vendor = parseHexField(u16, cols[4]) catch return .malformed,
|
||||||
|
.device = parseHexField(u16, cols[5]) catch return .malformed,
|
||||||
|
.subsystem = parseHexField(u32, cols[6]) catch return .malformed,
|
||||||
|
.hid = parseStringField(cols[7]),
|
||||||
|
.driver = driver,
|
||||||
|
} };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Parse a whole `/system/configuration/devices.csv` into `out_rules`. The string fields of the
|
||||||
|
/// returned rules point into `source`, which must outlive them.
|
||||||
|
pub fn parse(source: []const u8, out_rules: []Rule) ParseResult {
|
||||||
|
var result: ParseResult = .{ .count = 0, .malformed = 0, .truncated = false };
|
||||||
|
var lines = std.mem.splitScalar(u8, source, '\n');
|
||||||
|
while (lines.next()) |line| {
|
||||||
|
switch (parseLine(line)) {
|
||||||
|
.ignorable => {},
|
||||||
|
.malformed => result.malformed += 1,
|
||||||
|
.rule => |rule| {
|
||||||
|
if (result.count >= out_rules.len) {
|
||||||
|
result.truncated = true;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
out_rules[result.count] = rule;
|
||||||
|
result.count += 1;
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- tests -------------------------------------------------------------------
|
||||||
|
|
||||||
|
const testing = std.testing;
|
||||||
|
|
||||||
|
// The worked example from the design: a specific virtio-gpu rule (pins vendor +
|
||||||
|
// device) and a generic display rule (class only) both match the virtio card;
|
||||||
|
// the specific one must win. And a plain VGA adapter still falls to the generic
|
||||||
|
// rule. This is the whole point of widening the ABI to carry vendor/device.
|
||||||
|
test "virtio device rule beats the generic display rule" {
|
||||||
|
const text =
|
||||||
|
\\# bus, base, class, prog_if, vendor, device, subsystem, hid, driver
|
||||||
|
\\pci, 03, 00, 00, *, *, *, *, /system/drivers/display
|
||||||
|
\\pci, 03, 80, *, 1AF4, 1050, *, *, /system/drivers/virtio-gpu
|
||||||
|
;
|
||||||
|
var rules: [8]Rule = undefined;
|
||||||
|
const parsed = parse(text, &rules);
|
||||||
|
try testing.expectEqual(@as(usize, 2), parsed.count);
|
||||||
|
try testing.expectEqual(@as(usize, 0), parsed.malformed);
|
||||||
|
|
||||||
|
// The virtio-gpu function: display / other, vendor 1AF4 device 1050.
|
||||||
|
const virtio = matchDriver(rules[0..parsed.count], .{
|
||||||
|
.bus = .pci, .base = 0x03, .subclass = 0x80, .prog_if = 0x00,
|
||||||
|
.vendor = 0x1AF4, .device = 0x1050,
|
||||||
|
}).?;
|
||||||
|
try testing.expect(!virtio.ambiguous);
|
||||||
|
try testing.expectEqualStrings("/system/drivers/virtio-gpu", virtio.driver);
|
||||||
|
|
||||||
|
// A plain VGA adapter (display / VGA) still binds the generic display driver.
|
||||||
|
const vga = matchDriver(rules[0..parsed.count], .{
|
||||||
|
.bus = .pci, .base = 0x03, .subclass = 0x00, .prog_if = 0x00,
|
||||||
|
.vendor = 0x1234, .device = 0x1111,
|
||||||
|
}).?;
|
||||||
|
try testing.expectEqualStrings("/system/drivers/display", vga.driver);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "no matching row leaves the device unbound" {
|
||||||
|
const text = "pci, 0C, 03, 30, *, *, *, *, /system/drivers/usb-xhci-bus\n";
|
||||||
|
var rules: [8]Rule = undefined;
|
||||||
|
const parsed = parse(text, &rules);
|
||||||
|
try testing.expectEqual(@as(usize, 1), parsed.count);
|
||||||
|
|
||||||
|
// An AHCI controller (mass storage / SATA / AHCI) has no row — unbound.
|
||||||
|
const unmatched = matchDriver(rules[0..parsed.count], .{
|
||||||
|
.bus = .pci, .base = 0x01, .subclass = 0x06, .prog_if = 0x01,
|
||||||
|
});
|
||||||
|
try testing.expect(unmatched == null);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "acpi rows match on hid" {
|
||||||
|
const text =
|
||||||
|
\\acpi, *, *, *, *, *, *, PNP0303, /system/drivers/ps2-bus
|
||||||
|
\\acpi, *, *, *, *, *, *, PNP0F13, /system/drivers/ps2-bus
|
||||||
|
;
|
||||||
|
var rules: [8]Rule = undefined;
|
||||||
|
const parsed = parse(text, &rules);
|
||||||
|
try testing.expectEqual(@as(usize, 2), parsed.count);
|
||||||
|
|
||||||
|
const keyboard = matchDriver(rules[0..parsed.count], .{ .bus = .acpi, .hid = "PNP0303" }).?;
|
||||||
|
try testing.expectEqualStrings("/system/drivers/ps2-bus", keyboard.driver);
|
||||||
|
const nothing = matchDriver(rules[0..parsed.count], .{ .bus = .acpi, .hid = "PNP0A03" });
|
||||||
|
try testing.expect(nothing == null);
|
||||||
|
}
|
||||||
|
|
||||||
|
test "equally specific rules flag ambiguity" {
|
||||||
|
const text =
|
||||||
|
\\pci, 03, 00, 00, *, *, *, *, /system/drivers/display-a
|
||||||
|
\\pci, 03, 00, 00, *, *, *, *, /system/drivers/display-b
|
||||||
|
;
|
||||||
|
var rules: [8]Rule = undefined;
|
||||||
|
const parsed = parse(text, &rules);
|
||||||
|
const hit = matchDriver(rules[0..parsed.count], .{
|
||||||
|
.bus = .pci, .base = 0x03, .subclass = 0x00, .prog_if = 0x00,
|
||||||
|
}).?;
|
||||||
|
try testing.expect(hit.ambiguous);
|
||||||
|
try testing.expectEqualStrings("/system/drivers/display-a", hit.driver); // first wins
|
||||||
|
}
|
||||||
|
|
||||||
|
test "comments, blanks, and malformed lines" {
|
||||||
|
const text =
|
||||||
|
\\# a header comment
|
||||||
|
\\
|
||||||
|
\\pci, 0C, 03, 30, *, *, *, *, /system/drivers/usb-xhci-bus # trailing comment
|
||||||
|
\\pci, ZZ, 03, 30, *, *, *, *, /system/drivers/broken
|
||||||
|
\\pci, 03, 00, 00, *, *, *, *,
|
||||||
|
\\bogus-bus, *, *, *, *, *, *, *, /system/drivers/x
|
||||||
|
;
|
||||||
|
var rules: [8]Rule = undefined;
|
||||||
|
const parsed = parse(text, &rules);
|
||||||
|
try testing.expectEqual(@as(usize, 1), parsed.count); // only the xhci row is valid
|
||||||
|
try testing.expectEqual(@as(usize, 3), parsed.malformed); // bad hex, empty driver, bad bus
|
||||||
|
try testing.expectEqualStrings("/system/drivers/usb-xhci-bus", rules[0].driver);
|
||||||
|
try testing.expect(rules[0].hid == null); // trailing comment stripped, hid still wildcard
|
||||||
|
}
|
||||||
@@ -99,6 +99,19 @@ pub const Device = struct {
|
|||||||
return std.mem.bytesToValue(usb_transfer_protocol.InterruptSubscribeReply, reply[0..@sizeOf(usb_transfer_protocol.InterruptSubscribeReply)]).status == 0;
|
return std.mem.bytesToValue(usb_transfer_protocol.InterruptSubscribeReply, reply[0..@sizeOf(usb_transfer_protocol.InterruptSubscribeReply)]).status == 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Hand the controller a DMA-region capability (`handle` — from a `shareable`
|
||||||
|
/// dma_alloc, or forwarded from another process) so it binds that buffer into its
|
||||||
|
/// IOMMU domain. Must be called for every buffer whose physical address this device
|
||||||
|
/// will name in a `bulk` transfer, before the transfer. Harmless (and a no-op
|
||||||
|
/// success) when no IOMMU is enforcing. Returns false on failure.
|
||||||
|
pub fn attachDma(self: *Device, handle: ipc.Handle) bool {
|
||||||
|
var request = usb_transfer_protocol.DmaAttachRequest{ .device_token = self.token };
|
||||||
|
var reply: [@sizeOf(usb_transfer_protocol.DmaAttachReply)]u8 = undefined;
|
||||||
|
const result = ipc.callCap(self.bus, std.mem.asBytes(&request), &reply, handle) catch return false;
|
||||||
|
if (result.len < @sizeOf(usb_transfer_protocol.DmaAttachReply)) return false;
|
||||||
|
return std.mem.bytesToValue(usb_transfer_protocol.DmaAttachReply, reply[0..@sizeOf(usb_transfer_protocol.DmaAttachReply)]).status == 0;
|
||||||
|
}
|
||||||
|
|
||||||
/// One bulk transfer (IN or OUT per `endpoint_address`'s direction bit) to or
|
/// One bulk transfer (IN or OUT per `endpoint_address`'s direction bit) to or
|
||||||
/// from the caller's own DMA buffer at `physical`. Returns the bytes moved.
|
/// from the caller's own DMA buffer at `physical`. Returns the bytes moved.
|
||||||
pub fn bulk(self: *Device, endpoint_address: u8, physical: u64, length: u32) ?u32 {
|
pub fn bulk(self: *Device, endpoint_address: u8, physical: u64, length: u32) ?u32 {
|
||||||
|
|||||||
@@ -0,0 +1,104 @@
|
|||||||
|
//! The "kernel" library domain (library/kernel): the userspace private-ABI
|
||||||
|
//! library (kernel32-style), split by concern into directly-importable
|
||||||
|
//! modules. The graph is a DAG: memory depends on thread (heap needs
|
||||||
|
//! Thread.Mutex), and thread does its own raw mmap so there is no cycle.
|
||||||
|
//!
|
||||||
|
//! This package also exports `abi` — the kernel <-> user contract (SystemCall
|
||||||
|
//! numbers, mmap prot flags, page_size). Its source lives with the kernel in
|
||||||
|
//! system/abi.zig, outside this directory, but userspace's one view of it is
|
||||||
|
//! exported here so every consumer names the same module instance. Reaching
|
||||||
|
//! outside the package root means this package is valid only as an in-repo
|
||||||
|
//! path dependency (never fetchable by hash) — fine, since path dependencies
|
||||||
|
//! are the only way danos packages are consumed.
|
||||||
|
//!
|
||||||
|
//! The root shim (root.zig) and the user link script (user.ld) are plain
|
||||||
|
//! files, not modules; build-support reaches them through this package's
|
||||||
|
//! directory (Dependency.path).
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
pub fn build(b: *std.Build) void {
|
||||||
|
const protocol = b.dependency("protocol", .{});
|
||||||
|
|
||||||
|
const abi = b.addModule("abi", .{
|
||||||
|
.root_source_file = b.path("../../system/abi.zig"),
|
||||||
|
});
|
||||||
|
const system_call = b.addModule("system-call", .{
|
||||||
|
.root_source_file = b.path("system-call.zig"),
|
||||||
|
.imports = &.{.{ .name = "abi", .module = abi }},
|
||||||
|
});
|
||||||
|
const ipc = b.addModule("ipc", .{
|
||||||
|
.root_source_file = b.path("ipc.zig"),
|
||||||
|
.imports = &.{ .{ .name = "abi", .module = abi }, .{ .name = "system-call", .module = system_call } },
|
||||||
|
});
|
||||||
|
const time = b.addModule("time", .{
|
||||||
|
.root_source_file = b.path("time.zig"),
|
||||||
|
.imports = &.{.{ .name = "system-call", .module = system_call }},
|
||||||
|
});
|
||||||
|
const thread = b.addModule("thread", .{
|
||||||
|
.root_source_file = b.path("thread.zig"),
|
||||||
|
.imports = &.{ .{ .name = "abi", .module = abi }, .{ .name = "system-call", .module = system_call } },
|
||||||
|
});
|
||||||
|
const logging = b.addModule("logging", .{
|
||||||
|
.root_source_file = b.path("logging.zig"),
|
||||||
|
.imports = &.{ .{ .name = "abi", .module = abi }, .{ .name = "system-call", .module = system_call } },
|
||||||
|
});
|
||||||
|
const process = b.addModule("process", .{
|
||||||
|
.root_source_file = b.path("process.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "abi", .module = abi },
|
||||||
|
.{ .name = "system-call", .module = system_call },
|
||||||
|
.{ .name = "ipc", .module = ipc },
|
||||||
|
.{ .name = "time", .module = time },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
_ = b.addModule("file-system", .{
|
||||||
|
.root_source_file = b.path("file-system.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "abi", .module = abi },
|
||||||
|
.{ .name = "system-call", .module = system_call },
|
||||||
|
.{ .name = "ipc", .module = ipc },
|
||||||
|
.{ .name = "vfs-protocol", .module = protocol.module("vfs-protocol") },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
_ = b.addModule("memory", .{
|
||||||
|
.root_source_file = b.path("memory/memory.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "abi", .module = abi },
|
||||||
|
.{ .name = "system-call", .module = system_call },
|
||||||
|
.{ .name = "ipc", .module = ipc },
|
||||||
|
.{ .name = "thread", .module = thread },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
_ = b.addModule("service", .{
|
||||||
|
.root_source_file = b.path("service.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
.{ .name = "abi", .module = abi },
|
||||||
|
.{ .name = "ipc", .module = ipc },
|
||||||
|
.{ .name = "process", .module = process },
|
||||||
|
},
|
||||||
|
});
|
||||||
|
_ = b.addModule("start", .{
|
||||||
|
.root_source_file = b.path("start.zig"),
|
||||||
|
.imports = &.{ .{ .name = "process", .module = process }, .{ .name = "logging", .module = logging } },
|
||||||
|
});
|
||||||
|
|
||||||
|
// Standalone `zig build test` for this domain alone; the root build keeps
|
||||||
|
// its aggregate test step. time and thread pull in the syscall wrappers,
|
||||||
|
// which need the `abi` module; their danos seams fall back to host
|
||||||
|
// primitives off the danos target, so they run with real host threads.
|
||||||
|
const test_step = b.step("test", "Run the kernel library unit tests");
|
||||||
|
for ([_][]const u8{
|
||||||
|
"time.zig", // Instant/Duration arithmetic
|
||||||
|
"thread.zig", // Mutex/Condition/RwLock/WaitGroup state machines
|
||||||
|
}) |root| {
|
||||||
|
const kernel_tests = b.addTest(.{
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path(root),
|
||||||
|
.target = b.resolveTargetQuery(.{}),
|
||||||
|
.imports = &.{.{ .name = "abi", .module = abi }},
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
test_step.dependOn(&b.addRunArtifact(kernel_tests).step);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,11 @@
|
|||||||
|
.{
|
||||||
|
.name = .kernel,
|
||||||
|
.version = "0.0.0",
|
||||||
|
.fingerprint = 0x5dd29aab36503453, // Changing this has security and trust implications.
|
||||||
|
.minimum_zig_version = "0.16.0",
|
||||||
|
.dependencies = .{
|
||||||
|
// file-system speaks the VFS wire protocol.
|
||||||
|
.protocol = .{ .path = "../protocol" },
|
||||||
|
},
|
||||||
|
.paths = .{""},
|
||||||
|
}
|
||||||
@@ -305,7 +305,7 @@ pub fn makePath(path: []const u8) bool {
|
|||||||
while (end < path.len and path[end] != '/') end += 1;
|
while (end < path.len and path[end] != '/') end += 1;
|
||||||
const prefix = path[0..end];
|
const prefix = path[0..end];
|
||||||
if (prefix.len == 0 or (prefix.len == 1 and prefix[0] == '/')) continue;
|
if (prefix.len == 0 or (prefix.len == 1 and prefix[0] == '/')) continue;
|
||||||
// Best-effort per prefix: components at or above a mount point ("/mnt")
|
// Best-effort per prefix: components at or above a mount point ("/volumes")
|
||||||
// are router names, not filesystem nodes — they neither exist as nodes
|
// are router names, not filesystem nodes — they neither exist as nodes
|
||||||
// nor accept mkdir, and that is fine. Only the final verdict counts.
|
// nor accept mkdir, and that is fine. Only the final verdict counts.
|
||||||
if (!exists(prefix)) _ = makeDirectory(prefix);
|
if (!exists(prefix)) _ = makeDirectory(prefix);
|
||||||
@@ -348,8 +348,9 @@ pub fn mount(target: []const u8, backend: ipc.Handle) bool {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// As `mount`, with a backend-side rewrite prefix: a path under `target` reaches
|
/// As `mount`, with a backend-side rewrite prefix: a path under `target` reaches
|
||||||
/// the backend as `rewrite` + the mount-relative tail. How one volume serves two
|
/// the backend as `rewrite` + the mount-relative tail. How one volume serves
|
||||||
/// mounts ("/mnt/usb" from its root, "/var" from its /var subtree).
|
/// several mounts ("/volumes/usb" from its root, "/system/logs" from its
|
||||||
|
/// /system/logs subtree).
|
||||||
pub fn mountRewritten(target: []const u8, backend: ipc.Handle, rewrite: []const u8) bool {
|
pub fn mountRewritten(target: []const u8, backend: ipc.Handle, rewrite: []const u8) bool {
|
||||||
return fsMount(target, backend, rewrite);
|
return fsMount(target, backend, rewrite);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -34,6 +34,14 @@ pub fn register(id: abi.ServiceId, h: Handle) bool {
|
|||||||
return !failed(sc.systemCall2(.ipc_register, @intFromEnum(id), h));
|
return !failed(sc.systemCall2(.ipc_register, @intFromEnum(id), h));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Drop a capability handle (endpoint, shared-memory, or DMA-region) and free its table
|
||||||
|
/// slot. A forwarding hop closes a cap it passed on; a binder closes a DMA-region cap
|
||||||
|
/// once the binding holds its own reference — the 32-slot table is otherwise consumed by
|
||||||
|
/// repeated cap-passing.
|
||||||
|
pub fn close(h: Handle) bool {
|
||||||
|
return !failed(sc.systemCall1(.handle_close, h));
|
||||||
|
}
|
||||||
|
|
||||||
/// Find the endpoint published under `id`, installing a handle to it in this
|
/// Find the endpoint published under `id`, installing a handle to it in this
|
||||||
/// process.
|
/// process.
|
||||||
pub fn lookup(id: abi.ServiceId) ?Handle {
|
pub fn lookup(id: abi.ServiceId) ?Handle {
|
||||||
|
|||||||
@@ -12,12 +12,18 @@ const sc = @import("system-call");
|
|||||||
pub const coherent: usize = abi.dma_coherent;
|
pub const coherent: usize = abi.dma_coherent;
|
||||||
pub const write_combining: usize = abi.dma_write_combining;
|
pub const write_combining: usize = abi.dma_write_combining;
|
||||||
pub const below_4g: usize = abi.dma_below_4g;
|
pub const below_4g: usize = abi.dma_below_4g;
|
||||||
|
/// Ask for a capability handle (in `Region.handle`) so the buffer can be delegated to
|
||||||
|
/// another driver and bound into a device's IOMMU domain (`driver.dmaBind`). A driver's
|
||||||
|
/// private rings don't need it; a buffer whose physical address crosses IPC does.
|
||||||
|
pub const shareable: usize = abi.dma_shareable;
|
||||||
|
|
||||||
/// A DMA allocation: the `virtual` address the CPU touches, and the `physical` address
|
/// A DMA allocation: the `virtual` address the CPU touches, the `physical` address to
|
||||||
/// to program into the device's descriptor-ring / base registers.
|
/// program into the device's registers, and — when `shareable` was requested — a
|
||||||
|
/// capability `handle` naming the region for delegation (null otherwise).
|
||||||
pub const Region = struct {
|
pub const Region = struct {
|
||||||
virtual: usize,
|
virtual: usize,
|
||||||
physical: usize,
|
physical: usize,
|
||||||
|
handle: ?usize = null,
|
||||||
};
|
};
|
||||||
|
|
||||||
inline fn failed(r: usize) bool {
|
inline fn failed(r: usize) bool {
|
||||||
@@ -25,21 +31,23 @@ inline fn failed(r: usize) bool {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Allocate `len` bytes of DMA-capable memory with `flags` (e.g. `coherent`, or
|
/// Allocate `len` bytes of DMA-capable memory with `flags` (e.g. `coherent`, or
|
||||||
/// `coherent | below_4g`). Returns the virtual/physical pair, or null on failure. Two
|
/// `coherent | shareable`). Returns virtual/physical (and a handle when `shareable`), or
|
||||||
/// return values — the virtual address in rax, the physical address in rdx — so it
|
/// null on failure. Three return values — virtual in rax, physical in rdx, handle in r8
|
||||||
/// needs a hand-written stub.
|
/// — so it needs a hand-written stub.
|
||||||
pub fn alloc(len: usize, flags: usize) ?Region {
|
pub fn alloc(len: usize, flags: usize) ?Region {
|
||||||
var rax: usize = undefined;
|
var rax: usize = undefined;
|
||||||
var rdx: usize = undefined; // out: physical address
|
var rdx: usize = undefined; // out: physical address
|
||||||
|
var r8: usize = undefined; // out: capability handle (abi.no_cap unless shareable)
|
||||||
asm volatile ("syscall"
|
asm volatile ("syscall"
|
||||||
: [rax] "={rax}" (rax),
|
: [rax] "={rax}" (rax),
|
||||||
[rdx] "={rdx}" (rdx),
|
[rdx] "={rdx}" (rdx),
|
||||||
|
[r8] "={r8}" (r8),
|
||||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.dma_alloc)),
|
: [n] "{rax}" (@intFromEnum(abi.SystemCall.dma_alloc)),
|
||||||
[a0] "{rdi}" (len),
|
[a0] "{rdi}" (len),
|
||||||
[a1] "{rsi}" (flags),
|
[a1] "{rsi}" (flags),
|
||||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||||
if (failed(rax)) return null;
|
if (failed(rax)) return null;
|
||||||
return .{ .virtual = rax, .physical = rdx };
|
return .{ .virtual = rax, .physical = rdx, .handle = if (r8 == abi.no_cap) null else r8 };
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Release a region from a prior `alloc` (`virtual` and the same `len`).
|
/// Release a region from a prior `alloc` (`virtual` and the same `len`).
|
||||||
|
|||||||
@@ -38,6 +38,7 @@ pub const DmaRegion = dma.Region;
|
|||||||
pub const dma_coherent = dma.coherent;
|
pub const dma_coherent = dma.coherent;
|
||||||
pub const dma_write_combining = dma.write_combining;
|
pub const dma_write_combining = dma.write_combining;
|
||||||
pub const dma_below_4g = dma.below_4g;
|
pub const dma_below_4g = dma.below_4g;
|
||||||
|
pub const dma_shareable = dma.shareable;
|
||||||
pub const dmaAlloc = dma.alloc;
|
pub const dmaAlloc = dma.alloc;
|
||||||
pub const dmaFree = dma.free;
|
pub const dmaFree = dma.free;
|
||||||
|
|
||||||
|
|||||||
@@ -7,7 +7,9 @@
|
|||||||
//! buffer**, named by its physical address — the same physical-address handoff
|
//! buffer**, named by its physical address — the same physical-address handoff
|
||||||
//! usb-storage already uses toward the controller, one layer up. So a 512-byte
|
//! usb-storage already uses toward the controller, one layer up. So a 512-byte
|
||||||
//! sector never has to cross the 256-byte IPC boundary; only the small request /
|
//! sector never has to cross the 256-byte IPC boundary; only the small request /
|
||||||
//! reply headers do. (Safe while the IOMMU is unenforced; see docs/driver-model.md.)
|
//! reply headers do. Under an enforcing IOMMU the buffer's physical addresses are
|
||||||
|
//! only reachable by the device once the filesystem has `attach`ed the buffer's
|
||||||
|
//! capability (the block server forwards it to the controller); see docs/driver-model.md.
|
||||||
|
|
||||||
pub const Operation = enum(u32) {
|
pub const Operation = enum(u32) {
|
||||||
/// geometry() -> { block_size, block_count }
|
/// geometry() -> { block_size, block_count }
|
||||||
@@ -20,6 +22,11 @@ pub const Operation = enum(u32) {
|
|||||||
/// A filesystem calls this to make prior writes durable — e.g. before power-off,
|
/// A filesystem calls this to make prior writes durable — e.g. before power-off,
|
||||||
/// so a shutdown-time write isn't lost in the USB flash controller's cache.
|
/// so a shutdown-time write isn't lost in the USB flash controller's cache.
|
||||||
flush = 3,
|
flush = 3,
|
||||||
|
/// attach(): the caller's DMA-region capability rides the call's cap slot; the
|
||||||
|
/// block server forwards it to the controller so the buffer's physical addresses
|
||||||
|
/// (named in later read/write) are reachable by the device under an enforcing
|
||||||
|
/// IOMMU. Call once per buffer before using it in a transfer.
|
||||||
|
attach = 4,
|
||||||
};
|
};
|
||||||
|
|
||||||
pub const Request = extern struct {
|
pub const Request = extern struct {
|
||||||
|
|||||||
@@ -0,0 +1,46 @@
|
|||||||
|
//! The "protocol" library domain: the wire protocols — each service's public
|
||||||
|
//! interface, exposed as its own module (docs/driver-model.md). Both sides of
|
||||||
|
//! every conversation depend on the contract by name; neither reaches into the
|
||||||
|
//! other's files. Pure flat wire types: no protocol module imports anything.
|
||||||
|
//!
|
||||||
|
//! vfs-protocol : the VFS server <-> the file layer (unistd/stdio)
|
||||||
|
//! input-protocol : the input fan-out service <-> sources + subscribers
|
||||||
|
//! block-protocol : a filesystem <-> a block driver (usb-storage)
|
||||||
|
//! usb-transfer-protocol : a USB class driver <-> the xHCI bus driver
|
||||||
|
//! device-manager-protocol : the device manager <-> drivers + discovery
|
||||||
|
//! display-protocol : the compositor's client-facing surface
|
||||||
|
//! scanout-protocol : the compositor -> a native scanout driver (docs/display-v2.md)
|
||||||
|
//! power-protocol : system power's domain-named surface (docs/power.md)
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
pub fn build(b: *std.Build) void {
|
||||||
|
for ([_]struct { name: []const u8, root: []const u8 }{
|
||||||
|
.{ .name = "vfs-protocol", .root = "vfs/vfs-protocol.zig" },
|
||||||
|
.{ .name = "input-protocol", .root = "input/input-protocol.zig" },
|
||||||
|
.{ .name = "block-protocol", .root = "block/block-protocol.zig" },
|
||||||
|
.{ .name = "usb-transfer-protocol", .root = "usb-transfer/usb-transfer-protocol.zig" },
|
||||||
|
.{ .name = "device-manager-protocol", .root = "device-manager/device-manager-protocol.zig" },
|
||||||
|
.{ .name = "display-protocol", .root = "display/display-protocol.zig" },
|
||||||
|
.{ .name = "scanout-protocol", .root = "scanout/scanout-protocol.zig" },
|
||||||
|
.{ .name = "power-protocol", .root = "power/power-protocol.zig" },
|
||||||
|
}) |protocol| {
|
||||||
|
_ = b.addModule(protocol.name, .{ .root_source_file = b.path(protocol.root) });
|
||||||
|
}
|
||||||
|
|
||||||
|
// Standalone `zig build test` for this domain alone; the root build keeps
|
||||||
|
// its aggregate test step.
|
||||||
|
const test_step = b.step("test", "Run the protocol unit tests");
|
||||||
|
for ([_][]const u8{
|
||||||
|
"vfs/vfs-protocol.zig", // NodeKind / DirectoryEntry sizes + op values
|
||||||
|
"display/display-protocol.zig", // pack(): native pixel encoding per format
|
||||||
|
}) |root| {
|
||||||
|
const protocol_tests = b.addTest(.{
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path(root),
|
||||||
|
.target = b.resolveTargetQuery(.{}),
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
test_step.dependOn(&b.addRunArtifact(protocol_tests).step);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,8 @@
|
|||||||
|
.{
|
||||||
|
.name = .protocol,
|
||||||
|
.version = "0.0.0",
|
||||||
|
.fingerprint = 0xc8c0bc4c4d551283, // Changing this has security and trust implications.
|
||||||
|
.minimum_zig_version = "0.16.0",
|
||||||
|
.dependencies = .{},
|
||||||
|
.paths = .{""},
|
||||||
|
}
|
||||||
@@ -11,6 +11,19 @@
|
|||||||
/// startup instead of quiet corruption later.
|
/// startup instead of quiet corruption later.
|
||||||
pub const version: u16 = 1;
|
pub const version: u16 = 1;
|
||||||
|
|
||||||
|
/// Which bus a `child_added` came from — stated by the reporting bus driver so
|
||||||
|
/// the manager's /system/configuration/devices.csv matcher knows how to read the report's identity
|
||||||
|
/// (a PCI class triple vs a USB class triple are the same 24 bits but different
|
||||||
|
/// namespaces) and which `bus` column a rule must name to bind it. `unknown` is
|
||||||
|
/// the zero default, so an un-upgraded reporter fails to match rather than
|
||||||
|
/// binding to the wrong bus's rule.
|
||||||
|
pub const BusKind = enum(u8) {
|
||||||
|
unknown = 0,
|
||||||
|
pci = 1,
|
||||||
|
usb = 2,
|
||||||
|
acpi = 3,
|
||||||
|
};
|
||||||
|
|
||||||
/// What kind of driver is talking (docs/driver-model.md's shapes).
|
/// What kind of driver is talking (docs/driver-model.md's shapes).
|
||||||
pub const Role = enum(u8) {
|
pub const Role = enum(u8) {
|
||||||
/// Owns a controller and reports the devices behind it (`child_added`).
|
/// Owns a controller and reports the devices behind it (`child_added`).
|
||||||
@@ -66,7 +79,9 @@ pub const reply_size = @sizeOf(HelloReply);
|
|||||||
/// restarted instance rediscovers and re-reports.
|
/// restarted instance rediscovers and re-reports.
|
||||||
pub const ChildAdded = extern struct {
|
pub const ChildAdded = extern struct {
|
||||||
operation: u8 = @intFromEnum(Operation.child_added),
|
operation: u8 = @intFromEnum(Operation.child_added),
|
||||||
reserved0: u8 = 0,
|
/// A `BusKind` value: which bus reported this child, so the manager reads the
|
||||||
|
/// identity in the right namespace and matches against the right `bus` column.
|
||||||
|
bus: u8 = @intFromEnum(BusKind.unknown),
|
||||||
reserved1: u16 = 0,
|
reserved1: u16 = 0,
|
||||||
reserved2: u32 = 0,
|
reserved2: u32 = 0,
|
||||||
/// The reporting driver's own device (the controller) — the child's parent.
|
/// The reporting driver's own device (the controller) — the child's parent.
|
||||||
@@ -80,6 +95,18 @@ pub const ChildAdded = extern struct {
|
|||||||
/// manager hands a matched driver as its argv assignment — or `no_device`
|
/// manager hands a matched driver as its argv assignment — or `no_device`
|
||||||
/// for an unregistered leaf (a USB port before the descriptor track).
|
/// for an unregistered leaf (a USB port before the descriptor track).
|
||||||
device_id: u64 = no_device,
|
device_id: u64 = no_device,
|
||||||
|
/// The vendor id (PCI vendor / USB idVendor), or 0 when the bus has no such
|
||||||
|
/// concept (ACPI). Carried so the manager's /system/configuration/devices.csv matcher can bind
|
||||||
|
/// on vendor — a level the bus-native `identity` (a class triple) cannot express.
|
||||||
|
vendor: u16 = 0,
|
||||||
|
/// The device id (PCI device / USB idProduct), or 0. The most specific numeric
|
||||||
|
/// level: this is what lets one virtio-gpu (1AF4:1050) be told from any other
|
||||||
|
/// virtio display function without the driver re-confirming after it is spawned.
|
||||||
|
device: u16 = 0,
|
||||||
|
/// The PCI subsystem id, packed `(subsystem_vendor << 16) | subsystem_device`
|
||||||
|
/// (so it reads vendor-first, matching the CSV's `ssvid:ssid`), or 0 when the
|
||||||
|
/// device has no subsystem id (a bridge, or a non-PCI bus).
|
||||||
|
subsystem: u32 = 0,
|
||||||
/// The ACPI hardware id (`_HID`), EISA-decoded (e.g. "PNP0303"), for devices
|
/// The ACPI hardware id (`_HID`), EISA-decoded (e.g. "PNP0303"), for devices
|
||||||
/// discovered by firmware string rather than a numeric bus identity. Empty
|
/// discovered by firmware string rather than a numeric bus identity. Empty
|
||||||
/// (all zero) otherwise. Widens for FDT `compatible` strings later.
|
/// (all zero) otherwise. Widens for FDT `compatible` strings later.
|
||||||
|
|||||||
@@ -45,6 +45,11 @@ pub const Operation = enum(u32) {
|
|||||||
control = 1,
|
control = 1,
|
||||||
interrupt_subscribe = 2,
|
interrupt_subscribe = 2,
|
||||||
bulk = 3,
|
bulk = 3,
|
||||||
|
/// dma_attach: a class driver hands the controller a DMA-region capability (riding
|
||||||
|
/// the call's cap slot) so the controller binds that buffer into its IOMMU domain
|
||||||
|
/// and may then DMA to the physical addresses inside it. Needed once per buffer the
|
||||||
|
/// class driver will name in a `bulk` transfer (its own, or one forwarded to it).
|
||||||
|
dma_attach = 4,
|
||||||
};
|
};
|
||||||
|
|
||||||
/// The endpoint facts a class driver needs, lifted from the endpoint descriptor
|
/// The endpoint facts a class driver needs, lifted from the endpoint descriptor
|
||||||
@@ -138,6 +143,19 @@ pub const BulkReply = extern struct {
|
|||||||
actual_length: u32,
|
actual_length: u32,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
/// dma_attach: the region capability rides the call's cap slot; the body only carries
|
||||||
|
/// the device token (scoping) so the controller knows which caller is attaching.
|
||||||
|
pub const DmaAttachRequest = extern struct {
|
||||||
|
operation: u32 = @intFromEnum(Operation.dma_attach),
|
||||||
|
reserved: u32 = 0,
|
||||||
|
device_token: u64,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const DmaAttachReply = extern struct {
|
||||||
|
status: i32,
|
||||||
|
reserved: u32 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
/// An asynchronous interrupt report, pushed with `ipc.send` to a subscriber's
|
/// An asynchronous interrupt report, pushed with `ipc.send` to a subscriber's
|
||||||
/// endpoint. `Received.isMessage()` is set; there is no reply owed.
|
/// endpoint. `Received.isMessage()` is set; there is no reply owed.
|
||||||
pub const InterruptReport = extern struct {
|
pub const InterruptReport = extern struct {
|
||||||
|
|||||||
@@ -33,8 +33,8 @@ pub const Operation = enum(u32) {
|
|||||||
rename, // rename(old\0new payload) -> status
|
rename, // rename(old\0new payload) -> status
|
||||||
};
|
};
|
||||||
|
|
||||||
/// The type of a filesystem node, aligned to the FSH file-type table
|
/// The type of a filesystem node, aligned to the node-kind table
|
||||||
/// (docs/danos-file-system-hierarchy-FSH.md). Fills `FileStatus.kind` and
|
/// (docs/file-system-development/file-system-hierarchy.md). Fills `FileStatus.kind` and
|
||||||
/// `DirectoryEntry.kind`; `regular = 0` keeps the historical hardcoded value.
|
/// `DirectoryEntry.kind`; `regular = 0` keeps the historical hardcoded value.
|
||||||
pub const NodeKind = enum(u32) {
|
pub const NodeKind = enum(u32) {
|
||||||
regular = 0,
|
regular = 0,
|
||||||
|
|||||||
@@ -0,0 +1,31 @@
|
|||||||
|
//! The "xkeyboard-config" library domain: keyboard layouts compiled from the
|
||||||
|
//! X11 xkeyboard-config database into native Zig (keycode + modifiers ->
|
||||||
|
//! keysym/character). The `layouts` tables are generated by
|
||||||
|
//! tools/make-xkeyboard-config.py; `xkeyboard-config` is the hand-written API
|
||||||
|
//! over them.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
|
||||||
|
pub fn build(b: *std.Build) void {
|
||||||
|
const layouts = b.addModule("layouts", .{
|
||||||
|
.root_source_file = b.path("generated/layouts.zig"),
|
||||||
|
});
|
||||||
|
_ = b.addModule("xkeyboard-config", .{
|
||||||
|
.root_source_file = b.path("xkeyboard-config.zig"),
|
||||||
|
.imports = &.{.{ .name = "layouts", .module = layouts }},
|
||||||
|
});
|
||||||
|
|
||||||
|
// Standalone `zig build test` for this domain alone; the root build keeps
|
||||||
|
// its aggregate test step. The keycode->character assertions are the
|
||||||
|
// end-to-end proof that the xkb-data -> generator -> Zig-lookup pipeline
|
||||||
|
// is correct.
|
||||||
|
const test_step = b.step("test", "Run the xkeyboard-config unit tests");
|
||||||
|
const xkb_tests = b.addTest(.{
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path("xkeyboard-config.zig"),
|
||||||
|
.target = b.resolveTargetQuery(.{}),
|
||||||
|
.imports = &.{.{ .name = "layouts", .module = layouts }},
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
test_step.dependOn(&b.addRunArtifact(xkb_tests).step);
|
||||||
|
}
|
||||||
@@ -0,0 +1,8 @@
|
|||||||
|
.{
|
||||||
|
.name = .xkeyboard_config,
|
||||||
|
.version = "0.0.0",
|
||||||
|
.fingerprint = 0xea5abe82f08b6eae, // Changing this has security and trust implications.
|
||||||
|
.minimum_zig_version = "0.16.0",
|
||||||
|
.dependencies = .{},
|
||||||
|
.paths = .{""},
|
||||||
|
}
|
||||||
+6
-1
@@ -76,6 +76,10 @@ pub const SystemCall = enum(u64) {
|
|||||||
fs_node = 47, // fs_node(op, node_token, offset, buf_ptr, buf_len) -> bytes/0/-errno: read/status/readdir on a kernel-served node (op values mirror the vfs-protocol Operation numbers)
|
fs_node = 47, // fs_node(op, node_token, offset, buf_ptr, buf_len) -> bytes/0/-errno: read/status/readdir on a kernel-served node (op values mirror the vfs-protocol Operation numbers)
|
||||||
fs_mount = 48, // fs_mount(prefix_ptr, prefix_len, backend_handle, rewrite_ptr, rewrite_len) -> 0/-errno: mount a userspace filesystem's endpoint at an absolute prefix (possession of the handle is the capability)
|
fs_mount = 48, // fs_mount(prefix_ptr, prefix_len, backend_handle, rewrite_ptr, rewrite_len) -> 0/-errno: mount a userspace filesystem's endpoint at an absolute prefix (possession of the handle is the capability)
|
||||||
fs_unmount = 49, // fs_unmount(prefix_ptr, prefix_len) -> 0/-errno: remove a backend mount
|
fs_unmount = 49, // fs_unmount(prefix_ptr, prefix_len) -> 0/-errno: remove a backend mount
|
||||||
|
iommu_fault_drain = 50, // iommu_fault_drain() -> count: drain + log pending IOMMU translation faults (a diagnostic; the count of faults seen this call)
|
||||||
|
dma_bind = 51, // dma_bind(device_id, region_handle) -> 0/-errno: map a DMA-region capability into the claimed device's IOMMU domain (idempotent). The caller must own the device and hold the handle
|
||||||
|
dma_unbind = 52, // dma_unbind(device_id, region_handle) -> 0/-errno: unmap a previously bound region from the device's domain and invalidate
|
||||||
|
handle_close = 53, // handle_close(handle) -> 0/-errno: drop one capability handle and free its table slot (endpoints, shared-memory, DMA regions)
|
||||||
_,
|
_,
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -113,6 +117,7 @@ pub const msi_address_base: u64 = 0xFEE0_0000;
|
|||||||
pub const dma_coherent: u64 = 1; // strong-uncacheable — the default, the only portable one
|
pub const dma_coherent: u64 = 1; // strong-uncacheable — the default, the only portable one
|
||||||
pub const dma_write_combining: u64 = 2; // write-combining (framebuffers); needs PAT
|
pub const dma_write_combining: u64 = 2; // write-combining (framebuffers); needs PAT
|
||||||
pub const dma_below_4g: u64 = 4; // physical address must fit 32 bits (legacy DMA engines)
|
pub const dma_below_4g: u64 = 4; // physical address must fit 32 bits (legacy DMA engines)
|
||||||
|
pub const dma_shareable: u64 = 8; // return a capability handle (r8) so the region can be delegated + dma_bound
|
||||||
|
|
||||||
/// Set in the badge returned by `ipc_reply_wait` when what arrived is an
|
/// Set in the badge returned by `ipc_reply_wait` when what arrived is an
|
||||||
/// **asynchronous notification** (a device interrupt bound with `irq_bind`, or a
|
/// **asynchronous notification** (a device interrupt bound with `irq_bind`, or a
|
||||||
@@ -290,7 +295,7 @@ pub const ServiceId = enum(u32) {
|
|||||||
power = 5, // system power: events (button, lid, battery) + shutdown (docs/power.md; domain-named per docs/discovery.md — the acpi service registers it on x86, a PSCI service will on ARM)
|
power = 5, // system power: events (button, lid, battery) + shutdown (docs/power.md; domain-named per docs/discovery.md — the acpi service registers it on x86, a PSCI service will on ARM)
|
||||||
usb_bus = 6, // the xHCI host-controller driver's transfer endpoint; USB class drivers look it up and `callCap`-open their device to get a private per-device transfer channel (docs/driver-model.md)
|
usb_bus = 6, // the xHCI host-controller driver's transfer endpoint; USB class drivers look it up and `callCap`-open their device to get a private per-device transfer channel (docs/driver-model.md)
|
||||||
block = 7, // a block-device driver (USB mass storage today): read/write of fixed-size blocks, the storage a filesystem sits on
|
block = 7, // a block-device driver (USB mass storage today): read/write of fixed-size blocks, the storage a filesystem sits on
|
||||||
fat = 8, // the FAT filesystem server; the VFS mounts it and forwards paths under its mount point (/mnt/usb) to it
|
fat = 8, // the FAT filesystem server; the VFS mounts it and forwards paths under its mount point (/volumes/usb) to it
|
||||||
display = 9, // the display service: owns the framebuffer, composites a layer stack, presents frames (docs/display.md)
|
display = 9, // the display service: owns the framebuffer, composites a layer stack, presents frames (docs/display.md)
|
||||||
shared_memory_test = 10, // the shared-memory test server (V2): a client passes it a shared-memory capability, it maps + verifies (docs/display-v2.md)
|
shared_memory_test = 10, // the shared-memory test server (V2): a client passes it a shared-memory capability, it maps + verifies (docs/display-v2.md)
|
||||||
scanout = 11, // a native scanout driver (virtio-gpu): the compositor finds it here to upgrade off the GOP framebuffer (docs/display-v2.md)
|
scanout = 11, // a native scanout driver (virtio-gpu): the compositor finds it here to upgrade off the GOP framebuffer (docs/display-v2.md)
|
||||||
|
|||||||
@@ -0,0 +1,33 @@
|
|||||||
|
# /system/configuration/devices.csv — the device→driver registry.
|
||||||
|
#
|
||||||
|
# The device manager reads this at boot and binds each device a bus driver
|
||||||
|
# reports to the driver named here. It is AUTHORITATIVE: a device that no row
|
||||||
|
# matches goes unbound (logged), never guessed. Edit this file to teach the
|
||||||
|
# system new hardware — no recompile of the device manager required.
|
||||||
|
#
|
||||||
|
# One rule per line, nine comma-separated fields. '#' starts a comment
|
||||||
|
# (whole-line or trailing); blank lines are ignored. Whitespace around a field
|
||||||
|
# is trimmed, so columns may be padded for readability.
|
||||||
|
#
|
||||||
|
# bus which bus reported the device: pci | usb | acpi
|
||||||
|
# base PCI base class / USB class (hex)
|
||||||
|
# class PCI subclass / USB subclass (hex)
|
||||||
|
# prog_if PCI prog-IF / USB protocol (hex)
|
||||||
|
# vendor PCI vendor id / USB idVendor (hex)
|
||||||
|
# device PCI device id / USB idProduct (hex)
|
||||||
|
# subsystem PCI subsystem, packed (ssvid<<16)|ssid (hex)
|
||||||
|
# hid ACPI _HID string (e.g. PNP0303); blank for pci/usb
|
||||||
|
# driver full ramdisk path of the driver to spawn
|
||||||
|
#
|
||||||
|
# '*' or an empty field is a wildcard. When several rows match one device the
|
||||||
|
# MOST SPECIFIC wins (pinning vendor/device/hid beats pinning only a class), so
|
||||||
|
# a generic class rule and a precise vendor:device rule can coexist.
|
||||||
|
#
|
||||||
|
# bus base class prog_if vendor device subsystem hid driver
|
||||||
|
pci, 0C, 03, 30, *, *, *, *, /system/drivers/usb-xhci-bus
|
||||||
|
pci, 03, 80, *, 1AF4, 1050, *, *, /system/drivers/virtio-gpu
|
||||||
|
usb, 03, 01, 01, *, *, *, *, /system/drivers/usb-hid-keyboard
|
||||||
|
usb, 03, 01, 02, *, *, *, *, /system/drivers/usb-hid-mouse
|
||||||
|
usb, 08, 06, 50, *, *, *, *, /system/drivers/usb-storage
|
||||||
|
acpi, *, *, *, *, *, *, PNP0303, /system/drivers/ps2-bus
|
||||||
|
acpi, *, *, *, *, *, *, PNP0F13, /system/drivers/ps2-bus
|
||||||
|
@@ -0,0 +1,13 @@
|
|||||||
|
# /system/configuration/init.csv — diagnose variant (-Ddiagnose), bundled at
|
||||||
|
# /system/configuration/init.csv.
|
||||||
|
#
|
||||||
|
# The display stack (display, display-demo) is omitted so the kernel's timestamped
|
||||||
|
# on-screen boot transcript is never suppressed — the bring-up timeline (USB,
|
||||||
|
# storage, logger) stays readable on real hardware with no serial. See
|
||||||
|
# system/configuration/init.csv for the format; this file must otherwise track it.
|
||||||
|
#
|
||||||
|
# service args...
|
||||||
|
/system/services/input
|
||||||
|
/system/services/device-manager
|
||||||
|
/system/services/fat
|
||||||
|
/system/services/logger
|
||||||
|
@@ -0,0 +1,20 @@
|
|||||||
|
# /system/configuration/init.csv — the services init (PID 1) starts at boot, in order.
|
||||||
|
#
|
||||||
|
# init reads this at startup and spawns each service supervised (restarting it on
|
||||||
|
# a crash, up to a cap). Startup order is top->bottom; shutdown is the reverse, so
|
||||||
|
# the logger (last) goes down first and its final drain still has the fat server
|
||||||
|
# and the whole storage chain alive underneath it. It is AUTHORITATIVE — there is
|
||||||
|
# no hardcoded fallback list; a missing file means no services are started.
|
||||||
|
#
|
||||||
|
# '#' starts a comment (whole-line or trailing); blank lines are ignored. The
|
||||||
|
# first field is the service binary path; any fields after it are the service's
|
||||||
|
# argv. Drivers are absent on purpose — the device manager discovers hardware and
|
||||||
|
# spawns those (see /system/configuration/devices.csv).
|
||||||
|
#
|
||||||
|
# service args...
|
||||||
|
/system/services/input
|
||||||
|
/system/services/device-manager
|
||||||
|
/system/services/fat
|
||||||
|
/system/services/display
|
||||||
|
/system/services/display-demo
|
||||||
|
/system/services/logger
|
||||||
|
@@ -1,59 +0,0 @@
|
|||||||
//! /system/drivers/display - the generic display engine driver.
|
|
||||||
//! This driver is a non official driver for GPU vendors like Intel, NVIDIA, AMD. It provides basic
|
|
||||||
//! display engine features to the display engine protocol used by the display server, compositor
|
|
||||||
//! and graphical user interface libraries like Zooeee.
|
|
||||||
//!
|
|
||||||
//! This driver is acts like BUS driver, in that it detects the GPU, its capabilities and loads
|
|
||||||
//! sub-drivers for each device detected. Similar The device manager
|
|
||||||
//! finds display adaptor e.g. over the PCI/ACPI, and passes the buck on to this driver to handle.
|
|
||||||
//!
|
|
||||||
//! The display driver provides the low level part of identifying the device and launching the
|
|
||||||
//! generic device driver for a GPU vendor.
|
|
||||||
//!
|
|
||||||
//! It takes over the framebuffer feature that was setup during system boot.
|
|
||||||
const std = @import("std");
|
|
||||||
const device = @import("driver");
|
|
||||||
const ipc = @import("ipc");
|
|
||||||
const process = @import("process");
|
|
||||||
const service = @import("service");
|
|
||||||
const device_manager = @import("driver");
|
|
||||||
const logging = @import("logging");
|
|
||||||
const mmio = @import("mmio");
|
|
||||||
const display_protocol = @import("display-protocol");
|
|
||||||
const scanout_protocol = @import("scanout-protocol");
|
|
||||||
var device_id: u64 = 0;
|
|
||||||
|
|
||||||
fn initialise(endpoint: ipc.Handle) bool {
|
|
||||||
_ = endpoint;
|
|
||||||
// Hello the device manager (role: device — we claim one GPU's PCI function
|
|
||||||
// and serve its display engine; we report no children). Best-effort: without a
|
|
||||||
// manager the driver still runs standalone; when present, the manager marks us
|
|
||||||
// up before the hello deadline and restarts us if we die.
|
|
||||||
_ = device_manager.hello(.device, device_id);
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
|
||||||
_ = sender;
|
|
||||||
_ = capability;
|
|
||||||
_ = reply;
|
|
||||||
|
|
||||||
if (message.len < scanout_protocol.request_size) return 0;
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn main(init: process.Init) void {
|
|
||||||
const argument = init.arguments.get(1) orelse {
|
|
||||||
_ = logging.write("display: missing device id (argv[1])\n");
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
device_id = std.fmt.parseInt(u64, argument, 10) catch {
|
|
||||||
std.log.info("malformed device id '{s}'", .{argument});
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
service.run(256, .{
|
|
||||||
.service = .scanout,
|
|
||||||
.init = initialise,
|
|
||||||
.on_message = onMessage,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
@@ -1,69 +0,0 @@
|
|||||||
//! /system/drivers/display/intel-integrated - the intel 985 family display engine driver.
|
|
||||||
const std = @import("std");
|
|
||||||
const device = @import("driver");
|
|
||||||
const ipc = @import("ipc");
|
|
||||||
const process = @import("process");
|
|
||||||
const service = @import("service");
|
|
||||||
const logging = @import("logging");
|
|
||||||
const mmio = @import("mmio");
|
|
||||||
const display_protocol = @import("display-protocol");
|
|
||||||
const scanout_protocol = @import("scanout-protocol");
|
|
||||||
const device_manager_protocol = @import("device-manager-protocol");
|
|
||||||
var device_id: u64 = 0;
|
|
||||||
|
|
||||||
|
|
||||||
fn initialise(endpoint: ipc.Handle) bool {
|
|
||||||
_ = endpoint;
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
|
||||||
_ = sender;
|
|
||||||
_ = capability;
|
|
||||||
_ = reply;
|
|
||||||
|
|
||||||
if (message.len < scanout_protocol.request_size) return 0;
|
|
||||||
const request = std.mem.bytesToValue(scanout_protocol.Request, message[0..scanout_protocol.request_size]);
|
|
||||||
switch (request.operation) {
|
|
||||||
_ => return 0,
|
|
||||||
// TODO:
|
|
||||||
// @intFromEnum(sp.Operation.present) => return scanoutStatus(reply, presentFull()),
|
|
||||||
// @intFromEnum(sp.Operation.get_modes) => {
|
|
||||||
// var response = sp.ModesReply{ .status = 0, .count = offered_modes.len, .modes = undefined };
|
|
||||||
// for (0..sp.max_modes) |i| {
|
|
||||||
// response.modes[i] = if (i < offered_modes.len)
|
|
||||||
// .{ .width = offered_modes[i].width, .height = offered_modes[i].height }
|
|
||||||
// else
|
|
||||||
// .{ .width = 0, .height = 0 };
|
|
||||||
// }
|
|
||||||
// @memcpy(reply[0..sp.modes_reply_size], std.mem.asBytes(&response));
|
|
||||||
// return sp.modes_reply_size;
|
|
||||||
// },
|
|
||||||
// @intFromEnum(sp.Operation.set_mode) => {
|
|
||||||
// const w = request.width;
|
|
||||||
// const h = request.height;
|
|
||||||
// if (w == 0 or h == 0 or w > max_width or h > max_height) return scanoutStatus(reply, false);
|
|
||||||
// current_width = w;
|
|
||||||
// current_height = h;
|
|
||||||
// return scanoutStatus(reply, setScanoutRect());
|
|
||||||
// },
|
|
||||||
else => return 0,
|
|
||||||
}
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn main(init: process.Init) void {
|
|
||||||
const argument = init.arguments.get(1) orelse {
|
|
||||||
_ = logging.write("display/intel-985: missing device id (argv[1])\n");
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
device_id = std.fmt.parseInt(u64, argument, 10) catch {
|
|
||||||
std.log.info("malformed device id '{s}'", .{argument});
|
|
||||||
return;
|
|
||||||
};
|
|
||||||
service.run(256, .{
|
|
||||||
.service = .scanout,
|
|
||||||
.init = initialise,
|
|
||||||
.on_message = onMessage,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
@@ -0,0 +1,18 @@
|
|||||||
|
//! The pci-bus driver as a binary package (docs/build-packages-plan.md):
|
||||||
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const build_support = @import("build-support");
|
||||||
|
|
||||||
|
pub fn build(b: *std.Build) void {
|
||||||
|
const exe = build_support.userBinary(b, .{
|
||||||
|
.name = "pci-bus",
|
||||||
|
.root_source_file = b.path("pci-bus.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
"device-manager-protocol", "driver", "ipc", "logging", "memory", "pci-class",
|
||||||
|
"process", "service",
|
||||||
|
},
|
||||||
|
});
|
||||||
|
b.installArtifact(exe);
|
||||||
|
}
|
||||||
@@ -0,0 +1,16 @@
|
|||||||
|
.{
|
||||||
|
.name = .pci_bus,
|
||||||
|
.version = "0.0.0",
|
||||||
|
.fingerprint = 0x283fca121f0bb145, // Changing this has security and trust implications.
|
||||||
|
.minimum_zig_version = "0.16.0",
|
||||||
|
.dependencies = .{
|
||||||
|
// build-support supplies the shared recipe; kernel is implicit in
|
||||||
|
// every binary (the root shim + link script live there). The rest
|
||||||
|
// are exactly the homes of this binary's declared imports.
|
||||||
|
.@"build-support" = .{ .path = "../../../build-support" },
|
||||||
|
.kernel = .{ .path = "../../../library/kernel" },
|
||||||
|
.device = .{ .path = "../../../library/device" },
|
||||||
|
.protocol = .{ .path = "../../../library/protocol" },
|
||||||
|
},
|
||||||
|
.paths = .{""},
|
||||||
|
}
|
||||||
@@ -21,19 +21,20 @@ const logging = @import("logging");
|
|||||||
const device_manager_protocol = @import("device-manager-protocol");
|
const device_manager_protocol = @import("device-manager-protocol");
|
||||||
const pci_class = @import("pci-class");
|
const pci_class = @import("pci-class");
|
||||||
|
|
||||||
/// Log a discovered function with its (class / subclass / prog-IF) triple decoded
|
/// Log a discovered function as its would-be /system/configuration/devices.csv columns (bus, base,
|
||||||
/// to human names — the boot-log breadcrumb that says *what* the hardware is, so
|
/// class, prog_if, vendor, device, subsystem) followed by the human-readable
|
||||||
/// "class 0x01 (Mass Storage Controller) subclass 0x06 (Serial ATA Controller)
|
/// class/subclass/prog-IF names — so a row for a new driver reads straight off the
|
||||||
/// progif 0x01 (AHCI 1.0)" reads straight off the log when writing a new driver.
|
/// boot log. `subsystem` prints as `*` when the function has none, matching the CSV
|
||||||
/// A dedicated wider buffer than `writeLine`'s, since the decoded names are long.
|
/// wildcard. All read unclaimed, through the bridge's ECAM: the enumerator never
|
||||||
fn logFunction(bus: u64, dev: u64, function: u64, class_triple: u32) void {
|
/// claims the functions it probes (pci.zig's header — the device-owned pci.Function
|
||||||
|
/// view is what needs a claim, not this one). A wide buffer: the names are long.
|
||||||
|
fn logFunction(bus: u64, dev: u64, function: u64, class_triple: u32, vendor_id: u16, product_id: u16, subsystem: u32) void {
|
||||||
const cc = pci_class.ClassCode.unpack(@truncate(class_triple));
|
const cc = pci_class.ClassCode.unpack(@truncate(class_triple));
|
||||||
const pif = pci_class.progIfName(cc.base, cc.subclass, cc.prog_if);
|
const pif = pci_class.progIfName(cc.base, cc.subclass, cc.prog_if);
|
||||||
var line: [200]u8 = undefined;
|
var sub_buffer: [8]u8 = undefined;
|
||||||
const text = if (pif.len != 0)
|
const sub = if (subsystem == 0) "*" else std.fmt.bufPrint(&sub_buffer, "{X:0>8}", .{subsystem}) catch "*";
|
||||||
std.fmt.bufPrint(&line, "/system/drivers/pci-bus: {d}:{d}.{d} class 0x{x:0>2} ({s}) subclass 0x{x:0>2} ({s}) progif 0x{x:0>2} ({s})\n", .{ bus, dev, function, cc.base, pci_class.className(cc.base), cc.subclass, pci_class.subclassName(cc.base, cc.subclass), cc.prog_if, pif }) catch return
|
var line: [320]u8 = undefined;
|
||||||
else
|
const text = std.fmt.bufPrint(&line, "/system/drivers/pci-bus: {d}:{d}.{d} bus=pci base={X:0>2} class={X:0>2} prog_if={X:0>2} vendor={X:0>4} device={X:0>4} subsystem={s} — {s} / {s}{s}{s}\n", .{ bus, dev, function, cc.base, cc.subclass, cc.prog_if, vendor_id, product_id, sub, pci_class.className(cc.base), pci_class.subclassName(cc.base, cc.subclass), if (pif.len != 0) " / " else "", pif }) catch return;
|
||||||
std.fmt.bufPrint(&line, "/system/drivers/pci-bus: {d}:{d}.{d} class 0x{x:0>2} ({s}) subclass 0x{x:0>2} ({s}) progif 0x{x:0>2}\n", .{ bus, dev, function, cc.base, pci_class.className(cc.base), cc.subclass, pci_class.subclassName(cc.base, cc.subclass), cc.prog_if }) catch return;
|
|
||||||
_ = logging.write(text);
|
_ = logging.write(text);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -136,7 +137,6 @@ fn scan() void {
|
|||||||
if (vendor_device & 0xFFFF == 0xFFFF) continue;
|
if (vendor_device & 0xFFFF == 0xFFFF) continue;
|
||||||
const class_revision = configRead(bus, dev, function, 0x08);
|
const class_revision = configRead(bus, dev, function, 0x08);
|
||||||
found += 1;
|
found += 1;
|
||||||
logFunction(bus, dev, function, class_revision >> 8);
|
|
||||||
registerAndReport(bus, dev, function, class_revision >> 8);
|
registerAndReport(bus, dev, function, class_revision >> 8);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -153,6 +153,12 @@ fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void
|
|||||||
var descriptor = std.mem.zeroes(device.DeviceDescriptor);
|
var descriptor = std.mem.zeroes(device.DeviceDescriptor);
|
||||||
descriptor.class = @intFromEnum(device.DeviceClass.pci_device);
|
descriptor.class = @intFromEnum(device.DeviceClass.pci_device);
|
||||||
descriptor.pci_class = class_triple;
|
descriptor.pci_class = class_triple;
|
||||||
|
// Vendor/device from the first config dword (0x00): low half vendor, high half
|
||||||
|
// device. These carry to the manager's /system/configuration/devices.csv matcher so a function
|
||||||
|
// can bind on its exact 1AF4:1050 identity, not just its class triple.
|
||||||
|
const vendor_device = configRead(bus, dev, function, 0x00);
|
||||||
|
descriptor.vendor = @truncate(vendor_device);
|
||||||
|
descriptor.device = @truncate(vendor_device >> 16);
|
||||||
descriptor.resources[0] = .{
|
descriptor.resources[0] = .{
|
||||||
.kind = @intFromEnum(device.ResourceKind.memory),
|
.kind = @intFromEnum(device.ResourceKind.memory),
|
||||||
.start = ecam_physical + (((bus - start_bus) << 20) | (dev << 15) | (function << 12)),
|
.start = ecam_physical + (((bus - start_bus) << 20) | (dev << 15) | (function << 12)),
|
||||||
@@ -164,6 +170,11 @@ fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void
|
|||||||
// write all-ones, read the writable mask back, restore. Header type 0 only.
|
// write all-ones, read the writable mask back, restore. Header type 0 only.
|
||||||
const header_type = (configRead(bus, dev, function, 0x0C) >> 16) & 0x7F;
|
const header_type = (configRead(bus, dev, function, 0x0C) >> 16) & 0x7F;
|
||||||
if (header_type == 0) {
|
if (header_type == 0) {
|
||||||
|
// Subsystem id lives at 0x2C only on type-0 (device) headers, not on
|
||||||
|
// bridges: dword low half is subsystem-vendor, high half subsystem-device.
|
||||||
|
// Repack vendor-first so it reads like the CSV's `ssvid:ssid`.
|
||||||
|
const subsystem_dword = configRead(bus, dev, function, 0x2C);
|
||||||
|
descriptor.subsystem = (@as(u32, @truncate(subsystem_dword)) << 16) | @as(u32, @truncate(subsystem_dword >> 16));
|
||||||
const command = configRead16(bus, dev, function, 0x04);
|
const command = configRead16(bus, dev, function, 0x04);
|
||||||
configWrite16(bus, dev, function, 0x04, command & ~@as(u16, 0b11));
|
configWrite16(bus, dev, function, 0x04, command & ~@as(u16, 0b11));
|
||||||
var i: u64 = 0;
|
var i: u64 = 0;
|
||||||
@@ -210,15 +221,23 @@ fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void
|
|||||||
configWrite16(bus, dev, function, 0x04, command);
|
configWrite16(bus, dev, function, 0x04, command);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The devices.csv-column + friendly-name breadcrumb, now that vendor/device/
|
||||||
|
// subsystem are read. Every discovered function is logged, matched or not.
|
||||||
|
logFunction(bus, dev, function, class_triple, descriptor.vendor, descriptor.device, descriptor.subsystem);
|
||||||
|
|
||||||
const registered = device.register(bridge_id, &descriptor) orelse {
|
const registered = device.register(bridge_id, &descriptor) orelse {
|
||||||
std.log.info("register refused for {d}:{d}.{d}", .{ bus, dev, function });
|
std.log.info("register refused for {d}:{d}.{d}", .{ bus, dev, function });
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
const report = device_manager_protocol.ChildAdded{
|
const report = device_manager_protocol.ChildAdded{
|
||||||
|
.bus = @intFromEnum(device_manager_protocol.BusKind.pci),
|
||||||
.parent = bridge_id,
|
.parent = bridge_id,
|
||||||
.bus_address = (bus << 8) | (dev << 3) | function,
|
.bus_address = (bus << 8) | (dev << 3) | function,
|
||||||
.identity = class_triple,
|
.identity = class_triple,
|
||||||
.device_id = registered,
|
.device_id = registered,
|
||||||
|
.vendor = descriptor.vendor,
|
||||||
|
.device = descriptor.device,
|
||||||
|
.subsystem = descriptor.subsystem,
|
||||||
};
|
};
|
||||||
var reply: [device_manager_protocol.message_maximum]u8 = undefined;
|
var reply: [device_manager_protocol.message_maximum]u8 = undefined;
|
||||||
_ = ipc.call(manager_handle, std.mem.asBytes(&report), &reply) catch {
|
_ = ipc.call(manager_handle, std.mem.asBytes(&report), &reply) catch {
|
||||||
|
|||||||
@@ -0,0 +1,50 @@
|
|||||||
|
//! The ps2-bus driver as a binary package (docs/build-packages-plan.md):
|
||||||
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const build_support = @import("build-support");
|
||||||
|
|
||||||
|
pub fn build(b: *std.Build) void {
|
||||||
|
const ps2_bus_exe = build_support.userBinary(b, .{
|
||||||
|
.name = "ps2-bus",
|
||||||
|
.root_source_file = b.path("ps2-bus.zig"),
|
||||||
|
.imports = &.{ "acpi-ids", "driver", "ipc", "logging", "memory", "process", "service", "time" },
|
||||||
|
});
|
||||||
|
b.installArtifact(ps2_bus_exe);
|
||||||
|
|
||||||
|
const ps2_keyboard_exe = build_support.userBinary(b, .{
|
||||||
|
.name = "ps2-keyboard",
|
||||||
|
.root_source_file = b.path("keyboard.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
"acpi-ids", "driver", "input-client", "input-protocol", "ipc", "logging", "memory",
|
||||||
|
"process", "time", "xkeyboard-config",
|
||||||
|
},
|
||||||
|
});
|
||||||
|
b.installArtifact(ps2_keyboard_exe);
|
||||||
|
|
||||||
|
const ps2_mouse_exe = build_support.userBinary(b, .{
|
||||||
|
.name = "ps2-mouse",
|
||||||
|
.root_source_file = b.path("mouse.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
"acpi-ids", "driver", "input-client", "input-protocol", "ipc", "logging", "memory",
|
||||||
|
"process", "time",
|
||||||
|
},
|
||||||
|
});
|
||||||
|
b.installArtifact(ps2_mouse_exe);
|
||||||
|
|
||||||
|
// Standalone `zig build test`; the root aggregate depends on this step.
|
||||||
|
const test_step = b.step("test", "Run the ps2-bus unit tests");
|
||||||
|
for ([_][]const u8{
|
||||||
|
"scancode.zig", // set-2 decode + keyboard state machine
|
||||||
|
"mouse-packet.zig", // 3-byte mouse packet assembly
|
||||||
|
}) |test_root| {
|
||||||
|
const unit_tests = b.addTest(.{
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path(test_root),
|
||||||
|
.target = b.resolveTargetQuery(.{}),
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
test_step.dependOn(&b.addRunArtifact(unit_tests).step);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,18 @@
|
|||||||
|
.{
|
||||||
|
.name = .ps2_bus,
|
||||||
|
.version = "0.0.0",
|
||||||
|
.fingerprint = 0x642a365353bf7de9, // Changing this has security and trust implications.
|
||||||
|
.minimum_zig_version = "0.16.0",
|
||||||
|
.dependencies = .{
|
||||||
|
// build-support supplies the shared recipe; kernel is implicit in
|
||||||
|
// every binary (the root shim + link script live there). The rest
|
||||||
|
// are exactly the homes of this binary's declared imports.
|
||||||
|
.@"build-support" = .{ .path = "../../../build-support" },
|
||||||
|
.kernel = .{ .path = "../../../library/kernel" },
|
||||||
|
.client = .{ .path = "../../../library/client" },
|
||||||
|
.device = .{ .path = "../../../library/device" },
|
||||||
|
.protocol = .{ .path = "../../../library/protocol" },
|
||||||
|
.@"xkeyboard-config" = .{ .path = "../../../library/xkeyboard-config" },
|
||||||
|
},
|
||||||
|
.paths = .{""},
|
||||||
|
}
|
||||||
@@ -19,7 +19,7 @@ const device = @import("driver");
|
|||||||
const ipc = @import("ipc");
|
const ipc = @import("ipc");
|
||||||
const process = @import("process");
|
const process = @import("process");
|
||||||
const time = @import("time");
|
const time = @import("time");
|
||||||
const input = @import("input");
|
const input = @import("input-client");
|
||||||
const memory = @import("memory");
|
const memory = @import("memory");
|
||||||
const logging = @import("logging");
|
const logging = @import("logging");
|
||||||
const xkb = @import("xkeyboard-config");
|
const xkb = @import("xkeyboard-config");
|
||||||
|
|||||||
@@ -19,7 +19,7 @@ const device = @import("driver");
|
|||||||
const ipc = @import("ipc");
|
const ipc = @import("ipc");
|
||||||
const process = @import("process");
|
const process = @import("process");
|
||||||
const time = @import("time");
|
const time = @import("time");
|
||||||
const input = @import("input");
|
const input = @import("input-client");
|
||||||
const memory = @import("memory");
|
const memory = @import("memory");
|
||||||
const logging = @import("logging");
|
const logging = @import("logging");
|
||||||
const ps2 = @import("ps2-library.zig");
|
const ps2 = @import("ps2-library.zig");
|
||||||
|
|||||||
@@ -0,0 +1,42 @@
|
|||||||
|
//! The usb-hid driver as a binary package (docs/build-packages-plan.md):
|
||||||
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const build_support = @import("build-support");
|
||||||
|
|
||||||
|
pub fn build(b: *std.Build) void {
|
||||||
|
const usb_hid_keyboard_exe = build_support.userBinary(b, .{
|
||||||
|
.name = "usb-hid-keyboard",
|
||||||
|
.root_source_file = b.path("keyboard.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
"driver", "input-client", "input-protocol", "ipc", "logging", "process", "service",
|
||||||
|
"usb", "usb-abi", "xkeyboard-config",
|
||||||
|
},
|
||||||
|
});
|
||||||
|
b.installArtifact(usb_hid_keyboard_exe);
|
||||||
|
|
||||||
|
const usb_hid_mouse_exe = build_support.userBinary(b, .{
|
||||||
|
.name = "usb-hid-mouse",
|
||||||
|
.root_source_file = b.path("mouse.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
"driver", "input-client", "input-protocol", "ipc", "logging", "process", "service",
|
||||||
|
"usb", "usb-abi",
|
||||||
|
},
|
||||||
|
});
|
||||||
|
b.installArtifact(usb_hid_mouse_exe);
|
||||||
|
|
||||||
|
// Standalone `zig build test`; the root aggregate depends on this step.
|
||||||
|
const test_step = b.step("test", "Run the usb-hid unit tests");
|
||||||
|
for ([_][]const u8{
|
||||||
|
"hid-report.zig", // HID boot-report keyboard/mouse decode
|
||||||
|
}) |test_root| {
|
||||||
|
const unit_tests = b.addTest(.{
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path(test_root),
|
||||||
|
.target = b.resolveTargetQuery(.{}),
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
test_step.dependOn(&b.addRunArtifact(unit_tests).step);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,18 @@
|
|||||||
|
.{
|
||||||
|
.name = .usb_hid,
|
||||||
|
.version = "0.0.0",
|
||||||
|
.fingerprint = 0x66328b738fffff01, // Changing this has security and trust implications.
|
||||||
|
.minimum_zig_version = "0.16.0",
|
||||||
|
.dependencies = .{
|
||||||
|
// build-support supplies the shared recipe; kernel is implicit in
|
||||||
|
// every binary (the root shim + link script live there). The rest
|
||||||
|
// are exactly the homes of this binary's declared imports.
|
||||||
|
.@"build-support" = .{ .path = "../../../build-support" },
|
||||||
|
.kernel = .{ .path = "../../../library/kernel" },
|
||||||
|
.client = .{ .path = "../../../library/client" },
|
||||||
|
.device = .{ .path = "../../../library/device" },
|
||||||
|
.protocol = .{ .path = "../../../library/protocol" },
|
||||||
|
.@"xkeyboard-config" = .{ .path = "../../../library/xkeyboard-config" },
|
||||||
|
},
|
||||||
|
.paths = .{""},
|
||||||
|
}
|
||||||
@@ -18,7 +18,7 @@ const std = @import("std");
|
|||||||
const ipc = @import("ipc");
|
const ipc = @import("ipc");
|
||||||
const process = @import("process");
|
const process = @import("process");
|
||||||
const service = @import("service");
|
const service = @import("service");
|
||||||
const input = @import("input");
|
const input = @import("input-client");
|
||||||
const device_manager = @import("driver");
|
const device_manager = @import("driver");
|
||||||
const logging = @import("logging");
|
const logging = @import("logging");
|
||||||
const usb = @import("usb");
|
const usb = @import("usb");
|
||||||
|
|||||||
@@ -14,7 +14,7 @@ const std = @import("std");
|
|||||||
const ipc = @import("ipc");
|
const ipc = @import("ipc");
|
||||||
const process = @import("process");
|
const process = @import("process");
|
||||||
const service = @import("service");
|
const service = @import("service");
|
||||||
const input = @import("input");
|
const input = @import("input-client");
|
||||||
const device_manager = @import("driver");
|
const device_manager = @import("driver");
|
||||||
const logging = @import("logging");
|
const logging = @import("logging");
|
||||||
const usb = @import("usb");
|
const usb = @import("usb");
|
||||||
|
|||||||
@@ -0,0 +1,33 @@
|
|||||||
|
//! The usb-storage driver as a binary package (docs/build-packages-plan.md):
|
||||||
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const build_support = @import("build-support");
|
||||||
|
|
||||||
|
pub fn build(b: *std.Build) void {
|
||||||
|
const exe = build_support.userBinary(b, .{
|
||||||
|
.name = "usb-storage",
|
||||||
|
.root_source_file = b.path("usb-storage.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
"block-protocol", "driver", "ipc", "logging", "memory", "process", "service",
|
||||||
|
"time", "usb",
|
||||||
|
},
|
||||||
|
});
|
||||||
|
b.installArtifact(exe);
|
||||||
|
|
||||||
|
// Standalone `zig build test`; the root aggregate depends on this step.
|
||||||
|
const test_step = b.step("test", "Run the usb-storage unit tests");
|
||||||
|
for ([_][]const u8{
|
||||||
|
"bulk-only-transport.zig", // CBW/CSW wrapper sizes
|
||||||
|
"scsi.zig", // SCSI CDB encodings (big-endian)
|
||||||
|
}) |test_root| {
|
||||||
|
const unit_tests = b.addTest(.{
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path(test_root),
|
||||||
|
.target = b.resolveTargetQuery(.{}),
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
test_step.dependOn(&b.addRunArtifact(unit_tests).step);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,16 @@
|
|||||||
|
.{
|
||||||
|
.name = .usb_storage,
|
||||||
|
.version = "0.0.0",
|
||||||
|
.fingerprint = 0xce09fdc4c50bb4fe, // Changing this has security and trust implications.
|
||||||
|
.minimum_zig_version = "0.16.0",
|
||||||
|
.dependencies = .{
|
||||||
|
// build-support supplies the shared recipe; kernel is implicit in
|
||||||
|
// every binary (the root shim + link script live there). The rest
|
||||||
|
// are exactly the homes of this binary's declared imports.
|
||||||
|
.@"build-support" = .{ .path = "../../../build-support" },
|
||||||
|
.kernel = .{ .path = "../../../library/kernel" },
|
||||||
|
.device = .{ .path = "../../../library/device" },
|
||||||
|
.protocol = .{ .path = "../../../library/protocol" },
|
||||||
|
},
|
||||||
|
.paths = .{""},
|
||||||
|
}
|
||||||
@@ -90,9 +90,22 @@ fn initialise(endpoint: ipc.Handle) bool {
|
|||||||
bring_up_failed = true;
|
bring_up_failed = true;
|
||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
command_wrapper = memory.dmaAlloc(4096, memory.dma_coherent) orelse return false;
|
// Shareable, so each buffer's capability can be handed to the controller: usb-storage
|
||||||
status_wrapper = memory.dmaAlloc(4096, memory.dma_coherent) orelse return false;
|
// owns no device, so its buffers are not auto-bound anywhere — the controller reaches
|
||||||
command_data = memory.dmaAlloc(4096, memory.dma_coherent) orelse return false;
|
// them only once attached. (No-op binding when no IOMMU is enforcing.)
|
||||||
|
command_wrapper = memory.dmaAlloc(4096, memory.dma_coherent | memory.dma_shareable) orelse return false;
|
||||||
|
status_wrapper = memory.dmaAlloc(4096, memory.dma_coherent | memory.dma_shareable) orelse return false;
|
||||||
|
command_data = memory.dmaAlloc(4096, memory.dma_coherent | memory.dma_shareable) orelse return false;
|
||||||
|
for ([_]memory.DmaRegion{ command_wrapper, status_wrapper, command_data }) |region| {
|
||||||
|
if (region.handle) |handle| {
|
||||||
|
if (!device.attachDma(handle)) {
|
||||||
|
_ = logging.write("/system/drivers/usb-storage: could not attach a DMA buffer to the controller\n");
|
||||||
|
bring_up_failed = true;
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
_ = ipc.close(handle); // the binding holds its own reference now
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// Bring the LUN up: wait for it to be ready (clearing the initial unit-attention
|
// Bring the LUN up: wait for it to be ready (clearing the initial unit-attention
|
||||||
// with REQUEST SENSE), identify it, and read its capacity.
|
// with REQUEST SENSE), identify it, and read its capacity.
|
||||||
@@ -135,10 +148,17 @@ fn initialise(endpoint: ipc.Handle) bool {
|
|||||||
/// caller's DMA buffer (named by physical address).
|
/// caller's DMA buffer (named by physical address).
|
||||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||||
_ = sender;
|
_ = sender;
|
||||||
_ = capability;
|
|
||||||
if (message.len < block_protocol.request_size) return 0;
|
if (message.len < block_protocol.request_size) return 0;
|
||||||
const request = std.mem.bytesToValue(block_protocol.Request, message[0..block_protocol.request_size]);
|
const request = std.mem.bytesToValue(block_protocol.Request, message[0..block_protocol.request_size]);
|
||||||
switch (request.operation) {
|
switch (request.operation) {
|
||||||
|
@intFromEnum(block_protocol.Operation.attach) => {
|
||||||
|
// The filesystem's DMA buffer: forward its capability to the controller so
|
||||||
|
// the device can reach it, then release our copy (the binding holds a ref).
|
||||||
|
const handle = capability orelse return writeReply(reply, .{ .status = -1, .block_size = 0, .block_count = 0 });
|
||||||
|
const ok = device.attachDma(handle);
|
||||||
|
_ = ipc.close(handle);
|
||||||
|
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = 0, .block_count = 0 });
|
||||||
|
},
|
||||||
@intFromEnum(block_protocol.Operation.geometry) => {
|
@intFromEnum(block_protocol.Operation.geometry) => {
|
||||||
return writeReply(reply, .{ .status = 0, .block_size = block_size, .block_count = block_count });
|
return writeReply(reply, .{ .status = 0, .block_size = block_size, .block_count = block_count });
|
||||||
},
|
},
|
||||||
|
|||||||
@@ -0,0 +1,19 @@
|
|||||||
|
//! The usb-xhci-bus driver as a binary package (docs/build-packages-plan.md):
|
||||||
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const build_support = @import("build-support");
|
||||||
|
|
||||||
|
pub fn build(b: *std.Build) void {
|
||||||
|
const exe = build_support.userBinary(b, .{
|
||||||
|
.name = "usb-xhci-bus",
|
||||||
|
.root_source_file = b.path("usb-xhci-bus.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
"device-manager-protocol", "driver", "input-client", "ipc", "logging", "memory",
|
||||||
|
"mmio", "pci", "process", "service", "time", "usb-abi", "usb-ids",
|
||||||
|
"usb-transfer-protocol",
|
||||||
|
},
|
||||||
|
});
|
||||||
|
b.installArtifact(exe);
|
||||||
|
}
|
||||||
@@ -0,0 +1,17 @@
|
|||||||
|
.{
|
||||||
|
.name = .usb_xhci_bus,
|
||||||
|
.version = "0.0.0",
|
||||||
|
.fingerprint = 0x46d21373f05f6b20, // Changing this has security and trust implications.
|
||||||
|
.minimum_zig_version = "0.16.0",
|
||||||
|
.dependencies = .{
|
||||||
|
// build-support supplies the shared recipe; kernel is implicit in
|
||||||
|
// every binary (the root shim + link script live there). The rest
|
||||||
|
// are exactly the homes of this binary's declared imports.
|
||||||
|
.@"build-support" = .{ .path = "../../../build-support" },
|
||||||
|
.kernel = .{ .path = "../../../library/kernel" },
|
||||||
|
.client = .{ .path = "../../../library/client" },
|
||||||
|
.device = .{ .path = "../../../library/device" },
|
||||||
|
.protocol = .{ .path = "../../../library/protocol" },
|
||||||
|
},
|
||||||
|
.paths = .{""},
|
||||||
|
}
|
||||||
@@ -19,7 +19,7 @@ const ipc = @import("ipc");
|
|||||||
const process = @import("process");
|
const process = @import("process");
|
||||||
const service = @import("service");
|
const service = @import("service");
|
||||||
const time = @import("time");
|
const time = @import("time");
|
||||||
const input = @import("input");
|
const input = @import("input-client");
|
||||||
const device_manager = @import("driver");
|
const device_manager = @import("driver");
|
||||||
const memory = @import("memory");
|
const memory = @import("memory");
|
||||||
const logging = @import("logging");
|
const logging = @import("logging");
|
||||||
@@ -28,18 +28,37 @@ const usb_ids = @import("usb-ids");
|
|||||||
const usb_abi = @import("usb-abi");
|
const usb_abi = @import("usb-abi");
|
||||||
const usb_transfer_protocol = @import("usb-transfer-protocol");
|
const usb_transfer_protocol = @import("usb-transfer-protocol");
|
||||||
const library = @import("usb-xhci-library.zig");
|
const library = @import("usb-xhci-library.zig");
|
||||||
|
const pci = @import("pci");
|
||||||
|
|
||||||
/// The controller engine (reset, rings, transfers), stood up in `initialise`.
|
/// The controller engine (reset, rings, transfers), stood up in `initialise`.
|
||||||
var controller: ?library.Controller = null;
|
var controller: ?library.Controller = null;
|
||||||
|
|
||||||
/// This driver's service endpoint (registered as `.usb_bus`), where class-driver
|
/// This driver's service endpoint (registered as `.usb_bus`), where class-driver
|
||||||
/// requests, signals, and the interrupt-poll timer all arrive.
|
/// requests, signals, MSI notifications, and the poll/reconcile timer all arrive.
|
||||||
var service_endpoint: ipc.Handle = 0;
|
var service_endpoint: ipc.Handle = 0;
|
||||||
|
|
||||||
/// How often the driver drains the event ring for interrupt reports (~125 Hz),
|
/// How often the driver drains the event ring for interrupt reports (~125 Hz) when
|
||||||
/// re-armed each tick. Frequent enough for responsive input.
|
/// polling, re-armed each tick. Frequent enough for responsive input.
|
||||||
const poll_interval_ms: u64 = 8;
|
const poll_interval_ms: u64 = 8;
|
||||||
|
|
||||||
|
/// The timer interval in MSI mode: the ring is drained at interrupt time, and the tick
|
||||||
|
/// only reconciles root ports (real hardware delivers late USB2 companion-hub debounce
|
||||||
|
/// with no reliable port-change event — see onNotification) and un-wedges a lost MSI
|
||||||
|
/// edge (edge-triggered, no kernel mask/ack: a missed IP clear stalls, never storms).
|
||||||
|
const reconcile_interval_ms: u64 = 250;
|
||||||
|
|
||||||
|
/// The controller's own descriptor, kept at file scope because `pci.Function` holds a
|
||||||
|
/// pointer to it for the whole bring-up.
|
||||||
|
var controller_descriptor: device.DeviceDescriptor = undefined;
|
||||||
|
|
||||||
|
/// Non-null iff MSI mode is active: the vector whose notification badge means "the
|
||||||
|
/// controller interrupted". Null means the 8 ms polling fallback is running.
|
||||||
|
var msi_vector: ?u32 = null;
|
||||||
|
|
||||||
|
fn timerInterval() u64 {
|
||||||
|
return if (msi_vector != null) reconcile_interval_ms else poll_interval_ms;
|
||||||
|
}
|
||||||
|
|
||||||
/// The class driver endpoints that opened each device, so interrupt reports can
|
/// The class driver endpoints that opened each device, so interrupt reports can
|
||||||
/// be pushed back to them. Keyed by the device token (the interface's device id).
|
/// be pushed back to them. Keyed by the device token (the interface's device id).
|
||||||
const Open = struct {
|
const Open = struct {
|
||||||
@@ -95,6 +114,7 @@ fn initialise(endpoint: ipc.Handle) bool {
|
|||||||
std.log.info("device {d} not in the device tree", .{controller_id});
|
std.log.info("device {d} not in the device tree", .{controller_id});
|
||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
|
controller_descriptor = descriptor;
|
||||||
|
|
||||||
// The xHC's registers live behind the first memory BAR. Resource 0 is the
|
// The xHC's registers live behind the first memory BAR. Resource 0 is the
|
||||||
// function's ECAM configuration space (M15), so the walk starts at 1.
|
// function's ECAM configuration space (M15), so the walk starts at 1.
|
||||||
@@ -118,6 +138,14 @@ fn initialise(endpoint: ipc.Handle) bool {
|
|||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// Message-signalled interrupt setup comes BEFORE the controller bring-up, not
|
||||||
|
// after: Controller.init writes IMAN.IE, and QEMU's xhci only registers the MSI-X
|
||||||
|
// vector as in-use when that write happens with MSI-X already enabled (its
|
||||||
|
// intr_update callback early-outs on !msix_enabled, and msix_notify silently
|
||||||
|
// drops interrupts for an unused vector). Real hardware does not care about the
|
||||||
|
// order; QEMU requires it.
|
||||||
|
setupMsi();
|
||||||
|
|
||||||
// Bring the controller up: reset it, stand up the command and event rings,
|
// Bring the controller up: reset it, stand up the command and event rings,
|
||||||
// and start it running (the hardware half lives in usb-xhci-library.zig).
|
// and start it running (the hardware half lives in usb-xhci-library.zig).
|
||||||
controller = library.Controller.init(register_base) orelse {
|
controller = library.Controller.init(register_base) orelse {
|
||||||
@@ -146,12 +174,52 @@ fn initialise(endpoint: ipc.Handle) bool {
|
|||||||
|
|
||||||
scanPorts(handle);
|
scanPorts(handle);
|
||||||
|
|
||||||
// Arm the poll timer that drains interrupt reports from the event ring. It is
|
// Arm the timer: in polling mode it drains the event ring; in MSI mode it is the
|
||||||
// re-armed on each tick in onNotification; class drivers subscribe later.
|
// slower port-reconcile/safety-net tick. Re-armed on each tick in onNotification.
|
||||||
_ = time.timerOnce(service_endpoint, poll_interval_ms);
|
_ = time.timerOnce(service_endpoint, timerInterval());
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Switch the event ring from timer polling to message-signalled interrupts, if the
|
||||||
|
/// whole path is available: map the function's config space, bind a vector, program
|
||||||
|
/// the MSI capability — or, on an MSI-X-only function (QEMU's qemu-xhci is one: it
|
||||||
|
/// advertises MSI-X and PCIe but no plain MSI), entry 0 of the MSI-X table, which
|
||||||
|
/// takes the same kernel (address, data) pair (xHCI interrupter 0 raises vector 0).
|
||||||
|
/// Any step failing leaves `msi_vector` null and the 8 ms polling path exactly as it
|
||||||
|
/// was. The controller side needs nothing extra — IMAN.IE and USBCMD.INTE are already
|
||||||
|
/// set (see Controller.init: QEMU only writes runtime events with the interrupter
|
||||||
|
/// enabled).
|
||||||
|
///
|
||||||
|
/// After a supervised kill, the kernel drops the vector binding but the device still
|
||||||
|
/// has the interrupt enabled and fires the stale vector; the kernel EOIs it
|
||||||
|
/// harmlessly, and the respawned driver re-runs this with its fresh vector.
|
||||||
|
fn setupMsi() void {
|
||||||
|
var function = pci.Function.map(controller_id, &controller_descriptor) orelse {
|
||||||
|
std.log.info("config-space map failed; polling at {d} ms", .{poll_interval_ms});
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
function.enableMemoryAndBusMaster();
|
||||||
|
const message = device.msiBind(controller_id, service_endpoint) orelse {
|
||||||
|
std.log.info("msi_bind unavailable; polling at {d} ms", .{poll_interval_ms});
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
if (function.programMsi(message)) {
|
||||||
|
msi_vector = message.data;
|
||||||
|
std.log.info("msi active (vector {d}); reconcile tick at {d} ms", .{ message.data, reconcile_interval_ms });
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (function.msix()) |table| {
|
||||||
|
if (table.programEntry(0, message) and table.unmaskEntry(0)) {
|
||||||
|
table.enable();
|
||||||
|
function.setInterruptDisable();
|
||||||
|
msi_vector = message.data;
|
||||||
|
std.log.info("msix active (vector {d}); reconcile tick at {d} ms", .{ message.data, reconcile_interval_ms });
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
std.log.info("no msi/msi-x capability; polling at {d} ms", .{poll_interval_ms});
|
||||||
|
}
|
||||||
|
|
||||||
var register_base: usize = 0;
|
var register_base: usize = 0;
|
||||||
|
|
||||||
/// The xHCI default Protocol Speed IDs (the PORTSC port-speed field, bits 13:10)
|
/// The xHCI default Protocol Speed IDs (the PORTSC port-speed field, bits 13:10)
|
||||||
@@ -379,6 +447,7 @@ fn reportInterface(manager: ipc.Handle, port: u32, interface: library.InterfaceI
|
|||||||
};
|
};
|
||||||
|
|
||||||
const report = device_manager_protocol.ChildAdded{
|
const report = device_manager_protocol.ChildAdded{
|
||||||
|
.bus = @intFromEnum(device_manager_protocol.BusKind.usb),
|
||||||
.parent = controller_id,
|
.parent = controller_id,
|
||||||
.bus_address = (@as(u64, port) << 8) | interface.number,
|
.bus_address = (@as(u64, port) << 8) | interface.number,
|
||||||
.identity = identity,
|
.identity = identity,
|
||||||
@@ -389,13 +458,16 @@ fn reportInterface(manager: ipc.Handle, port: u32, interface: library.InterfaceI
|
|||||||
std.log.info("child report for port {d} interface {d} failed", .{ port, interface.number });
|
std.log.info("child report for port {d} interface {d} failed", .{ port, interface.number });
|
||||||
return null;
|
return null;
|
||||||
};
|
};
|
||||||
std.log.info("port {d} interface {d}: {s} ({d}/{d}/{d}) registered as device {d}", .{
|
// The devices.csv columns (bus=usb, and the class triple as base/class/prog_if)
|
||||||
|
// then the human-readable interface name — a would-be /system/configuration/devices.csv row read
|
||||||
|
// straight off the boot log.
|
||||||
|
std.log.info("port {d} interface {d} bus=usb base={X:0>2} class={X:0>2} prog_if={X:0>2} — {s} registered as device {d}", .{
|
||||||
port,
|
port,
|
||||||
interface.number,
|
interface.number,
|
||||||
usb_ids.interfaceName(interface.class, interface.subclass, interface.protocol),
|
|
||||||
interface.class,
|
interface.class,
|
||||||
interface.subclass,
|
interface.subclass,
|
||||||
interface.protocol,
|
interface.protocol,
|
||||||
|
usb_ids.interfaceName(interface.class, interface.subclass, interface.protocol),
|
||||||
registered,
|
registered,
|
||||||
});
|
});
|
||||||
return registered;
|
return registered;
|
||||||
@@ -412,10 +484,22 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Han
|
|||||||
@intFromEnum(usb_transfer_protocol.Operation.control) => handleControl(message, reply),
|
@intFromEnum(usb_transfer_protocol.Operation.control) => handleControl(message, reply),
|
||||||
@intFromEnum(usb_transfer_protocol.Operation.interrupt_subscribe) => handleSubscribe(message, reply),
|
@intFromEnum(usb_transfer_protocol.Operation.interrupt_subscribe) => handleSubscribe(message, reply),
|
||||||
@intFromEnum(usb_transfer_protocol.Operation.bulk) => handleBulk(message, reply),
|
@intFromEnum(usb_transfer_protocol.Operation.bulk) => handleBulk(message, reply),
|
||||||
|
@intFromEnum(usb_transfer_protocol.Operation.dma_attach) => handleDmaAttach(message, reply, capability),
|
||||||
else => 0,
|
else => 0,
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// dma_attach: bind the class driver's DMA-region capability into the controller's IOMMU
|
||||||
|
/// domain, so the controller may DMA to the physical addresses inside that buffer. The
|
||||||
|
/// binding holds its own kernel reference, so the forwarded capability is closed here.
|
||||||
|
fn handleDmaAttach(message: []const u8, reply: []u8, capability: ?ipc.Handle) usize {
|
||||||
|
if (message.len < @sizeOf(usb_transfer_protocol.DmaAttachRequest)) return writeReply(reply, usb_transfer_protocol.DmaAttachReply{ .status = -1 });
|
||||||
|
const handle = capability orelse return writeReply(reply, usb_transfer_protocol.DmaAttachReply{ .status = -1 });
|
||||||
|
const ok = device.dmaBind(controller_id, handle);
|
||||||
|
_ = ipc.close(handle);
|
||||||
|
return writeReply(reply, usb_transfer_protocol.DmaAttachReply{ .status = if (ok) 0 else -1 });
|
||||||
|
}
|
||||||
|
|
||||||
fn writeReply(reply: []u8, value: anytype) usize {
|
fn writeReply(reply: []u8, value: anytype) usize {
|
||||||
const bytes = std.mem.asBytes(&value);
|
const bytes = std.mem.asBytes(&value);
|
||||||
@memcpy(reply[0..bytes.len], bytes);
|
@memcpy(reply[0..bytes.len], bytes);
|
||||||
@@ -497,10 +581,27 @@ fn handleBulk(message: []const u8, reply: []u8) usize {
|
|||||||
return writeReply(reply, usb_transfer_protocol.BulkReply{ .status = if (transferred != null) 0 else -1, .actual_length = transferred orelse 0 });
|
return writeReply(reply, usb_transfer_protocol.BulkReply{ .status = if (transferred != null) 0 else -1, .actual_length = transferred orelse 0 });
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The poll timer landed: drain any interrupt reports off the event ring and push
|
/// A timer tick or an MSI landed: drain the event ring, reconcile ports, and fan out.
|
||||||
/// each to the class driver that subscribed, then re-arm the timer.
|
/// The timer arm re-arms itself (8 ms drain when polling, 250 ms reconcile under MSI);
|
||||||
|
/// the MSI arm clears the interrupter's pending bit FIRST, then drains — so an event
|
||||||
|
/// arriving after the drain takes IP 0→1 and fires a fresh edge instead of being
|
||||||
|
/// swallowed until the reconcile tick.
|
||||||
fn onNotification(badge: u64) void {
|
fn onNotification(badge: u64) void {
|
||||||
if (badge & ipc.notify_timer_bit == 0) return;
|
if (badge & ipc.notify_timer_bit != 0) {
|
||||||
|
serviceController();
|
||||||
|
_ = time.timerOnce(service_endpoint, timerInterval());
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const vector = msi_vector orelse return;
|
||||||
|
if (badge & ~ipc.notify_badge_bit != vector) return;
|
||||||
|
if (controller) |*engine| engine.acknowledgeInterrupt();
|
||||||
|
serviceController();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Everything one servicing pass does, shared verbatim by the poll/reconcile tick and
|
||||||
|
/// the MSI notification: drain the event ring, reconcile root ports, service hub
|
||||||
|
/// changes, and push interrupt reports to their class drivers.
|
||||||
|
fn serviceController() void {
|
||||||
if (controller) |*engine| {
|
if (controller) |*engine| {
|
||||||
engine.pump();
|
engine.pump();
|
||||||
// Poll every root port and reconcile — a device present but not yet
|
// Poll every root port and reconcile — a device present but not yet
|
||||||
@@ -560,7 +661,6 @@ fn onNotification(badge: u64) void {
|
|||||||
_ = ipc.send(report.report_endpoint, std.mem.asBytes(&message));
|
_ = ipc.send(report.report_endpoint, std.mem.asBytes(&message));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
_ = time.timerOnce(service_endpoint, poll_interval_ms);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn main(init: process.Init) void {
|
pub fn main(init: process.Init) void {
|
||||||
|
|||||||
@@ -1563,6 +1563,15 @@ pub const Controller = struct {
|
|||||||
return report;
|
return report;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Clear interrupter 0's pending bit (IMAN.IP). IP is write-1-to-clear, and the
|
||||||
|
/// read-back carries IE (plain read-write) through unchanged. In MSI mode the
|
||||||
|
/// driver clears IP **before** draining the ring: an event that lands after the
|
||||||
|
/// drain then takes IP 0→1 and fires a fresh edge, where clearing afterwards would
|
||||||
|
/// leave a race in which a new event finds IP already set and raises nothing.
|
||||||
|
pub fn acknowledgeInterrupt(self: *const Controller) void {
|
||||||
|
write32(self.interrupter(interrupter_management), read32(self.interrupter(interrupter_management)) | 1);
|
||||||
|
}
|
||||||
|
|
||||||
/// Drain any events currently on the event ring: interrupt reports into the
|
/// Drain any events currently on the event ring: interrupt reports into the
|
||||||
/// report queue, PORT STATUS CHANGES into the port-change queue (hot-plug —
|
/// report queue, PORT STATUS CHANGES into the port-change queue (hot-plug —
|
||||||
/// these were silently dropped before M20). Non-blocking — called on the
|
/// these were silently dropped before M20). Non-blocking — called on the
|
||||||
|
|||||||
@@ -0,0 +1,33 @@
|
|||||||
|
//! The virtio-gpu driver as a binary package (docs/build-packages-plan.md):
|
||||||
|
//! this file names the binary and EXACTLY the modules its source imports —
|
||||||
|
//! build-support resolves each name from the domains this zon declares.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const build_support = @import("build-support");
|
||||||
|
|
||||||
|
pub fn build(b: *std.Build) void {
|
||||||
|
const exe = build_support.userBinary(b, .{
|
||||||
|
.name = "virtio-gpu",
|
||||||
|
.root_source_file = b.path("virtio-gpu.zig"),
|
||||||
|
.imports = &.{
|
||||||
|
"display-protocol", "driver", "ipc", "logging", "memory", "mmio", "pci", "process",
|
||||||
|
"scanout-protocol", "service", "time",
|
||||||
|
},
|
||||||
|
});
|
||||||
|
b.installArtifact(exe);
|
||||||
|
|
||||||
|
// Standalone `zig build test`; the root aggregate depends on this step.
|
||||||
|
const test_step = b.step("test", "Run the virtio-gpu unit tests");
|
||||||
|
for ([_][]const u8{
|
||||||
|
"virtio-gpu-protocol.zig", // virtio-gpu command struct sizes
|
||||||
|
"virtio-pci.zig", // virtio 1.0 PCI transport struct sizes
|
||||||
|
}) |test_root| {
|
||||||
|
const unit_tests = b.addTest(.{
|
||||||
|
.root_module = b.createModule(.{
|
||||||
|
.root_source_file = b.path(test_root),
|
||||||
|
.target = b.resolveTargetQuery(.{}),
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
test_step.dependOn(&b.addRunArtifact(unit_tests).step);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,16 @@
|
|||||||
|
.{
|
||||||
|
.name = .virtio_gpu,
|
||||||
|
.version = "0.0.0",
|
||||||
|
.fingerprint = 0xfb704899c18b9a23, // Changing this has security and trust implications.
|
||||||
|
.minimum_zig_version = "0.16.0",
|
||||||
|
.dependencies = .{
|
||||||
|
// build-support supplies the shared recipe; kernel is implicit in
|
||||||
|
// every binary (the root shim + link script live there). The rest
|
||||||
|
// are exactly the homes of this binary's declared imports.
|
||||||
|
.@"build-support" = .{ .path = "../../../build-support" },
|
||||||
|
.kernel = .{ .path = "../../../library/kernel" },
|
||||||
|
.device = .{ .path = "../../../library/device" },
|
||||||
|
.protocol = .{ .path = "../../../library/protocol" },
|
||||||
|
},
|
||||||
|
.paths = .{""},
|
||||||
|
}
|
||||||
@@ -33,11 +33,6 @@ const vg = @import("virtio-gpu-protocol.zig");
|
|||||||
/// the compositor in the announce so it packs colours in the surface's byte order.
|
/// the compositor in the announce so it packs colours in the surface's byte order.
|
||||||
const display_format_bgrx: u32 = 1;
|
const display_format_bgrx: u32 = 1;
|
||||||
|
|
||||||
/// The PCI vendor/device ids of a modern virtio-gpu (Red Hat / virtio; GPU is a
|
|
||||||
/// virtio-1.0-only device, so the id is always the modern 0x1050 — no legacy variant).
|
|
||||||
const virtio_vendor: u16 = 0x1AF4;
|
|
||||||
const virtio_gpu_device: u16 = 0x1050;
|
|
||||||
|
|
||||||
/// The scanout resource + shared surface are sized to the *largest* mode we offer; a mode
|
/// The scanout resource + shared surface are sized to the *largest* mode we offer; a mode
|
||||||
/// change (V5) re-points the scanout rectangle within it, so the resource, its backing, and
|
/// change (V5) re-points the scanout rectangle within it, so the resource, its backing, and
|
||||||
/// the shared surface never churn — and the surface's row stride is always `max_width`, which
|
/// the shared surface never churn — and the surface's row stride is always `max_width`, which
|
||||||
@@ -210,19 +205,15 @@ fn initialise(endpoint: ipc.Handle) bool {
|
|||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
|
|
||||||
// Config space is resource 0. Confirm it really is a virtio-gpu, then enable memory-space
|
// Config space is resource 0. The registry (/system/configuration/devices.csv) bound this driver by the
|
||||||
// decode + bus mastering (the device DMAs the ring and backing out of RAM); pci-bus only
|
// exact virtio-gpu identity (vendor 0x1AF4 / device 0x1050), so there is no re-confirm to
|
||||||
// preserves whatever the firmware left, and a secondary display is often left disabled.
|
// do here any more — map config space and enable memory-space decode + bus mastering (the
|
||||||
|
// device DMAs the ring and backing out of RAM; pci-bus only preserves whatever the firmware
|
||||||
|
// left, and a secondary display is often left disabled).
|
||||||
var function = pci.Function.map(device_id, descriptor) orelse {
|
var function = pci.Function.map(device_id, descriptor) orelse {
|
||||||
std.log.info("config-space map failed", .{});
|
std.log.info("config-space map failed", .{});
|
||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
const vendor = function.vendorId();
|
|
||||||
const dev = function.deviceId();
|
|
||||||
if (vendor != virtio_vendor or dev != virtio_gpu_device) {
|
|
||||||
std.log.info("not a virtio-gpu (vendor 0x{x} device 0x{x})", .{ vendor, dev });
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
function.enableMemoryAndBusMaster();
|
function.enableMemoryAndBusMaster();
|
||||||
|
|
||||||
// Walk the capability list for the virtio common-config and notify structures (V3 needs
|
// Walk the capability list for the virtio common-config and notify structures (V3 needs
|
||||||
@@ -332,6 +323,15 @@ fn initialise(endpoint: ipc.Handle) bool {
|
|||||||
std.log.info("could not resolve the scanout surface physical address", .{});
|
std.log.info("could not resolve the scanout surface physical address", .{});
|
||||||
return false;
|
return false;
|
||||||
};
|
};
|
||||||
|
// Bind the shared surface into this device's IOMMU domain so the GPU may DMA the
|
||||||
|
// framebuffer (attach_backing points it here). The ring/command buffers are
|
||||||
|
// dma_alloc'd and auto-bound; a shared-memory surface needs an explicit bind. We keep
|
||||||
|
// the handle (it is also passed to the display service), so do not close it. No-op
|
||||||
|
// without an IOMMU.
|
||||||
|
if (!device.dmaBind(device_id, surface.handle)) {
|
||||||
|
std.log.info("could not bind the scanout surface for DMA", .{});
|
||||||
|
return false;
|
||||||
|
}
|
||||||
{
|
{
|
||||||
const request = requestAt(vg.ResourceAttachBacking);
|
const request = requestAt(vg.ResourceAttachBacking);
|
||||||
request.* = .{
|
request.* = .{
|
||||||
|
|||||||
+153
-25
@@ -95,18 +95,46 @@ pub const PlatformInformation = struct {
|
|||||||
override_count: usize = 0,
|
override_count: usize = 0,
|
||||||
/// Whether an IOMMU (VT-d DMA-remapping unit) was found in the ACPI DMAR table.
|
/// Whether an IOMMU (VT-d DMA-remapping unit) was found in the ACPI DMAR table.
|
||||||
/// When false, `device_claim` on a DMA-capable device is equivalent to granting
|
/// When false, `device_claim` on a DMA-capable device is equivalent to granting
|
||||||
/// ring 0 — a device can DMA to any physical address (docs/driver-model.md M16).
|
/// ring 0 — a device can DMA to any physical address (docs/driver-model.md M16),
|
||||||
/// Detection is the first step; per-device domain enforcement lands with the first
|
/// and the kernel says so at every boot (the fail-open platform log line). When
|
||||||
/// DMA driver.
|
/// true, the IOMMU core builds per-device translation domains from this record.
|
||||||
iommu_present: bool = false,
|
iommu_present: bool = false,
|
||||||
/// MMIO base of the first DMA-remapping hardware unit (DMAR DRHD), when present.
|
/// true when the present unit is AMD-Vi (from IVRS) rather than Intel VT-d (DMAR).
|
||||||
|
/// The two are mutually exclusive on real hardware; the IOMMU core picks the backend.
|
||||||
|
iommu_is_amd: bool = false,
|
||||||
|
/// MMIO base of the selected DMA-remapping hardware unit — the VT-d DRHD with
|
||||||
|
/// INCLUDE_PCI_ALL (the catch-all unit; falls back to the first), or the AMD-Vi
|
||||||
|
/// IOMMU's control-register base from the first IVHD.
|
||||||
iommu_base: u64 = 0,
|
iommu_base: u64 = 0,
|
||||||
/// The unit's Version register (offset 0x00) — its low byte is major.minor;
|
/// Whether the selected unit carries INCLUDE_PCI_ALL. False means every unit is
|
||||||
/// reading it back nonzero confirms a real, mappable VT-d unit.
|
/// device-scoped (unusual) — the core still enables on the selected unit but
|
||||||
iommu_version: u32 = 0,
|
/// devices outside its scope remain untranslated.
|
||||||
/// The unit's Capability register (offset 0x08): supported address widths, number
|
iommu_include_all: bool = false,
|
||||||
/// of domains, etc. Recorded now; consumed when enforcement is built.
|
/// DRHD units in the DMAR beyond the selected one. Devices scoped to those units
|
||||||
iommu_capabilities: u64 = 0,
|
/// (typically the integrated GPU) are NOT translated by v1 — the boot log warns.
|
||||||
|
iommu_extra_units: u8 = 0,
|
||||||
|
/// Reserved-memory regions (DMAR RMRRs): firmware-owned buffers a named device
|
||||||
|
/// keeps DMAing into across the OS handoff (classically the xHC keyboard-emulation
|
||||||
|
/// buffer). These must be identity-mapped in the device's domain BEFORE translation
|
||||||
|
/// enables, or platform firmware breaks. Only single-path endpoint scopes are
|
||||||
|
/// recorded; anything fancier is skipped with a loud log at parse time.
|
||||||
|
rmrr: [maximum_rmrr]RmrrRegion = undefined,
|
||||||
|
rmrr_count: usize = 0,
|
||||||
|
/// RMRR device scopes the parser could not record (multi-hop paths, sub-hierarchy
|
||||||
|
/// types, or table overflow). Non-zero means a device keeps an unmapped firmware
|
||||||
|
/// buffer — the kernel boot log warns loudly (the platform module itself is
|
||||||
|
/// log-free by design; it records, the kernel reports).
|
||||||
|
rmrr_skipped: u8 = 0,
|
||||||
|
};
|
||||||
|
|
||||||
|
pub const maximum_rmrr = 8;
|
||||||
|
|
||||||
|
/// One recorded RMRR: the device (requester id) and the inclusive physical range it
|
||||||
|
/// must always be allowed to reach.
|
||||||
|
pub const RmrrRegion = struct {
|
||||||
|
bdf: u16,
|
||||||
|
base: u64,
|
||||||
|
limit: u64,
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Filled in by `discover`; the architecture layer reads it during bring-up.
|
/// Filled in by `discover`; the architecture layer reads it during bring-up.
|
||||||
@@ -248,6 +276,8 @@ const SLIT: [4]u8 = "SLIT".*;
|
|||||||
const SRAT: [4]u8 = "SRAT".*;
|
const SRAT: [4]u8 = "SRAT".*;
|
||||||
/// Secondary System Description Table (SSDT)
|
/// Secondary System Description Table (SSDT)
|
||||||
const DMAR: [4]u8 = "DMAR".*;
|
const DMAR: [4]u8 = "DMAR".*;
|
||||||
|
/// I/O Virtualization Reporting Structure (IVRS) — the AMD-Vi analogue of DMAR.
|
||||||
|
const IVRS: [4]u8 = "IVRS".*;
|
||||||
const SSDT: [4]u8 = "SSDT".*;
|
const SSDT: [4]u8 = "SSDT".*;
|
||||||
/// Serial Port Console Redirection table (SPCR) — the firmware's console UART.
|
/// Serial Port Console Redirection table (SPCR) — the firmware's console UART.
|
||||||
const SPCR: [4]u8 = "SPCR".*;
|
const SPCR: [4]u8 = "SPCR".*;
|
||||||
@@ -466,7 +496,9 @@ fn handleTable(device_tree: *DeviceTree, hal: Hal, sdt_physical: u64) !void {
|
|||||||
} else if (std.mem.eql(u8, &sig, &SPCR)) {
|
} else if (std.mem.eql(u8, &sig, &SPCR)) {
|
||||||
parseSpcr(header);
|
parseSpcr(header);
|
||||||
} else if (std.mem.eql(u8, &sig, &DMAR)) {
|
} else if (std.mem.eql(u8, &sig, &DMAR)) {
|
||||||
parseDmar(hal, header);
|
parseDmar(header);
|
||||||
|
} else if (std.mem.eql(u8, &sig, &IVRS)) {
|
||||||
|
parseIvrs(header);
|
||||||
} else if (std.mem.eql(u8, &sig, &SSDT)) {
|
} else if (std.mem.eql(u8, &sig, &SSDT)) {
|
||||||
// Secondary namespace bytecode — collect it to publish for the ring-3 parse.
|
// Secondary namespace bytecode — collect it to publish for the ring-3 parse.
|
||||||
addAmlBlock(sdt_physical);
|
addAmlBlock(sdt_physical);
|
||||||
@@ -757,19 +789,34 @@ fn parseSpcr(header: *const SystemDescriptorTableHeader) void {
|
|||||||
|
|
||||||
// DMAR remapping-structure layout (Intel VT-d spec §8): the DMAR-specific header is 12
|
// DMAR remapping-structure layout (Intel VT-d spec §8): the DMAR-specific header is 12
|
||||||
// bytes (host-address-width, flags, 10 reserved), then a list of {type u16, length u16}
|
// bytes (host-address-width, flags, 10 reserved), then a list of {type u16, length u16}
|
||||||
// structures. Type 0 is a DRHD (DMA Remapping Hardware Unit Definition), whose 64-bit
|
// structures. Type 0 is a DRHD (DMA Remapping Hardware Unit Definition): flags byte at
|
||||||
// register base sits at offset 8 within it.
|
// offset 4 (bit 0 = INCLUDE_PCI_ALL, the catch-all unit), 64-bit register base at
|
||||||
|
// offset 8. Type 1 is an RMRR (Reserved Memory Region Reporting): a physical range at
|
||||||
|
// offsets 8/16 (base / inclusive limit) that the device(s) named by the trailing
|
||||||
|
// device-scope entries keep DMAing into across the firmware→OS handoff.
|
||||||
const dmar_structures_offset = 48; // 36-byte ACPI header + 12-byte DMAR header
|
const dmar_structures_offset = 48; // 36-byte ACPI header + 12-byte DMAR header
|
||||||
const dmar_type_drhd: u16 = 0;
|
const dmar_type_drhd: u16 = 0;
|
||||||
|
const dmar_type_rmrr: u16 = 1;
|
||||||
|
const drhd_flags_offset = 4;
|
||||||
|
const drhd_include_pci_all: u8 = 1;
|
||||||
const drhd_register_base_offset = 8;
|
const drhd_register_base_offset = 8;
|
||||||
|
const rmrr_base_offset = 8;
|
||||||
|
const rmrr_limit_offset = 16;
|
||||||
|
const rmrr_scopes_offset = 24;
|
||||||
|
// Device-scope entry (within DRHD/RMRR structures): type 1 = PCI endpoint; the path is
|
||||||
|
// (device, function) byte pairs from offset 6, one pair per bridge hop plus the leaf.
|
||||||
|
const scope_type_pci_endpoint: u8 = 1;
|
||||||
|
const scope_start_bus_offset = 5;
|
||||||
|
const scope_path_offset = 6;
|
||||||
|
|
||||||
/// DMAR -> detect the IOMMU. Find the first DMA-remapping hardware unit, map its
|
/// DMAR -> the VT-d unit(s) and reserved memory regions. Walks every remapping
|
||||||
/// register block, and record its version and capabilities. This is *detection only*:
|
/// structure: selects the INCLUDE_PCI_ALL DRHD (the catch-all covering all devices not
|
||||||
/// it tells the system an IOMMU exists (so `device_claim` on a DMA device could one day
|
/// scoped elsewhere — commonly the SECOND unit on real machines, after an iGPU-scoped
|
||||||
/// be gated by a per-device translation domain), but no domains are programmed yet —
|
/// one), counts the rest so the boot log can warn that their devices stay untranslated,
|
||||||
/// enforcement is built with the first DMA driver, which is what there is to protect and
|
/// and records single-path endpoint RMRRs for the IOMMU core to pre-map before it
|
||||||
/// test against. See docs/driver-model.md (M16), the honest caveat.
|
/// enables translation. Multi-hop RMRR scopes are skipped loudly: better a named gap
|
||||||
fn parseDmar(hal: Hal, header: *const SystemDescriptorTableHeader) void {
|
/// than a silent one.
|
||||||
|
fn parseDmar(header: *const SystemDescriptorTableHeader) void {
|
||||||
const base: [*]align(1) const u8 = @ptrCast(header);
|
const base: [*]align(1) const u8 = @ptrCast(header);
|
||||||
const total: usize = header.length;
|
const total: usize = header.length;
|
||||||
|
|
||||||
@@ -780,13 +827,94 @@ fn parseDmar(hal: Hal, header: *const SystemDescriptorTableHeader) void {
|
|||||||
if (length < 4 or off + length > total) break; // malformed; stop rather than loop
|
if (length < 4 or off + length > total) break; // malformed; stop rather than loop
|
||||||
if (kind == dmar_type_drhd) {
|
if (kind == dmar_type_drhd) {
|
||||||
const register_base = fadt(u64, base, total, off + drhd_register_base_offset) orelse 0;
|
const register_base = fadt(u64, base, total, off + drhd_register_base_offset) orelse 0;
|
||||||
|
const include_all = ((fadt(u8, base, total, off + drhd_flags_offset) orelse 0) & drhd_include_pci_all) != 0;
|
||||||
if (register_base != 0) {
|
if (register_base != 0) {
|
||||||
const regs = hal.mapMmio(register_base, abi.page_size, true);
|
// Selection: the INCLUDE_PCI_ALL unit wins; otherwise keep the first
|
||||||
|
// seen. A later catch-all replaces an earlier scoped unit.
|
||||||
|
const replace = !platform_information.iommu_present or
|
||||||
|
(include_all and !platform_information.iommu_include_all);
|
||||||
|
if (replace) {
|
||||||
|
// Table facts only: the unit's registers are the IOMMU
|
||||||
|
// backend's business (it maps and validates them at
|
||||||
|
// detect) — discovery records where they live, never
|
||||||
|
// reads them.
|
||||||
|
if (platform_information.iommu_present) platform_information.iommu_extra_units += 1;
|
||||||
|
platform_information.iommu_present = true;
|
||||||
|
platform_information.iommu_base = register_base;
|
||||||
|
platform_information.iommu_include_all = include_all;
|
||||||
|
} else {
|
||||||
|
platform_information.iommu_extra_units += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
} else if (kind == dmar_type_rmrr) {
|
||||||
|
parseRmrr(base, total, off, length);
|
||||||
|
}
|
||||||
|
off += length;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One RMRR structure: record a {bdf, base, limit} per single-path endpoint scope.
|
||||||
|
fn parseRmrr(base: [*]align(1) const u8, total: usize, off: usize, length: usize) void {
|
||||||
|
const range_base = fadt(u64, base, total, off + rmrr_base_offset) orelse return;
|
||||||
|
const range_limit = fadt(u64, base, total, off + rmrr_limit_offset) orelse return;
|
||||||
|
if (range_limit < range_base) return;
|
||||||
|
|
||||||
|
var scope = off + rmrr_scopes_offset;
|
||||||
|
const end = off + length;
|
||||||
|
while (scope + 6 <= end) {
|
||||||
|
const scope_type = fadt(u8, base, total, scope) orelse break;
|
||||||
|
const scope_length = fadt(u8, base, total, scope + 1) orelse break;
|
||||||
|
if (scope_length < 6 or scope + scope_length > end) break;
|
||||||
|
if (scope_type == scope_type_pci_endpoint and scope_length == scope_path_offset + 2) {
|
||||||
|
// Single (device, function) pair: a directly-reachable endpoint.
|
||||||
|
const bus = fadt(u8, base, total, scope + scope_start_bus_offset) orelse 0;
|
||||||
|
const device = fadt(u8, base, total, scope + scope_path_offset) orelse 0;
|
||||||
|
const function = fadt(u8, base, total, scope + scope_path_offset + 1) orelse 0;
|
||||||
|
if (platform_information.rmrr_count < maximum_rmrr) {
|
||||||
|
platform_information.rmrr[platform_information.rmrr_count] = .{
|
||||||
|
.bdf = (@as(u16, bus) << 8) | (@as(u16, device) << 3) | function,
|
||||||
|
.base = range_base,
|
||||||
|
.limit = range_limit,
|
||||||
|
};
|
||||||
|
platform_information.rmrr_count += 1;
|
||||||
|
} else {
|
||||||
|
platform_information.rmrr_skipped +|= 1; // table full
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
platform_information.rmrr_skipped +|= 1; // multi-hop path or non-endpoint scope
|
||||||
|
}
|
||||||
|
scope += scope_length;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// IVRS layout (AMD I/O Virtualization spec): 36-byte ACPI header, IVinfo u32 @36,
|
||||||
|
// 8 reserved @40, then IVHD/IVMD blocks from @48. An IVHD common header is type u8 @0,
|
||||||
|
// flags u8 @1, length u16 @2, device id u16 @4, capability offset u16 @6, IOMMU base
|
||||||
|
// address u64 @8, PCI segment u16 @16, IOMMU info u16 @18.
|
||||||
|
const ivrs_blocks_offset = 48;
|
||||||
|
const ivhd_type_10: u8 = 0x10;
|
||||||
|
const ivhd_type_11: u8 = 0x11;
|
||||||
|
const ivhd_base_offset = 8;
|
||||||
|
|
||||||
|
/// IVRS -> detect an AMD-Vi IOMMU. Record the control-register base from the first IVHD
|
||||||
|
/// of type 0x10/0x11. Per-device entries and IVMD (the AMD analogue of RMRR) are ignored
|
||||||
|
/// in v1 — the default-deny device table is what we build anyway, and QEMU emits no IVMD;
|
||||||
|
/// a real machine that needs them is flagged untested on AMD regardless.
|
||||||
|
fn parseIvrs(header: *const SystemDescriptorTableHeader) void {
|
||||||
|
const base: [*]align(1) const u8 = @ptrCast(header);
|
||||||
|
const total: usize = header.length;
|
||||||
|
var off: usize = ivrs_blocks_offset;
|
||||||
|
while (off + 4 <= total) {
|
||||||
|
const kind = fadt(u8, base, total, off) orelse break;
|
||||||
|
const length = fadt(u16, base, total, off + 2) orelse break;
|
||||||
|
if (length < 4 or off + length > total) break;
|
||||||
|
if (kind == ivhd_type_10 or kind == ivhd_type_11) {
|
||||||
|
const iommu_base = fadt(u64, base, total, off + ivhd_base_offset) orelse 0;
|
||||||
|
if (iommu_base != 0) {
|
||||||
platform_information.iommu_present = true;
|
platform_information.iommu_present = true;
|
||||||
platform_information.iommu_base = register_base;
|
platform_information.iommu_is_amd = true;
|
||||||
platform_information.iommu_version = @as(*const volatile u32, @ptrFromInt(regs + 0x00)).*;
|
platform_information.iommu_base = iommu_base;
|
||||||
platform_information.iommu_capabilities = @as(*const volatile u64, @ptrFromInt(regs + 0x08)).*;
|
return; // first IVHD is enough; multi-unit is future work
|
||||||
return; // first unit is enough for detection; multi-unit is future
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
off += length;
|
off += length;
|
||||||
|
|||||||
@@ -17,6 +17,11 @@ const io = @import("io.zig");
|
|||||||
const smp = @import("smp.zig");
|
const smp = @import("smp.zig");
|
||||||
const pcpu = @import("per-cpu.zig");
|
const pcpu = @import("per-cpu.zig");
|
||||||
|
|
||||||
|
/// The x86-64 IOMMU backends (Intel VT-d, AMD-Vi) behind their dispatch
|
||||||
|
/// surface — the architecture-neutral IOMMU core (system/kernel/iommu.zig)
|
||||||
|
/// reaches the hardware only through this.
|
||||||
|
pub const iommu = @import("iommu.zig");
|
||||||
|
|
||||||
/// The saved register/trap frame passed to a fault handler.
|
/// The saved register/trap frame passed to a fault handler.
|
||||||
pub const CpuState = idt.CpuState;
|
pub const CpuState = idt.CpuState;
|
||||||
|
|
||||||
@@ -252,6 +257,16 @@ pub fn translate(root: u64, virtual: u64) ?u64 {
|
|||||||
return paging.translateIn(root, virtual);
|
return paging.translateIn(root, virtual);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// `translate` for an address the kernel is about to touch *on a process's
|
||||||
|
/// behalf*: the walk additionally demands the permission ring 3 would need — the
|
||||||
|
/// leaf user-accessible (U/S set at every level), and writable (R/W at every
|
||||||
|
/// level) when `for_write`. Null means "the process itself could not do this",
|
||||||
|
/// which the checked copy layer (system/kernel/user-memory.zig) turns into
|
||||||
|
/// -EFAULT instead of a kernel dereference.
|
||||||
|
pub fn translateUser(root: u64, virtual: u64, for_write: bool) ?u64 {
|
||||||
|
return paging.translateUserIn(root, virtual, for_write);
|
||||||
|
}
|
||||||
|
|
||||||
/// Map a page accessible from ring 3 (U/S bit at every level). The caller keeps
|
/// Map a page accessible from ring 3 (U/S bit at every level). The caller keeps
|
||||||
/// W^X: code read-only + executable, data writable + no-execute.
|
/// W^X: code read-only + executable, data writable + no-execute.
|
||||||
pub fn mapUserPage(virtual: u64, physical: u64, writable: bool, executable: bool) void {
|
pub fn mapUserPage(virtual: u64, physical: u64, writable: bool, executable: bool) void {
|
||||||
|
|||||||
@@ -0,0 +1,265 @@
|
|||||||
|
//! AMD-Vi (AMD I/O Virtualization) backend for the IOMMU core, behind the
|
||||||
|
//! architecture boundary. The AMD analogue of iommu-intel.zig: it supplies
|
||||||
|
//! the architecture-neutral core's `Backend` vtable (iommu.zig beside this file)
|
||||||
|
//! with AMD-Vi's page-table entry bits and drives the device table, command buffer, and
|
||||||
|
//! event log.
|
||||||
|
//!
|
||||||
|
//! **UNTESTED on real AMD hardware.** danos is developed on Intel; this backend is
|
||||||
|
//! validated only against QEMU's `-device amd-iommu,dma-remap=on`, whose AMD-Vi
|
||||||
|
//! emulation is far less exercised than its Intel one. Every code path here should be
|
||||||
|
//! read as "QEMU-verified, real-AMD-unverified" until an AMD machine confirms it.
|
||||||
|
//!
|
||||||
|
//! Interrupt remapping is left off (the DTE forwards interrupts unmapped), so MSI writes
|
||||||
|
//! to the 0xFEE00000 range reach the APIC untranslated, exactly as on the Intel path.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const boot_handoff = @import("boot-handoff");
|
||||||
|
const paging = @import("paging.zig");
|
||||||
|
const iommu = @import("iommu.zig");
|
||||||
|
|
||||||
|
const page_size = abi.page_size;
|
||||||
|
|
||||||
|
// MMIO register offsets from the IOMMU control-register base.
|
||||||
|
const reg_device_table_base = 0x00; // base | (pages-1) in bits 8:0
|
||||||
|
const reg_command_buffer_base = 0x08; // base | (ComLen << 56)
|
||||||
|
const reg_event_log_base = 0x10; // base | (EventLen << 56)
|
||||||
|
const reg_control = 0x18;
|
||||||
|
const reg_command_head = 0x2000;
|
||||||
|
const reg_command_tail = 0x2008;
|
||||||
|
const reg_event_head = 0x2010;
|
||||||
|
const reg_event_tail = 0x2018;
|
||||||
|
const reg_status = 0x2020;
|
||||||
|
|
||||||
|
const control_iommu_enable: u64 = 1 << 0;
|
||||||
|
const control_event_log_enable: u64 = 1 << 2;
|
||||||
|
const control_command_buffer_enable: u64 = 1 << 12;
|
||||||
|
|
||||||
|
// Device table: one 32-byte DTE (4 qwords) per requester id, indexed by bdf.
|
||||||
|
const device_table_pages = 512; // 2 MiB = 65536 entries (a full 256-bus segment)
|
||||||
|
const dte_qwords = 4;
|
||||||
|
const dte_valid: u64 = 1 << 0; // V
|
||||||
|
const dte_translation_valid: u64 = 1 << 1; // TV
|
||||||
|
const dte_mode_shift = 9; // bits 11:9 — page-table levels
|
||||||
|
const dte_read: u64 = 1 << 61; // IR
|
||||||
|
const dte_write: u64 = 1 << 62; // IW
|
||||||
|
const dte_intctl_forward: u64 = @as(u64, 1) << 60; // qword2 bits 61:60 = 01b: forward interrupts unmapped
|
||||||
|
|
||||||
|
// Page-table entry bits (AMD native format).
|
||||||
|
const pte_present: u64 = 1 << 0; // PR
|
||||||
|
const pte_next_level_shift = 9; // bits 11:9: 0 = leaf, N = pointer to a level-N table
|
||||||
|
const pte_read: u64 = 1 << 61; // IR
|
||||||
|
const pte_write: u64 = 1 << 62; // IW
|
||||||
|
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||||
|
|
||||||
|
// Command buffer / event log: one 4 KiB frame each = 256 entries.
|
||||||
|
const ring_entries = 256;
|
||||||
|
const ring_length_code: u64 = 8; // log2(256), the ComLen/EventLen field value
|
||||||
|
const command_opcode_shift = 60; // opcode in bits 63:60 of qword 0
|
||||||
|
const command_completion_wait: u64 = 0x01;
|
||||||
|
const command_invalidate_devtab: u64 = 0x02;
|
||||||
|
const command_invalidate_pages: u64 = 0x03;
|
||||||
|
const completion_wait_store: u64 = 1 << 1; // S: store `data` to the supplied address
|
||||||
|
const invalidate_pages_all: u64 = 0x000F_FFFF_FFFF_F000 | 1; // address bits 51:12 all-ones + S
|
||||||
|
|
||||||
|
const levels: u8 = 4; // 48-bit IOVA, matching the Intel 4-level path
|
||||||
|
|
||||||
|
var register_base: usize = 0;
|
||||||
|
var device_table: u64 = 0; // physical base of the device table
|
||||||
|
var command_buffer: u64 = 0;
|
||||||
|
var event_log: u64 = 0;
|
||||||
|
var completion_frame: u64 = 0; // COMPLETION_WAIT store target
|
||||||
|
var command_tail: u32 = 0; // our software copy of the command tail (bytes)
|
||||||
|
|
||||||
|
var fault_log_budget: u32 = 32;
|
||||||
|
var completion_warned = false;
|
||||||
|
|
||||||
|
fn read64(offset: usize) u64 {
|
||||||
|
return @as(*const volatile u64, @ptrFromInt(register_base + offset)).*;
|
||||||
|
}
|
||||||
|
fn write64(offset: usize, value: u64) void {
|
||||||
|
@as(*volatile u64, @ptrFromInt(register_base + offset)).* = value;
|
||||||
|
}
|
||||||
|
fn ram(physical: u64) [*]volatile u64 {
|
||||||
|
return @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Map the register window, allocate the device table / command buffer / event log.
|
||||||
|
/// Returns the vtable, or null if the boot-time allocations fail.
|
||||||
|
pub fn detect(discovery: iommu.Discovery) ?iommu.Backend {
|
||||||
|
register_base = paging.mapMmio(discovery.register_base, 16 * 1024, true);
|
||||||
|
|
||||||
|
device_table = iommu.environment.allocateContiguous(device_table_pages, ~@as(u64, 0)) orelse return null;
|
||||||
|
zero(device_table, device_table_pages); // all-zero DTE = V=0 = deny every device
|
||||||
|
command_buffer = allocZeroedFrame() orelse return null;
|
||||||
|
event_log = allocZeroedFrame() orelse return null;
|
||||||
|
completion_frame = allocZeroedFrame() orelse return null;
|
||||||
|
|
||||||
|
return iommu.Backend{
|
||||||
|
.levels = levels,
|
||||||
|
.supports_huge_pages = false, // 4 KiB leaves only (AMD superpage encoding deferred)
|
||||||
|
.enable = enable,
|
||||||
|
.makeLeaf = makeLeaf,
|
||||||
|
.makeTable = makeTable,
|
||||||
|
.isPresent = isPresent,
|
||||||
|
.flushStructure = flushStructure,
|
||||||
|
.attach = attach,
|
||||||
|
.detach = detach,
|
||||||
|
.invalidateDomain = invalidateDomain,
|
||||||
|
.faultDrain = faultDrain,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Program the base registers and enable translation. The device table is already
|
||||||
|
/// zeroed (every device denied) except any entries `attach` wrote for RMRR/claimed
|
||||||
|
/// devices, so turning translation on blocks all other DMA and logs it.
|
||||||
|
fn enable() void {
|
||||||
|
write64(reg_device_table_base, (device_table & address_mask) | (device_table_pages - 1));
|
||||||
|
write64(reg_command_buffer_base, (command_buffer & address_mask) | (ring_length_code << 56));
|
||||||
|
write64(reg_event_log_base, (event_log & address_mask) | (ring_length_code << 56));
|
||||||
|
write64(reg_command_head, 0);
|
||||||
|
write64(reg_command_tail, 0);
|
||||||
|
write64(reg_event_head, 0);
|
||||||
|
write64(reg_event_tail, 0);
|
||||||
|
command_tail = 0;
|
||||||
|
// Buffers first, then the master enable.
|
||||||
|
write64(reg_control, control_command_buffer_enable | control_event_log_enable);
|
||||||
|
write64(reg_control, control_command_buffer_enable | control_event_log_enable | control_iommu_enable);
|
||||||
|
|
||||||
|
iommu.environment.write("/system/kernel: iommu online (AMD-Vi) — UNTESTED on real AMD hardware (QEMU-verified only)\n");
|
||||||
|
var buffer: [48]u8 = undefined;
|
||||||
|
if (std.fmt.bufPrint(&buffer, " levels : {d} (48-bit)\n", .{levels})) |line|
|
||||||
|
iommu.environment.write(line)
|
||||||
|
else |_| {}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- Backend vtable ------------------------------------------------------------------
|
||||||
|
|
||||||
|
fn makeLeaf(physical: u64, huge: bool) u64 {
|
||||||
|
_ = huge; // 4 KiB only
|
||||||
|
return (physical & address_mask) | pte_present | pte_read | pte_write; // Next Level 0 = leaf
|
||||||
|
}
|
||||||
|
fn makeTable(table_physical: u64, level: u8) u64 {
|
||||||
|
// This entry (at `level`) points to a table one level down; AMD's Next Level names
|
||||||
|
// the pointed-to table's level.
|
||||||
|
const next_level: u64 = @as(u64, level) - 1;
|
||||||
|
return (table_physical & address_mask) | pte_present | pte_read | pte_write | (next_level << pte_next_level_shift);
|
||||||
|
}
|
||||||
|
fn isPresent(entry: u64) bool {
|
||||||
|
return (entry & pte_present) != 0;
|
||||||
|
}
|
||||||
|
fn flushStructure(address: usize) void {
|
||||||
|
_ = address; // AMD-Vi reads its structures coherently
|
||||||
|
}
|
||||||
|
|
||||||
|
fn attach(bdf: u16, domain: u16, page_table_root: u64) void {
|
||||||
|
const dte = ram(device_table) + @as(usize, bdf) * dte_qwords;
|
||||||
|
dte[0] = (page_table_root & address_mask) | dte_valid | dte_translation_valid |
|
||||||
|
(@as(u64, levels) << dte_mode_shift) | dte_read | dte_write;
|
||||||
|
dte[1] = @as(u64, domain); // DomainID in bits 15:0
|
||||||
|
dte[2] = dte_intctl_forward; // forward interrupts unmapped (no remapping)
|
||||||
|
dte[3] = 0;
|
||||||
|
invalidateDevice(bdf);
|
||||||
|
invalidateDomain(domain);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn detach(bdf: u16) void {
|
||||||
|
const dte = ram(device_table) + @as(usize, bdf) * dte_qwords;
|
||||||
|
dte[0] = 0; // V=0: deny
|
||||||
|
dte[1] = 0;
|
||||||
|
dte[2] = 0;
|
||||||
|
dte[3] = 0;
|
||||||
|
invalidateDevice(bdf);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn invalidateDomain(domain: u16) void {
|
||||||
|
submitCommand((command_invalidate_pages << command_opcode_shift) | (@as(u64, domain) << 32), invalidate_pages_all);
|
||||||
|
completeAndWait();
|
||||||
|
}
|
||||||
|
|
||||||
|
fn faultDrain() usize {
|
||||||
|
const tail = read64(reg_event_tail) & 0xFFFF_FFF0;
|
||||||
|
var head = read64(reg_event_head) & 0xFFFF_FFF0;
|
||||||
|
if (head == tail) return 0;
|
||||||
|
var seen: usize = 0;
|
||||||
|
while (head != tail) {
|
||||||
|
const entry = ram(event_log) + (head / 8);
|
||||||
|
const code: u4 = @truncate(entry[0] >> command_opcode_shift);
|
||||||
|
if (code == 0x2) { // IO_PAGE_FAULT
|
||||||
|
const source: u16 = @truncate(entry[0]);
|
||||||
|
logFault(source, entry[1]);
|
||||||
|
}
|
||||||
|
seen += 1;
|
||||||
|
head += 16;
|
||||||
|
if (head >= ring_entries * 16) head = 0;
|
||||||
|
}
|
||||||
|
write64(reg_event_head, head);
|
||||||
|
return seen;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn logFault(source: u16, address: u64) void {
|
||||||
|
if (fault_log_budget == 0) return;
|
||||||
|
fault_log_budget -= 1;
|
||||||
|
var buffer: [128]u8 = undefined;
|
||||||
|
if (std.fmt.bufPrint(&buffer, "DANOS-IOMMU-FAULT: bdf={x:0>2}:{x:0>2}.{d} addr=0x{x} reason=amd-io-page-fault\n", .{
|
||||||
|
source >> 8,
|
||||||
|
(source >> 3) & 0x1F,
|
||||||
|
source & 0x7,
|
||||||
|
address,
|
||||||
|
})) |line| iommu.environment.write(line) else |_| {}
|
||||||
|
if (fault_log_budget == 0) iommu.environment.write("DANOS-IOMMU-FAULT: (further faults suppressed)\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- command ring --------------------------------------------------------------------
|
||||||
|
|
||||||
|
fn invalidateDevice(bdf: u16) void {
|
||||||
|
submitCommand((command_invalidate_devtab << command_opcode_shift) | @as(u64, bdf), 0);
|
||||||
|
completeAndWait();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Append a 128-bit command (two qwords) to the ring and advance the tail.
|
||||||
|
fn submitCommand(qword0: u64, qword1: u64) void {
|
||||||
|
const slot = ram(command_buffer) + (command_tail / 8);
|
||||||
|
slot[0] = qword0;
|
||||||
|
slot[1] = qword1;
|
||||||
|
command_tail += 16;
|
||||||
|
if (command_tail >= ring_entries * 16) command_tail = 0;
|
||||||
|
write64(reg_command_tail, command_tail);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Append a COMPLETION_WAIT (store form) and spin until the IOMMU writes our sentinel to
|
||||||
|
/// the completion frame. QEMU consumes the command buffer synchronously on the tail-
|
||||||
|
/// register write, so by the time we poll the prior invalidation is already applied; the
|
||||||
|
/// store confirmation is belt-and-suspenders for real hardware. If it never lands
|
||||||
|
/// (QEMU's amd-iommu does not implement the store form), warn ONCE and proceed — the
|
||||||
|
/// invalidation itself has happened.
|
||||||
|
fn completeAndWait() void {
|
||||||
|
const sentinel: u64 = 0xC0FFEE;
|
||||||
|
ram(completion_frame)[0] = 0;
|
||||||
|
submitCommand(
|
||||||
|
(command_completion_wait << command_opcode_shift) | (completion_frame & 0x000F_FFFF_FFFF_FFF8) | completion_wait_store,
|
||||||
|
sentinel,
|
||||||
|
);
|
||||||
|
var spins: u64 = 0;
|
||||||
|
while (@as(*const volatile u64, @ptrFromInt(boot_handoff.physicalToVirtual(completion_frame))).* != sentinel) {
|
||||||
|
spins += 1;
|
||||||
|
if (spins > 100_000) {
|
||||||
|
if (!completion_warned) {
|
||||||
|
completion_warned = true;
|
||||||
|
iommu.environment.write("/system/kernel: AMD-Vi COMPLETION_WAIT store not observed — proceeding (QEMU processes commands synchronously)\n");
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn allocZeroedFrame() ?u64 {
|
||||||
|
const frame = iommu.environment.allocateFrame() orelse return null;
|
||||||
|
zero(frame, 1);
|
||||||
|
return frame;
|
||||||
|
}
|
||||||
|
fn zero(physical: u64, pages: usize) void {
|
||||||
|
const words = ram(physical);
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < pages * page_size / 8) : (i += 1) words[i] = 0;
|
||||||
|
}
|
||||||
@@ -0,0 +1,327 @@
|
|||||||
|
//! Intel VT-d backend for the IOMMU core, behind the architecture boundary.
|
||||||
|
//!
|
||||||
|
//! Provides the architecture-neutral core (system/kernel/iommu.zig) with the VT-d
|
||||||
|
//! hardware specifics behind the `Backend` vtable (iommu.zig beside this file): second-level page-table entry bits, the root/context table structure, the
|
||||||
|
//! translation-enable and invalidation register sequences, and the fault drain. The
|
||||||
|
//! core owns the domain table and the page-table walk; this file owns the registers.
|
||||||
|
//!
|
||||||
|
//! Register model (VT-d spec §10-11): offsets from the DRHD register base. The unit is
|
||||||
|
//! programmed once at enable (root table + Translation Enable), then touched only for
|
||||||
|
//! per-device context changes, per-domain invalidations, and fault draining. Interrupt
|
||||||
|
//! remapping is deliberately left OFF (GCMD.IRE stays 0): with it off, upstream writes
|
||||||
|
//! to 0xFEE0_0000-0xFEEF_FFFF are treated as interrupt requests and bypass second-level
|
||||||
|
//! translation, so the existing MSI contract survives unchanged.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const boot_handoff = @import("boot-handoff");
|
||||||
|
const paging = @import("paging.zig");
|
||||||
|
const iommu = @import("iommu.zig");
|
||||||
|
|
||||||
|
const page_size = abi.page_size;
|
||||||
|
|
||||||
|
// Register offsets from the unit's base.
|
||||||
|
const reg_cap = 0x08; // Capability (64)
|
||||||
|
const reg_ecap = 0x10; // Extended Capability (64)
|
||||||
|
const reg_gcmd = 0x18; // Global Command (32, write-only)
|
||||||
|
const reg_gsts = 0x1C; // Global Status (32, read-only)
|
||||||
|
const reg_rtaddr = 0x20; // Root Table Address (64)
|
||||||
|
const reg_ccmd = 0x28; // Context Command (64)
|
||||||
|
const reg_fsts = 0x34; // Fault Status (32)
|
||||||
|
|
||||||
|
const gcmd_te: u32 = 1 << 31; // Translation Enable
|
||||||
|
const gcmd_srtp: u32 = 1 << 30; // Set Root Table Pointer
|
||||||
|
const gsts_tes: u32 = 1 << 31; // Translation Enable Status
|
||||||
|
const gsts_rtps: u32 = 1 << 30; // Root Table Pointer Status
|
||||||
|
|
||||||
|
const cap_cm: u64 = 1 << 7; // Caching Mode
|
||||||
|
const cap_sagaw_shift = 8; // Supported Adjusted Guest Address Widths, bits 12:8
|
||||||
|
const cap_sagaw_39bit: u64 = 1 << 9; // 3-level
|
||||||
|
const cap_sagaw_48bit: u64 = 1 << 10; // 4-level
|
||||||
|
const cap_fro_shift = 24; // Fault-Recording Register Offset, bits 33:24 (×16)
|
||||||
|
const cap_nfr_shift = 40; // Number of Fault Recording regs, bits 47:40 (+1)
|
||||||
|
const ecap_coherent: u64 = 1 << 0; // hardware snoops CPU caches reading its structures
|
||||||
|
const ecap_iro_shift = 8; // IOTLB Register Offset, bits 17:8 (×16)
|
||||||
|
|
||||||
|
const ccmd_icc: u64 = 1 << 63; // Invalidate Context-Cache
|
||||||
|
const ccmd_cirg_global: u64 = @as(u64, 1) << 61; // global granularity
|
||||||
|
const ccmd_cirg_device: u64 = @as(u64, 3) << 61; // device-selective
|
||||||
|
|
||||||
|
const iotlb_ivt: u64 = 1 << 63; // Invalidate IOTLB
|
||||||
|
const iotlb_iirg_global: u64 = @as(u64, 1) << 60;
|
||||||
|
const iotlb_iirg_domain: u64 = @as(u64, 2) << 60;
|
||||||
|
const iotlb_dr: u64 = 1 << 49; // drain reads
|
||||||
|
const iotlb_dw: u64 = 1 << 48; // drain writes
|
||||||
|
|
||||||
|
const fsts_ppf: u32 = 1 << 1; // Primary Pending Fault
|
||||||
|
|
||||||
|
// Second-level PTE bits.
|
||||||
|
const slpte_read: u64 = 1 << 0;
|
||||||
|
const slpte_write: u64 = 1 << 1;
|
||||||
|
const slpte_page_size: u64 = 1 << 7; // a 2 MiB leaf (== iommu.huge_leaf_bit)
|
||||||
|
|
||||||
|
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||||
|
|
||||||
|
var register_base: usize = 0;
|
||||||
|
var version: u32 = 0;
|
||||||
|
var capabilities: u64 = 0;
|
||||||
|
var extended_capabilities: u64 = 0;
|
||||||
|
var coherent: bool = true; // ECAP.C — whether clflush is unnecessary
|
||||||
|
var levels: u8 = 4;
|
||||||
|
var context_aw: u64 = 2; // context-entry AW field (001=3-level, 010=4-level)
|
||||||
|
var gcmd_shadow: u32 = 0; // sticky GCMD bits (TE etc.), for the write-only register
|
||||||
|
|
||||||
|
var root_table: u64 = 0; // physical base of the 256-entry root table
|
||||||
|
var context_table: [256]u64 = .{0} ** 256; // per-bus context table physical, 0 = none
|
||||||
|
|
||||||
|
var fault_log_budget: u32 = 32; // rate-limit: log this many faults, then just count
|
||||||
|
var faults_suppressed: u64 = 0;
|
||||||
|
|
||||||
|
fn read32(offset: usize) u32 {
|
||||||
|
return @as(*const volatile u32, @ptrFromInt(register_base + offset)).*;
|
||||||
|
}
|
||||||
|
fn write32(offset: usize, value: u32) void {
|
||||||
|
@as(*volatile u32, @ptrFromInt(register_base + offset)).* = value;
|
||||||
|
}
|
||||||
|
fn read64(offset: usize) u64 {
|
||||||
|
return @as(*const volatile u64, @ptrFromInt(register_base + offset)).*;
|
||||||
|
}
|
||||||
|
fn write64(offset: usize, value: u64) void {
|
||||||
|
@as(*volatile u64, @ptrFromInt(register_base + offset)).* = value;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn tableAt(physical: u64) [*]volatile u64 {
|
||||||
|
return @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Map the register window, read caps, pick the address width. Returns the vtable, or
|
||||||
|
/// null when the unit is not live or advertises no address width danos can drive.
|
||||||
|
pub fn detect(discovery: iommu.Discovery) ?iommu.Backend {
|
||||||
|
// Map 16 KiB: FRCD and IOTLB registers can sit past the first page (CAP.FRO /
|
||||||
|
// ECAP.IRO are 16-byte-unit offsets).
|
||||||
|
register_base = paging.mapMmio(discovery.register_base, 16 * 1024, true);
|
||||||
|
// The Version register's low byte is major.minor; reading it back nonzero is
|
||||||
|
// the live-mappable-unit sanity check (previously a kernel-test assertion).
|
||||||
|
version = read32(0x00);
|
||||||
|
if (version == 0) return null;
|
||||||
|
capabilities = read64(reg_cap);
|
||||||
|
extended_capabilities = read64(reg_ecap);
|
||||||
|
coherent = (extended_capabilities & ecap_coherent) != 0;
|
||||||
|
|
||||||
|
const sagaw = capabilities >> cap_sagaw_shift;
|
||||||
|
if (sagaw & cap_sagaw_48bit != 0) {
|
||||||
|
levels = 4;
|
||||||
|
context_aw = 2; // 010b
|
||||||
|
} else if (sagaw & cap_sagaw_39bit != 0) {
|
||||||
|
levels = 3;
|
||||||
|
context_aw = 1; // 001b
|
||||||
|
} else {
|
||||||
|
return null; // no width we build tables for
|
||||||
|
}
|
||||||
|
|
||||||
|
root_table = allocZeroed() orelse return null;
|
||||||
|
|
||||||
|
return iommu.Backend{
|
||||||
|
.levels = levels,
|
||||||
|
.supports_huge_pages = true,
|
||||||
|
.enable = enable,
|
||||||
|
.makeLeaf = makeLeaf,
|
||||||
|
.makeTable = makeTable,
|
||||||
|
.isPresent = isPresent,
|
||||||
|
.flushStructure = flushStructure,
|
||||||
|
.attach = attach,
|
||||||
|
.detach = detach,
|
||||||
|
.invalidateDomain = invalidateDomain,
|
||||||
|
.faultDrain = faultDrain,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Program the root table and turn Translation Enable on. The core has already created
|
||||||
|
/// and populated the RMRR domains (their context entries are live via `attach`), so at
|
||||||
|
/// this instant every OTHER device's context entry is not-present and will fault — which
|
||||||
|
/// for stale firmware bus-mastering is the desired evidence, not a bug.
|
||||||
|
fn enable() void {
|
||||||
|
write64(reg_rtaddr, root_table); // legacy mode (bits 11:10 = 00)
|
||||||
|
setGlobalCommand(gcmd_srtp);
|
||||||
|
spinStatus(gsts_rtps);
|
||||||
|
globalInvalidate();
|
||||||
|
setGlobalCommand(gcmd_te);
|
||||||
|
spinStatus(gsts_tes);
|
||||||
|
gcmd_shadow |= gcmd_te;
|
||||||
|
|
||||||
|
iommu.environment.write("/system/kernel: iommu online (Intel VT-d)\n");
|
||||||
|
var buffer: [64]u8 = undefined;
|
||||||
|
if (std.fmt.bufPrint(&buffer, " version : 0x{x}\n", .{version})) |line|
|
||||||
|
iommu.environment.write(line)
|
||||||
|
else |_| {}
|
||||||
|
if (std.fmt.bufPrint(&buffer, " agaw : {d} levels\n", .{levels})) |line|
|
||||||
|
iommu.environment.write(line)
|
||||||
|
else |_| {}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- Backend vtable ------------------------------------------------------------------
|
||||||
|
|
||||||
|
fn makeLeaf(physical: u64, huge: bool) u64 {
|
||||||
|
return (physical & address_mask) | slpte_read | slpte_write | (if (huge) slpte_page_size else 0);
|
||||||
|
}
|
||||||
|
fn makeTable(table_physical: u64, level: u8) u64 {
|
||||||
|
_ = level;
|
||||||
|
return (table_physical & address_mask) | slpte_read | slpte_write;
|
||||||
|
}
|
||||||
|
fn isPresent(entry: u64) bool {
|
||||||
|
return (entry & (slpte_read | slpte_write)) != 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn flushStructure(address: usize) void {
|
||||||
|
if (coherent) return; // the unit snoops CPU caches; no flush needed (QEMU)
|
||||||
|
asm volatile ("clflush (%[p])"
|
||||||
|
:
|
||||||
|
: [p] "r" (address),
|
||||||
|
: .{ .memory = true });
|
||||||
|
}
|
||||||
|
|
||||||
|
fn attach(bdf: u16, domain: u16, page_table_root: u64) void {
|
||||||
|
const bus: u8 = @intCast(bdf >> 8);
|
||||||
|
const devfn: u8 = @intCast(bdf & 0xFF);
|
||||||
|
|
||||||
|
// Lazily allocate this bus's context table and link it into the root table.
|
||||||
|
if (context_table[bus] == 0) {
|
||||||
|
const table = allocZeroed() orelse return;
|
||||||
|
context_table[bus] = table;
|
||||||
|
const root_entry = &tableAt(root_table)[@as(usize, bus) * 2]; // 16-byte entries
|
||||||
|
root_entry.* = (table & address_mask) | 1; // present
|
||||||
|
flushStructure(@intFromPtr(root_entry));
|
||||||
|
}
|
||||||
|
|
||||||
|
const context = tableAt(context_table[bus]);
|
||||||
|
const low = &context[@as(usize, devfn) * 2];
|
||||||
|
const high = &context[@as(usize, devfn) * 2 + 1];
|
||||||
|
high.* = (context_aw & 0x7) | (@as(u64, domain) << 8); // AW + DID
|
||||||
|
low.* = (page_table_root & address_mask) | 1; // present, TT=00 (use second-level)
|
||||||
|
flushStructure(@intFromPtr(high));
|
||||||
|
flushStructure(@intFromPtr(low));
|
||||||
|
|
||||||
|
invalidateContextDevice(bdf, domain);
|
||||||
|
invalidateDomain(domain);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn detach(bdf: u16) void {
|
||||||
|
const bus: u8 = @intCast(bdf >> 8);
|
||||||
|
const devfn: u8 = @intCast(bdf & 0xFF);
|
||||||
|
if (context_table[bus] == 0) return;
|
||||||
|
const context = tableAt(context_table[bus]);
|
||||||
|
context[@as(usize, devfn) * 2] = 0; // not present
|
||||||
|
context[@as(usize, devfn) * 2 + 1] = 0;
|
||||||
|
flushStructure(@intFromPtr(&context[@as(usize, devfn) * 2]));
|
||||||
|
invalidateContextDevice(bdf, 0);
|
||||||
|
globalIotlb();
|
||||||
|
}
|
||||||
|
|
||||||
|
fn invalidateDomain(domain: u16) void {
|
||||||
|
const iotlb_offset = iotlbOffset();
|
||||||
|
write64(iotlb_offset, iotlb_ivt | iotlb_iirg_domain | iotlb_dr | iotlb_dw | (@as(u64, domain) << 32));
|
||||||
|
spin64(iotlb_offset, iotlb_ivt);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn faultDrain() usize {
|
||||||
|
const fsts = read32(reg_fsts);
|
||||||
|
if (fsts & fsts_ppf == 0) return 0;
|
||||||
|
|
||||||
|
const fro = (capabilities >> cap_fro_shift) & 0x3FF;
|
||||||
|
const nfr = ((capabilities >> cap_nfr_shift) & 0xFF) + 1;
|
||||||
|
const frcd_base = @as(usize, @intCast(fro)) * 16;
|
||||||
|
|
||||||
|
var seen: usize = 0;
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < nfr) : (i += 1) {
|
||||||
|
const off = frcd_base + i * 16;
|
||||||
|
const high = read64(off + 8);
|
||||||
|
if (high & (@as(u64, 1) << 63) == 0) continue; // F: no fault recorded here
|
||||||
|
const low = read64(off);
|
||||||
|
const address = low & ~@as(u64, 0xFFF);
|
||||||
|
const source: u16 = @intCast(high & 0xFFFF);
|
||||||
|
const reason: u8 = @intCast((high >> 32) & 0xFF);
|
||||||
|
const is_read = (high >> 62) & 1; // T: 1 = read request
|
||||||
|
logFault(source, address, reason, is_read == 1);
|
||||||
|
write64(off + 8, @as(u64, 1) << 63); // RW1C: clear F
|
||||||
|
seen += 1;
|
||||||
|
}
|
||||||
|
write32(reg_fsts, fsts); // clear PPF/PFO (write-1-to-clear)
|
||||||
|
return seen;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn logFault(source: u16, address: u64, reason: u8, is_read: bool) void {
|
||||||
|
if (fault_log_budget > 0) {
|
||||||
|
fault_log_budget -= 1;
|
||||||
|
var buffer: [128]u8 = undefined;
|
||||||
|
if (std.fmt.bufPrint(&buffer, "DANOS-IOMMU-FAULT: bdf={x:0>2}:{x:0>2}.{d} addr=0x{x} reason=0x{x} write={d}\n", .{
|
||||||
|
source >> 8,
|
||||||
|
(source >> 3) & 0x1F,
|
||||||
|
source & 0x7,
|
||||||
|
address,
|
||||||
|
reason,
|
||||||
|
@intFromBool(!is_read),
|
||||||
|
})) |line| iommu.environment.write(line) else |_| {}
|
||||||
|
if (fault_log_budget == 0)
|
||||||
|
iommu.environment.write("DANOS-IOMMU-FAULT: (further faults suppressed)\n");
|
||||||
|
} else {
|
||||||
|
faults_suppressed += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- register helpers ----------------------------------------------------------------
|
||||||
|
|
||||||
|
fn setGlobalCommand(one_shot: u32) void {
|
||||||
|
// GCMD is write-only: every write must carry the full sticky state plus the one-shot
|
||||||
|
// bit being requested, or a set sticky bit (TE) would be cleared as a side effect.
|
||||||
|
write32(reg_gcmd, gcmd_shadow | one_shot);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn spinStatus(bit: u32) void {
|
||||||
|
var spins: u64 = 0;
|
||||||
|
while (read32(reg_gsts) & bit == 0) {
|
||||||
|
spins += 1;
|
||||||
|
if (spins > 10_000_000) {
|
||||||
|
iommu.environment.write("/system/kernel: WARNING VT-d status bit never set — translation may be incomplete\n");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn spin64(offset: usize, bit: u64) void {
|
||||||
|
var spins: u64 = 0;
|
||||||
|
while (read64(offset) & bit != 0) {
|
||||||
|
spins += 1;
|
||||||
|
if (spins > 10_000_000) return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn globalInvalidate() void {
|
||||||
|
write64(reg_ccmd, ccmd_icc | ccmd_cirg_global);
|
||||||
|
spin64(reg_ccmd, ccmd_icc);
|
||||||
|
globalIotlb();
|
||||||
|
}
|
||||||
|
|
||||||
|
fn globalIotlb() void {
|
||||||
|
const iotlb_offset = iotlbOffset();
|
||||||
|
write64(iotlb_offset, iotlb_ivt | iotlb_iirg_global | iotlb_dr | iotlb_dw);
|
||||||
|
spin64(iotlb_offset, iotlb_ivt);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn invalidateContextDevice(bdf: u16, domain: u16) void {
|
||||||
|
write64(reg_ccmd, ccmd_icc | ccmd_cirg_device | (@as(u64, bdf) << 16) | domain);
|
||||||
|
spin64(reg_ccmd, ccmd_icc);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn iotlbOffset() usize {
|
||||||
|
const iro = (extended_capabilities >> ecap_iro_shift) & 0x3FF;
|
||||||
|
return @as(usize, @intCast(iro)) * 16 + 8; // IOTLB register sits at IRO*16 + 8
|
||||||
|
}
|
||||||
|
|
||||||
|
fn allocZeroed() ?u64 {
|
||||||
|
const frame = iommu.environment.allocateFrame() orelse return null;
|
||||||
|
const table = tableAt(frame);
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < 512) : (i += 1) table[i] = 0;
|
||||||
|
return frame;
|
||||||
|
}
|
||||||
@@ -0,0 +1,77 @@
|
|||||||
|
//! x86-64 IOMMU backends, behind the architecture boundary: Intel VT-d
|
||||||
|
//! (iommu-intel.zig) and AMD-Vi (iommu-amd.zig). The architecture-neutral
|
||||||
|
//! core (system/kernel/iommu.zig) owns the domain table and the shared
|
||||||
|
//! page-table walker; it hands this file the firmware discovery facts and an
|
||||||
|
//! environment (frame allocation + the log sink, injected the same way
|
||||||
|
//! enablePaging receives its frame hooks), and gets back a hardware vtable.
|
||||||
|
//! A new architecture supplies its own unit (ARM: the SMMU) from its own
|
||||||
|
//! directory with no core change.
|
||||||
|
|
||||||
|
const intel = @import("iommu-intel.zig");
|
||||||
|
const amd = @import("iommu-amd.zig");
|
||||||
|
|
||||||
|
/// What the platform's firmware tables reported: where the unit's registers
|
||||||
|
/// live, and which programming model its table implies (an IVRS table
|
||||||
|
/// describes AMD-Vi; a DMAR table describes Intel VT-d).
|
||||||
|
pub const Discovery = struct {
|
||||||
|
register_base: u64,
|
||||||
|
amd: bool,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// What the backends need from the generic kernel, injected at detect so this
|
||||||
|
/// module never imports kernel internals: physical-frame allocation for the
|
||||||
|
/// hardware structures, and the kernel log sink (fault reports, warnings, the
|
||||||
|
/// enable banner).
|
||||||
|
pub const Environment = struct {
|
||||||
|
allocateFrame: *const fn () ?u64,
|
||||||
|
allocateContiguous: *const fn (count: usize, max_physical: u64) ?u64,
|
||||||
|
write: *const fn (bytes: []const u8) void,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The bit encodings and hardware operations a backend supplies to the shared
|
||||||
|
/// core. Entry helpers build the raw page-table entries for the backend's
|
||||||
|
/// format; the core walks the tree with them. The hardware ops act on a whole
|
||||||
|
/// domain (identified by its hardware domain id = core index + 1) or device
|
||||||
|
/// (by requester id / bdf).
|
||||||
|
pub const Backend = struct {
|
||||||
|
/// Number of page-table levels (3 or 4) the backend selected from hardware caps.
|
||||||
|
levels: u8,
|
||||||
|
/// Largest leaf the walker may emit: 4 KiB always, 2 MiB when the backend allows.
|
||||||
|
supports_huge_pages: bool,
|
||||||
|
|
||||||
|
/// Raw entry bits for a leaf mapping `physical` (with the given size), and for a
|
||||||
|
/// non-leaf entry pointing at `table_physical` at `level` (level counts down to 1
|
||||||
|
/// at the leaf's parent). `isPresent` tests a read-back entry.
|
||||||
|
makeLeaf: *const fn (physical: u64, huge: bool) u64,
|
||||||
|
makeTable: *const fn (table_physical: u64, level: u8) u64,
|
||||||
|
isPresent: *const fn (entry: u64) bool,
|
||||||
|
/// Flush a cache line holding IOMMU structures the hardware reads non-coherently
|
||||||
|
/// (VT-d with ECAP.C==0). A no-op where the unit snoops caches.
|
||||||
|
flushStructure: *const fn (address: usize) void,
|
||||||
|
|
||||||
|
/// Turn translation on (the core has already seeded any pre-claim domains)
|
||||||
|
/// and write the unit's identity lines to the log — the core follows with
|
||||||
|
/// the neutral posture lines.
|
||||||
|
enable: *const fn () void,
|
||||||
|
/// Point `bdf`'s translation structure at `domain` (hardware id) and invalidate the
|
||||||
|
/// context/device caches so the change takes effect.
|
||||||
|
attach: *const fn (bdf: u16, domain: u16, page_table_root: u64) void,
|
||||||
|
/// Return `bdf`'s translation structure to not-present + invalidate — all its DMA
|
||||||
|
/// faults afterward.
|
||||||
|
detach: *const fn (bdf: u16) void,
|
||||||
|
/// Invalidate cached translations for `domain` (after a map or unmap).
|
||||||
|
invalidateDomain: *const fn (domain: u16) void,
|
||||||
|
/// Pull pending faults out of the hardware, log them (rate-limited), return the
|
||||||
|
/// count seen this call.
|
||||||
|
faultDrain: *const fn () usize,
|
||||||
|
};
|
||||||
|
|
||||||
|
/// The injected kernel services, stored for the backends at detect time.
|
||||||
|
pub var environment: Environment = undefined;
|
||||||
|
|
||||||
|
/// Probe the discovered unit and return its vtable, or null when it is
|
||||||
|
/// unusable (the core stays fail-open and says so).
|
||||||
|
pub fn detect(discovery: Discovery, injected: Environment) ?Backend {
|
||||||
|
environment = injected;
|
||||||
|
return if (discovery.amd) amd.detect(discovery) else intel.detect(discovery);
|
||||||
|
}
|
||||||
@@ -524,6 +524,58 @@ pub fn translateIn(pml4: u64, virtual: u64) ?u64 {
|
|||||||
return (pte & address_mask) | (virtual & (page_size - 1));
|
return (pte & address_mask) | (virtual & (page_size - 1));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// `translateIn` with the ring-3 permission bits enforced: the walk accumulates
|
||||||
|
/// the protection flags of every level it descends through and refuses the
|
||||||
|
/// translation unless the *effective* permission allows the access ring 3 would
|
||||||
|
/// be allowed — U/S set at every level, and (for `for_write`) R/W set at every
|
||||||
|
/// level too. A bit cleared anywhere on the path denies, which is exactly how
|
||||||
|
/// the MMU combines them, so a checked kernel copy sees the same permissions the
|
||||||
|
/// process itself does.
|
||||||
|
///
|
||||||
|
/// This is the walk `system/kernel/user-memory.zig` copies through, and the
|
||||||
|
/// reason a kernel copy can never be steered at a kernel-only mapping or made to
|
||||||
|
/// write a read-only user page (a process's own text, say).
|
||||||
|
///
|
||||||
|
/// Huge pages: a 2 MiB PDE leaf resolves like `translateIn`, with its own U/S and
|
||||||
|
/// R/W folded into the accumulator first. A PDPTE with PS set (a 1 GiB leaf) is
|
||||||
|
/// refused rather than descended into — danos never builds one, and denying is
|
||||||
|
/// the safe direction for a permission-checked walk.
|
||||||
|
pub fn translateUserIn(pml4: u64, virtual: u64, for_write: bool) ?u64 {
|
||||||
|
// Start all-ones and AND in each level: a cleared bit at any level denies.
|
||||||
|
var effective: u64 = ~@as(u64, 0);
|
||||||
|
|
||||||
|
const pml4e = tableAt(pml4)[(virtual >> 39) & 0x1FF];
|
||||||
|
if (pml4e & present == 0) return null;
|
||||||
|
effective &= pml4e;
|
||||||
|
|
||||||
|
const pdpte = tableAt(pml4e & address_mask)[(virtual >> 30) & 0x1FF];
|
||||||
|
if (pdpte & present == 0) return null;
|
||||||
|
if (pdpte & page_size_bit != 0) return null; // 1 GiB leaf: never built here, refuse
|
||||||
|
effective &= pdpte;
|
||||||
|
|
||||||
|
const pde = tableAt(pdpte & address_mask)[(virtual >> 21) & 0x1FF];
|
||||||
|
if (pde & present == 0) return null;
|
||||||
|
effective &= pde;
|
||||||
|
if (pde & page_size_bit != 0) { // 2 MiB huge leaf: frame base is bits 51:21
|
||||||
|
if (!permits(effective, for_write)) return null;
|
||||||
|
return (pde & address_mask & ~@as(u64, huge_page_size - 1)) | (virtual & (huge_page_size - 1));
|
||||||
|
}
|
||||||
|
|
||||||
|
const pte = tableAt(pde & address_mask)[(virtual >> 12) & 0x1FF];
|
||||||
|
if (pte & present == 0) return null;
|
||||||
|
effective &= pte;
|
||||||
|
if (!permits(effective, for_write)) return null;
|
||||||
|
return (pte & address_mask) | (virtual & (page_size - 1));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether accumulated walk flags allow a ring-3 access: user-accessible always,
|
||||||
|
/// and writable when the access is a store.
|
||||||
|
fn permits(effective: u64, for_write: bool) bool {
|
||||||
|
if (effective & user == 0) return false;
|
||||||
|
if (for_write and effective & writable == 0) return false;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
fn invalidate(virtual: u64) void {
|
fn invalidate(virtual: u64) void {
|
||||||
// invlpg needs its operand via a register-indirect memory reference that Zig
|
// invlpg needs its operand via a register-indirect memory reference that Zig
|
||||||
// inline asm won't form directly, so stage the address in a register first.
|
// inline asm won't form directly, so stage the address in a register first.
|
||||||
|
|||||||
@@ -132,13 +132,31 @@ fn record(node: *platform.Device, parent_id: u64) u64 {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Copy up to `out.len` device descriptors into `out`; returns the total count
|
/// Copy up to `out.len` device descriptors into `out`; returns the total count
|
||||||
/// available (which may exceed `out.len`).
|
/// available (which may exceed `out.len`). For kernel callers with a buffer big
|
||||||
|
/// enough to take the whole table in one go.
|
||||||
pub fn enumerate(out: []device_abi.DeviceDescriptor) usize {
|
pub fn enumerate(out: []device_abi.DeviceDescriptor) usize {
|
||||||
const n = @min(count, out.len);
|
_ = enumerateFrom(0, out);
|
||||||
@memcpy(out[0..n], devices[0..n]);
|
|
||||||
return count;
|
return count;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// How many devices the table holds — the total `device_enumerate` reports back
|
||||||
|
/// however few of them fit in the caller's buffer.
|
||||||
|
pub fn deviceCount() usize {
|
||||||
|
return count;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Copy up to `out.len` descriptors starting at table index `start`, returning how
|
||||||
|
/// many were filled (0 once `start` reaches the end). The chunked form: the
|
||||||
|
/// `device_enumerate` system call bounces the table out through a small kernel
|
||||||
|
/// buffer, one chunk at a time, because a descriptor is far too big to stage a
|
||||||
|
/// whole user-requested array of them on a 16 KiB kernel stack.
|
||||||
|
pub fn enumerateFrom(start: usize, out: []device_abi.DeviceDescriptor) usize {
|
||||||
|
if (start >= count) return 0;
|
||||||
|
const n = @min(count - start, out.len);
|
||||||
|
@memcpy(out[0..n], devices[start..][0..n]);
|
||||||
|
return n;
|
||||||
|
}
|
||||||
|
|
||||||
/// Take exclusive ownership of device `id` for task `owner`. Fails if the id is
|
/// Take exclusive ownership of device `id` for task `owner`. Fails if the id is
|
||||||
/// out of range or already claimed.
|
/// out of range or already claimed.
|
||||||
pub fn claim(id: u64, owner: u32) bool {
|
pub fn claim(id: u64, owner: u32) bool {
|
||||||
@@ -167,6 +185,20 @@ pub fn releaseAllOwnedBy(owner: u32) void {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Release the claim on `id` iff `owner` holds it — the rollback for a claim that
|
||||||
|
/// cannot be confined (the IOMMU domain could not be created/attached). Returns true
|
||||||
|
/// when a claim was actually cleared.
|
||||||
|
pub fn unclaim(id: u64, owner: u32) bool {
|
||||||
|
if (id >= count) return false;
|
||||||
|
if (claimed[@intCast(id)]) |o| {
|
||||||
|
if (o == owner) {
|
||||||
|
claimed[@intCast(id)] = null;
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
/// Resource `index` of device `id`, or null if out of range.
|
/// Resource `index` of device `id`, or null if out of range.
|
||||||
pub fn resourceOf(id: u64, index: u64) ?device_abi.ResourceDescriptor {
|
pub fn resourceOf(id: u64, index: u64) ?device_abi.ResourceDescriptor {
|
||||||
if (id >= count) return null;
|
if (id >= count) return null;
|
||||||
@@ -175,6 +207,50 @@ pub fn resourceOf(id: u64, index: u64) ?device_abi.ResourceDescriptor {
|
|||||||
return d.resources[@intCast(index)];
|
return d.resources[@intCast(index)];
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The PCI requester id (bus<<8 | device<<3 | function) of device `id`, derived from
|
||||||
|
/// its config-space slice against its host bridge's ECAM window — the identity a VT-d
|
||||||
|
/// context entry / AMD-Vi DTE is keyed by. null when `id` is not a PCI function or the
|
||||||
|
/// geometry doesn't decode. The kernel never stored the BDF (the descriptor has no such
|
||||||
|
/// field); pci-bus encodes it into resource 0's physical base as
|
||||||
|
/// `ecam_base + ((bus - start_bus) << 20 | device << 15 | function << 12)`, and the
|
||||||
|
/// requester id the device emits uses the absolute bus, so we add `start_bus << 8` back.
|
||||||
|
pub fn pciAddressOf(id: u64) ?u16 {
|
||||||
|
if (id >= count) return null;
|
||||||
|
const d = &devices[@intCast(id)];
|
||||||
|
if (d.class != @intFromEnum(device_abi.DeviceClass.pci_device)) return null;
|
||||||
|
if (d.resource_count == 0) return null;
|
||||||
|
const config = d.resources[0];
|
||||||
|
if (config.kind != @intFromEnum(device_abi.ResourceKind.memory) or config.len != 4096) return null;
|
||||||
|
|
||||||
|
// Walk up to the host bridge, whose resource 0 is the segment's ECAM window and
|
||||||
|
// resource 1 the bus_range (start_bus, bus_count).
|
||||||
|
var parent = d.parent;
|
||||||
|
while (parent != device_abi.no_parent and parent < count) {
|
||||||
|
const p = &devices[@intCast(parent)];
|
||||||
|
if (p.class == @intFromEnum(device_abi.DeviceClass.pci_host_bridge)) {
|
||||||
|
if (p.resource_count < 2) return null;
|
||||||
|
const ecam = p.resources[0];
|
||||||
|
const bus_range = p.resources[1];
|
||||||
|
if (config.start < ecam.start or config.start >= ecam.start + ecam.len) return null;
|
||||||
|
const offset = config.start - ecam.start;
|
||||||
|
const start_bus: u16 = @intCast(bus_range.start & 0xFF);
|
||||||
|
return @intCast((offset >> 12) + (@as(u64, start_bus) << 8));
|
||||||
|
}
|
||||||
|
parent = p.parent;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Call `visit(id, bdf)` for every PCI function in the table — the IOMMU core's boot
|
||||||
|
/// sweep to place every device under a domain. Only functions whose BDF decodes are
|
||||||
|
/// visited.
|
||||||
|
pub fn forEachPciFunction(visit: *const fn (id: u64, bdf: u16) void) void {
|
||||||
|
var id: u64 = 0;
|
||||||
|
while (id < count) : (id += 1) {
|
||||||
|
if (pciAddressOf(id)) |bdf| visit(id, bdf);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Is `child` wholly inside `parent`? For a range (memory, io_port, bus_range) that's
|
/// Is `child` wholly inside `parent`? For a range (memory, io_port, bus_range) that's
|
||||||
/// interval containment; for an irq it's equality, since an interrupt line is not
|
/// interval containment; for an irq it's equality, since an interrupt line is not
|
||||||
/// divisible. Zero-length child ranges are refused — an empty window is meaningless
|
/// divisible. Zero-length child ranges are refused — an empty window is meaningless
|
||||||
|
|||||||
@@ -0,0 +1,397 @@
|
|||||||
|
//! system/kernel/iommu.zig — architecture-neutral IOMMU core: per-device DMA
|
||||||
|
//! translation domains over a backend the architecture module supplies
|
||||||
|
//! (x86-64: Intel VT-d or AMD-Vi, behind architecture/x86_64/iommu.zig).
|
||||||
|
//!
|
||||||
|
//! The problem this closes: without an IOMMU, a claimed bus-mastering device can DMA to
|
||||||
|
//! ANY physical address, so a compromised or buggy driver reaches all of memory through
|
||||||
|
//! its device — driver isolation stops at the CPU's MMU. This core gives each claimed
|
||||||
|
//! PCI function its own translation domain; a device reaches only the physical ranges
|
||||||
|
//! mapped into its domain, and nothing else (kernel, page tables, other processes) is
|
||||||
|
//! visible to it.
|
||||||
|
//!
|
||||||
|
//! Design:
|
||||||
|
//! - **Identity mappings** (IOVA == physical). `dma_alloc` already hands drivers the
|
||||||
|
//! physical address they program into hardware; a domain simply makes that same
|
||||||
|
//! address the ONLY thing the device can reach. No IOVA allocator, and every
|
||||||
|
//! driver's register-programming code is untouched.
|
||||||
|
//! - **Architecture-neutral**: this file owns the domain table and a shared
|
||||||
|
//! 512-entry page-table walker; the architecture module's `Backend` vtable
|
||||||
|
//! supplies the hardware specifics — the entry-bit encodings, the
|
||||||
|
//! enable/invalidate register dances, and the fault drain — with the frame
|
||||||
|
//! allocator and log sink injected the other way.
|
||||||
|
//! - **Fail-open**: when no IOMMU is found, nothing activates and every entry point is a
|
||||||
|
//! success no-op, so callers in process.zig stay unconditional and behavior is
|
||||||
|
//! byte-for-byte the pre-IOMMU kernel. The boot log states the posture.
|
||||||
|
//!
|
||||||
|
//! All entry points run under the big kernel lock (the caller holds it); no internal
|
||||||
|
//! locking. All memory comes from `pmm` reached through the physmap, like paging.zig.
|
||||||
|
|
||||||
|
const std = @import("std");
|
||||||
|
const abi = @import("abi");
|
||||||
|
const boot_handoff = @import("boot-handoff");
|
||||||
|
const pmm = @import("pmm.zig");
|
||||||
|
const platform = @import("platform");
|
||||||
|
const architecture = @import("architecture");
|
||||||
|
const devices_broker = @import("devices-broker.zig");
|
||||||
|
const log = @import("log.zig");
|
||||||
|
|
||||||
|
const page_size: u64 = abi.page_size;
|
||||||
|
const page_mask: u64 = page_size - 1;
|
||||||
|
const huge_page_size: u64 = 2 * 1024 * 1024;
|
||||||
|
|
||||||
|
/// One domain per claimed PCI function. 64 mirrors devices-broker's device cap.
|
||||||
|
pub const maximum_domains = 64;
|
||||||
|
pub const invalid_domain: u16 = 0xFFFF;
|
||||||
|
|
||||||
|
const Domain = struct {
|
||||||
|
in_use: bool = false,
|
||||||
|
owner: u32 = 0, // task that owns the attached device
|
||||||
|
bdf: u16 = 0, // requester id of the attached device
|
||||||
|
page_table_root: u64 = 0, // physical address of the top-level table
|
||||||
|
rmrr: bool = false, // a firmware reserved-region domain (persists across claims)
|
||||||
|
};
|
||||||
|
|
||||||
|
var active: bool = false;
|
||||||
|
var backend: architecture.iommu.Backend = undefined;
|
||||||
|
var domains: [maximum_domains]Domain = .{Domain{}} ** maximum_domains;
|
||||||
|
|
||||||
|
pub fn enabled() bool {
|
||||||
|
return active;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Detect the IOMMU (the architecture module probes the discovered unit and
|
||||||
|
/// returns its backend), pre-map firmware reserved regions, and enable
|
||||||
|
/// translation. Fail-open (nothing activates) when no usable unit exists — the
|
||||||
|
/// caller logs the posture. Must run after platform discovery and before any
|
||||||
|
/// user process starts.
|
||||||
|
pub fn init() void {
|
||||||
|
const info = platform.platformInformation();
|
||||||
|
if (!info.iommu_present) return;
|
||||||
|
// A present-but-unusable unit stays fail-open with a logged reason rather
|
||||||
|
// than half-enabling. The backend receives the kernel services it needs
|
||||||
|
// (frames, the log sink) here — it never imports kernel internals.
|
||||||
|
backend = architecture.iommu.detect(.{
|
||||||
|
.register_base = info.iommu_base,
|
||||||
|
.amd = info.iommu_is_amd,
|
||||||
|
}, .{
|
||||||
|
.allocateFrame = pmm.alloc,
|
||||||
|
.allocateContiguous = pmm.allocContiguous,
|
||||||
|
.write = log.write,
|
||||||
|
}) orelse {
|
||||||
|
log.write("/system/kernel: WARNING IOMMU present but unusable — staying fail-open\n");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
active = true;
|
||||||
|
|
||||||
|
// The translation structures start empty: every device is denied until its driver
|
||||||
|
// claims it (confineDevice gives it a private domain). PCI functions are enumerated
|
||||||
|
// post-boot by the ring-3 pci-bus driver, so there is nothing to attach at init.
|
||||||
|
// The backend writes its identity lines; the neutral posture lines follow.
|
||||||
|
backend.enable();
|
||||||
|
logPosture(info);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Per-claimed-device record: its private domain, so a driver's death tears down
|
||||||
|
/// exactly the domains it held.
|
||||||
|
const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0, domain: u16 = invalid_domain };
|
||||||
|
var confined: [maximum_domains]Confined = .{Confined{}} ** maximum_domains;
|
||||||
|
|
||||||
|
/// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`: give
|
||||||
|
/// it a private empty domain, seed it with the device's own firmware reserved region,
|
||||||
|
/// and attach. Its DMA buffers arrive afterward as explicit grants — the owner's own
|
||||||
|
/// `dma_alloc`'d regions are bound by the claim path (`mapForDevice`), and cross-process
|
||||||
|
/// buffers by `dma_bind`. false only if a domain can't be allocated — the caller rolls
|
||||||
|
/// the claim back (a claim that can't be confined must not stand). No-op success when no
|
||||||
|
/// IOMMU exists (fail-open).
|
||||||
|
pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool {
|
||||||
|
if (!active) return true;
|
||||||
|
if (device_id >= confined.len) return true; // unusual id; leave it to fail-open
|
||||||
|
const domain = domainCreate(owner, bdf) orelse return false;
|
||||||
|
|
||||||
|
// Firmware reserved region for this device, if any (real hardware; QEMU has none).
|
||||||
|
const info = platform.platformInformation();
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < info.rmrr_count) : (i += 1) {
|
||||||
|
if (info.rmrr[i].bdf == bdf)
|
||||||
|
_ = map(domain, info.rmrr[i].base, info.rmrr[i].limit - info.rmrr[i].base + 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
attachDevice(domain, bdf);
|
||||||
|
confined[@intCast(device_id)] = .{ .active = true, .owner = owner, .bdf = bdf, .domain = domain };
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The confined record for `device_id`, or null if the device is not confined.
|
||||||
|
fn confinedOf(device_id: u64) ?*Confined {
|
||||||
|
if (device_id >= confined.len) return null;
|
||||||
|
const c = &confined[@intCast(device_id)];
|
||||||
|
return if (c.active) c else null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Map a DMA region into a specific claimed device's domain (the device owner binding a
|
||||||
|
/// granted buffer). false if the device is not confined. No-op success without an IOMMU.
|
||||||
|
pub fn mapForDevice(device_id: u64, physical: u64, len: u64) bool {
|
||||||
|
if (!active) return true;
|
||||||
|
const c = confinedOf(device_id) orelse return false;
|
||||||
|
return map(c.domain, physical, len);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Unmap a DMA region from a specific claimed device's domain. No-op if not confined.
|
||||||
|
pub fn unmapForDevice(device_id: u64, physical: u64, len: u64) void {
|
||||||
|
if (!active) return;
|
||||||
|
const c = confinedOf(device_id) orelse return;
|
||||||
|
unmap(c.domain, physical, len);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Map a region into every claimed device owned by `owner` — the auto-bind of a task's
|
||||||
|
/// own freshly-`dma_alloc`'d buffer into the devices it drives.
|
||||||
|
pub fn mapRegionForOwner(owner: u32, physical: u64, len: u64) void {
|
||||||
|
if (!active) return;
|
||||||
|
for (&confined) |*c| {
|
||||||
|
if (c.active and c.owner == owner) _ = map(c.domain, physical, len);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Unmap a region from EVERY claimed device's domain — the freed-region sweep. MUST run
|
||||||
|
/// before the frames return to pmm: a device translating to a reallocated frame is the
|
||||||
|
/// use-after-free this prevents. Cross-device because a granted buffer may be bound in a
|
||||||
|
/// domain other than its owner's.
|
||||||
|
pub fn unmapRegionEverywhere(physical: u64, len: u64) void {
|
||||||
|
if (!active) return;
|
||||||
|
for (&confined) |*c| {
|
||||||
|
if (c.active) unmap(c.domain, physical, len);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A driver died or released its devices: tear down every domain it held (detach the
|
||||||
|
/// device, free the tables) so their DMA is blocked again and a restarted driver
|
||||||
|
/// re-claims cleanly. Runs BEFORE the broker claims and the DMA frames are released.
|
||||||
|
pub fn releaseAllOwnedBy(owner: u32) void {
|
||||||
|
if (!active) return;
|
||||||
|
for (&confined) |*c| {
|
||||||
|
if (c.active and c.owner == owner) {
|
||||||
|
detachDevice(c.bdf);
|
||||||
|
domainDestroy(c.domain);
|
||||||
|
c.* = .{};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
_ = faultDrain(); // log any faults a mid-DMA device raised as it was cut off
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Allocate an empty domain (an empty top-level table). null when the table is full.
|
||||||
|
pub fn domainCreate(owner: u32, bdf: u16) ?u16 {
|
||||||
|
if (!active) return 0; // fail-open: a dummy id the no-op ops ignore
|
||||||
|
for (&domains, 0..) |*d, index| {
|
||||||
|
if (d.in_use) continue;
|
||||||
|
const root = allocTable() orelse return null;
|
||||||
|
d.* = .{ .in_use = true, .owner = owner, .bdf = bdf, .page_table_root = root };
|
||||||
|
return @intCast(index);
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Free a domain's page-table frames and its slot. Precondition: no device attached
|
||||||
|
/// (detach first).
|
||||||
|
pub fn domainDestroy(domain: u16) void {
|
||||||
|
if (!active) return;
|
||||||
|
const d = &domains[domain];
|
||||||
|
if (!d.in_use) return;
|
||||||
|
freeTables(d.page_table_root, backend.levels);
|
||||||
|
d.* = .{};
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Attach `bdf`'s device to `domain` and pre-load any RMRR range recorded for it.
|
||||||
|
pub fn attachDevice(domain: u16, bdf: u16) void {
|
||||||
|
if (!active) return;
|
||||||
|
const d = &domains[domain];
|
||||||
|
d.bdf = bdf;
|
||||||
|
backend.attach(bdf, hardwareId(domain), d.page_table_root);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Return `bdf`'s device to not-present + invalidate.
|
||||||
|
pub fn detachDevice(bdf: u16) void {
|
||||||
|
if (!active) return;
|
||||||
|
backend.detach(bdf);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Identity-map [physical, physical+len) into `domain` (read+write) and invalidate.
|
||||||
|
/// Unconditional domain-selective invalidation after every map — correct under VT-d
|
||||||
|
/// caching-mode and free otherwise.
|
||||||
|
pub fn map(domain: u16, physical: u64, len: u64) bool {
|
||||||
|
if (!active) return true;
|
||||||
|
const d = &domains[domain];
|
||||||
|
if (!d.in_use) return false;
|
||||||
|
if (!mapRange(d.page_table_root, physical, len)) return false;
|
||||||
|
backend.invalidateDomain(hardwareId(domain));
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Unmap [physical, physical+len) from `domain` and invalidate. MUST finish its
|
||||||
|
/// invalidation before the caller returns the frames to pmm — a stale IOTLB entry
|
||||||
|
/// pointing at a reallocated frame is the use-after-free this ordering prevents.
|
||||||
|
pub fn unmap(domain: u16, physical: u64, len: u64) void {
|
||||||
|
if (!active) return;
|
||||||
|
const d = &domains[domain];
|
||||||
|
if (!d.in_use) return;
|
||||||
|
unmapRange(d.page_table_root, physical, len);
|
||||||
|
backend.invalidateDomain(hardwareId(domain));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Poll the hardware for translation faults, log them, return the count. Called by the
|
||||||
|
/// IOMMU test case and opportunistically after a device detaches.
|
||||||
|
pub fn faultDrain() usize {
|
||||||
|
if (!active) return 0;
|
||||||
|
return backend.faultDrain();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The physical address `virtual` maps to in `domain`, or null if unmapped — a test
|
||||||
|
/// helper that walks the domain's page tables (identity mappings return `virtual`).
|
||||||
|
pub fn translationOf(domain: u16, virtual: u64) ?u64 {
|
||||||
|
if (!active) return virtual;
|
||||||
|
const d = &domains[domain];
|
||||||
|
if (!d.in_use) return null;
|
||||||
|
var table = d.page_table_root;
|
||||||
|
var level = backend.levels;
|
||||||
|
while (level > 1) : (level -= 1) {
|
||||||
|
const entry = tableAt(table)[indexAt(virtual, level)];
|
||||||
|
if (!backend.isPresent(entry)) return null;
|
||||||
|
if (level == 2 and isHugeLeaf(entry))
|
||||||
|
return (entry & address_mask) | (virtual & (huge_page_size - 1));
|
||||||
|
table = entry & address_mask;
|
||||||
|
}
|
||||||
|
const leaf = tableAt(table)[indexAt(virtual, 1)];
|
||||||
|
if (!backend.isPresent(leaf)) return null;
|
||||||
|
return (leaf & address_mask) | (virtual & page_mask);
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- the shared page-table walker -----------------------------------------------------
|
||||||
|
// 512-entry, 9-bits-per-level, 4 KiB tables reached through the physmap — the shape both
|
||||||
|
// VT-d second-level and AMD-Vi native tables share. The backend supplies the entry bits.
|
||||||
|
|
||||||
|
fn tableAt(physical: u64) [*]volatile u64 {
|
||||||
|
return @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
||||||
|
}
|
||||||
|
|
||||||
|
fn allocTable() ?u64 {
|
||||||
|
const frame = pmm.alloc() orelse return null;
|
||||||
|
const table = tableAt(frame);
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < 512) : (i += 1) table[i] = 0;
|
||||||
|
return frame;
|
||||||
|
}
|
||||||
|
|
||||||
|
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||||
|
|
||||||
|
fn indexAt(virtual: u64, level: u8) usize {
|
||||||
|
// level 1 is the leaf table; shift = 12 + 9*(level-1).
|
||||||
|
const shift: u6 = @intCast(12 + 9 * (@as(u32, level) - 1));
|
||||||
|
return @intCast((virtual >> shift) & 0x1FF);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Descend to (allocating) the next-level table below `entry_ptr`, returning its
|
||||||
|
/// physical base. null on out-of-memory.
|
||||||
|
fn descend(entry_ptr: *volatile u64, level: u8) ?u64 {
|
||||||
|
const entry = entry_ptr.*;
|
||||||
|
if (backend.isPresent(entry)) return entry & address_mask;
|
||||||
|
const table = allocTable() orelse return null;
|
||||||
|
backend.flushStructure(@intFromPtr(tableAt(table)));
|
||||||
|
entry_ptr.* = backend.makeTable(table, level);
|
||||||
|
backend.flushStructure(@intFromPtr(entry_ptr));
|
||||||
|
return table;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn mapRange(root: u64, physical: u64, len: u64) bool {
|
||||||
|
const start = physical & ~page_mask;
|
||||||
|
const end = (physical + len + page_mask) & ~page_mask;
|
||||||
|
var addr = start;
|
||||||
|
while (addr < end) {
|
||||||
|
// 2 MiB leaf when the backend allows it and both address and remaining span are
|
||||||
|
// huge-aligned — keeps table memory sane for the blanket-identity and real-PC
|
||||||
|
// cases without a separate superpage path per backend.
|
||||||
|
const huge = backend.supports_huge_pages and
|
||||||
|
addr % huge_page_size == 0 and (end - addr) >= huge_page_size;
|
||||||
|
if (!mapOne(root, addr, huge)) return false;
|
||||||
|
addr += if (huge) huge_page_size else page_size;
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn mapOne(root: u64, addr: u64, huge: bool) bool {
|
||||||
|
const leaf_level: u8 = if (huge) 2 else 1;
|
||||||
|
var table = root;
|
||||||
|
var level = backend.levels;
|
||||||
|
while (level > leaf_level) : (level -= 1) {
|
||||||
|
const entry_ptr = &tableAt(table)[indexAt(addr, level)];
|
||||||
|
table = descend(entry_ptr, level) orelse return false;
|
||||||
|
}
|
||||||
|
const leaf_ptr = &tableAt(table)[indexAt(addr, leaf_level)];
|
||||||
|
leaf_ptr.* = backend.makeLeaf(addr, huge);
|
||||||
|
backend.flushStructure(@intFromPtr(leaf_ptr));
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn unmapRange(root: u64, physical: u64, len: u64) void {
|
||||||
|
const start = physical & ~page_mask;
|
||||||
|
const end = (physical + len + page_mask) & ~page_mask;
|
||||||
|
var addr = start;
|
||||||
|
while (addr < end) {
|
||||||
|
const huge = backend.supports_huge_pages and
|
||||||
|
addr % huge_page_size == 0 and (end - addr) >= huge_page_size;
|
||||||
|
unmapOne(root, addr, huge);
|
||||||
|
addr += if (huge) huge_page_size else page_size;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn unmapOne(root: u64, addr: u64, huge: bool) void {
|
||||||
|
const leaf_level: u8 = if (huge) 2 else 1;
|
||||||
|
var table = root;
|
||||||
|
var level = backend.levels;
|
||||||
|
while (level > leaf_level) : (level -= 1) {
|
||||||
|
const entry = tableAt(table)[indexAt(addr, level)];
|
||||||
|
if (!backend.isPresent(entry)) return; // nothing mapped here
|
||||||
|
table = entry & address_mask;
|
||||||
|
}
|
||||||
|
const leaf_ptr = &tableAt(table)[indexAt(addr, leaf_level)];
|
||||||
|
leaf_ptr.* = 0;
|
||||||
|
backend.flushStructure(@intFromPtr(leaf_ptr));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Post-order free of a domain's whole table tree.
|
||||||
|
fn freeTables(root: u64, level: u8) void {
|
||||||
|
if (level > 1) {
|
||||||
|
const table = tableAt(root);
|
||||||
|
var i: usize = 0;
|
||||||
|
while (i < 512) : (i += 1) {
|
||||||
|
const entry = table[i];
|
||||||
|
if (!backend.isPresent(entry)) continue;
|
||||||
|
// A 2 MiB leaf sits at level 2 and points at RAM, not a sub-table.
|
||||||
|
if (level == 2 and isHugeLeaf(entry)) continue;
|
||||||
|
freeTables(entry & address_mask, level - 1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
pmm.free(root);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn isHugeLeaf(entry: u64) bool {
|
||||||
|
// Both backends set a page-size bit (VT-d bit 7, AMD leaf next-level=0 at level 2).
|
||||||
|
// The backend's makeLeaf encodes it; the walker only needs "is this a leaf, not a
|
||||||
|
// pointer" at level 2, which huge leaves are by construction.
|
||||||
|
return entry & huge_leaf_bit != 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The size-bit the backends set on a 2 MiB leaf (VT-d SL-PTE PS bit 7; AMD encodes a
|
||||||
|
/// leaf as next-level 0, so the core marks huge leaves with this software bit — an
|
||||||
|
/// ignored bit in both formats — to tell them apart from table pointers when freeing).
|
||||||
|
const huge_leaf_bit: u64 = 1 << 7;
|
||||||
|
|
||||||
|
fn hardwareId(domain: u16) u16 {
|
||||||
|
return domain + 1; // id 0 is reserved by both architectures
|
||||||
|
}
|
||||||
|
|
||||||
|
fn logPosture(info: platform.PlatformInformation) void {
|
||||||
|
log.print(" rmrr : {d} region(s) premapped\n", .{info.rmrr_count});
|
||||||
|
if (info.rmrr_skipped > 0)
|
||||||
|
log.print(" rmrr : WARNING {d} scope(s) skipped — a device keeps an unmapped firmware buffer\n", .{info.rmrr_skipped});
|
||||||
|
if (info.iommu_extra_units > 0)
|
||||||
|
log.print(" units : WARNING {d} other unit(s) — their scoped devices are NOT translated\n", .{info.iommu_extra_units});
|
||||||
|
}
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user