Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
198767bb30 |
@@ -1,131 +0,0 @@
|
||||
//! The danos build API (docs/build-packages-plan.md): the one shared recipe
|
||||
//! for building a user-space binary. A binary package's build.zig names its
|
||||
//! binary and EXACTLY the modules its source imports — the moral equivalent
|
||||
//! of a C file's include list — and `userBinary` resolves each name from the
|
||||
//! library domain package that exports it. Nothing is pre-wired: an @import
|
||||
//! the package did not declare is a compile error, and a domain none of the
|
||||
//! imports come from never appears in the package's manifest. The only
|
||||
//! implicit dependency is the kernel package, because the shared root shim
|
||||
//! (root.zig, user.ld) lives there and itself reaches start + logging.
|
||||
//!
|
||||
//! Consumers declare this package in their build.zig.zon (as "build-support")
|
||||
//! and @import its build.zig from their own build.zig; nothing is compiled
|
||||
//! from this package itself — it exports build-time functions only.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
_ = b; // nothing to build: this package exports build-time functions only
|
||||
}
|
||||
|
||||
/// The freestanding x86-64 target every danos binary (kernel and user) is
|
||||
/// built for. SSE2 is part of the x86_64 baseline and UEFI leaves it enabled
|
||||
/// at handoff, so we keep it: disabling it forces soft-float and makes the
|
||||
/// compiler unable to encode the vector ops that std's formatting/runtime
|
||||
/// still emit.
|
||||
pub fn freestandingTarget(b: *std.Build) std.Build.ResolvedTarget {
|
||||
return b.resolveTargetQuery(.{
|
||||
.cpu_arch = .x86_64,
|
||||
.os_tag = .freestanding,
|
||||
.abi = .none,
|
||||
});
|
||||
}
|
||||
|
||||
/// Resolve one imported module by searching the packages this binary DECLARED
|
||||
/// in its own build.zig.zon — the C include path made literal: an import can
|
||||
/// only be satisfied by a domain the binary claims, and each domain's own
|
||||
/// build.zig (its addModule exports) is the single statement of who owns
|
||||
/// what. There is no name table here to drift.
|
||||
fn moduleFromDeclaredDependencies(b: *std.Build, name: []const u8) *std.Build.Module {
|
||||
for (b.available_deps) |declared| {
|
||||
const dependency = b.dependency(declared[0], .{});
|
||||
if (dependency.builder.modules.get(name)) |module| return module;
|
||||
}
|
||||
@panic(b.fmt(
|
||||
"no declared dependency exports a module named '{s}' — declare the domain that owns it in this package's build.zig.zon",
|
||||
.{name},
|
||||
));
|
||||
}
|
||||
|
||||
/// What `userBinary` needs to know about one user binary.
|
||||
pub const UserBinaryOptions = struct {
|
||||
name: []const u8,
|
||||
/// The program's own source file — it becomes the `program` module the
|
||||
/// root shim imports; a program only defines `pub fn main`.
|
||||
root_source_file: std.Build.LazyPath,
|
||||
/// Exactly the modules the program's source @imports (directly or through
|
||||
/// its same-directory files) — no more, no less. Order is free; sorted
|
||||
/// reads best. An undeclared @import fails the compile; a name no
|
||||
/// declared domain exports fails the build graph, naming the miss.
|
||||
imports: []const []const u8,
|
||||
/// Built multi-threaded (`single_threaded = false`) so real atomics/TLS
|
||||
/// work — required before a binary may call `Thread.spawn`
|
||||
/// (docs/threading.md). Threads are a deliberate per-binary opt-in.
|
||||
threaded: bool = false,
|
||||
};
|
||||
|
||||
/// Build one user-space binary the same way for every program (init, the
|
||||
/// services, the drivers): freestanding, ReleaseSmall, `.large` code model
|
||||
/// (the image base is above 4 GiB — smaller models emit 32-bit relocations
|
||||
/// that can't reach), linked with the shared user link script. Pinned to
|
||||
/// LLVM + LLD so the script's PHDRS (segment permissions) are authoritative —
|
||||
/// the kernel's W^X user-ELF loader requires exact perms.
|
||||
///
|
||||
/// The compilation root is not the program's own file but the shared shim
|
||||
/// (the kernel package's root.zig), which supplies the root declarations
|
||||
/// (`main` re-export, panic handler, `_start` pull) so a program only defines
|
||||
/// `pub fn main`. The program's file becomes the `program` module the shim
|
||||
/// imports; reach it through `programModule` to add per-binary non-library
|
||||
/// modules (compile-time options).
|
||||
pub fn userBinary(b: *std.Build, options: UserBinaryOptions) *std.Build.Step.Compile {
|
||||
const kernel = b.dependency("kernel", .{});
|
||||
var imports: std.ArrayListUnmanaged(std.Build.Module.Import) = .empty;
|
||||
for (options.imports) |name| {
|
||||
imports.append(b.allocator, .{
|
||||
.name = name,
|
||||
.module = moduleFromDeclaredDependencies(b, name),
|
||||
}) catch @panic("OOM");
|
||||
}
|
||||
// Settings (target, optimize, code model, ...) live on the root module
|
||||
// only; the program module inherits them.
|
||||
const program_module = b.createModule(.{
|
||||
.root_source_file = options.root_source_file,
|
||||
.imports = imports.items,
|
||||
});
|
||||
const exe = b.addExecutable(.{
|
||||
.name = options.name,
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = kernel.path("root.zig"),
|
||||
.target = freestandingTarget(b),
|
||||
.optimize = .ReleaseSmall,
|
||||
.code_model = .large,
|
||||
.single_threaded = !options.threaded, // a threaded binary needs real atomics/TLS
|
||||
.sanitize_c = .off,
|
||||
.stack_check = false,
|
||||
.stack_protector = false,
|
||||
// The root shim itself imports only start (_start + panic) and
|
||||
// logging (std_options) — straight from the kernel package, so a
|
||||
// program's own import list stays exactly its own.
|
||||
.imports = &.{
|
||||
.{ .name = "start", .module = kernel.module("start") },
|
||||
.{ .name = "logging", .module = kernel.module("logging") },
|
||||
.{ .name = "program", .module = program_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
exe.setLinkerScript(kernel.path("user.ld"));
|
||||
exe.entry = .{ .symbol_name = "_start" };
|
||||
exe.image_base = 0x7000_0000_0000;
|
||||
exe.use_llvm = true;
|
||||
exe.use_lld = true;
|
||||
return exe;
|
||||
}
|
||||
|
||||
/// The `program` module of a binary built by `userBinary` — the module rooted
|
||||
/// at the program's own source file. Per-binary non-library modules (an
|
||||
/// addOptions build_options) go here, not on the root shim: module imports
|
||||
/// are not transitive, so an import added to the root would be invisible to
|
||||
/// the program's code.
|
||||
pub fn programModule(exe: *std.Build.Step.Compile) *std.Build.Module {
|
||||
return exe.root_module.import_table.get("program").?;
|
||||
}
|
||||
@@ -1,8 +0,0 @@
|
||||
.{
|
||||
.name = .build_support,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0xad91962994f4be41, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{},
|
||||
.paths = .{""},
|
||||
}
|
||||
+8
-48
@@ -32,53 +32,6 @@
|
||||
// Once all dependencies are fetched, `zig build` no longer requires
|
||||
// internet connectivity.
|
||||
.dependencies = .{
|
||||
// The danos build API — the shared user-binary recipe every build file
|
||||
// (root and per-binary packages) consumes (docs/build-packages-plan.md).
|
||||
.@"build-support" = .{ .path = "build-support" },
|
||||
// The library domains, each a package exporting its modules.
|
||||
.kernel = .{ .path = "library/kernel" },
|
||||
.device = .{ .path = "library/device" },
|
||||
.client = .{ .path = "library/client" },
|
||||
.protocol = .{ .path = "library/protocol" },
|
||||
.csv = .{ .path = "library/csv" },
|
||||
.@"xkeyboard-config" = .{ .path = "library/xkeyboard-config" },
|
||||
// Binary packages (phase 2), consumed as artifacts for the boot image.
|
||||
.@"pci-bus" = .{ .path = "system/drivers/pci-bus" },
|
||||
.init = .{ .path = "system/services/init" },
|
||||
.fat = .{ .path = "system/services/fat" },
|
||||
.display = .{ .path = "system/services/display" },
|
||||
.@"display-demo" = .{ .path = "system/services/display-demo" },
|
||||
.@"device-manager" = .{ .path = "system/services/device-manager" },
|
||||
.input = .{ .path = "system/services/input" },
|
||||
.logger = .{ .path = "system/services/logger" },
|
||||
// The discovery pair and the /test fixtures are lazy: only what a
|
||||
// given build actually ships gets its build file loaded and compiled
|
||||
// (-Ddiscovery picks one of the pair; -Dtest-case pulls the fixtures).
|
||||
.acpi = .{ .path = "system/services/acpi", .lazy = true },
|
||||
.fdt = .{ .path = "system/services/fdt", .lazy = true },
|
||||
.@"ps2-bus" = .{ .path = "system/drivers/ps2-bus" },
|
||||
.@"usb-xhci-bus" = .{ .path = "system/drivers/usb-xhci-bus" },
|
||||
.@"usb-hid" = .{ .path = "system/drivers/usb-hid" },
|
||||
.@"usb-storage" = .{ .path = "system/drivers/usb-storage" },
|
||||
.@"virtio-gpu" = .{ .path = "system/drivers/virtio-gpu" },
|
||||
.@"vfs-test" = .{ .path = "test/system/services/vfs-test", .lazy = true },
|
||||
.@"fat-test" = .{ .path = "test/system/services/fat-test", .lazy = true },
|
||||
.@"badge-scope-test" = .{ .path = "test/system/services/badge-scope-test", .lazy = true },
|
||||
.@"shared-memory-server" = .{ .path = "test/system/services/shared-memory-server", .lazy = true },
|
||||
.@"shared-memory-client" = .{ .path = "test/system/services/shared-memory-client", .lazy = true },
|
||||
.@"crash-test" = .{ .path = "test/system/services/crash-test", .lazy = true },
|
||||
.@"device-list" = .{ .path = "test/system/services/device-list", .lazy = true },
|
||||
.@"pci-cap-test" = .{ .path = "test/system/services/pci-cap-test", .lazy = true },
|
||||
.@"iommu-fault-test" = .{ .path = "test/system/services/iommu-fault-test", .lazy = true },
|
||||
.@"input-source" = .{ .path = "test/system/services/input-source", .lazy = true },
|
||||
.@"input-test" = .{ .path = "test/system/services/input-test", .lazy = true },
|
||||
.@"args-echo" = .{ .path = "test/system/services/args-echo", .lazy = true },
|
||||
.@"process-test" = .{ .path = "test/system/services/process-test", .lazy = true },
|
||||
.@"thread-test" = .{ .path = "test/system/services/thread-test", .lazy = true },
|
||||
.@"user-memory-test" = .{ .path = "test/system/services/user-memory-test", .lazy = true },
|
||||
.@"protocol-registry-test" = .{ .path = "test/system/services/protocol-registry-test", .lazy = true },
|
||||
.@"protocol-denied-test" = .{ .path = "test/system/services/protocol-denied-test", .lazy = true },
|
||||
.@"protocol-conformance-test" = .{ .path = "test/system/services/protocol-conformance-test", .lazy = true },
|
||||
// See `zig fetch --save <url>` for a command-line interface for adding dependencies.
|
||||
//.example = .{
|
||||
// // When updating this field to a new URL, be sure to delete the corresponding
|
||||
@@ -117,5 +70,12 @@
|
||||
// Paths are relative to the build root. Use the empty string (`""`) to refer to
|
||||
// the build root itself.
|
||||
// A directory listed here means that all files within, recursively, are included.
|
||||
.paths = .{""},
|
||||
.paths = .{
|
||||
"build.zig",
|
||||
"build.zig.zon",
|
||||
"src",
|
||||
// For example...
|
||||
//"LICENSE",
|
||||
//"README.md",
|
||||
},
|
||||
}
|
||||
|
||||
@@ -1,166 +0,0 @@
|
||||
//! Boot-image assembly (docs/build-packages-plan.md, phase 3): everything
|
||||
//! between "here are the built binaries" and "here is a bootable volume".
|
||||
//! The FHS-shaped zig-out install tree, the boot manifest, the boot capsule,
|
||||
//! the FAT32 USB image (+ its serial-enabled twin for the QEMU run steps),
|
||||
//! and the release ISO — with their check steps. The root build.zig decides
|
||||
//! WHAT ships (the bundled list); this file owns HOW it becomes an image.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// One user binary and its FHS home on the boot volume (and in zig-out).
|
||||
pub const BundledBinary = struct { path: []const u8, binary: std.Build.LazyPath };
|
||||
|
||||
pub const Options = struct {
|
||||
/// The installed/flashable kernel (serial follows the root -Dserial).
|
||||
kernel: *std.Build.Step.Compile,
|
||||
/// The serial-enabled kernel variant the `run-x86-64` image boots.
|
||||
kernel_serial: *std.Build.Step.Compile,
|
||||
/// The UEFI loader (BOOTX64).
|
||||
efi: *std.Build.Step.Compile,
|
||||
/// Every user binary and data file at its FHS path.
|
||||
bundled: []const BundledBinary,
|
||||
};
|
||||
|
||||
/// Wire up the install tree, both FAT32 boot images, the release ISO, and the
|
||||
/// check steps. Returns the serial-enabled FAT image for the QEMU run steps.
|
||||
pub fn addImageSteps(b: *std.Build, options: Options) std.Build.LazyPath {
|
||||
// Everything installs into a FHS-shaped zig-out: it IS the danos filesystem *and*
|
||||
// the boot volume. Each binary lands at its addressed, leaf-collapsed path — the
|
||||
// kernel at zig-out/system/kernel (from system/kernel/kernel.zig), init at
|
||||
// zig-out/system/services/init, and so on (see docs/README.md). The bootloader
|
||||
// then loads these FHS paths off the volume.
|
||||
const kernel_install = b.addInstallArtifact(options.kernel, .{ .dest_dir = .{ .override = .{ .custom = "system" } } });
|
||||
b.getInstallStep().dependOn(&kernel_install.step);
|
||||
|
||||
// UEFI firmware requires the removable-media loader at exactly \EFI\BOOT\BOOTX64.efi,
|
||||
// so that path is fixed by the firmware (it is /boot's EFI stub, conceptually).
|
||||
const efi_install = b.addInstallArtifact(options.efi, .{ .dest_dir = .{ .override = .{ .custom = "EFI/BOOT" } } });
|
||||
b.getInstallStep().dependOn(&efi_install.step);
|
||||
|
||||
// The boot manifest: the FHS path of every bundled binary, one per line. The
|
||||
// EFI loader reads THIS by name and opens each listed path by name — FAT
|
||||
// name lookup is case-insensitive and firmware-portable, unlike directory
|
||||
// ENUMERATION, whose returned names vary by firmware (bare 8.3 entries come
|
||||
// back uppercase on some FAT drivers). The tree walk remains only as the
|
||||
// loader's fallback for hand-assembled sticks without a manifest.
|
||||
var manifest_text: std.ArrayListUnmanaged(u8) = .empty;
|
||||
for (options.bundled) |item| {
|
||||
manifest_text.append(b.allocator, '/') catch @panic("OOM");
|
||||
manifest_text.appendSlice(b.allocator, item.path) catch @panic("OOM");
|
||||
manifest_text.append(b.allocator, '\n') catch @panic("OOM");
|
||||
}
|
||||
const manifest_files = b.addWriteFiles();
|
||||
const manifest_file = manifest_files.add("manifest", manifest_text.items);
|
||||
const manifest_install = b.addInstallFileWithDir(manifest_file, .prefix, "system/manifest");
|
||||
b.getInstallStep().dependOn(&manifest_install.step);
|
||||
|
||||
// The boot capsule: the same bundled list packed into ONE file (v2
|
||||
// initial_ramdisk format), because a single open + sequential read is the
|
||||
// only firmware file I/O shape that is fast everywhere — a per-file tree
|
||||
// walk measured MINUTES on real firmware. The loader tries this first,
|
||||
// then the manifest, then the walk; the running system cannot tell the
|
||||
// difference (it always receives the same in-RAM table). Derived from the
|
||||
// tree in the same build graph, so the two cannot drift.
|
||||
const mk_capsule = b.addSystemCommand(&.{"python3"});
|
||||
mk_capsule.addFileArg(b.path("tools/pack-system-image.py"));
|
||||
const capsule_img = mk_capsule.addOutputFileArg("system.img");
|
||||
for (options.bundled) |item| {
|
||||
mk_capsule.addArg(item.path);
|
||||
mk_capsule.addFileArg(item.binary);
|
||||
}
|
||||
const capsule_install = b.addInstallFile(capsule_img, "boot/system.img");
|
||||
b.getInstallStep().dependOn(&capsule_install.step);
|
||||
|
||||
// Install every bundled binary to its FHS home, so zig-out is a true image of
|
||||
// the filesystem — the same tree make-fat-image.py lays out on the boot volume.
|
||||
for (options.bundled) |item| {
|
||||
const install = b.addInstallFileWithDir(item.binary, .prefix, item.path);
|
||||
b.getInstallStep().dependOn(&install.step);
|
||||
}
|
||||
|
||||
// --- danos-usb.img: the bootable FAT32 USB image ---
|
||||
// Format a real FAT32 image (the in-repo Python builder, no external tools)
|
||||
// holding the EFI stub, the kernel, and the whole /system tree of user
|
||||
// binaries at their FHS paths. QEMU presents this image as a USB mass-storage
|
||||
// device the guest boots from (see run-x86-64 and the test harness), and the
|
||||
// danos fat driver mounts the same image at /volumes/usb.
|
||||
const fat_image = addBootImage(b, options.kernel.getEmittedBin(), options.efi.getEmittedBin(), manifest_file, capsule_img, options.bundled);
|
||||
const fat_image_install = b.addInstallFile(fat_image, "danos-usb.img");
|
||||
b.getInstallStep().dependOn(&fat_image_install.step);
|
||||
|
||||
// The image `run-x86-64` boots: identical to the flashable one but with the
|
||||
// serial log sink compiled in, so a developer always gets the machine-readable
|
||||
// log captured to serial0 — without baking serial into the image users flash.
|
||||
// Built lazily (only when `run-x86-64` is requested), and never installed.
|
||||
const fat_image_serial = addBootImage(b, options.kernel_serial.getEmittedBin(), options.efi.getEmittedBin(), manifest_file, capsule_img, options.bundled);
|
||||
|
||||
// `zig build check-fat-image` — validate the produced image is a real FAT32
|
||||
// with the EFI stub present (the builder's own --verify, no external tools).
|
||||
const check_fat = b.addSystemCommand(&.{"python3"});
|
||||
check_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||
check_fat.addArg("--verify");
|
||||
check_fat.addFileArg(fat_image);
|
||||
const check_fat_step = b.step("check-fat-image", "Verify the FAT32 USB image is valid and bootable");
|
||||
check_fat_step.dependOn(&check_fat.step);
|
||||
|
||||
// --- release-x86-64: danos-x86-64.iso, the flashable release image ---
|
||||
// Wrap the FAT32 boot volume in a hybrid ISO (the in-repo Python builder
|
||||
// again, no xorriso/isohybrid): an ISO9660 whose El Torito EFI boot entry
|
||||
// and MBR ESP partition entry both point at the embedded FAT image. One
|
||||
// file then boots every way release media is consumed — flashed raw to a
|
||||
// USB stick with Etcher or dd, or burned to optical media — while
|
||||
// danos-usb.img stays the raw superfloppy QEMU and the test harness boot.
|
||||
const mk_iso = b.addSystemCommand(&.{"python3"});
|
||||
mk_iso.addFileArg(b.path("tools/make-iso-image.py"));
|
||||
const iso_image = mk_iso.addOutputFileArg("danos-x86-64.iso");
|
||||
mk_iso.addFileArg(fat_image);
|
||||
const iso_install = b.addInstallFile(iso_image, "danos-x86-64.iso");
|
||||
const release_step = b.step("release-x86-64", "Build the flashable x86-64 release ISO (zig-out/danos-x86-64.iso; flash with Etcher or dd)");
|
||||
release_step.dependOn(&iso_install.step);
|
||||
|
||||
// `zig build check-iso-image` — the ISO builder's own --verify (mirroring
|
||||
// check-fat-image): the MBR partition, the El Torito catalog, and the
|
||||
// embedded FAT32 image must all agree.
|
||||
const check_iso = b.addSystemCommand(&.{"python3"});
|
||||
check_iso.addFileArg(b.path("tools/make-iso-image.py"));
|
||||
check_iso.addArg("--verify");
|
||||
check_iso.addFileArg(iso_image);
|
||||
const check_iso_step = b.step("check-iso-image", "Verify the release ISO is a valid hybrid (MBR ESP partition + El Torito EFI entry)");
|
||||
check_iso_step.dependOn(&check_iso.step);
|
||||
|
||||
return fat_image_serial;
|
||||
}
|
||||
|
||||
/// Assemble the bootable FAT32 image (the in-repo Python builder) holding the
|
||||
/// EFI stub, the kernel, and every user binary at its FHS path — the volume's
|
||||
/// /system tree IS the system image; the EFI loader walks it at boot and builds
|
||||
/// the in-RAM initial_ramdisk from it. Factored so the serial-enabled
|
||||
/// `run-x86-64` variant can bundle its own serial kernel while sharing the
|
||||
/// loader and user tree (the loader's boot breadcrumbs and init's heartbeat both
|
||||
/// follow the top-level -Dserial). Returns the image's LazyPath.
|
||||
fn addBootImage(
|
||||
b: *std.Build,
|
||||
kernel_bin: std.Build.LazyPath,
|
||||
efi_bin: std.Build.LazyPath,
|
||||
manifest: std.Build.LazyPath,
|
||||
capsule: std.Build.LazyPath,
|
||||
bundled: []const BundledBinary,
|
||||
) std.Build.LazyPath {
|
||||
const mk_fat = b.addSystemCommand(&.{"python3"});
|
||||
mk_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||
const fat_image = mk_fat.addOutputFileArg("danos-usb.img");
|
||||
mk_fat.addArg("64"); // MiB
|
||||
mk_fat.addArg("EFI/BOOT/BOOTX64.efi");
|
||||
mk_fat.addFileArg(efi_bin);
|
||||
mk_fat.addArg("system/kernel");
|
||||
mk_fat.addFileArg(kernel_bin);
|
||||
mk_fat.addArg("system/manifest");
|
||||
mk_fat.addFileArg(manifest);
|
||||
mk_fat.addArg("boot/system.img");
|
||||
mk_fat.addFileArg(capsule);
|
||||
for (bundled) |item| {
|
||||
mk_fat.addArg(item.path);
|
||||
mk_fat.addFileArg(item.binary);
|
||||
}
|
||||
return fat_image;
|
||||
}
|
||||
-176
@@ -1,176 +0,0 @@
|
||||
//! The QEMU run steps (docs/build-packages-plan.md, phase 3): `run-x86-64`
|
||||
//! boots the serial-enabled FAT image via UEFI/OVMF; `run-x86-64-gpu` adds a
|
||||
//! virtio-gpu adapter for the native-present display path. OVMF firmware is
|
||||
//! probed across distro/OS layouts (-Dovmf-code / -Dovmf-vars override).
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// Wire up the `run-x86-64` and `run-x86-64-gpu` steps around the given
|
||||
/// serial-enabled boot image (the guest boots that self-contained image
|
||||
/// attached as USB storage, not the installed FHS zig-out).
|
||||
pub fn addRunSteps(b: *std.Build, fat_image_serial: std.Build.LazyPath) void {
|
||||
// Firmware lives in different places per OS/distro, so probe the known
|
||||
// layouts (Architecture, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
|
||||
// that exists. Override with -Dovmf-code / -Dovmf-vars if yours is elsewhere.
|
||||
const ovmf_code = b.option(
|
||||
[]const u8,
|
||||
"ovmf-code",
|
||||
"Path to the OVMF_CODE firmware image",
|
||||
) orelse firstExisting(b.graph.io, &.{
|
||||
"/usr/share/edk2/x64/OVMF_CODE.4m.fd", // Architecture
|
||||
"/usr/share/OVMF/OVMF_CODE_4M.fd", // Debian/Ubuntu
|
||||
"/usr/share/OVMF/OVMF_CODE.fd", // older Debian/Ubuntu
|
||||
"/usr/share/edk2-ovmf/x64/OVMF_CODE.fd", // Fedora
|
||||
"/opt/homebrew/share/qemu/edk2-x86_64-code.fd", // macOS Homebrew (Apple Silicon)
|
||||
"/usr/local/share/qemu/edk2-x86_64-code.fd", // macOS Homebrew (Intel)
|
||||
});
|
||||
const ovmf_vars = b.option(
|
||||
[]const u8,
|
||||
"ovmf-vars",
|
||||
"Path to the OVMF_VARS firmware image (a writable copy is made)",
|
||||
) orelse firstExisting(b.graph.io, &.{
|
||||
"/usr/share/edk2/x64/OVMF_VARS.4m.fd", // Architecture
|
||||
"/usr/share/OVMF/OVMF_VARS_4M.fd", // Debian/Ubuntu
|
||||
"/usr/share/OVMF/OVMF_VARS.fd", // older Debian/Ubuntu
|
||||
"/usr/share/edk2-ovmf/x64/OVMF_VARS.fd", // Fedora
|
||||
"/opt/homebrew/share/qemu/edk2-i386-vars.fd", // macOS Homebrew (Apple Silicon)
|
||||
"/usr/local/share/qemu/edk2-i386-vars.fd", // macOS Homebrew (Intel)
|
||||
});
|
||||
|
||||
// The firmware needs to write NVRAM, so give it a writable copy of the vars.
|
||||
const vars_copy = b.addSystemCommand(&.{ "cp", "-f", ovmf_vars });
|
||||
const vars_out = vars_copy.addOutputFileArg("OVMF_VARS.4m.fd");
|
||||
|
||||
// Capture the guest's serial0 (danos's machine-readable log) to the qemu-test
|
||||
// scratch area — a dev/host artifact, kept out of the boot volume we mount.
|
||||
// (/system/logs on the volume belongs to the guest's own logger.) One
|
||||
// timestamped file per run.
|
||||
const log_dir = b.fmt("{s}/qemu-test", .{b.install_path});
|
||||
const make_log_dir = b.addSystemCommand(&.{ "mkdir", "-p", log_dir });
|
||||
|
||||
// --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF ---
|
||||
const run_efi = b.addSystemCommand(&.{
|
||||
"qemu-system-x86_64",
|
||||
"-device",
|
||||
"qemu-xhci,id=xhci",
|
||||
"-device",
|
||||
"usb-mouse,bus=xhci.0",
|
||||
"-device",
|
||||
"usb-kbd,bus=xhci.0",
|
||||
"-machine",
|
||||
"q35",
|
||||
"-m",
|
||||
"128M",
|
||||
"-drive",
|
||||
b.fmt("if=pflash,format=raw,readonly=on,file={s}", .{ovmf_code}),
|
||||
});
|
||||
run_efi.addArg("-drive");
|
||||
run_efi.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
||||
// Boot off the FAT32 USB image: a mass-storage device on the same xHCI bus as
|
||||
// the keyboard and mouse. OVMF finds \EFI\BOOT\BOOTX64.efi on it and boots.
|
||||
// The serial-enabled variant, so serial0 carries the log for this dev boot.
|
||||
run_efi.addArg("-drive");
|
||||
run_efi.addPrefixedFileArg("if=none,id=bootusb,format=raw,file=", fat_image_serial);
|
||||
run_efi.addArgs(&.{
|
||||
"-device",
|
||||
"usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
||||
"-net",
|
||||
"none",
|
||||
// Emulated display advertising 1280x720 as its native (EDID preferred)
|
||||
// resolution, so the kernel's native-resolution switch has something to
|
||||
// find. `-vga none` avoids a second, default adapter.
|
||||
"-vga",
|
||||
"none",
|
||||
"-device",
|
||||
"VGA,edid=on,xres=1280,yres=720",
|
||||
});
|
||||
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
||||
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
|
||||
// We boot the self-contained `fat_image_serial` (added as a file arg above, so
|
||||
// it's already a dependency) — not the installed FHS zig-out — so `run-x86-64`
|
||||
// builds only the serial kernel, never the flashable one. Just make the serial
|
||||
// scratch dir first.
|
||||
run_efi.step.dependOn(&make_log_dir.step);
|
||||
|
||||
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/qemu-test/run-x86-64-serial0-<timestamp>.log");
|
||||
run_efi_step.dependOn(&run_efi.step);
|
||||
|
||||
// --- run-x86-64-gpu: the same boot plus a virtio-gpu adapter ---
|
||||
// The VGA device still supplies the boot (GOP) framebuffer the compositor starts
|
||||
// on; the virtio-gpu function is discovered by the device-manager stack, its
|
||||
// driver announces a shared scanout, and the compositor upgrades off the GOP
|
||||
// floor to fenced, tear-free native presents (docs/display-v2.md).
|
||||
// This is the interactive twin of the `display-native` test case, and 512M
|
||||
// matches it (the whole driver stack + the compositor's surfaces at once).
|
||||
// QEMU shows one head per adapter: pick the virtio-gpu head in the View menu
|
||||
// to watch the native output.
|
||||
const run_gpu = b.addSystemCommand(&.{
|
||||
"qemu-system-x86_64",
|
||||
"-device",
|
||||
"qemu-xhci,id=xhci",
|
||||
"-device",
|
||||
"usb-mouse,bus=xhci.0",
|
||||
"-device",
|
||||
"usb-kbd,bus=xhci.0",
|
||||
"-machine",
|
||||
"q35",
|
||||
"-m",
|
||||
"512M",
|
||||
"-drive",
|
||||
b.fmt("if=pflash,format=raw,readonly=on,file={s}", .{ovmf_code}),
|
||||
});
|
||||
run_gpu.addArg("-drive");
|
||||
run_gpu.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
||||
run_gpu.addArg("-drive");
|
||||
run_gpu.addPrefixedFileArg("if=none,id=bootusb,format=raw,file=", fat_image_serial);
|
||||
run_gpu.addArgs(&.{
|
||||
"-device",
|
||||
"usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
||||
"-net",
|
||||
"none",
|
||||
"-vga",
|
||||
"none",
|
||||
"-device",
|
||||
"VGA,edid=on,xres=1280,yres=720",
|
||||
"-device",
|
||||
"virtio-gpu-pci",
|
||||
});
|
||||
const gpu_serial_log = b.fmt("{s}/run-x86-64-gpu-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
||||
run_gpu.addArgs(&.{ "-serial", b.fmt("file:{s}", .{gpu_serial_log}) });
|
||||
run_gpu.step.dependOn(&make_log_dir.step);
|
||||
|
||||
const run_gpu_step = b.step("run-x86-64-gpu", "Boot in QEMU with a virtio-gpu adapter: the compositor upgrades to fenced (tear-free) native presents; watch the virtio-gpu head in QEMU's View menu");
|
||||
run_gpu_step.dependOn(&run_gpu.step);
|
||||
}
|
||||
|
||||
/// Return the first path in `candidates` that exists on the build host, else the
|
||||
/// first candidate as a fallback so a missing-firmware error still names a
|
||||
/// concrete (and, by convention, the primary) path. Used to locate OVMF firmware
|
||||
/// across distro/OS layouts without configuration.
|
||||
fn firstExisting(io: std.Io, candidates: []const []const u8) []const u8 {
|
||||
for (candidates) |path| {
|
||||
std.Io.Dir.accessAbsolute(io, path, .{}) catch continue;
|
||||
return path;
|
||||
}
|
||||
return candidates[0];
|
||||
}
|
||||
|
||||
/// A UTC timestamp like "20260708-153045", for naming a per-run artifact so
|
||||
/// repeated runs don't clobber each other's logs. Resolved when `zig build`
|
||||
/// runs, which is moments before QEMU launches.
|
||||
fn timestamp(b: *std.Build) []const u8 {
|
||||
const ns = std.Io.Clock.now(.real, b.graph.io).nanoseconds;
|
||||
const secs: u64 = @intCast(@divFloor(ns, std.time.ns_per_s));
|
||||
const es = std.time.epoch.EpochSeconds{ .secs = secs };
|
||||
const yd = es.getEpochDay().calculateYearDay();
|
||||
const md = yd.calculateMonthDay();
|
||||
const ds = es.getDaySeconds();
|
||||
return b.fmt("{d:0>4}{d:0>2}{d:0>2}-{d:0>2}{d:0>2}{d:0>2}", .{
|
||||
yd.year,
|
||||
md.month.numeric(),
|
||||
@as(u32, md.day_index) + 1,
|
||||
ds.getHoursIntoDay(),
|
||||
ds.getMinutesIntoHour(),
|
||||
ds.getSecondsIntoMinute(),
|
||||
});
|
||||
}
|
||||
+6
-17
@@ -55,7 +55,9 @@ rather than restate it. Roughly in the order things happen at runtime:
|
||||
15. **[drivers.md](device-driver-development/drivers.md) — writing a driver.** The payoff: a driver is an
|
||||
ordinary ring-3 process that claims a device, maps its registers, and **sleeps
|
||||
until its hardware interrupts it**. The claim is the capability; `irq_ack` is the
|
||||
unmask.
|
||||
unmask. The condensed version: the
|
||||
[new-driver checklist](device-driver-development/new-driver-checklist.md) —
|
||||
the minimum steps from boot-log line to mapped registers.
|
||||
16. **[driver-model.md](device-driver-development/driver-model.md) — buses, classes and host controllers.** How
|
||||
real driver stacks factor into three shapes and how families share code. The
|
||||
three primitives it proposed are long since built (M13 capability passing,
|
||||
@@ -208,7 +210,7 @@ the whole reason for the arrangement ([vision.md](vision.md)).
|
||||
danos is a **monorepo of sub-projects**. Each service or driver is a directory that is
|
||||
its own Zig module — it can hold as many files as it needs, and other sub-projects
|
||||
reach it *by module name*, never by a path into its files. The source tree deliberately
|
||||
**mirrors the runtime file-system hierarchy** ([file-system-hierarchy.md](file-system-development/file-system-hierarchy.md)):
|
||||
**mirrors the runtime FHS** ([danos-file-system-hierarchy-FSH.md](file-system-development/danos-file-system-hierarchy-FSH.md)):
|
||||
what you see under `system/` in the source is what a running danos represents under
|
||||
`/system`.
|
||||
|
||||
@@ -216,7 +218,7 @@ what you see under `system/` in the source is what a running danos represents un
|
||||
name.** `system/services/init/` contains `init.zig` (its root), and produces a binary
|
||||
addressed as **`system/services/init`** — the repeated leaf resolves away:
|
||||
|
||||
| Source (root file) | Addressed as (module / binary / hierarchy path) |
|
||||
| Source (root file) | Addressed as (module / binary / FHS path) |
|
||||
|----------------------------------------|--------------------------------------------|
|
||||
| `system/services/init/init.zig` | `system/services/init` → `/system/services/init` |
|
||||
| `system/drivers/ps2-bus/ps2-bus.zig` | `system/drivers/ps2-bus` → `/system/drivers/ps2-bus` |
|
||||
@@ -263,20 +265,9 @@ test/ → /test the test tree: the QEMU harness (qemu_test.py, h
|
||||
system/services/ beside the on-image test fixtures — vfs-test/ thread-test/
|
||||
crash-test/ … — whose repo path IS their boot-volume path
|
||||
(/test/system/services/<name>)
|
||||
build-support/ the danos build API (build-time only, nothing on the image):
|
||||
the shared user-binary recipe + default-import wiring every
|
||||
build file consumes (docs/build-packages-plan.md)
|
||||
build/ root-build helpers: image assembly (images.zig) + the QEMU
|
||||
run steps (qemu.zig)
|
||||
tools/ host-side build scripts
|
||||
```
|
||||
|
||||
**Builds are packages** (docs/build-packages-plan.md): each `library/` domain owns a
|
||||
`build.zig`/`build.zig.zon` exporting its modules (with a standalone `zig build test`),
|
||||
every binary directory is a ~15-line package build, and the root `build.zig`
|
||||
orchestrates — the kernel + loader, what ships, and the aggregate test step — with
|
||||
image assembly in `build/images.zig` and the QEMU run steps in `build/qemu.zig`.
|
||||
|
||||
**Wire protocols live in `library/protocol/`**, one module per directory
|
||||
(`library/protocol/vfs/vfs-protocol.zig` is the `vfs-protocol` module), imported by module
|
||||
name. A protocol is the seam between a low-level driver and the higher-level service it
|
||||
@@ -331,7 +322,5 @@ exception in [coding-standards.md](coding-standards.md) applies to that seam.
|
||||
| System services (init, the `fat` filesystem, the device-manager) | `system/services/` |
|
||||
| Device drivers, one sub-project each (`pci-bus`, `ps2-bus`, `usb-xhci-bus` bus drivers) | `system/drivers/` |
|
||||
| On-image test fixtures for the QEMU cases (`vfs-test`, `crash-test`, `thread-test`, …) → `/test/system/services` | `test/system/services/` |
|
||||
| Build orchestration (kernel + loader, what ships, the aggregate test step) | `build.zig` (root; the shared user-binary recipe is `build-support/`, and each `library/` domain + binary package carries its own `build.zig`) |
|
||||
| Image assembly + `release-x86-64` (the flashable ISO) | `build/images.zig` |
|
||||
| `run-x86-64` / `run-x86-64-gpu` (QEMU/OVMF) | `build/qemu.zig` |
|
||||
| Build + `run-x86-64` (QEMU/OVMF) + `release-x86-64` (the flashable ISO) | `build.zig` |
|
||||
| QEMU integration test harness | `test/qemu_test.py` |
|
||||
|
||||
@@ -1,179 +0,0 @@
|
||||
# Plan: packages — hierarchical builds for libraries and binaries
|
||||
|
||||
**Status: complete** (branch `claude/build-packages-plan-174144`). Phase 0
|
||||
(`build-support`), phase 1 (all six library domains), phase 2 (every binary —
|
||||
the pci-bus pilot first, then services, drivers, and test fixtures in waves;
|
||||
multi-binary directories like ps2-bus and usb-hid are one package exporting
|
||||
several artifacts, and the acpi/fdt discovery pair each export an artifact
|
||||
named "discovery" that the root's -Ddiscovery picks between), and phase 3 (the
|
||||
root split into `build/images.zig` + `build/qemu.zig`; the root `build.zig` is
|
||||
~460 lines of orchestration, down from ~1,250). Every phase landed green: unit
|
||||
tests, the QEMU suite at parity with main, boot-image file list unchanged.
|
||||
The `lazyDependency` payoff (What-this-buys #4) is in too: the /test fixtures
|
||||
and the unselected discovery package are lazy — a build loads and compiles
|
||||
only what it ships. And imports are exact: the pre-wired default set is gone;
|
||||
every binary names precisely the modules its source imports and carries only
|
||||
those domains in its manifest (rule 1 below).
|
||||
|
||||
## Why
|
||||
|
||||
`build.zig` was ~1,250 lines, growing by three hand-written stanzas per binary;
|
||||
at a driver per device family that does not scale. More fundamentally: in one
|
||||
monolithic build every binary compiles against library *source*, so a library
|
||||
interface break is silently absorbed by whoever edits everything in one commit —
|
||||
the interface never has to be honest. danos is about isolation; the build should
|
||||
mirror it.
|
||||
|
||||
A **package** here is a build-time unit only — a directory owning a `build.zig`
|
||||
(recipe: what it exports, how to test it) and a `build.zig.zon` (manifest: name
|
||||
+ dependencies). Binaries remain fully static freestanding ELFs; packages change
|
||||
who declares what, not what links to what. Source code is untouched: `@import`
|
||||
uses module names (`"pci"`, `"service"`) exactly as today — only build files
|
||||
know where anything lives.
|
||||
|
||||
## Target shape
|
||||
|
||||
```
|
||||
build-support/ package: the danos build API (userBinary(), defaultImports(), targets)
|
||||
library/kernel/ package "kernel": modules abi, ipc, service, memory, process, logging, time, ... (depends on protocol)
|
||||
library/device/ package "device": modules driver, pci, usb-abi, model, ... (depends on kernel, protocol, csv)
|
||||
library/protocol/ package "protocol": the wire protocols
|
||||
library/client/ package "client" (depends on kernel, protocol)
|
||||
library/csv/ package "csv"
|
||||
library/xkeyboard-config/ package "xkeyboard-config"
|
||||
system/services/<name>/ one package per binary: ~15-line build.zig + zon
|
||||
system/drivers/<name>/ one package per binary
|
||||
build.zig (root) orchestrator: dependency() per binary, image assembly, QEMU, test steps
|
||||
```
|
||||
|
||||
The three shared contracts: `boot-handoff` stays a root module (only the
|
||||
loader↔kernel pair speaks it); `abi` is exported by the kernel package from
|
||||
`../../system/abi.zig` (the source stays with the kernel; userspace's one view
|
||||
of it lives in the package, so every consumer names the same module instance);
|
||||
`device-abi` is exported by device. Reaching outside the package root means the
|
||||
kernel package is valid only as an in-repo path dependency — it could never be
|
||||
fetched by hash — which is fine: path dependencies are the only way any of
|
||||
these packages is consumed.
|
||||
|
||||
Rules:
|
||||
|
||||
- **Imports are exact and per binary.** A binary's build.zig names precisely
|
||||
the modules its source `@import`s — the moral equivalent of a C file's
|
||||
include list — and its zon names only the domains those modules come from
|
||||
(plus `build-support` and `kernel`, which is implicit in every binary: the
|
||||
root shim and user link script live there). Nothing is pre-wired: an
|
||||
undeclared `@import` is a compile error, and build-support resolves each
|
||||
name by searching the packages the zon declares — the domains' own
|
||||
addModule exports are the single statement of who owns what, with no name
|
||||
table anywhere to drift. Availability
|
||||
never meant bloat — Zig only compiles what a program actually imports — but
|
||||
exactness makes the declared interface honest and machine-checked.
|
||||
- **Modules export source, not artifacts** — each consumer compiles libraries
|
||||
with its own flags, so per-binary optimization choices keep working; Zig's
|
||||
cache deduplicates.
|
||||
- **Zon paths are relative and that is accepted.** Binaries sit exactly three
|
||||
levels deep, so the `../../../` prefix is a constant idiom; a library-domain
|
||||
move is a rare, already-breaking event fixed by one sed across manifests, and
|
||||
a stale path fails loudly before anything compiles.
|
||||
- **Cross-cutting build changes live in `build-support` only** — that is the
|
||||
contract that keeps per-binary build files declarative.
|
||||
|
||||
## What this buys
|
||||
|
||||
1. Library interfaces become machine-checked: a consumer can only import what
|
||||
it declared — per binary, down to the single module — and each domain's zon
|
||||
declares what it needs (claim-before-touch, applied to source). A keyboard
|
||||
driver carries `xkeyboard-config` in its manifest; nothing else does.
|
||||
2. Each library domain gets a standalone `zig build test` — runtime-library
|
||||
stability testing in isolation.
|
||||
3. Adding a binary = adding a directory (source + two small files), not editing
|
||||
three places in a 1,250-line file.
|
||||
4. `lazyDependency` lets an image target build only what it ships: the /test
|
||||
fixtures resolve only under -Dtest-case, and only the -Ddiscovery-selected
|
||||
discovery package ever loads.
|
||||
|
||||
## Phases
|
||||
|
||||
Each phase ends green: `zig build test` passes (88/88 QEMU) and the boot
|
||||
image's file list is unchanged. Byte-identical binaries are expected but not
|
||||
required (module reorganization can perturb symbol order); file list is the
|
||||
hard gate.
|
||||
|
||||
**Phase 0 — `build-support`.** Extract `addUserBinary`/`addThreadedUserBinary`,
|
||||
the freestanding target setup, and the default-import wiring into the
|
||||
`build-support` package. Root build consumes it; nothing else moves. This is
|
||||
the cross-cutting-change home, so it lands first.
|
||||
|
||||
**Phase 1 — library domains become packages.** In dependency order: `protocol`
|
||||
and `csv` (the roots) → `kernel` (depends on protocol: file-system speaks
|
||||
vfs-protocol) → `device`, `client`; `xkeyboard-config` stands alone. Each gets
|
||||
build.zig + zon + a standalone test step (client's is empty until its modules
|
||||
grow host tests — kept for uniformity, since the root aggregate depends on
|
||||
every domain's test step). The root build swaps its `createModule` calls for
|
||||
`b.dependency("<domain>").module("<name>")`. **No binary moves in this phase**
|
||||
— the root build is the pilot consumer, which proves the packages without
|
||||
touching 30 binaries.
|
||||
|
||||
**Phase 2 — binaries become packages, in waves.** The template was shaken out
|
||||
by the pci-bus pilot (see Status). Wave A: services (done). Wave B: the
|
||||
remaining drivers (done). Wave C: test fixtures (done). Root build shrank to
|
||||
orchestration per wave. init's `-Dserial` heartbeat flag rides a dependency
|
||||
option; a directory with several binaries (ps2-bus, usb-hid) is one package
|
||||
exporting several artifacts.
|
||||
|
||||
**Phase 3 — root cleanup (done).** What remained of the root build split into
|
||||
`build/images.zig` (the FHS install tree, boot manifest + capsule, FAT32
|
||||
images, release ISO, check steps) and `build/qemu.zig` (the run steps + OVMF
|
||||
probing), imported by a short root `build.zig`.
|
||||
|
||||
**Afterwards** (outside this plan): the intel-uhd-graphics-750 driver is
|
||||
(re)created as a greenfield package. The new-driver checklist's build step
|
||||
(docs/device-driver-development/new-driver-checklist.md, step 2) is already
|
||||
rewritten against the package template.
|
||||
|
||||
## Execution notes (the finished shape)
|
||||
|
||||
- The shared recipe lives in `build-support/build.zig`: `userBinary` (what
|
||||
every binary package calls; each named import resolves by searching the
|
||||
packages the binary's zon declares) and `programModule` (for per-binary
|
||||
addOptions modules). The `start` root shim and `user.ld` are named through the kernel
|
||||
package (Dependency.path).
|
||||
- Adding a binary = adding a directory with source + a ~15-line build.zig +
|
||||
zon (copy any existing binary package, e.g.
|
||||
`system/drivers/pci-bus/build.zig`) listing exactly the modules the source
|
||||
imports and the domains they come from, then one dependency + one bundled
|
||||
entry in the root build.zig and one zon line.
|
||||
- The boot-tree array in the root (search `"etc/init.csv"` or
|
||||
`.getEmittedBin()`) is the image file list — the authoritative comparison
|
||||
target for any future build change.
|
||||
- Package unit tests live in each package's own `test` step; the root
|
||||
aggregate depends on every test-bearing package's step, so `zig build test`
|
||||
at the root still runs everything.
|
||||
|
||||
Verification per phase:
|
||||
|
||||
- Unit tests: `zig build test`.
|
||||
- QEMU integration suite: `python3 test/qemu_test.py` (docs/testing.md; the
|
||||
full suite, all cases must pass).
|
||||
- Image file list: the boot-tree array is the source of truth — snapshot it
|
||||
(paths only) before phase 0 and diff after each phase; `zig build
|
||||
check-fat-image` must also stay green.
|
||||
|
||||
Context a fresh session should read first: this doc, docs/testing.md,
|
||||
docs/coding-standards.md (kebab-case names, no abbreviations), and the
|
||||
`userBinary`/`userBinaryFromImports` bodies in build-support/build.zig. Commit
|
||||
style: no Co-Authored-By trailers.
|
||||
|
||||
## Risks / notes
|
||||
|
||||
- Zig version churn: the package API (`b.dependency`, zon schema) has moved
|
||||
between releases; the work pins against the repo's current Zig and any
|
||||
upgrade lands separately, never mid-phase.
|
||||
- The QEMU size-check tests hardcode source paths (e.g. virtio-gpu protocol
|
||||
struct sizes) — they moved into their binaries' packages with their waves,
|
||||
discharging the carry-along obligation.
|
||||
- Doc updates ride each phase: docs/README.md (repo layout + source map),
|
||||
docs/device-driver-development/new-driver-checklist.md (step 2) and
|
||||
devices-csv.md ("Adding a driver"), and the docs that cite the build recipe
|
||||
(driver-model.md, threading.md, system-requirements.md) reference build
|
||||
shapes that keep changing.
|
||||
@@ -1,205 +0,0 @@
|
||||
# The C library compatibility layer
|
||||
|
||||
A design note and milestone plan for **libdanos-c** — the mini C library that lets
|
||||
`zig cc` cross-compile C programs for danos. It is milestone **P0** of
|
||||
[python-on-danos-milestones.md](python-on-danos-milestones.md), expanded here the
|
||||
way [character-devices-and-tty.md](character-devices-and-tty.md) expands P1.
|
||||
CPython is the driving consumer, but the layer is general: any portable C program
|
||||
within its surface should build.
|
||||
|
||||
## What it is — and the three things it is not
|
||||
|
||||
The deliverable is a **sysroot**: a set of C headers plus a static `libdanos-c.a`,
|
||||
handed to `zig cc -target x86_64-freestanding-none` via `-isystem` and linked into
|
||||
every C binary. Three explicit non-goals keep it small:
|
||||
|
||||
- **Not a musl port.** Whole-musl assumes Linux syscall semantics at its bottom
|
||||
(the door the Zig roadmap deferred, twice now). We *lift* musl's pure-computation
|
||||
source files and *write* a danos-native bottom — see the layer split below.
|
||||
- **Not full POSIX — *yet*.** Stage 1's surface is "what CPython's minimal
|
||||
configuration and ordinary portable C need" — roughly 100–150 functions — and
|
||||
at that stage absence is a *feature*: configure scripts probe and adapt, and a
|
||||
linker error is honest. But the end state is a **full C compatibility layer**
|
||||
(see "The road to full coverage" below); the absence table is a schedule of
|
||||
arrivals, not a wall.
|
||||
- **Not a second runtime.** The library is a thin C-ABI re-spelling of the same
|
||||
danos-native surface `runtime` already provides. It contains no policy of its
|
||||
own; when the Zig track's `runtime.os` seam is authored, the libc bottom
|
||||
re-targets it near-mechanically — the fourth appearance of the roadmap's "same
|
||||
surface" symmetry.
|
||||
|
||||
One scoping rule sits above all three — the **size doctrine**: this layer serves
|
||||
**applications only**. The kernel and the system services never link libdanos-c;
|
||||
they stay danos-native Zig over `runtime`, small and static, because leanness is
|
||||
an operating-system property. Applications have their own budget and may be as
|
||||
big as they need to be. The libc is how big software *lands on* danos, never how
|
||||
danos itself is built.
|
||||
|
||||
## The layer split: lift the mathematics, write the plumbing
|
||||
|
||||
The realization that makes 100–150 functions tractable: a libc is two very
|
||||
different kinds of code, and the hard kind is portable.
|
||||
|
||||
| Layer | Contents | Source |
|
||||
|-------|----------|--------|
|
||||
| **Pure computation** | `string.h`/`memcpy` family, all of libm, `strtod`/`dtoa`, `strtol`, `qsort`, `ctype` tables, `gmtime` calendar math, the `printf`/`scanf` engines, `setjmp` (a dozen instructions of x86-64 asm) | **Lift from musl**, vendored under `library/c/third-party/musl/` (MIT; files compile standalone) |
|
||||
| **OS plumbing** | fds (`open`/`read`/`write`/`close`/`lseek`/`stat`/`getcwd`/`chdir`/`isatty`), `mmap`/`munmap`, clocks, `exit`, `getenv`, `getentropy` | **Write in Zig**, exporting C ABI over the `runtime` syscall + VFS client surface |
|
||||
| **The middle** | `malloc` over danos `mmap` (simple free-list; CPython's arenas sit above), `FILE*` buffering, `errno` | **Write in Zig** (small, danos-shaped) |
|
||||
| **Entry** | `crt0`: the existing danos entry shim ([sysv.md](os-development/sysv.md)) bridged to C `main(argc, argv, envp)`, `environ` initialised, `exit` flushing stdio | **Write** |
|
||||
|
||||
Two liftings deserve their own line because getting them wrong is silent
|
||||
corruption rather than a linker error:
|
||||
|
||||
- **`strtod`/float formatting.** Python's float `repr` guarantees shortest
|
||||
round-trip; that property lives entirely in these routines. musl's are correct;
|
||||
an improvised one would be subtly wrong for years. Lift, never write.
|
||||
- **The stdio engines.** musl's `vfprintf`/`vfscanf` are self-contained around
|
||||
its `FILE` abstraction (function-pointer read/write slots), so the whole
|
||||
formatted-I/O engine lifts too — we implement only the fd-backed slots
|
||||
(`__stdio_write`-shaped) and the buffering glue.
|
||||
|
||||
## Header policy
|
||||
|
||||
Hand-write the headers as danos's own minimal set rather than importing musl's
|
||||
(musl's are entangled with Linux ABI details), borrowing declarations freely.
|
||||
Freestanding compiler headers (`stdint.h`, `stddef.h`, `stdarg.h`, `stdbool.h`,
|
||||
`float.h`, `limits.h`) come from clang via `zig cc` — do not duplicate them.
|
||||
`errno.h` values are the danos errno enum re-spelled with POSIX names; there is no
|
||||
Linux numbering to be compatible with, so the enum is the truth.
|
||||
|
||||
Deliberate absences, and their planned arrivals — this table is the
|
||||
compatibility matrix, and "the road to full coverage" below is the schedule
|
||||
that empties it:
|
||||
|
||||
| Absent | Arrives with |
|
||||
|--------|--------------|
|
||||
| `pthread.h` | the post-P5 pthread subset over `thread_spawn`/futex — but see the risk below |
|
||||
| real `signal.h` (beyond no-op `signal()`/`raise` stubs) | M17 signals-over-IPC in the libc |
|
||||
| `dlfcn.h` | [dynamic-libraries.md](dynamic-libraries.md) D1 |
|
||||
| `fork`/`exec*`/`wait*` | P5 exposes danos spawn as `posix_spawn`; `fork` itself never (see below) |
|
||||
| `socket.h` | a future networking track |
|
||||
| locale beyond `"C"` | stage 3 evaluation (CPython is UTF-8-mode happy without it) |
|
||||
| pipes (`pipe()`) | P5 process-control cluster |
|
||||
|
||||
## The road to full coverage
|
||||
|
||||
The layer grows in three stages; only stage 1 is a current milestone (P0), but
|
||||
the stages exist so stage-1 decisions never have to be unmade:
|
||||
|
||||
- **Stage 1 — CPython-minimal** (P0, the slicing below): ~100–150 functions,
|
||||
static-only, absences honest.
|
||||
- **Stage 2 — the danos-complete layer**: the full hosted C11 standard library,
|
||||
plus every POSIX facility danos semantics support, landing as its enabling
|
||||
milestone lands — pipes and `posix_spawn` at P5, real signals at M17, the
|
||||
pthread subset after P5, `dlfcn.h` at
|
||||
[dynamic-libraries](dynamic-libraries.md) D1, sockets with networking. Stage 2
|
||||
is not one milestone but the standing rule that **every system capability
|
||||
gets its C spelling when it ships**, so the matrix above drains as the OS
|
||||
grows.
|
||||
- **Stage 3 — ecosystem grade**: the point where "portable C program" generally
|
||||
means "builds on danos" (autotools-style probing included). Reaching it is
|
||||
mostly stage 2 compounding, plus the long tail (locale, wide-char,
|
||||
`fnmatch`/`glob`/`regex` — the last three lift from musl like the rest). At
|
||||
this stage, re-evaluate hand-grown-vs-musl-port once with real data; the
|
||||
standing recommendation remains danos-native — musl's bottom assumes Linux
|
||||
syscall semantics, and by stage 3 the danos bottom exists and is tested —
|
||||
with musl continuing as the quarry for computation code.
|
||||
|
||||
Two boundaries are permanent and worth stating at every stage: **`fork` never
|
||||
comes** — danos is a spawn-shaped OS, and `fork`'s address-space-duplication
|
||||
semantics are hostile to everything from capabilities to threads; software that
|
||||
hard-requires `fork` (not `posix_spawn`) stays off the platform. And the
|
||||
**public ABI stays the vDSO + IPC protocols** — a full libc is a compatibility
|
||||
*layer*, not a second stable system ABI.
|
||||
|
||||
## Milestone slicing
|
||||
|
||||
1. **sysroot-skeleton** — layout under `library/c/` (a build package:
|
||||
`include/`, Zig sources, vendored musl subtree); `crt0`; string/mem +
|
||||
`ctype` lifted; a `build.zig` step making C binaries first-class targets.
|
||||
*Test:* a C program using only computation links and runs in QEMU
|
||||
(`c-hello` printing via a raw `write` extern to `debug_write`).
|
||||
2. **fd-plumbing** — `errno`; open/read/write/close/lseek/stat/unlink/mkdir/
|
||||
rename over the `runtime` VFS client; `getcwd`/`chdir`/`getenv`/
|
||||
`getentropy` arriving as P1 lands them (stubbed truthfully until then:
|
||||
`getenv` empty, `getentropy` `ENOSYS`). *Test:* QEMU `c-file-io` — create,
|
||||
write, reopen, read back, stat size + mtime through FAT.
|
||||
3. **malloc** — free-list allocator over danos `mmap`; `calloc`/`realloc`/
|
||||
`free`; alignment guarantees documented. *Test:* host + QEMU allocator
|
||||
torture (interleaved sizes, realloc growth, alignment asserts).
|
||||
4. **stdio** — `FILE*`, buffering modes, the lifted printf/scanf engines wired
|
||||
to the fd slots; `snprintf` family; stdin/stdout/stderr over fd 0/1/2.
|
||||
*Test:* host round-trip suite for format engines (especially `%.17g`
|
||||
float round-trip); QEMU `c-stdio` cooked-line echo once P1's console exists.
|
||||
5. **mathematics-and-time** — libm lifted wholesale; `strtod`/`strtol`;
|
||||
`clock_gettime` (monotonic + realtime over `clock`/`wall_clock`);
|
||||
`gmtime`/`mktime`/`strftime` (UTC only — no timezone database);
|
||||
`setjmp`/`longjmp`; `qsort`/`bsearch`; `abort`/`assert`. *Test:* host
|
||||
`strtod`/`dtoa` vectors against known-hard cases; QEMU `c-time` sanity
|
||||
against the wall clock.
|
||||
|
||||
Slices 1, 3, 4-host, and 5-host have **no dependency on P1** and can start
|
||||
immediately; slice 2 and the QEMU halves interleave with P1 as it lands.
|
||||
|
||||
**Exit for the layer as a whole** (= P0's exit): `c-hello` and `c-file-io` green
|
||||
in the QEMU suite, and the host-side computation tests green — at which point P2
|
||||
(CPython configure) becomes the layer's real integration test.
|
||||
|
||||
## Testing strategy: two targets, on purpose
|
||||
|
||||
The computation layer is target-independent, so it is unit-tested **on the host**
|
||||
(built for the host triple, compared against the host libc's answers —
|
||||
thousands of cheap oracle checks for `strtod`, `printf`, libm edge cases). The
|
||||
plumbing layer only means anything **on danos**, so it is tested in the QEMU
|
||||
suite like every other subsystem. Keeping the split explicit stops the slow-QEMU
|
||||
suite from absorbing tests that a host `zig test` runs in milliseconds.
|
||||
|
||||
## Risks and gotchas
|
||||
|
||||
- **CPython's configure may insist on pthreads.** WASI-class targets build
|
||||
threadless, but verify this *first* in P2 bring-up; the fallback is a
|
||||
truthfully-single-threaded `pthread.h` stub set (create returns `EAGAIN`,
|
||||
mutexes are no-ops — valid when only one thread can exist). Decide from
|
||||
evidence, not assumption.
|
||||
- **`long double` is x87 80-bit on x86-64.** musl's libm handles it, but keep
|
||||
CPython away from it (`configure` uses `double` throughout by default);
|
||||
don't hand-write anything touching x87.
|
||||
- **errno is a contract, not a convention.** The Zig plumbing must map every
|
||||
`runtime` error to a POSIX name consistently — CPython turns errno into
|
||||
exception types (`FileNotFoundError` is `ENOENT`). One table, tested.
|
||||
- **`malloc` alignment**: 16-byte minimum on x86-64 (SSE spills in
|
||||
compiled C). The free-list must guarantee it from day one; retrofitting
|
||||
alignment bugs out of an allocator is misery.
|
||||
- **Vendoring discipline.** The musl subtree is lift-only — never edited in
|
||||
place (patches live beside it if ever needed), pinned to one musl release,
|
||||
with the file list documented so a version bump is a re-copy, not an
|
||||
archaeology dig.
|
||||
- **stdio buffering vs. crashes.** Buffered stdout + a crashing program eats
|
||||
output — the classic debugging trap. `stderr` stays unbuffered (per C
|
||||
standard) and `exit`/`abort` flush; document that `_exit` does not.
|
||||
|
||||
## Decisions needing sign-off
|
||||
|
||||
- **Lift-from-musl for all pure computation** (vendored, pinned, unedited) rather
|
||||
than writing or porting whole-musl.
|
||||
- **Hand-written danos-native headers**; danos errno values are the numbering.
|
||||
- **`library/c/` as a build package** producing both the sysroot and the
|
||||
first-class C-binary build step.
|
||||
- The **deliberate-absence table** as the living compatibility matrix, drained
|
||||
by the three-stage road above — with exactly one permanent "never": `fork`.
|
||||
- **Full coverage as the end state** (stage 3), reached by the standing rule
|
||||
that every system capability ships with its C spelling — not by a musl port.
|
||||
|
||||
## Related
|
||||
|
||||
- [python-on-danos-milestones.md](python-on-danos-milestones.md) — this is P0.
|
||||
- [dynamic-libraries.md](dynamic-libraries.md) — ships in this sysroot
|
||||
(`dlfcn.h` + the loader) once its D1 lands.
|
||||
- [python-on-danos.md](python-on-danos.md) — the design note that scoped the
|
||||
layer.
|
||||
- [character-devices-and-tty.md](character-devices-and-tty.md) — P1; supplies
|
||||
the console that makes stdio interactive.
|
||||
- [zig-self-hosting.md](zig-self-hosting.md) — the `runtime.os` seam the
|
||||
plumbing layer will re-target when it exists.
|
||||
- [os-development/sysv.md](os-development/sysv.md) — the entry stack `crt0`
|
||||
bridges.
|
||||
@@ -1,174 +0,0 @@
|
||||
# Character devices, the console, and the tty question
|
||||
|
||||
A design note for the **stream** half of the device world. danos has block devices
|
||||
(the USB storage service behind the FAT mount) but no character devices — and three
|
||||
tracks now need them at once: the terminal application, Zig self-hosting Phase 1
|
||||
("wire fd 0/1/2 to a console byte stream"), and [Python on danos](python-on-danos.md)
|
||||
Phase 1. This note settles what a character device *is* on danos before any of those
|
||||
tracks build one.
|
||||
|
||||
## The Unix picture, briefly
|
||||
|
||||
Unix splits devices in two: **block devices** are seekable arrays of fixed-size
|
||||
sectors (disks); **character devices** are unseekable byte streams (keyboards,
|
||||
serial ports, terminals, `/dev/null`, entropy). A **tty** is the canonical
|
||||
character device — a byte stream plus a *line discipline* (echo, line buffering,
|
||||
erase handling, Ctrl-C-to-signal) that lives in the kernel. A **pty** is a pair of
|
||||
character devices (master/slave) that exists so a *userspace* program — a terminal
|
||||
emulator — can impersonate terminal hardware to the kernel's in-kernel line
|
||||
discipline.
|
||||
|
||||
The identification asked for and confirmed: yes, tty and pty are character
|
||||
devices in this taxonomy.
|
||||
|
||||
## The realization that shapes everything: danos already has the mechanism
|
||||
|
||||
A Unix character device is an in-kernel dispatch table: major/minor numbers route
|
||||
`read()`/`write()` to a driver. danos already has exactly that dispatch — the VFS:
|
||||
`fs_resolve` routes a path to a mounted backend service, and `Operation.mount`
|
||||
attaches a backend *endpoint* at a prefix. What is missing is not a device model;
|
||||
it is **one node kind with stream semantics**. And the protocol already reserved
|
||||
it: `NodeKind.character_device = 2` sits unimplemented in
|
||||
[vfs-protocol.zig](../library/protocol/vfs/vfs-protocol.zig), exactly like
|
||||
`symbolic_link`.
|
||||
|
||||
So the design is small:
|
||||
|
||||
**A character device on danos is a VFS node, served by an ordinary service over
|
||||
the existing VFS wire protocol, whose read/write have stream semantics.**
|
||||
|
||||
No device numbers, no `/dev` special casing, no new syscalls, no new protocol —
|
||||
a service is reachable at a path, clients open it with `runtime.fs` like any
|
||||
file, and the node kind says what it is. (Since the protocol namespace landed
|
||||
in design, that path is `/protocol/console` — a protocol node, see
|
||||
[os-development/protocol-namespace.md](os-development/protocol-namespace.md) —
|
||||
rather than a mounted device file; the stream semantics below are unchanged.)
|
||||
|
||||
### Stream semantics (the actual contract change)
|
||||
|
||||
For a node whose kind is `character_device`:
|
||||
|
||||
- **`offset` is ignored** on read and write; there is no seek position. (`lseek`,
|
||||
when the C layer exists, returns `ESPIPE`.)
|
||||
- **Reads block** until at least one byte is available, then return what is there —
|
||||
**short reads are normal**, not EOF. A zero-length read reply means the stream
|
||||
is closed (hangup), not end-of-file-at-size.
|
||||
- **`FileStatus.size` is 0** and means nothing; `mtime` may be 0.
|
||||
- Writes may be short if the service's buffer is full; the client loops as it
|
||||
already must for the 256-byte message cap.
|
||||
|
||||
This is a semantics note on existing operations, not a wire change — the `Request`
|
||||
and `Reply` structs are untouched. The one true protocol addition is a **`control`
|
||||
operation** (appended to `Operation`, values stable): a typed request the stream's
|
||||
service interprets. Deliberately *not* an `ioctl` grab-bag — the control payloads
|
||||
are enumerated per protocol, starting with the terminal set below.
|
||||
|
||||
## The first character device is a pseudo-device
|
||||
|
||||
The first device is deliberately **not hardware**: an in-memory **loopback** — a
|
||||
byte queue served over the stream contract, where bytes written to one end are
|
||||
read from the other. It is the reference implementation of the semantics above
|
||||
(blocking reads, short reads, hangup on close, the `control` round-trip), it
|
||||
tests deterministically with no QEMU serial scripting, and it keeps hardware off
|
||||
the critical path entirely. `null` and `zero` come along nearly for free as
|
||||
degenerate cases. This is a decision, not a convenience: the dead-COM1 boot bug
|
||||
on real hardware already proved serial cannot be assumed present or alive, so
|
||||
**nothing in this milestone writes to COM1**. (A serial-backed stream node can
|
||||
exist *later* as one more optional backend for headless debugging; it is on
|
||||
nobody's critical path.)
|
||||
|
||||
The loopback is also not throwaway — it is the seed of P5's `pipe()`, which is
|
||||
the same object with two fds.
|
||||
|
||||
## The console service
|
||||
|
||||
A `console` service owns the line discipline — **in userspace**, where a
|
||||
microkernel wants it, not in the kernel as Unix has it:
|
||||
|
||||
- **The discipline is a pure library first**: bytes and key events in, bytes
|
||||
out, no I/O of its own — developed and host-tested against in-memory buffers,
|
||||
then shared verbatim between the console and the future terminal application.
|
||||
- **Input**: subscribes to keyboard `InputEvent` IPC (the structured events that
|
||||
exist today) and cooks them into bytes. Cooked mode is the default: echo, line
|
||||
buffering, backspace/erase, so a line is delivered on Enter. Raw mode delivers
|
||||
bytes as they come (the REPL's line editor and any full-screen program need it).
|
||||
- **Output is a pluggable sink**, and the stream contract is independent of it:
|
||||
the bring-up sink is in-memory (readable back by tests, mirrored to the boot
|
||||
log), and the real one is the framebuffer text renderer when the display
|
||||
track's font work lands.
|
||||
- **Control set** (the `control` payloads): mode raw/cooked, echo on/off, and
|
||||
window-size query — the minimal termios. Ctrl-C-to-signal joins when M17
|
||||
signals-over-IPC lands; until then Ctrl-C is just a byte.
|
||||
- Mounts itself at `/device/console` as a `character_device` node.
|
||||
|
||||
**fd 0/1/2** then stop being special: spawn hands the child three open handles
|
||||
(console by default; anything else if the parent chooses), and `runtime`'s fd
|
||||
table maps 0/1/2 to them. `isatty` is simply "does `status` say
|
||||
`character_device`" — no side channel needed.
|
||||
|
||||
## The pty answer: there is no pty
|
||||
|
||||
The pty exists in Unix *because the line discipline is in the kernel* — userspace
|
||||
terminal emulators need a kernel gadget to impersonate hardware. On danos the
|
||||
terminal emulator is already a userspace server, so the pair collapses:
|
||||
|
||||
**The graphical terminal application serves the VFS stream protocol itself and
|
||||
hands its own endpoints to the children it spawns as their fd 0/1/2.**
|
||||
|
||||
The terminal *is* the console service for its children — same protocol, same
|
||||
control set, same line discipline code (shared as a library with the boot
|
||||
console). No master/slave device pair, no `/dev/pts`, no new kernel object. When
|
||||
CPython arrives, the libc's `isatty`/read/write see a character device and are
|
||||
none the wiser; when xonsh eventually wants job control, that lands as control
|
||||
messages + M17 signals, still with no pty object.
|
||||
|
||||
What this costs: programs that *specifically* manipulate Unix ptys
|
||||
(`os.openpty()`, `pexpect`-style tools) have no direct equivalent — the danos
|
||||
answer is "spawn the child yourself with your own stream endpoints," which is the
|
||||
same capability with less machinery. Accepted.
|
||||
|
||||
## Milestone slicing
|
||||
|
||||
1. **pseudo-devices** — VFS honors `character_device` semantics end to end;
|
||||
`Operation.control` added; the in-memory **loopback** (plus `null`/`zero`)
|
||||
as the first device. QEMU test: one client writes, another reads — open,
|
||||
offsetless read/write, blocking read, short read, hangup on close, control
|
||||
round-trip. No hardware anywhere.
|
||||
2. **console-service** — the line-discipline library (host-tested, pure) plus
|
||||
the console composing keyboard `InputEvent`s with an in-memory output sink;
|
||||
mounted at `/device/console`. QEMU test injects key events and reads cooked
|
||||
lines and raw bytes back through the sink.
|
||||
3. **fd-inheritance** — spawn passes 0/1/2 handles; `runtime` fd table; `isatty`
|
||||
via `status`; existing binaries' stdout migrates from `debug_write` to fd 1
|
||||
(the logger keeps its own path).
|
||||
4. **terminal-as-server** — deferred to the terminal application milestone
|
||||
(Python track P3): the terminal reuses the discipline library and serves its
|
||||
children directly.
|
||||
|
||||
Steps 1–3 are exactly the shared seam that Zig self-hosting Phase 1 and Python
|
||||
Phase 1 both list; neither track repeats them.
|
||||
|
||||
## Decisions needing sign-off
|
||||
|
||||
- **No pty object; the terminal serves its children directly** (the section
|
||||
above) — the load-bearing simplification.
|
||||
- **`control` as an enumerated, typed operation** rather than an ioctl-style
|
||||
opaque pass-through.
|
||||
- **Line discipline in userspace services** (console + terminal, shared library),
|
||||
never in the kernel.
|
||||
|
||||
## Related
|
||||
|
||||
- [python-on-danos.md](python-on-danos.md) — consumes this as its Phase 1.
|
||||
- [zig-self-hosting.md](zig-self-hosting.md) — ditto ("stdio as fds").
|
||||
- [file-system-development/vfs-protocol.md](file-system-development/vfs-protocol.md) —
|
||||
the wire protocol this note extends.
|
||||
- [file-system-development/file-system-hierarchy.md](file-system-development/file-system-hierarchy.md)
|
||||
— the tree the console surfaces in.
|
||||
- [os-development/protocol-namespace.md](os-development/protocol-namespace.md) —
|
||||
supersedes this note's device-node naming: the console lands as a protocol
|
||||
(`/protocol/console`, a protocol node), not a `/dev`-style device file. The
|
||||
stream semantics designed here (line discipline, cooked/raw modes) carry over
|
||||
unchanged.
|
||||
- [device-driver-development/input.md](device-driver-development/input.md) — the
|
||||
`InputEvent` stream the console cooks.
|
||||
@@ -182,26 +182,6 @@ test for "is this an abbreviation I must expand" is simply: *is there a longer w
|
||||
is a clipped form of?* If yes, write the word. If it's an initialism standing in for a
|
||||
phrase, leave it.
|
||||
|
||||
## Kernel code touches user memory only through `user-memory`
|
||||
|
||||
A syscall argument is an attacker-controlled integer. Kernel code never
|
||||
dereferences one: every read of a process's memory goes through
|
||||
`copyFromUser` and every write through `copyToUser`
|
||||
(`system/kernel/user-memory.zig`), which walk that address space's page tables
|
||||
and move the bytes through the physmap, with the permissions ring 3 itself
|
||||
would face. A bad pointer then fails the call instead of faulting the kernel,
|
||||
and a struct pulled in once cannot change underneath the checks that follow it.
|
||||
|
||||
This is not a review convention — CR4.SMAP enforces it in hardware
|
||||
(`docs/os-development/smep-smap.md`), so a raw dereference of a user address is
|
||||
a #PF with a kernel instruction pointer the first time the QEMU suite reaches
|
||||
it. Which is also why **there is no `stac` in this tree, and never should be**:
|
||||
`stac` suspends exactly that enforcement, the copy layer needs no such window
|
||||
by construction, and a change that adds one has removed the guarantee rather
|
||||
than worked around a limitation. The same goes for the boot-time `clac` patch
|
||||
at the interrupt entry — it exists so that ring 3 cannot suspend SMAP either,
|
||||
by taking an interrupt with `EFLAGS.AC` set.
|
||||
|
||||
## Zen of Zig
|
||||
|
||||
* Communicate intent precisely.
|
||||
|
||||
@@ -64,48 +64,19 @@ restarted instance to rebuild exactly the same ids.
|
||||
|
||||
## The protocol
|
||||
|
||||
A `device-manager-protocol` module, defined through the
|
||||
[envelope](../os-development/protocol-namespace.md): every packet — request,
|
||||
reply, and pushed event alike — begins with the folded `Header`, and **the device
|
||||
id is `Header.target`**, the manager's object addressing. The contract is bound at
|
||||
`/protocol/device-manager`; the kernel-stamped badge tells the manager who is
|
||||
A `device-manager-protocol` module (the vfs-protocol pattern): extern-struct
|
||||
messages, a version in the handshake, reserved fields everywhere. The manager is a
|
||||
well-known endpoint (`ipc.register(.device_manager)`); the badge tells it who is
|
||||
talking; the same endpoint receives its children's exit notifications — one loop,
|
||||
one world.
|
||||
|
||||
| Direction | Packet | Purpose |
|
||||
| Direction | Message | Purpose |
|
||||
|---|---|---|
|
||||
| driver → manager | `hello { role, version }` @ the assigned device | confirms the argv assignment, starts the deadline clock |
|
||||
| bus → manager | `child_added { parent, bus_address, identity, bus, vendor, device, subsystem, hid }` @ the registered device id | one node the bus discovered |
|
||||
| driver → manager | `hello { version, role, device_id }` | confirms the argv assignment, starts the deadline clock |
|
||||
| bus → manager | `child_added { parent, bus_address, identity, device_id, hid }` | one node the bus discovered |
|
||||
| bus → manager | `child_removed { parent, bus_address }` | unplug, or the bus lost it |
|
||||
| app → manager | `enumerate` (reserved verb 1) | snapshot of the tree: one `ChildEntry` per record in the reply's tail |
|
||||
| app → manager | `subscribe` (reserved verb 2) | receive published add/remove events; the subscriber's endpoint rides as the call's capability |
|
||||
| manager → app | `child_added` / `child_removed` events | the same two structs, pushed rather than called |
|
||||
|
||||
The watcher table behind those last two rows is the **service harness's**
|
||||
(`service.Subscribers`, shared with input and power), not the manager's: it
|
||||
answers `subscribe`/`unsubscribe`, frames each event once for the fan-out, and
|
||||
sweeps a watcher on its exit notification — where the manager previously had no
|
||||
sweep for watchers at all. Its own supervised-driver exits are a different thing
|
||||
and unchanged, except that a driver's death now arrives twice (the manager is
|
||||
both its supervisor and a subscriber to published exits), so the manager retires
|
||||
a dead driver's process id as it handles the first and the second finds nothing
|
||||
to act on.
|
||||
|
||||
Two of what used to be the manager's own operations are the envelope's **reserved**
|
||||
verbs, which mean the same thing at every provider in the system, so this protocol
|
||||
numbers only three of its own (`hello` = 16, `child_added` = 17,
|
||||
`child_removed` = 18) and its two events in their own space (`child_added` = 16,
|
||||
`child_removed` = 17). No reply carries a status field: that is the `Status` every
|
||||
reply begins with.
|
||||
|
||||
`child_added` is the one struct that travels both ways — a bus *calls* it, the
|
||||
manager *pushes* it — which is why the operation and event numbering spaces are
|
||||
separate: one encoding, both directions, told apart by which way the packet went.
|
||||
Folding the operation byte and the device id out of it is also what makes it fit:
|
||||
a pushed event is 64 bytes at most, header included, and this one lands exactly on
|
||||
that floor. `child_removed` is the single message whose target stays 0, because it
|
||||
is addressed by the composite (parent, bus address) and no single `u64` carries a
|
||||
pair.
|
||||
| app → manager | `enumerate` | snapshot of the tree (read-only) |
|
||||
| app → manager | `subscribe` | receive published add/remove events |
|
||||
|
||||
`hello` is the one deadline the manager enforces itself: spawned and silent past the
|
||||
deadline means wrong binary, wrong protocol version, or wedged before main — apply
|
||||
|
||||
@@ -96,15 +96,7 @@ in different namespaces — against the right `bus` column.
|
||||
|
||||
## Adding a driver
|
||||
|
||||
(The step-by-step walkthrough with a worked example is
|
||||
[new-driver-checklist.md](new-driver-checklist.md).)
|
||||
|
||||
1. Create `system/drivers/<name>/` with the driver source plus a ~15-line
|
||||
package `build.zig` + `build.zig.zon` (copy an existing driver package,
|
||||
e.g. `system/drivers/pci-bus/`; per-driver extras go through
|
||||
`build_support.programModule`). Then bundle it at `/system/drivers/<name>`:
|
||||
one dependency + one bundled entry in the root `build.zig`, one line in the
|
||||
root `build.zig.zon`.
|
||||
1. Build the driver binary and bundle it at `/system/drivers/<name>` (build.zig).
|
||||
2. Add a row to `etc/devices.csv` naming the identity it binds and its full path.
|
||||
|
||||
No device-manager change is required — the registry is the seam.
|
||||
|
||||
@@ -20,10 +20,9 @@ Read [display.md](display.md) first for the *why*; this is the *what* and the *o
|
||||
|
||||
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym abbreviations in
|
||||
full, kebab-case file names, no `Co-Authored-By` trailers on commits. New user binaries
|
||||
go through build-support's shared user-binary recipe and get packed into the
|
||||
initial-ramdisk; protocols are modules exported by the `library/protocol` package.
|
||||
(This section predates the build-packages split; see
|
||||
[build-packages-plan.md](../build-packages-plan.md) for the current build shape.)
|
||||
go through `addUserBinary` in [build.zig](../../build.zig) and get packed into the
|
||||
initial-ramdisk; protocols are `b.addModule("…-protocol", …)` and imported into the
|
||||
`runtime` module.
|
||||
|
||||
## How to verify along the way
|
||||
|
||||
|
||||
@@ -18,11 +18,9 @@ lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run,
|
||||
|
||||
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym abbreviations,
|
||||
kebab-case file names, no `Co-Authored-By` trailers. New user binaries go through
|
||||
build-support's shared user-binary recipe and get packed into the initial-ramdisk;
|
||||
protocols are modules exported by the `library/protocol` package; new syscalls extend
|
||||
[abi.zig](../../system/abi.zig) `SystemCall` + a `library/kernel` wrapper.
|
||||
(This section predates the build-packages split; see
|
||||
[build-packages-plan.md](../build-packages-plan.md) for the current build shape.)
|
||||
`addUserBinary` and get packed into the initial-ramdisk; protocols are
|
||||
`b.addModule("…-protocol", …)` imported into `runtime`; new syscalls extend
|
||||
[abi.zig](../../system/abi.zig) `SystemCall` + a `library/runtime` wrapper.
|
||||
|
||||
## How to verify along the way
|
||||
|
||||
|
||||
@@ -29,7 +29,7 @@ which one you're holding decides what you can do.
|
||||
|
||||
- **The PCI class-0x03 device is the raw controller** — BARs, config space, registers,
|
||||
IO ports. It is what you actually *own* after boot. On QEMU's emulated adapter
|
||||
([`-device VGA,edid=on`](../../build/qemu.zig), the Bochs VBE/DISPI model) the `base` GOP handed
|
||||
([`-device VGA,edid=on`](../../build.zig), the Bochs VBE/DISPI model) the `base` GOP handed
|
||||
you *is* that device's linear-framebuffer BAR — the same physical memory, seen through
|
||||
a different door. On a real discrete GPU, GOP's `base` is an aperture inside the GPU's
|
||||
VRAM BAR. danos already decodes this device
|
||||
@@ -87,13 +87,13 @@ rest of the system hasn't had to face:
|
||||
│ (ResourceKind.memory = [base, height*pitch], write-combining hint,
|
||||
│ plus DisplayInfo{width, height, pitch, format, refresh_hz})
|
||||
▼
|
||||
display service (system/services/display/, /protocol/display) ← the compositor
|
||||
display service (system/services/display/, ServiceId.display) ← the compositor
|
||||
│ device.claim(display node) → mmio_map(WRITE-COMBINING) = FRONT buffer (the LFB)
|
||||
│ mmap(cacheable) a BACK buffer of the same geometry
|
||||
│ owns: an ordered LAYER STACK + a per-frame DAMAGE tracker (rect list or tile grid)
|
||||
│ loop: composite dirty layers → back buffer → present dirty rects → front
|
||||
│ backend is an INTERNAL interface: {gop-fb} at boot; {virtio-gpu} on hot-attach (v2)
|
||||
▼ reached by name (open /protocol/display); clients drive it over the display protocol
|
||||
▼ reached by name (ipc_lookup); clients drive it over the display protocol
|
||||
┌────────────────────────────────────┬──────────────────────────────────────┐
|
||||
drawing clients (v1) surface clients (deferred)
|
||||
display commands: display surfaces:
|
||||
@@ -106,10 +106,8 @@ The bring-up sequence mirrors a hardware driver's — it is the
|
||||
[`usb-xhci-bus` `initialise`](../../system/drivers/usb-xhci-bus/usb-xhci-bus.zig) shape
|
||||
(claim → `mmio_map` → run loop) — and the request/reply service shell is the
|
||||
[FAT](../../system/services/fat/fat.zig) / [input](../../system/services/input/input.zig) shape
|
||||
([`service.run`](../../library/kernel/service.zig) over the dispatch table its
|
||||
protocol module generates through
|
||||
[`envelope.Define`](../os-development/protocol-namespace.md) — one request and
|
||||
reply type per verb, and the layer id in the packet header's `target`).
|
||||
([`service.run`](../../library/kernel/service.zig) with a `protocol.zig` of
|
||||
`extern struct` messages and an `Operation` tag).
|
||||
|
||||
**One process, for now.** v1 is a *single* service that both owns the framebuffer and
|
||||
composites — it does not split a "framebuffer driver" from a "compositor" the way input
|
||||
@@ -211,18 +209,6 @@ shell, a terminal, a cursor, and a wallpaper:
|
||||
| `damage` | mark a region of a layer dirty |
|
||||
| `present` | request a repaint: composited at the next frame-clock tick |
|
||||
|
||||
**A layer belongs to the client that created it.** The id is a slot in a
|
||||
sixteen-entry table — small, dense, guessable — so every verb above that names one is
|
||||
answered only for the task whose `create_layer` produced it, and a layer that is
|
||||
somebody else's is refused exactly as one that never existed (`-ENOENT`), so a client
|
||||
cannot use the refusal to learn which ids are live
|
||||
([protocol-namespace.md](../os-development/protocol-namespace.md): handles are scoped
|
||||
per client, validated against the badge). The compositor's own layers — the cursor
|
||||
sprite and the startup self-check's pair — are marked service-owned and are created by
|
||||
direct call rather than over the protocol, so no client can move or destroy the
|
||||
cursor. A dead client's layers are released on its exit notification, the same sweep
|
||||
the FAT server runs for open files.
|
||||
|
||||
Text is intentionally *not* an operation — a client renders glyphs by blitting tiles
|
||||
(the [PSF font](../../system/kernel/font.psf) path the console already uses can move into a
|
||||
client). Keeping the protocol to rectangles and tiles keeps the compositor small and the
|
||||
|
||||
@@ -138,17 +138,13 @@ a higher-level service (block ↔ filesystem, a scanout driver ↔ the composito
|
||||
private wire to its *hardware* — virtio-gpu's command set — is not that; it stays a
|
||||
driver-private file, like the virtio-pci transport beside it.
|
||||
|
||||
The build side of this has since landed: every binary owns a package whose
|
||||
~15-line `build.zig` names EXACTLY the modules its source imports — the moral
|
||||
equivalent of a C file's include list — and the shared recipe in
|
||||
[`build-support/build.zig`](../../build-support/build.zig) (`userBinary`)
|
||||
resolves each name from the library domain that exports it (kernel's concern
|
||||
modules, the device driver libraries, the service clients, the protocols). An
|
||||
undeclared `@import` is a compile error, and a domain none of the imports come
|
||||
from never appears in the binary's manifest — a keyboard driver declares
|
||||
`xkeyboard-config`; nothing else does (see
|
||||
[build-packages-plan.md](../build-packages-plan.md)). That's the *entire*
|
||||
mechanism — Zig modules already give you everything else.
|
||||
The build side of this has since landed: [`addUserBinary`](build.zig) injects the
|
||||
default modules — the library/kernel concern modules (`ipc`, `memory`, `process`, `time`,
|
||||
`logging`, `file-system`, `thread`, `service`), the device/service clients (`driver`,
|
||||
`block`, `display`, `input`), plus `mmio`, `xkeyboard-config`, `acpi-ids` — into every user
|
||||
binary, and per-binary extras — protocol modules, bus logic — are added with
|
||||
`programModule(exe).addImport(...)`. That's the *entire* mechanism — Zig modules
|
||||
already give you everything else.
|
||||
|
||||
The discipline that makes this work: **a class driver must not import a bus's *hardware*
|
||||
logic module.** `usb-hid` imports `usb` (the transfer client) and `input-protocol`, never
|
||||
|
||||
@@ -4,7 +4,9 @@ In a monolithic kernel a driver is a function call away from everything: it runs
|
||||
ring 0, dereferences any physical address, and its interrupt handler *is* the ISR. In
|
||||
danos a driver is **an ordinary ring-3 process**. It has its own address space, it
|
||||
can crash without taking the kernel with it, and — the point of this document — it
|
||||
can be restarted ([resilience](../os-development/resilience.md)).
|
||||
can be restarted ([resilience](../os-development/resilience.md)). This document is
|
||||
the reasoning; the condensed do-this-then-that version is the
|
||||
[new-driver checklist](new-driver-checklist.md).
|
||||
|
||||
That leaves three questions the kernel has to answer, because a process can't answer
|
||||
them for itself:
|
||||
|
||||
@@ -22,22 +22,12 @@ event:
|
||||
- `JoystickEvent` — `axis` moves (a signed value on a `control` index) and
|
||||
`button_down`/`button_up`.
|
||||
|
||||
A source publishes any of the three as one **`InputEvent`** tagged with a `DeviceKind`, so
|
||||
`publish` is a single verb; decode one with `asKeyboard()` / `asMouse()` / `asJoystick()`
|
||||
(each returns null unless the tag matches). On the *delivery* wire the class is the
|
||||
packet's own operation instead — the protocol declares one event per class
|
||||
([protocol-namespace.md](../os-development/protocol-namespace.md)), so a pushed packet is
|
||||
the 16-byte header plus the typed event and nothing carries a tag twice. The client
|
||||
helpers re-tag what arrives back into an `InputEvent`, so a subscriber can still take a
|
||||
mix of classes on one stream. A subscriber names the classes it wants with a
|
||||
**`device_mask`**, and the service routes each event only to subscribers whose mask
|
||||
includes its class — so a mouse-only listener never wakes for keystrokes.
|
||||
|
||||
**`subscribe` is not this protocol's verb.** Its shape — a synchronous call whose attached
|
||||
capability is the subscriber's own endpoint — is what the envelope's *reserved* subscribe
|
||||
means at every provider in the system, so the input protocol adopts it rather than
|
||||
defining a second spelling of the same thing. The interest mask rides as the packet's
|
||||
tail. `publish` is the one verb the protocol defines for itself.
|
||||
All three travel in one **`InputEvent` envelope** tagged with a `DeviceKind`, so the
|
||||
fan-out is a single code path and a subscriber can take a mix of classes on one stream.
|
||||
Decode an envelope with `asKeyboard()` / `asMouse()` / `asJoystick()` (each returns null
|
||||
unless the tag matches). A subscriber names the classes it wants with a **`device_mask`**,
|
||||
and the service routes each event only to subscribers whose mask includes its class — so a
|
||||
mouse-only listener never wakes for keystrokes.
|
||||
|
||||
## Why this needed a new kernel primitive
|
||||
|
||||
@@ -108,25 +98,12 @@ This is the async counterpart of `ipc_call`, and the input service is its first
|
||||
`publishJoystickEvent`. Publishing is a short synchronous `ipc_call` the service answers at
|
||||
once; the service's own fan-out is asynchronous, so publishing never blocks on a slow
|
||||
subscriber.
|
||||
- The **service** ([input.zig](../../system/services/input/input.zig)) owns none of that
|
||||
machinery any more: the subscriber table (endpoint handle + owning task + interest mask),
|
||||
the reserved `subscribe`/`unsubscribe` verbs, the fan-out, and the dead-subscriber sweep
|
||||
are the shared harness's (`service.Subscribers` in
|
||||
[service.zig](../../library/kernel/service.zig)), so every event stream in the system has
|
||||
identical semantics. What is left in this file is what is actually about input: which
|
||||
class an event belongs to, and which classes a subscriber asked for. On `publish` it names
|
||||
the event's class and the harness `ipc_send`s the packet — framed once — to every
|
||||
subscriber whose mask includes it.
|
||||
- **A dead subscriber goes away on its exit notification**, not on a poll. The service used
|
||||
to walk `process_enumerate` on every subscribe and drop slots whose owner had gone; it now
|
||||
subscribes to the kernel's published exits like the FAT server and the compositor do
|
||||
([process-lifecycle.md](../os-development/process-lifecycle.md)), which reclaims the slot
|
||||
*and* closes the endpoint capability in it promptly rather than at the next subscribe.
|
||||
(The fan-out also drops a subscriber whose `ipc_send` fails, as a backstop for a
|
||||
notification a full ring dropped.)
|
||||
- The service runs on the shared harness like every other, so it answers the universal ping
|
||||
and exits on `terminate`; it was the last hand-rolled receive loop in the tree, and the
|
||||
last service a shutdown had to kill rather than ask.
|
||||
- The **service** ([input.zig](../../system/services/input/input.zig)) keeps a small subscriber
|
||||
table (endpoint handle + owning task id + `device_mask`). On `publish` it `ipc_send`s the
|
||||
event to every subscriber whose mask includes the event's device class. On `subscribe` it
|
||||
stores the passed capability and mask and, as housekeeping, prunes any slot whose owning
|
||||
process has exited (checked against `process_enumerate`) — not for correctness (an async
|
||||
send to an orphaned endpoint is harmless) but to reclaim the slot.
|
||||
|
||||
Publisher and subscriber must be **separate processes**: a single thread that both
|
||||
published and serviced its own subscription would deadlock (its `publish` call blocks until
|
||||
|
||||
@@ -1,70 +1,34 @@
|
||||
# IPC: the kernel-ipc transport
|
||||
# IPC: message-passing channels
|
||||
|
||||
Inter-process communication is the **backbone of a microkernel**. Once drivers and
|
||||
services run isolated in their own address spaces ([vision](../vision.md)), they can't
|
||||
just call each other — a request becomes bytes on a wire. In a microkernel, whatever
|
||||
just call each other — a request becomes a **message**. In a microkernel, whatever
|
||||
was a function call across a monolithic kernel is IPC, so it's a first-class
|
||||
concern, not an afterthought.
|
||||
|
||||
This document describes **one transport** — the bottom layer (L0) of the
|
||||
communication stack defined in
|
||||
[communication.md](../os-development/communication.md), which owns the model
|
||||
and the vocabulary (*protocol*, *channel*, *packet*, *signal*, *endpoint*).
|
||||
kernel-ipc is the **first** transport, not the only possible one: in
|
||||
buffer-plus-doorbell terms it is a kernel-owned mailbox with the scheduler as
|
||||
the doorbell. Its distinguishing properties, which the layers above may rely
|
||||
on where they say so:
|
||||
There are two layers, built a milestone apart:
|
||||
|
||||
- **Rendezvous.** A call is a synchronous meeting, copied sender-page to
|
||||
receiver-page — natural backpressure, no queue to size.
|
||||
- **Capability carriage.** The *only* transport that can move a handle
|
||||
between processes. Channels are therefore always established over
|
||||
kernel-ipc, and it remains every channel's control path even when bulk
|
||||
data is negotiated onto a fatter transport (a shared-memory ring).
|
||||
- **Verified source.** Every delivery carries the kernel-stamped badge — the
|
||||
identity the channel layer attaches to received packets.
|
||||
- **Bounded packets.** 256 bytes call/reply, 64 pushed — the floor every
|
||||
protocol may assume on any transport.
|
||||
- **`system/kernel/ipc.zig`** — a bounded blocking channel between *kernel threads*,
|
||||
described below. The primitive, and where the blocking discipline was worked out.
|
||||
- **`system/kernel/ipc-synchronous.zig`** — synchronous call/reply between *processes*, across
|
||||
address spaces. What user-space servers and drivers actually talk over. It's the
|
||||
second half of this document.
|
||||
|
||||
Three properties keep the networking analogy honest — kernel-ipc is
|
||||
networking-*shaped*, not TCP:
|
||||
## The channel
|
||||
|
||||
- **Channels over it are RPC-shaped, not streams.** Packets, call/reply,
|
||||
datagram pushes — closer to UDP plus RPC than to a byte stream. Ordering
|
||||
exists per exchange (a reply answers its call), not across a channel.
|
||||
- **Possession is the connection.** There is no handshake state in the
|
||||
kernel: holding the capability *is* having the channel. A provider's one
|
||||
endpoint terminates every client's channel at once, demultiplexed by badge
|
||||
— like every client sharing the server's listening socket, with
|
||||
per-connection state living in the provider, keyed by badge. A *private*
|
||||
channel (a dedicated endpoint pair) is built when wanted: that is exactly
|
||||
what `subscribe` does.
|
||||
- **Packets never fragment.** If it doesn't fit in a packet, it isn't a
|
||||
packet: bulk data lives in shared memory and a packet (or signal) is the
|
||||
doorbell. The display path already works this way.
|
||||
|
||||
The rest of this document is the implementation, bottom-up: the kernel-thread
|
||||
queue the blocking discipline was worked out on, then endpoints — this
|
||||
transport's termination points.
|
||||
|
||||
## The kernel-thread queue
|
||||
|
||||
The first form is a **bounded blocking queue** (`system/kernel/ipc.zig`): a
|
||||
fixed-size ring buffer of messages with a producer/consumer rendezvous, built
|
||||
on the scheduler's [wait queues](../os-development/scheduling.md). (Its type
|
||||
is still named `Channel(T, capacity)` — it predates the vocabulary above, and
|
||||
is a *queue between kernel threads in one address space*, not a channel in
|
||||
the model's sense; a rename can ride a later flag-day.)
|
||||
The first form is a **bounded blocking channel** (`system/kernel/ipc.zig`): a fixed-size
|
||||
ring buffer of messages with a producer/consumer rendezvous, built on the
|
||||
scheduler's [wait queues](../os-development/scheduling.md).
|
||||
|
||||
`Channel(T, capacity)` is generic over the message type and buffer size. It holds a
|
||||
ring buffer, a count, and two wait queues:
|
||||
|
||||
- **`send(msg)`** — if the queue is full, block on the *not-full* queue; otherwise
|
||||
- **`send(msg)`** — if the channel is full, block on the *not-full* queue; otherwise
|
||||
write the message, bump the count, and wake a waiting receiver.
|
||||
- **`receive()`** — if the queue is empty, block on the *not-empty* queue; otherwise
|
||||
- **`receive()`** — if the channel is empty, block on the *not-empty* queue; otherwise
|
||||
take a message, drop the count, and wake a waiting sender.
|
||||
|
||||
Neither side busy-waits: a full queue parks the sender, an empty one parks the
|
||||
Neither side busy-waits: a full channel parks the sender, an empty one parks the
|
||||
receiver, and each operation wakes the other side when it makes progress possible.
|
||||
|
||||
Two details make it correct:
|
||||
@@ -81,53 +45,47 @@ Two details make it correct:
|
||||
CPU. `waitLocked` / `wakeLocked` are the variants that assume the caller already
|
||||
holds that critical section.
|
||||
|
||||
### Verifying it
|
||||
## Verifying it
|
||||
|
||||
The `ipc` test (see [testing.md](../testing.md)) runs a producer and a consumer passing
|
||||
**100 messages through a 4-slot queue**. The small buffer means the queue goes
|
||||
**100 messages through a 4-slot channel**. The small buffer means the channel goes
|
||||
full and empty over and over, so both the blocking-send and blocking-receive paths are
|
||||
exercised heavily. The messages arrive intact and in order (their sum is the
|
||||
expected `5050`), and neither task busy-waits — they block and wake each other.
|
||||
|
||||
## Endpoints: the termination points
|
||||
## Endpoints: call/reply across address spaces
|
||||
|
||||
A queue connects two kernel threads sharing one address space. Real providers are
|
||||
*processes*, so a packet has to cross an address-space boundary. That's
|
||||
A channel connects two kernel threads sharing one address space. Real servers are
|
||||
*processes*, so the payload has to cross an address-space boundary. That's
|
||||
`system/kernel/ipc-synchronous.zig`, and its shape is L4's: a synchronous **rendezvous** at an
|
||||
`Endpoint`, with the packet copied directly from the sender's pages to the receiver's
|
||||
`Endpoint`, with the message copied directly from the sender's pages to the receiver's
|
||||
(`copyAcross` walks both sets of page tables through the physmap — no CR3 switch, no
|
||||
bounce buffer).
|
||||
|
||||
Two syscalls carry the request/reply exchange:
|
||||
Two syscalls carry it:
|
||||
|
||||
- **`ipc_call(h, msg, reply)`** — copy the request packet to the provider, block
|
||||
until the reply packet comes back.
|
||||
- **`ipc_call(h, msg, reply)`** — copy `msg` to the server, block until it replies.
|
||||
- **`ipc_reply_wait(h, reply, recv)`** — reply to the client you're still holding (if
|
||||
any), then block for the next request. One syscall, because a provider's steady state
|
||||
any), then block for the next request. One syscall, because a server's steady state
|
||||
is *always* "finish the last one, wait for the next".
|
||||
|
||||
An endpoint is reached by **handle** — a small integer index into the process's handle
|
||||
table (`Task.handles`), exactly like a file descriptor, and just as unforgeable.
|
||||
The provider never learns the client's identity beyond the **badge** delivered
|
||||
alongside each packet: the caller's task id, stamped by the kernel —
|
||||
unforgeable source addressing, a property a network's source field lacks.
|
||||
table (`Task.handles`), exactly like a file descriptor, and just as unforgeable. The
|
||||
bootstrap problem (how do you get the first handle?) is solved by a tiny name registry:
|
||||
a server calls `ipc_register(service_id, h)` under a well-known small integer, and a
|
||||
client calls `ipc_lookup(service_id)`.
|
||||
|
||||
The bootstrap problem — how a channel is first established — is the subject of
|
||||
[protocol-namespace.md](../os-development/protocol-namespace.md): a protocol is
|
||||
resolved by name and the channel arrives as a capability. (The mechanism it
|
||||
replaced — `ipc_register`/`ipc_lookup` under compile-time `ServiceId` integers —
|
||||
is gone: both syscalls and the enum were deleted when the registry landed, and
|
||||
their syscall numbers are left vacant.)
|
||||
The server never learns the client's identity beyond a **badge**, delivered alongside
|
||||
the message: the caller's task id.
|
||||
|
||||
### Interrupts are signals
|
||||
### Interrupts are messages too
|
||||
|
||||
`notifyFromIsr` posts an *asynchronous* signal to an endpoint — no payload, no
|
||||
`notifyFromIsr` posts an *asynchronous* notification to an endpoint — no payload, no
|
||||
reply owed — and wakes whoever is blocked in `reply_wait`. Its badge has the top bit
|
||||
set (`notify_badge_bit`), which is how a driver's single event loop distinguishes "a
|
||||
client wants something" from "the hardware wants something". Signals sit in a
|
||||
client wants something" from "the hardware wants something". Notifications sit in a
|
||||
small coalescing ring on the endpoint, so an interrupt taken while the driver was busy
|
||||
elsewhere is not lost — coalesced, never dropped, which is exactly a signal's
|
||||
contract (the *count* may collapse; the *fact* may not).
|
||||
elsewhere is not lost.
|
||||
|
||||
This is what makes a user-space driver possible at all, and it's the subject of
|
||||
[drivers.md](drivers.md).
|
||||
@@ -135,65 +93,40 @@ This is what makes a user-space driver possible at all, and it's the subject of
|
||||
## What's next (partly done since)
|
||||
|
||||
- **Priority inheritance** through IPC — still open: a high-priority client
|
||||
blocked on a low-priority provider suffers unbounded priority inversion.
|
||||
blocked on a low-priority server suffers unbounded priority inversion.
|
||||
- **Handle transfer.** *Landed as cap-passing (M13)*: `ipc_call` and
|
||||
`ipc_reply_wait` carry an optional capability alongside the bytes (`send_cap`),
|
||||
copying an endpoint or shared-memory handle into the peer's table — the
|
||||
mechanism by which channels are established and private channels built. First
|
||||
user: [input](input.md) subscribers register by handing over their own
|
||||
endpoint, and class drivers get a private channel to one device.
|
||||
copying an endpoint or shared-memory handle into the peer's table. First user:
|
||||
[input](input.md) subscribers register by handing over their own endpoint, and
|
||||
class drivers get a private channel to one device.
|
||||
- **Asynchronous / buffered send** for the cases where a rendezvous is the wrong
|
||||
shape (logging, event fan-out). *Landed as `ipc_send`* — a
|
||||
non-blocking post of an event packet (≤ 64 bytes) to an endpoint's bounded
|
||||
queue, delivered through `reply_wait` (badge bit `notify_message_bit`). Built
|
||||
for, and first used by, the [input service](input.md)'s keyboard-event
|
||||
broadcast, where a synchronous push would let one dead subscriber hang the
|
||||
fan-out. A full queue drops the oldest — event packets are droppable by
|
||||
design ([protocol-namespace.md](../os-development/protocol-namespace.md)'s
|
||||
wiring section states the rule).
|
||||
- **A bounded reply** — half landed. The copy is still one packet
|
||||
(256 bytes) under the big kernel lock, but bulk transfer got its shared
|
||||
shape (logging, notifications between servers). *Landed as `ipc_send`* — a
|
||||
non-blocking post to an endpoint's bounded payload queue, delivered through
|
||||
`reply_wait` as a buffered message (badge bit `notify_message_bit`). Built for, and
|
||||
first used by, the [input service](input.md)'s keyboard-event broadcast, where a
|
||||
synchronous push would let one dead subscriber hang the fan-out. A full queue drops
|
||||
the oldest (discrete messages, not a coalescing level like the notification ring).
|
||||
- **A bounded reply** — half landed. The copy is still 256 bytes
|
||||
(`MESSAGE_MAXIMUM`) under the big kernel lock, but bulk transfer got its shared
|
||||
pages: `shared_memory_create`/`map`/`physical`, the region handle delegated as
|
||||
a capability (above) — the packets-never-fragment rule in practice.
|
||||
virtio-gpu's scanout surface is the first user
|
||||
a capability (above). virtio-gpu's scanout surface is the first user
|
||||
([display-v2.md](display-v2.md)).
|
||||
|
||||
## Lifecycle conventions over IPC (M17)
|
||||
|
||||
Three conventions from [process-lifecycle.md](../os-development/process-lifecycle.md) ride the
|
||||
signal mechanism:
|
||||
notification mechanism:
|
||||
|
||||
- **Process signals** arrive as endpoint signals on the endpoint a process
|
||||
nominated with `signal_bind` (`process.bindSignals`): badge = the signal bit
|
||||
plus the coalesced pending mask (`process.signalsFrom` decodes). Statements,
|
||||
- **Signals** arrive as notifications on the endpoint a process nominated with
|
||||
`signal_bind` (`process.bindSignals`): badge = the signal bit plus the
|
||||
coalesced pending mask (`process.signalsFrom` decodes). Statements,
|
||||
never questions; no payload, no reply.
|
||||
- **One-shot timers** (`timer_bind`, `time.timerOnce`) land as a
|
||||
timer-bit signal — the timed wait: a service arms a deadline and keeps
|
||||
timer-bit notification — the timed wait: a service arms a deadline and keeps
|
||||
serving, instead of blocking in sleep.
|
||||
- **Kernel notifications go only to your own endpoint.** `signal_bind`,
|
||||
`timer_bind`, `process_subscribe`, `irq_bind`, `msi_bind`, and spawn's exit
|
||||
endpoint all *nominate where the kernel will speak*, and all of them refuse an
|
||||
endpoint the caller did not create (`-EPERM`; the check is `ipc.ownedBy`,
|
||||
normalized to the process, so any thread may nominate an endpoint a sibling
|
||||
created). Holding a handle is not enough, because holding a handle is cheap:
|
||||
`fs_resolve` installs a mounted backend's capability in *any* caller's table,
|
||||
so every process holds a handle to PID 1's mailbox. Without the rule, "bind
|
||||
init's endpoint, then signal yourself" is a genuine, kernel-stamped `terminate`
|
||||
badge in PID 1's queue — a shutdown a receiver has no way to disbelieve — and
|
||||
timers, which carry no identity at all, multiply any loop that re-arms on its
|
||||
own landing.
|
||||
- **A capability that arrives belongs to the turn.** The kernel installs a sent
|
||||
capability in the receiver's table whatever the message's length or kind, so a
|
||||
receive loop must dispose of one on *every* path — the ping, the notification,
|
||||
the malformed request. The service harness (`service.run`) and PID 1 both hold
|
||||
it in an `ipc.Arrival`, released by a `defer`, and a handler that means to keep
|
||||
it says `take()`: forgetting closes, keeping is explicit. The reverse
|
||||
arrangement leaks a handle-table slot per request, and thirty-two unauthorized
|
||||
zero-length pings then end a service's ability to accept any capability —
|
||||
no subscribe, no shared-memory handover — for the rest of the boot.
|
||||
- **The universal ping**: a **zero-length request is the liveness probe**,
|
||||
answered with a zero-length reply by the service harness itself
|
||||
(`service.run`). No protocol's requests start at length zero, so the
|
||||
encoding cannot collide, and a wedged service simply fails to answer — which
|
||||
is the diagnosis. Deep health ("can I reach my hardware?") stays a per-service
|
||||
protocol packet.
|
||||
protocol message.
|
||||
|
||||
@@ -1,218 +0,0 @@
|
||||
# New driver: the minimum steps
|
||||
|
||||
The shortest path from "a device shows up in the boot log" to "my process is
|
||||
running with its registers mapped". This is the checklist; the reasoning behind
|
||||
every step lives in [Writing a driver](drivers.md), the matching rules in
|
||||
[devices.csv](devices-csv.md), and interrupts in
|
||||
[device interrupts](device-interrupts.md).
|
||||
|
||||
Worked example throughout: the Intel UHD 750 iGPU, which the boot log reports as
|
||||
|
||||
```
|
||||
pci-bus: 0:2.0 bus=pci base=03 class=00 prog_if=00 vendor=8086 device=4C8A ...
|
||||
```
|
||||
|
||||
## 1. Create the source file
|
||||
|
||||
`system/drivers/<name>/<name>.zig` — kebab-case, abbreviations spelled out
|
||||
([coding standards](../coding-standards.md)). The directory name, the binary
|
||||
name, and the `devices.csv` driver path must all agree; a mismatch fails
|
||||
silently (the device-manager logs the spawn failure, nothing else happens).
|
||||
|
||||
The complete minimal driver — claims its device, logs every resource, maps the
|
||||
register window, then sleeps in the harness loop:
|
||||
|
||||
```zig
|
||||
//! /system/drivers/intel-uhd-graphics-750 — spawned by the device manager with
|
||||
//! the device-tree id as argv[1]; claims that device and no other.
|
||||
|
||||
const std = @import("std");
|
||||
const device = @import("driver");
|
||||
const ipc = @import("ipc");
|
||||
const memory = @import("memory");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
|
||||
/// No protocol yet: the kernel's IPC ceiling (MESSAGE_MAXIMUM) sizes the buffers.
|
||||
const message_maximum = 256;
|
||||
|
||||
var controller_id: u64 = 0;
|
||||
var register_base: usize = 0;
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
_ = endpoint; // needed later, for irq binding and timers
|
||||
|
||||
if (!device.claim(controller_id)) {
|
||||
std.log.err("unable to claim device {d}", .{controller_id});
|
||||
return false;
|
||||
}
|
||||
|
||||
// Fetch our own descriptor back for the device's resources.
|
||||
const buffer = memory.allocator().alloc(device.DeviceDescriptor, 64) catch return false;
|
||||
defer memory.allocator().free(buffer);
|
||||
const total = device.enumerate(buffer);
|
||||
const descriptor = for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||
if (d.id == controller_id) break d;
|
||||
} else {
|
||||
std.log.err("device {d} not in the device tree", .{controller_id});
|
||||
return false;
|
||||
};
|
||||
|
||||
// Log every resource BEFORE choosing one (see step 5).
|
||||
var register_index: u64 = 0;
|
||||
for (descriptor.resources[0..@intCast(descriptor.resource_count)], 0..) |resource, index| {
|
||||
std.log.info("resource {d}: kind={d} start=0x{x} len=0x{x}", .{
|
||||
index, resource.kind, resource.start, resource.len,
|
||||
});
|
||||
// The 16 MiB window is GTTMMADR, the register BAR (this device also has
|
||||
// a 256 MiB memory BAR, GMADR — "first memory resource" would be wrong).
|
||||
if (resource.kind == @intFromEnum(device.ResourceKind.memory) and
|
||||
resource.len == 16 * 1024 * 1024) register_index = index;
|
||||
}
|
||||
if (register_index == 0) {
|
||||
std.log.err("register BAR not found", .{});
|
||||
return false;
|
||||
}
|
||||
|
||||
register_base = device.mmioMap(controller_id, register_index) orelse {
|
||||
std.log.err("mmio_map failed", .{});
|
||||
return false;
|
||||
};
|
||||
std.log.info("registers mapped at 0x{x}", .{register_base});
|
||||
return true;
|
||||
}
|
||||
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||
_ = message;
|
||||
_ = reply;
|
||||
_ = sender;
|
||||
_ = capability;
|
||||
return 0; // no protocol yet; the zero-length ping is answered by the harness
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
const argument = init.arguments.get(1) orelse {
|
||||
std.log.err("missing device id (argv[1])", .{});
|
||||
return;
|
||||
};
|
||||
controller_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
std.log.err("malformed device id '{s}'", .{argument});
|
||||
return;
|
||||
};
|
||||
service.run(message_maximum, .{
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
// .on_notification only once an IRQ or timer is bound
|
||||
});
|
||||
}
|
||||
```
|
||||
|
||||
`claim` is the capability gate: MMIO mapping, DMA grants, and IRQ binding all
|
||||
require it, and it pins the IOMMU domain to this process
|
||||
([drivers.md — claim before touch](drivers.md#the-capability-claim-before-touch)).
|
||||
|
||||
## 2. Create the build package and register it in the root build
|
||||
|
||||
The driver directory is its own build package
|
||||
([build-packages-plan.md](../build-packages-plan.md)): a ~15-line `build.zig`
|
||||
plus a `build.zig.zon` beside the source. Copy both from an existing driver —
|
||||
`system/drivers/pci-bus/` is the template — and adjust the name, root source
|
||||
file, and the import list. The list names EXACTLY the modules the driver's
|
||||
source `@import`s (the moral equivalent of its include list; an undeclared
|
||||
import is a compile error):
|
||||
|
||||
```zig
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "intel-uhd-graphics-750",
|
||||
.root_source_file = b.path("intel-uhd-graphics-750.zig"),
|
||||
.imports = &.{ "driver", "ipc", "memory", "process", "service" },
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
}
|
||||
```
|
||||
|
||||
The zon declares `build-support`, `kernel` (implicit in every binary: the root
|
||||
shim lives there), and the homes of the listed imports — for the minimal
|
||||
driver above that is kernel alone plus `device` (for `driver`); add
|
||||
`protocol`, `client`, ... only when an import comes from them (again, copy
|
||||
pci-bus's zon and adjust). For the `.fingerprint` field, leave the copied
|
||||
value in place and `zig build` will reject it and suggest the fresh one to
|
||||
paste.
|
||||
|
||||
Then three one-liners in the root build register the package: the dependency
|
||||
and a row in the boot-tree array in `build.zig` (search for
|
||||
`virtio_gpu_package` to land in the right places),
|
||||
|
||||
```zig
|
||||
const intel_uhd_graphics_750_exe = b.dependency("intel-uhd-graphics-750", .{}).artifact("intel-uhd-graphics-750");
|
||||
```
|
||||
|
||||
```zig
|
||||
.{ .path = "system/drivers/intel-uhd-graphics-750", .binary = intel_uhd_graphics_750_exe.getEmittedBin() },
|
||||
```
|
||||
|
||||
and the path entry in the root `build.zig.zon`:
|
||||
|
||||
```zig
|
||||
.@"intel-uhd-graphics-750" = .{ .path = "system/drivers/intel-uhd-graphics-750" },
|
||||
```
|
||||
|
||||
Without the boot-tree row the binary never reaches the image and the
|
||||
device-manager has nothing to spawn. (The package also builds standalone:
|
||||
`cd system/drivers/intel-uhd-graphics-750 && zig build`.)
|
||||
|
||||
## 3. Add the match rule to `etc/devices.csv`
|
||||
|
||||
One row: bus, class triplet, vendor/device, driver path. **Copy the class
|
||||
triplet from the pci-bus boot log line, not from another row** — for the iGPU
|
||||
above the correct rule is
|
||||
|
||||
```
|
||||
pci, 03, 00, 00, 8086, 4C8A, *, *, /system/drivers/intel-uhd-graphics-750
|
||||
```
|
||||
|
||||
Field-by-field rules and the most-specific-wins policy: [devices.csv](devices-csv.md).
|
||||
The registry is authoritative: an unmatched device is logged unbound, never
|
||||
guessed — so a wrong nibble here means the driver simply never starts.
|
||||
|
||||
## 4. First contact: read, predict, verify
|
||||
|
||||
Before writing any register, read one whose value you can predict from state
|
||||
the firmware already programmed (for a display controller: the pipe source
|
||||
size of the live mode). Registers are volatile loads at `register_base +
|
||||
offset`, where `offset` is what the device's manual lists:
|
||||
|
||||
```zig
|
||||
fn read32(offset: usize) u32 {
|
||||
return @as(*volatile u32, @ptrFromInt(register_base + offset)).*;
|
||||
}
|
||||
```
|
||||
|
||||
A matching read proves the whole chain — CSV match, spawn, claim, BAR choice,
|
||||
mapping — with zero risk to the hardware.
|
||||
|
||||
## 5. Verify the plumbing
|
||||
|
||||
- `zig build test` still passes.
|
||||
- On the image: `/var/log/<boot-stamp>/system/services/device-manager.log`
|
||||
shows `spawned <name> for device <N>`, and
|
||||
`/var/log/<boot-stamp>/system/drivers/<name>.log` holds the resource list and
|
||||
your first read.
|
||||
- If the driver did not spawn, diagnose in this order: binary on the image
|
||||
(step 2) → CSV row matches the log line exactly (step 3) → path identical in
|
||||
both (step 1).
|
||||
|
||||
## Later, when the device needs them
|
||||
|
||||
- **Interrupts**: MSI/MSI-X via the `pci` module, delivered as notifications to
|
||||
`on_notification` — see [device interrupts](device-interrupts.md) and the
|
||||
xHCI driver's `setupMsi` (QEMU trap documented there: enable MSI-X before
|
||||
unmasking the device's own interrupt-enable bit).
|
||||
- **DMA**: grant-backed buffers, bounded by the IOMMU domain established at
|
||||
claim time ([driver model](driver-model.md)).
|
||||
- **Children**: a bus driver publishes what it finds via `device_register`
|
||||
([drivers.md — publishing children](drivers.md#publishing-children-device_register)).
|
||||
- **A protocol**: replace `message_maximum` with the protocol's own maximum and
|
||||
dispatch on the operation word in `onMessage` — every service under
|
||||
`system/services/` is an example.
|
||||
@@ -1,124 +0,0 @@
|
||||
# Dynamic libraries on danos
|
||||
|
||||
A design note and milestone plan for shared objects: building them, loading them
|
||||
with `dlopen`, and — the part that needs kernel work — actually *sharing* them
|
||||
between processes. Directional, post-P5 of
|
||||
[python-on-danos-milestones.md](python-on-danos-milestones.md); nothing on the
|
||||
CPython bring-up path depends on it.
|
||||
|
||||
## Reconciling the earlier "rejected"
|
||||
|
||||
Dynamic libraries were evaluated once before and rejected — but as an answer to a
|
||||
*different question*: whether they could claw back ReleaseSafe's measured ~2×
|
||||
code size. They cannot (the safety checks inline at every call site; no library
|
||||
scheme dedups them), and that verdict stands for that question. The reasons to
|
||||
build them now are the ones that investigation never weighed:
|
||||
|
||||
- **`ctypes` and runtime FFI** — Python calling into a danos library without
|
||||
rebuilding the interpreter. This is the piece that makes Python prototyping
|
||||
self-serve: drop a `.so` on the image, `ctypes.CDLL` it, iterate.
|
||||
- **Loadable CPython extension modules** — today every C extension means
|
||||
relinking the interpreter (`Modules/Setup`); with `dlopen`, an extension is a
|
||||
file.
|
||||
- **One interpreter image, many Python services** — a statically-linked CPython
|
||||
is tens of megabytes *per process*. A shared `libpython` mapped read-only once
|
||||
(milestone D3 below) makes Python services cheap enough to be the default way
|
||||
to prototype one.
|
||||
- **Plugin-shaped applications** — the UI toolkit and the terminal will want
|
||||
them eventually.
|
||||
|
||||
The scoping that dissolves the apparent contradiction is the **size doctrine**:
|
||||
leanness is an *operating-system* property — the kernel and system services stay
|
||||
small and statically linked, and none of them ever link the loader — while
|
||||
*applications* have their own budget and may be big. Dynamic libraries are an
|
||||
**application-layer facility**, full stop.
|
||||
|
||||
What also does **not** change: the public ABI stays the vDSO + the IPC
|
||||
protocols. Shared objects are artifacts *within* one system image, versioned by
|
||||
the build — not a new stable ABI surface for the OS.
|
||||
|
||||
## Design
|
||||
|
||||
- **Format and codegen are free.** ELF shared objects with position-independent
|
||||
code; `zig cc -fPIC -shared` against the [libdanos-c](c-library-compatibility.md)
|
||||
sysroot already emits them. The work is entirely on the loading side.
|
||||
- **The loader lives in userspace, inside the libc.** `dlopen` reads the `.so`
|
||||
through the VFS, maps its segments, applies relocations, resolves symbols
|
||||
against the process and the `DT_NEEDED` dependency graph, runs constructors,
|
||||
returns a handle. No kernel loader changes in v1 — segments land in anonymous
|
||||
`mmap` as private copies.
|
||||
- **Bind-now, always.** All relocations resolved at `dlopen` time
|
||||
(`RTLD_NOW` semantics only). Lazy PLT binding buys startup latency danos does
|
||||
not care about, at the price of a writable GOT dance and a much subtler
|
||||
loader. Not worth it; keep it out permanently.
|
||||
- **W^X from day one.** Map, relocate, then flip text pages read-execute —
|
||||
which requires memory-protection change (`mprotect`-shaped) in the danos
|
||||
`mmap` surface if it is not already there. No page is ever writable and
|
||||
executable at once.
|
||||
- **TLS in shared objects is deferred.** Thread-local storage models
|
||||
(initial-exec vs. general-dynamic) are the deep end of every dynamic linker.
|
||||
v1 refuses a `.so` with a TLS segment; revisit alongside the post-P5 pthread
|
||||
subset, which is when it could matter.
|
||||
- **Executables stay static until D4.** v1 is "a static binary that can
|
||||
`dlopen`" — no `PT_INTERP`, no program interpreter, no dynamically-linked
|
||||
`main` binaries. That keeps process startup untouched.
|
||||
|
||||
## Milestones
|
||||
|
||||
1. **D1 — dlopen in-process.** The `.so` build target; the loader in libdanos-c:
|
||||
map, relocate (`RELATIVE`/`GLOB_DAT`/`JUMP_SLOT`), resolve, constructors;
|
||||
`dlopen`/`dlsym`/`dlerror`/`dlclose`; private anonymous mappings; no TLS.
|
||||
*Test:* QEMU `dlopen-hello` — load a `.so`, call a symbol, unload, reload.
|
||||
2. **D2 — the FFI payoff.** `DT_NEEDED` dependency graphs; a **libffi port**
|
||||
(x86-64 SysV assembly is upstream; the port is its closure-allocation paths,
|
||||
which must respect W^X); CPython's `ctypes` enabled; extension modules
|
||||
loadable from file. *Test:* QEMU — a Python script `ctypes.CDLL`s a danos
|
||||
`.so` and round-trips a call; a `.so` extension module imports.
|
||||
3. **D3 — actual sharing (the kernel milestone).** Shared read-only file-backed
|
||||
mappings — a page-cache-shaped facility so N processes mapping `libpython`
|
||||
hold one physical copy. This is the memory-win milestone and the only one
|
||||
touching the kernel; design it with the existing shm machinery in view
|
||||
(the shared-fate walks already locked the relevant paths). *Test:* N Python
|
||||
services up; measure physical pages against N× the static baseline.
|
||||
4. **D4 — dynamically-linked executables** (optional, evaluate after D3):
|
||||
`PT_INTERP`, a danos program interpreter, and the spawn path teaching the
|
||||
loader about it. Only worth it if the image-size or update story demands it.
|
||||
|
||||
## Risks and gotchas
|
||||
|
||||
- **Scope creep is the failure mode.** Every dynamic linker grows toward glibc.
|
||||
The fences: bind-now only, no lazy binding ever, no TLS until pthreads demand
|
||||
it, no dlopen-from-memory, no versioned symbols. Each fence removed is a
|
||||
design discussion, not a patch.
|
||||
- **Code loading is a security event.** `dlopen` turns file bytes into executable
|
||||
code, so W^X discipline is table stakes and *what may be dlopened* is a
|
||||
capability question — the natural danos answer is that loadability follows VFS
|
||||
readability of the `.so`, and services' images are supervised like any other
|
||||
artifact. Revisit explicitly at D3 when mappings become shared.
|
||||
- **`dlclose` is where loaders go to die.** Constructors/destructors,
|
||||
dangling function pointers, re-open identity. Keep v1 semantics honest and
|
||||
simple: `dlclose` runs destructors and unmaps; holding pointers past it is
|
||||
undefined; no reference-counted deferral cleverness.
|
||||
- **The ReleaseSafe fact still applies to `.so`s** — a ReleaseSafe shared object
|
||||
carries its inlined checks like any static code; D3's sharing saves *copies*,
|
||||
not check overhead. Size expectations should be set accordingly.
|
||||
|
||||
## Decisions needing sign-off
|
||||
|
||||
- Dynamic libraries join the roadmap at all (this note exists because the
|
||||
earlier size-motivated rejection was re-opened for ABI/sharing reasons).
|
||||
- **Bind-now only; no lazy binding, permanently.**
|
||||
- **Loader in userspace libc; kernel involvement only at D3** (shared read-only
|
||||
mappings).
|
||||
- **Static executables until D4**, and D4 only on demonstrated need.
|
||||
|
||||
## Related
|
||||
|
||||
- [c-library-compatibility.md](c-library-compatibility.md) — the sysroot the
|
||||
loader ships in; its absence table gains `dlfcn.h` at D1.
|
||||
- [python-on-danos.md](python-on-danos.md) — the `ctypes` story this unlocks.
|
||||
- [python-on-danos-milestones.md](python-on-danos-milestones.md) — sequencing;
|
||||
this work is post-P5.
|
||||
- [os-development/memory-map.md](os-development/memory-map.md) /
|
||||
[os-development/paging.md](os-development/paging.md) — where W^X and shared
|
||||
mappings land.
|
||||
@@ -0,0 +1,128 @@
|
||||
# DanOS Filesystem Hierarchy Standard (DFHS)
|
||||
|
||||
Most modern Unix and Unix-like operating systems follow the FHS. DanOS has its own FHS structure which extends the unix FHS. Root path resolution is provided by the kernel-resident VFS root (`fs_resolve`, `system/kernel/vfs.zig`); mounted filesystem servers serve the subtrees they own.
|
||||
|
||||
## Directory structure
|
||||
|
||||
| Path | Description |
|
||||
|------------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| / | Primary hierarchy root and root directory of the entire file system hierarchy. |
|
||||
| /bin | Essential command binaries that need to be available in single-user mode, including to bring up the system or repair it, for all users (e.g., cat, ls, cp). |
|
||||
| /boot | Boot loader files (e.g., EFI, initial-ramdisk.img ). |
|
||||
| /dev | POSIX Device files (e.g., /dev/null, /dev/disk0, /dev/tty, /dev/random). |
|
||||
| /etc | Host-specific system-wide configuration files. |
|
||||
| /home | Users' home directories, containing saved files, personal settings, etc. |
|
||||
| /lib | Libraries essential for the binaries in /bin and /sbin. eg realtime, system, ipc etc. |
|
||||
| /sbin | Essential system binaries (e.g init) |
|
||||
| /srv | Site-specific data served by this system, such as data and scripts for web servers, data offered by FTP servers, and repositories for version control systems |
|
||||
| /system | DanOS operating system files (similar idea to C:\Windows). A true representation of danos — its layout mirrors the source tree, so `/system` is what danos *is*. |
|
||||
| /system/devices | danos virtual device tree e.g. similar to /sys on linux but with danos device tree conventions (the structures in the devices module) |
|
||||
| /system/drivers | driver binaries, one sub-project each (e.g. /system/drivers/pci-bus, /system/drivers/ps2-bus) |
|
||||
| /system/services | system-service binaries — init, the FAT server, and other user-mode servers (e.g. /system/services/init, /system/services/fat) |
|
||||
| /system/kernel | the kernel image |
|
||||
| /test | Test fixtures for the QEMU integration suite. Read-only and initrd-backed like /system, and its layout likewise mirrors the source tree (the repo's test/ directory). Present on development and test images; a volume without it still boots. |
|
||||
| /test/system/services | test-fixture binaries (e.g. /test/system/services/vfs-test, /test/system/services/thread-test) — the same path in the repo source tree and on the boot volume |
|
||||
| /tmp | Directory for temporary files (see also /var/tmp). Often not preserved between system reboots and may be severely size-restricted. |
|
||||
| /usr | Secondary hierarchy for read-only user data; contains the majority of (multi-)user utilities and applications. Should be shareable and read-only. |
|
||||
| /var | Variable files: files whose content is expected to continually change during normal operation of the system, such as logs, spool files, and temporary e-mail files. |
|
||||
|
||||
## File types
|
||||
|
||||
POSIX specifies the long format of the ls command to represent the Unix file type as the first letter for an entry.
|
||||
|
||||
| type | symbol | Description |
|
||||
|-------------------|--------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| regular | - | An ordinary file holding an uninterpreted byte stream. Reads and writes are positional, and the file grows on demand (e.g., a binary in /bin, a config file in /etc). |
|
||||
| directory | d | A container mapping names to other files. It may only be modified through directory operations, never written to directly. |
|
||||
| symbolic link | l | A file whose contents are a path that is resolved in its place. The target need not exist, and may cross mount points. |
|
||||
| FIFO special | p | A named pipe: an in-order byte stream between processes, where writers block until a reader opens the other end. |
|
||||
| block special | b | A device node addressed in fixed-size blocks with the kernel free to buffer and reorder access (e.g., /dev/disk0). |
|
||||
| character special | c | A device node addressed as an unbuffered byte stream, delivered to the driver in order (e.g., /dev/tty, /dev/null). |
|
||||
| socket | s | A named endpoint for bidirectional message-passing between processes, bound to a path rather than an address. |
|
||||
|
||||
## /dev
|
||||
|
||||
`/dev` holds the names through which processes reach devices. It is deliberately not
|
||||
the device tree: the tree — every node discovered by ACPI or PCI enumeration, with its
|
||||
resources and its parent — lives under [/system/devices](#directory-structure) and is
|
||||
addressed by device id. `/dev` is the much smaller set of devices that have a driver
|
||||
willing to serve them, addressed by name.
|
||||
|
||||
A device node is not a file the VFS can read. The bytes live in a driver process
|
||||
([drivers.md](../device-driver-development/drivers.md)), so opening a `/dev` name has to resolve to that driver's
|
||||
IPC endpoint, and subsequent reads and writes are calls against it. Resolve-to-endpoint
|
||||
is exactly what the kernel's `fs_resolve` already does for any mounted backend, and
|
||||
`FileStatus.kind` is the field that marks a device node; **what is not implemented today
|
||||
is `/dev` itself** — no service mounts it. (The flat eight-node ramfs this section once
|
||||
described is retired: the kernel-resident VFS root in `system/kernel/vfs.zig` serves a
|
||||
read-only initrd mount per top-level tree — `/system`, and `/test` on images that carry
|
||||
the fixtures — with real directories and node kinds, and filesystem
|
||||
backends such as the FAT server mount the rest.) The three sections below describe the
|
||||
intended shape, and are honest about which parts the kernel can already support.
|
||||
|
||||
### Character devices
|
||||
|
||||
A character device is a byte stream with no addressable position: bytes are delivered
|
||||
to the driver in the order written, and a read consumes what is there. Terminals,
|
||||
serial lines, keyboards and mice are all of this shape. These are the natural first
|
||||
device nodes in danos, because a character driver needs nothing the kernel doesn't
|
||||
already provide — it claims its device, maps its registers with `mmio_map`, and blocks
|
||||
on `replyWait` for either an interrupt or a client request. `system/drivers/ps2-bus/ps2-bus.zig`
|
||||
is already that program, minus the file-node client half.
|
||||
|
||||
The obstacle was never the file type; it is which hardware a ring-3 driver can reach.
|
||||
Direct `in`/`out` from user space is still a #GP (no TSS I/O bitmap, IOPL never raised),
|
||||
but a driver no longer needs it: **`io_read`/`io_write`** grant port access the same way
|
||||
`mmio_map` grants memory — gated by `device_claim` and the device's discovered `io_port`
|
||||
resource. So the 16550 UART at `0x3F8` and the PS/2 controller at `0x60`/`0x64` (and thus
|
||||
`/dev/ttyS0` and a keyboard node) are now writable as ordinary ring-3 drivers; the
|
||||
low-rate legacy hardware that needs port I/O is fine with a syscall per access. A
|
||||
memory-mapped device such as the framebuffer, needing no port I/O at all, remains the
|
||||
easiest first entry.
|
||||
|
||||
### Block devices
|
||||
|
||||
A block device is addressed in fixed-size blocks and, unlike a character device, the
|
||||
layer above is free to buffer, reorder, coalesce and retry requests against it. Disks
|
||||
and other persistent storage are the whole population of this class.
|
||||
|
||||
A block driver is now **writable, but not yet memory-safe.** Every storage controller
|
||||
worth naming is a bus master: it is programmed by handing it the physical address of a
|
||||
descriptor ring and left to read and write memory on its own. That ring is exactly what
|
||||
**`dma_alloc`** now provides — physically contiguous, pinned, uncacheable, with its
|
||||
physical address disclosed — and **`/lib/device/mmio`**'s barriers order the descriptor writes
|
||||
against the doorbell, and **`msi_bind`** delivers completions. So an AHCI or NVMe driver
|
||||
can be written today (the M14/M15 work in [driver-model.md](../device-driver-development/driver-model.md); the earlier
|
||||
"cannot host a block driver at all" is no longer true).
|
||||
|
||||
What is *not* yet true is that it is safe. A device programmed with an arbitrary physical
|
||||
address writes to arbitrary physical memory, and page tables do not sit between a device
|
||||
and RAM — an IOMMU does. The IOMMU is now *detected* (M16), but no translation domains
|
||||
are programmed, so granting a DMA-capable device to a driver process is still equivalent
|
||||
to granting ring 0. Until per-device domains confine a driver's DMA to the buffers it
|
||||
`dma_alloc`'d, a block driver works but forfeits the isolation that motivates user-space
|
||||
drivers — enforcement is the next step, and lands with that first driver. A ramdisk over
|
||||
the initial ramdisk remains the one block-shaped thing that needs no driver process at all.
|
||||
|
||||
### Pseudo-devices
|
||||
|
||||
A pseudo-device has the interface of a device and no hardware behind it: `/dev/null`
|
||||
discarding writes and reading as end-of-file, `/dev/zero` reading as an endless run of
|
||||
zero bytes, `/dev/full` failing writes with `ENOSPC`, `/dev/random` and `/dev/urandom`
|
||||
yielding unpredictable bytes.
|
||||
|
||||
These are the only `/dev` entries danos can implement immediately, and they are the
|
||||
sensible place to start, because they are exactly the entries that need no driver
|
||||
process, no `device_claim`, no MMIO grant and no interrupt. A future pseudo-device
|
||||
service would answer them out of its own address space — `null` and `zero` are a few
|
||||
lines each in its `read` and `write` handlers — and mount itself at `/dev` the way the
|
||||
FAT server mounts `/mnt/usb`. The two pieces of structure every later device node
|
||||
depends on (and that the flat ramfs of the time lacked) exist now: directories, so that
|
||||
`/dev/null` is a path rather than a name; and a populated `FileStatus.kind`, so that a
|
||||
caller can tell a character device from a regular file.
|
||||
|
||||
`/dev/random` is the one that is not free. It needs an entropy source, and the honest
|
||||
options on this kernel are `RDRAND`/`RDSEED` where CPUID advertises them, and the HPET
|
||||
counter's low bits as a poor fallback. Neither is a seeded CSPRNG, and a `/dev/random`
|
||||
that is merely unpredictable-looking is worse than none — nothing should be keyed from
|
||||
it until it is a real one.
|
||||
@@ -1,90 +0,0 @@
|
||||
# The danos file-system hierarchy
|
||||
|
||||
danos is not unix, and its tree does not follow the unix FHS. Paths are the
|
||||
system's universal namespace — files, the device inventory, and protocol
|
||||
endpoints all live in one tree — but what a path *yields* differs by subtree:
|
||||
bytes, facts, or a connection. Root path resolution is provided by the
|
||||
kernel-resident VFS root (`fs_resolve`, `system/kernel/vfs.zig`); mounted
|
||||
backends serve the subtrees they own.
|
||||
|
||||
Naming follows the codebase conventions: kebab-case, full words, no
|
||||
abbreviations. Every top-level name says what its subtree *is*.
|
||||
|
||||
## The tree
|
||||
|
||||
| Path | What it is |
|
||||
|-------------------------|---------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| `/` | The root of the one namespace. |
|
||||
| `/applications` | Installed applications, one directory per application — the directory is the identity, the same rule as source sub-projects. *(Planned; empty today.)* |
|
||||
| `/protocol` | The contract namespace: one protocol node per contract, grouped into directories by domain (`/protocol/display`, `/protocol/networking/ip`). Synthetic — no bytes; opening a name yields a connection to the current provider. See [protocol-namespace.md](../os-development/protocol-namespace.md). |
|
||||
| `/system` | The operating system — what danos *is*. Its program subtrees mirror the source tree exactly. |
|
||||
| `/system/kernel` | The kernel image. |
|
||||
| `/system/drivers` | Driver binaries, one per sub-project (`/system/drivers/pci-bus`, `/system/drivers/ps2-bus`). |
|
||||
| `/system/services` | System-service binaries (`/system/services/init`, `/system/services/fat`). |
|
||||
| `/system/devices` | The device inventory: every node hardware discovery found, with its resources and parent — the structures of the devices module, as a browsable virtual tree. Informational only; you *read about* hardware here and *talk to* it through `/protocol`. *(Planned; served by device-manager.)* |
|
||||
| `/system/configuration` | Machine configuration (`init.csv`, `devices.csv`). Writable, served from the boot volume. |
|
||||
| `/system/logs` | Per-boot logs: `/system/logs/<boot-stamp>/<binary-path>.log`. Writable, served from the boot volume. |
|
||||
| `/test` | Test fixtures for the QEMU integration suite. Read-only and initrd-backed like the program subtrees of `/system`, mirroring the repo's `test/` directory. Present on development and test images; a volume without it still boots. |
|
||||
| `/volumes` | Attached storage volumes, one directory per volume (`/volumes/usb`). A volume's own tree appears beneath its name. |
|
||||
|
||||
Read-only and writable halves of `/system`: the program subtrees (`kernel`,
|
||||
`drivers`, `services`) and the future `devices` are immutable at runtime —
|
||||
initrd-backed or synthetic — while `configuration` and `logs` are mutable
|
||||
machine state served by the boot-volume FAT backend. The kernel's
|
||||
reserved-prefix rule (no mount may shadow `/system`, `/test`, or `/protocol`)
|
||||
needs a carve-out for exactly these two writable subtrees; that lands with the
|
||||
path migration below.
|
||||
|
||||
Deliberately not defined yet: a temporary-files location and per-application
|
||||
mutable storage. Both belong to the `/applications` design and will be
|
||||
specified there, not guessed at here.
|
||||
|
||||
## Node kinds
|
||||
|
||||
What a path resolves to. These fill `FileStatus.kind` and
|
||||
`DirectoryEntry.kind` in the [vfs protocol](vfs-protocol.md)
|
||||
(`library/protocol/vfs/vfs-protocol.zig`); enum values are append-only.
|
||||
|
||||
| Kind | Meaning |
|
||||
|--------------------|-------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| `regular` | An ordinary file: an uninterpreted byte stream, positional reads and writes, grows on demand. |
|
||||
| `directory` | A container mapping names to nodes; modified only through directory operations. |
|
||||
| `character_device` | A node whose read/write have **stream semantics**: unseekable, reads block until bytes exist, size is meaningless. The console and every tty-shaped node ([character-devices-and-tty.md](../character-devices-and-tty.md)); what a POSIX layer's `isatty` detects. |
|
||||
| `block_device` | A node addressed in fixed-size sectors — a raw volume. Reserved: recognized, nothing serves one yet. |
|
||||
| `symbolic_link` | Reserved: a recognized value, not implemented by any backend. |
|
||||
| `fifo` | Reserved for the future pipe object (wanted by the POSIX compatibility layer); not implemented. |
|
||||
| `protocol` | A node naming a contract: `open` yields an IPC connection (an endpoint capability) instead of a file id — the kind of every leaf under `/protocol`. *(Being added; see protocol-namespace.md.)* |
|
||||
|
||||
Note the layering: `protocol` says what *opening the name* does (you get a
|
||||
conversation); `character_device`/`block_device` say what *read and write
|
||||
mean* on a node a provider serves you. The two compose — `/protocol/console`
|
||||
is a protocol node in the registry, and the node opened over that connection
|
||||
reports `character_device`, which is what gives it stream semantics. Only
|
||||
`socket` is retired (its value stays reserved for wire stability): a named
|
||||
rendezvous point is exactly what a protocol node is.
|
||||
|
||||
## What is deliberately absent
|
||||
|
||||
There is no `/bin`, `/boot`, `/dev`, `/etc`, `/home`, `/lib`, `/mnt`, `/sbin`,
|
||||
`/srv`, `/tmp`, `/usr`, or `/var`. These encode unix history — the
|
||||
binary/library split of small disks, configuration-as-scattered-text, devices
|
||||
as magic files — that danos does not carry. A POSIX compatibility layer (the
|
||||
Python track's mini-libc) may *present* whichever of these its programs
|
||||
expect, mapped onto the real tree; the tree itself stays danos-native.
|
||||
|
||||
## Migration
|
||||
|
||||
The tree above is the specification; some code still writes the unix paths it
|
||||
replaced. The flag-day converting them:
|
||||
|
||||
| Today (in code) | Becomes | Where |
|
||||
|------------------------------------------|-------------------------------------------|-----------------------------------------------------------------|
|
||||
| `/etc/init.csv` | `/system/configuration/init.csv` | `system/services/init/init.zig` |
|
||||
| `/etc/devices.csv` | `/system/configuration/devices.csv` | `system/services/device-manager/device-manager.zig` |
|
||||
| `/var/log/...` | `/system/logs/...` | `system/services/logger/logger.zig`, the FAT server's `/var` mount |
|
||||
| `/mnt/usb` | `/volumes/usb` | `system/services/fat/fat.zig`, the fat/vfs tests |
|
||||
| `ServiceId` lookup | resolve + open under `/protocol` | every service and client; [protocol-namespace.md](../os-development/protocol-namespace.md) |
|
||||
|
||||
The boot-image builder and the on-volume directory layout move in the same
|
||||
change, so a freshly written image and the paths the services expect never
|
||||
disagree.
|
||||
@@ -6,10 +6,7 @@
|
||||
> serves it directly (the read-only /system initrd mount, via `fs_node`) or
|
||||
> redirects the caller to the owning backend's endpoint plus the rewritten
|
||||
> mount-relative path — after which the client speaks THIS protocol to the
|
||||
> backend, unchanged. Since P4a the contract is expressed through
|
||||
> `envelope.Define` (docs/os-development/protocol-namespace.md), so every
|
||||
> packet begins with the universal 16-byte prefix and the open-node id rides
|
||||
> in it. The Zig source of truth is `library/protocol/vfs/vfs-protocol.zig`
|
||||
> backend, unchanged. The Zig source of truth is `library/protocol/vfs/vfs-protocol.zig`
|
||||
> (the `vfs-protocol` module), whose unit test pins a sample of the sizes
|
||||
> and values below. This page is the **language-neutral wire specification**
|
||||
> of that contract — what a Rust or C client implements ([vdso.md](../os-development/vdso.md)
|
||||
@@ -25,17 +22,12 @@ also hands back the path rewritten relative to the mount — not from a
|
||||
registry lookup. (Service id 1, the old userspace router, is retired.)
|
||||
|
||||
- A message is at most **256 bytes** (`message_maximum`).
|
||||
- Every packet begins with the 16-byte **envelope prefix**
|
||||
([protocol-namespace.md](../os-development/protocol-namespace.md)): a
|
||||
`Header` on a request, a `Status` on a reply. The prefix is **folded, not
|
||||
stacked** — the verb and the object being addressed live in it, and no
|
||||
request or reply below repeats either.
|
||||
- A request is the header, then the verb's own fixed part (0–16 bytes), then
|
||||
an inline tail of at most **224 bytes** (`maximum_payload`) — a path, or
|
||||
write bytes. There is no multi-message request: paths and single
|
||||
reads/writes must fit, and larger transfers loop (see *read* / *write*).
|
||||
- A reply is the status, then the verb's own fixed part, then an inline tail
|
||||
— read bytes, or a directory entry's name.
|
||||
- A request is a fixed 32-byte **Request** header followed by an inline
|
||||
payload of at most **224 bytes** (`maximum_payload`) — a path, or write
|
||||
bytes. There is no multi-message request: paths and single reads/writes
|
||||
must fit, and larger transfers loop (see *read* / *write*).
|
||||
- A reply is a fixed 24-byte **Reply** header followed by an inline payload —
|
||||
read bytes, a `FileStatus`, or a `DirectoryEntry`.
|
||||
- All integers are **little-endian**; layouts are C layout for x86-64
|
||||
(`extern struct`), offsets given below so nothing need be inferred.
|
||||
|
||||
@@ -45,108 +37,91 @@ With clients holding backend node ids directly, a backend records each open
|
||||
handle's owner and sweeps a dead client's handles via the published process
|
||||
exit events.
|
||||
|
||||
## Request header — 16 bytes
|
||||
|
||||
The envelope's `Header`, identical in every danos protocol:
|
||||
## Request header — 32 bytes
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
| 0 | 4 | `operation` | an **Operation** value (below); 0–15 are the reserved universal verbs |
|
||||
| 0 | 4 | `operation` | an **Operation** value (below) |
|
||||
| 4 | 4 | — | padding |
|
||||
| 8 | 8 | `target` | **the open-node id** from a prior `open`; 0 for `open` itself and the path-based verbs |
|
||||
| 8 | 8 | `node` | the server-side open-node id from a prior `open`; 0 for path-based operations |
|
||||
| 16 | 8 | `offset` | byte position for read/write; entry index (cursor) for readdir; else 0 |
|
||||
| 24 | 4 | `len` | payload length for path/write operations; requested byte count for read |
|
||||
| 28 | 4 | `flags` | open flags (below); else 0 |
|
||||
|
||||
## Reply header — 16 bytes
|
||||
|
||||
The envelope's `Status`:
|
||||
## Reply header — 24 bytes
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
| 0 | 4 | `status` | **0 = success**, negative = failure (signed) |
|
||||
| 4 | 4 | — | padding |
|
||||
| 8 | 4 | `len` | reply bytes following this header: the verb's fixed part plus its tail |
|
||||
| 12 | 4 | — | padding |
|
||||
| 8 | 8 | `node` | the new open-node id (for `open`); else 0 |
|
||||
| 16 | 4 | `len` | reply payload length in bytes |
|
||||
| 20 | 4 | — | padding |
|
||||
|
||||
A failing backend replies with the status alone (`len` = 0) and no fixed
|
||||
part, and that reply reaches the client directly — there is no party between
|
||||
them on the wire. (Kernel-served paths produce no wire replies at all:
|
||||
`fs_resolve`/`fs_node` failures are syscall register statuses.) The errno
|
||||
vocabulary is the kernel's, continued by the envelope: `ENOENT` = 4 is what a
|
||||
backend answers for anything it cannot find or cannot do, `ENOSYS` = 10 for a
|
||||
verb it does not implement, `EPROTO` = 11 for a packet shorter than the verb
|
||||
it names. Clients must treat *any* negative status as failure rather than
|
||||
matching a particular one.
|
||||
On failure the backend replies `status = -1`, and that reply reaches the
|
||||
client directly — there is no party between them on the wire. (Kernel-served
|
||||
paths produce no wire replies at all: `fs_resolve`/`fs_node` failures are
|
||||
syscall register statuses.) A richer errno vocabulary is future work —
|
||||
clients must treat *any* negative status as failure, not match on -1.
|
||||
|
||||
## Operations
|
||||
|
||||
Values number from 16 (`first_protocol_operation`) in declaration order, and
|
||||
are frozen once shipped. Values 0–15 are the envelope's reserved universal
|
||||
verbs, which mean the same thing at every provider in the system: `describe`
|
||||
(0) answers the protocol's name and version and is implemented by the
|
||||
envelope itself, so every backend answers it. A verb outside this table is
|
||||
answered `-ENOSYS`; it is never a safety check any more, because the
|
||||
dispatch compares numbers rather than decoding an enum.
|
||||
Values are append-only and never renumbered (the same evolution rule every
|
||||
danos protocol follows). Send only values from this table: the shipped server
|
||||
decodes the operation into an exhaustive enum, so an out-of-range value is
|
||||
not answered with a `status = -1` reply — it trips a safety check in safe
|
||||
builds and is undefined otherwise. (The `-1` replies cover recognised but
|
||||
refused operations, such as `mount` sent to a backend.)
|
||||
|
||||
Each row's *request* and *reply* name the bytes **after** the 16-byte prefix.
|
||||
|
||||
| value | operation | request | tail | reply | reply tail |
|
||||
|------:|-----------|---------|------|-------|-----------|
|
||||
| 16 | `open` | `flags` (4 bytes, below) | the path | `node` (8 bytes) = the open-node id | — |
|
||||
| 17 | `close` | — | — | — | — |
|
||||
| 18 | `read` | `offset` (8), `len` (4) = wanted count | — | — | the bytes read; `Status.len` 0 at end of file |
|
||||
| 19 | `write` | `offset` (8), `len` (4) = count | the bytes | `count` (4) = bytes accepted (may be short — loop) | — |
|
||||
| 20 | `status` | — | — | **FileStatus** (24 bytes) | — |
|
||||
| 21 | `readdir` | `cursor` (8) | — | one **DirectoryEntry** (16 bytes) | the name |
|
||||
| 22 | `mount` | — | the mount-point path; the backend endpoint rides as the call's **capability** | — | — |
|
||||
| 23 | `unmount` | — | the mount-point path | — | — |
|
||||
| 24 | `mkdir` | — | the path | — | — |
|
||||
| 25 | `unlink` | — | the path | — | — |
|
||||
| 26 | `rename` | — | old path, one `0x00`, new path | — | — |
|
||||
| 27 | `bind` | — | the contract name; the provider's endpoint rides as the call's **capability** | — | — |
|
||||
| value | operation | request payload | reply |
|
||||
|------:|-----------|-----------------|-------|
|
||||
| 0 | `open` | the path (`len` = its length), `flags` as below | `node` = open-node id |
|
||||
| 1 | `close` | — (`node` set) | status only |
|
||||
| 2 | `read` | — (`node`, `offset`, `len` = wanted count) | `len` bytes read, payload = the bytes; `len` 0 at end of file |
|
||||
| 3 | `write` | the bytes (`node`, `offset`, `len` = count) | `len` = bytes accepted (may be short — loop) |
|
||||
| 4 | `status` | — (`node` set) | payload = **FileStatus** (24 bytes) |
|
||||
| 5 | `readdir` | — (`node` = a directory, `offset` = cursor) | payload = one **DirectoryEntry** + name; `len` 0 at end |
|
||||
| 6 | `mount` | the mount-point path; the backend endpoint rides as the call's **capability** | status only |
|
||||
| 7 | `unmount` | the mount-point path | status only |
|
||||
| 8 | `mkdir` | the path | status only |
|
||||
| 9 | `unlink` | the path | status only |
|
||||
| 10 | `rename` | old path, one `0x00`, new path (`len` = total) | status only |
|
||||
|
||||
Notes per operation:
|
||||
|
||||
- **open** — the path is the mount-relative path `fs_resolve` handed back
|
||||
(absolute-shaped: `/notes.txt` under fat's `/volumes/usb` mount). Bare names
|
||||
(absolute-shaped: `/notes.txt` under fat's `/mnt/usb` mount). Bare names
|
||||
(`greeting`) resolve nowhere — the flat ramfs is retired, and `fs_resolve`
|
||||
refuses non-absolute paths. The returned `node` is the *backend's* own
|
||||
open-node id: with the router in the kernel there is no forwarding table,
|
||||
and clients hold backend ids directly (see *Lifetimes and trust*). Every
|
||||
later packet carries it in `Header.target` — the path is spoken once, here,
|
||||
and integers do the rest.
|
||||
and clients hold backend ids directly (see *Lifetimes and trust*).
|
||||
- **read / write** — a single exchange moves at most 224 bytes
|
||||
(`maximum_payload`); the client loops, advancing its own offset by what
|
||||
came back, until done (read) or the slice is written (write). A `write`
|
||||
reply shorter than requested is progress, not an error; a count of 0 means
|
||||
no forward progress — stop rather than spin.
|
||||
- **readdir** — `cursor` is the **entry index**, not a byte position. Each
|
||||
call returns exactly one entry; the client increments the cursor by 1. **A
|
||||
`name_len` of 0 is end-of-directory** — the reply's own length cannot say
|
||||
so, because the envelope always sends the fixed reply part. The directory
|
||||
must have been opened with the `directory` flag.
|
||||
(`maximum_payload`); the client loops, advancing `offset` by the returned
|
||||
`len`, until done (read) or the slice is written (write). A `write` reply
|
||||
shorter than requested is progress, not an error; a `len` of 0 means no
|
||||
forward progress — stop rather than spin.
|
||||
- **readdir** — `offset` is a **cursor: the entry index**, not a byte
|
||||
position. Each call returns exactly one entry; the client increments the
|
||||
cursor by 1. A reply with `len` 0 is end-of-directory. The directory must
|
||||
have been opened with the `directory` flag.
|
||||
- **mount / unmount** — RETIRED from the wire: mounting is the `fs_mount`
|
||||
syscall now (a filesystem server passes its endpoint handle; possession is
|
||||
the capability, exactly the trust of the old cap-passing op). The verb
|
||||
the capability, exactly the trust of the old cap-passing op). The op
|
||||
numbers stay reserved. Mount-prefix semantics are unchanged: prefixes
|
||||
match at path boundaries only (`/volumes/usb` never captures
|
||||
`/volumes/usbextra`), the longest matching prefix wins, and an optional
|
||||
backend-side rewrite prefix maps a mount into the backend's namespace (fat
|
||||
serves `/volumes/usb` from its volume root and `/system/logs` from its
|
||||
`/system/logs` subtree).
|
||||
match at path boundaries only (`/mnt/usb` never captures `/mnt/usbextra`),
|
||||
the longest matching prefix wins, and an optional backend-side rewrite
|
||||
prefix maps a mount into the backend's namespace (fat serves `/mnt/usb`
|
||||
from its volume root and `/var` from its `/var` subtree).
|
||||
- **rename** — same-directory rename only: the backend compares the old and
|
||||
new parent paths and refuses a mismatch. The client (`file_system`) refuses
|
||||
earlier when the two paths resolve to different backend endpoints, but that
|
||||
check is coarser than "one mount" — one endpoint can serve several mounts
|
||||
(fat serves `/volumes/usb`, `/system/configuration` and `/system/logs`), so
|
||||
a cross-mount rename reaches the backend and fails on its same-directory
|
||||
check.
|
||||
- **bind** — the protocol registry's claim verb, implemented only by the
|
||||
synthetic `/protocol` backend inside PID 1
|
||||
([protocol-namespace.md](../os-development/protocol-namespace.md)). A file
|
||||
backend answers `-ENOSYS`.
|
||||
(fat serves `/mnt/usb` and `/var`), so a cross-mount rename reaches the
|
||||
backend and fails on its same-directory check.
|
||||
|
||||
## Open flags
|
||||
|
||||
Bitwise OR in `open`'s `flags`, meaningful for `open` only:
|
||||
Bitwise OR in `Request.flags`, meaningful for `open` only:
|
||||
|
||||
| bit | name | meaning |
|
||||
|----:|------|---------|
|
||||
@@ -154,7 +129,7 @@ Bitwise OR in `open`'s `flags`, meaningful for `open` only:
|
||||
| 2 | `directory` | open a directory node for `readdir` rather than a file |
|
||||
| 4 | `truncate` | truncate an existing file to zero length on open (replace, don't overwrite in place) |
|
||||
|
||||
## FileStatus — 24 bytes (the `status` reply's fixed part)
|
||||
## FileStatus — 24 bytes (the `status` reply payload)
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
@@ -163,19 +138,19 @@ Bitwise OR in `open`'s `flags`, meaningful for `open` only:
|
||||
| 12 | 4 | — | padding |
|
||||
| 16 | 8 | `mtime` | modification time, Unix epoch seconds UTC; 0 if the backend keeps none |
|
||||
|
||||
## DirectoryEntry — 16 bytes + name (the `readdir` reply)
|
||||
## DirectoryEntry — 16 bytes + name (the `readdir` reply payload)
|
||||
|
||||
| offset | size | field | meaning |
|
||||
|-------:|-----:|-------|---------|
|
||||
| 0 | 4 | `kind` | a **NodeKind** value |
|
||||
| 4 | 4 | `name_len` | length of the name that follows; **0 means end of directory** |
|
||||
| 4 | 4 | `name_len` | length of the name that follows |
|
||||
| 8 | 8 | `size` | the entry's size in bytes |
|
||||
| 16 | `name_len` | name | the entry's name, not NUL-terminated |
|
||||
|
||||
## NodeKind
|
||||
|
||||
Aligned to the node-kind table in the file-system hierarchy
|
||||
(docs/file-system-development/file-system-hierarchy.md):
|
||||
Aligned to the FSH file-type table
|
||||
(docs/danos-file-system-hierarchy-FSH.md):
|
||||
|
||||
| value | kind |
|
||||
|------:|------|
|
||||
@@ -186,25 +161,9 @@ Aligned to the node-kind table in the file-system hierarchy
|
||||
| 4 | symbolic link |
|
||||
| 5 | fifo |
|
||||
| 6 | socket |
|
||||
| 7 | protocol |
|
||||
|
||||
Clients should map unknown values to *regular* rather than reject — the
|
||||
table can grow. Kind 6 (`socket`) keeps its wire value but is retired from
|
||||
the design — a named rendezvous point is exactly what a `protocol` node is,
|
||||
landed as value 7 with the protocol namespace
|
||||
(docs/os-development/protocol-namespace.md). `character_device` (stream
|
||||
semantics — the tty/console shape) and `block_device` (raw sector-addressed
|
||||
volumes, reserved) remain part of the design.
|
||||
|
||||
## An open reply may carry a capability
|
||||
|
||||
`open` rides `ipc_call`, whose reply direction can hand back an endpoint
|
||||
capability alongside the reply. A file backend never uses it — FAT
|
||||
answers with a node id and nothing else — but a **synthetic** backend does:
|
||||
opening a `protocol` node returns the provider's endpoint, and possession of
|
||||
that endpoint *is* the channel. The convention is per-backend, not
|
||||
per-operation, so a client that opens an ordinary file simply receives no
|
||||
capability, exactly as before.
|
||||
table can grow.
|
||||
|
||||
## Lifetimes and trust
|
||||
|
||||
@@ -212,24 +171,10 @@ Open-node ids live in the backend. A client that dies without closing leaks
|
||||
nothing permanently: the backend (the FAT server) subscribes to the kernel's
|
||||
published process-exit events (docs/process-lifecycle.md) and releases a dead
|
||||
client's handles. The kernel VFS root needs no sweep at all — its node tokens
|
||||
are permanent for a boot and carry no open state.
|
||||
|
||||
Ids are plain integers rather than capabilities, so the backend **scopes them
|
||||
to the caller's badge**: an open node belongs to the task that opened it, and
|
||||
every verb that names one — read, write, status, readdir, close — is answered
|
||||
only for that task. A node id is a small number drawn from a table of
|
||||
thirty-two, trivially guessable, and until this rule a backend honoured every
|
||||
client's ids from every other client
|
||||
(docs/os-development/protocol-namespace.md: *handles must be scoped per
|
||||
client — validated against the badge, or drawn from a per-client id
|
||||
namespace*).
|
||||
|
||||
The refusal is deliberately **identical to absence**: a node that is somebody
|
||||
else's answers `-ENOENT`, exactly as one that was never opened, so a prober
|
||||
learns nothing about which ids are live — the same discipline the protocol
|
||||
namespace applies to a refused open. The owner is a *task*, because the badge
|
||||
is: a threaded client uses a node from the thread that opened it, which is
|
||||
already the granularity of the exit sweep that releases it.
|
||||
are permanent for a boot and carry no open state. Ids are plain integers, not
|
||||
capabilities — a backend trusts its callers with each other's ids today, which
|
||||
is acceptable while every client is part of the system image and worth
|
||||
revisiting (per-client id namespaces) before third-party binaries arrive.
|
||||
|
||||
## Evolution rules
|
||||
|
||||
@@ -237,15 +182,11 @@ What a non-Zig implementation may rely on, and what it must not:
|
||||
|
||||
- Operation values, flag bits, `NodeKind` values, and struct layouts are
|
||||
**append-only and frozen once shipped**. The unit test in
|
||||
`library/protocol/vfs/vfs-protocol.zig` pins a sample of them (the
|
||||
`DirectoryEntry` size, `NodeKind` 0–1 and 6–7, `Operation` values 16–21, 26
|
||||
and 27); this page is the full record of the frozen values.
|
||||
*The one renumbering this contract has had was the rebase onto the envelope
|
||||
(P4a), which moved every verb above the reserved range — a deliberate
|
||||
flag-day across a system with no third-party clients yet, not a precedent.*
|
||||
`library/protocol/vfs/vfs-protocol.zig` pins a sample of them (the `DirectoryEntry`
|
||||
size, `NodeKind` 0–1, `Operation` values 0, 4 and 5); this page is the
|
||||
full record of the frozen values.
|
||||
- The 256-byte message ceiling is a property of the current IPC transport,
|
||||
not a promise; clients should read `maximum_payload`-shaped limits from the
|
||||
reply lengths they actually get (loop-until-done), not hard-code 224.
|
||||
- Success is exactly 0, and the negative statuses come from one system-wide
|
||||
errno vocabulary (the kernel's, continued by the envelope) rather than from
|
||||
this protocol.
|
||||
- Negative statuses beyond -1 will appear (an errno vocabulary); success is
|
||||
exactly 0.
|
||||
|
||||
@@ -1,120 +0,0 @@
|
||||
# Communication: the four layers
|
||||
|
||||
*Design, agreed 2026-07-31. The model document — the vocabulary and layering
|
||||
every other communication document speaks.*
|
||||
|
||||
danos separates **what is said** from **how the bytes move**, so that the
|
||||
mechanism is replaceable. The shape is a network stack's, cut into four
|
||||
layers; a program only ever touches the top two.
|
||||
|
||||
```
|
||||
L3 namespace /protocol/... names establishment points protocol-namespace.md
|
||||
L2 protocol the language: packet schemas, verbs, targets the envelope, library/protocol/*
|
||||
L1 channel two ends exchanging packets and signals the client library's Channel
|
||||
L0 transport a buffer + a doorbell: moves the bytes ipc.md (kernel-ipc), later shm-ring, …
|
||||
```
|
||||
|
||||
## Vocabulary
|
||||
|
||||
| Term | Meaning |
|
||||
|---|---|
|
||||
| **protocol** | The language: which packets exist, what their fields mean, which verbs a provider answers. Defined transport-independently in a `library/protocol/*` module. |
|
||||
| **channel** | An open conversation between two processes, speaking one protocol. Established by opening a `/protocol/...` name; both ends can send and receive. |
|
||||
| **packet** | The unit a protocol transmits: a bounded, atomic header+payload. Never fragmented — if it doesn't fit, it isn't a packet; bulk data rides shared memory with a packet as the doorbell. |
|
||||
| **signal** | A payload-less poke below the packet layer: "something happened, come look." Coalescing — the count may collapse, the fact may not. |
|
||||
| **transport** | What moves the bytes of one channel: a buffer plus a doorbell. Chosen (and upgradable) at establishment, invisible above L1. |
|
||||
| **endpoint** | A termination point where a transport delivers. The kernel-ipc transport's endpoint is its kernel mailbox object. |
|
||||
|
||||
## Addressing: parties by channel, objects by target
|
||||
|
||||
There are no network-style addresses in a packet. The two questions addresses
|
||||
answer are answered at different layers:
|
||||
|
||||
- **Who am I talking to?** The **channel**, decided once at establishment.
|
||||
Opening `/protocol/input` yields a channel; every packet sent on it goes to
|
||||
the peer. Nothing to route per-packet — like TCP, where no HTTP request
|
||||
carries the server's IP.
|
||||
- **Who sent this?** Attached to every received packet **by the channel
|
||||
layer**, from identity the transport can verify — under kernel-ipc, the
|
||||
kernel-stamped badge. The sender never writes a source field, which is what
|
||||
makes source unforgeable (the property a network's spoofable source header
|
||||
lacks).
|
||||
- **Which of your things?** The packet's **`target`** field: *object*
|
||||
addressing within the already-chosen peer — the vfs protocol's node id, the
|
||||
display protocol's layer id, a block volume. `target = 0` addresses the
|
||||
provider itself; a protocol without objects never uses it.
|
||||
|
||||
`target` is how instance multiplicity stays out of the namespace. Ten USB
|
||||
sticks and the namespace still holds exactly one name, `/protocol/block`: a
|
||||
channel to the provider, `enumerate` lists the current volumes as targets, a
|
||||
`targets_changed` signal announces hotplug, and a read names its volume in
|
||||
`target`. The unix `/dev/sda`,`/dev/sdb` problem is dissolved, not renamed.
|
||||
|
||||
If a future transport genuinely routes between machines, *it* carries real
|
||||
source/destination addressing internally at L0 — the way IP runs under TCP —
|
||||
and none of it surfaces into the packet header. Protocols stay ignorant of
|
||||
distance.
|
||||
|
||||
## The transport (L0): a buffer and a doorbell
|
||||
|
||||
Strip any transport to its skeleton and the same two parts remain:
|
||||
|
||||
| Transport | Buffer | Doorbell | Status |
|
||||
|---|---|---|---|
|
||||
| **kernel-ipc** | kernel-owned mailbox (the `Endpoint`) | the scheduler (rendezvous wake) | the first transport — [ipc.md](../device-driver-development/ipc.md) |
|
||||
| **shm-ring** | user-owned shared-memory ring | a signal | exists ad hoc (display bulk); to be formalized — the unlock for the 256-byte ceiling |
|
||||
| network | NIC queue | an interrupt | someday, when danos networks |
|
||||
|
||||
Transports differ in their **properties**, which the channel layer exposes and
|
||||
the protocol layer may depend on:
|
||||
|
||||
- **packet ceiling** — kernel-ipc: 256 bytes request/reply, 64 pushed. An
|
||||
shm-ring's ceiling is its slot size. Kernel-ipc's 256 is the *floor* every
|
||||
protocol may assume everywhere.
|
||||
- **synchrony** — kernel-ipc's call is a rendezvous: natural backpressure, no
|
||||
queue to size. An asynchronous transport buffers, so a channel over one
|
||||
needs explicit flow control. Backpressure is a *transport property*, not a
|
||||
channel guarantee — protocols that rely on it say so.
|
||||
- **droppability** — pushed event packets may drop when a ring fills;
|
||||
request/reply may not.
|
||||
- **capability carriage** — **only kernel-ipc can move a capability.**
|
||||
Handles are kernel objects; a user-space ring cannot transfer one. So
|
||||
kernel-ipc is always the *establishment and control* transport — channels
|
||||
are born on it, capabilities ride it — even when a channel's data is
|
||||
negotiated onto something fatter.
|
||||
|
||||
That negotiation is the upgrade path: a channel starts on kernel-ipc; the
|
||||
protocol's handshake may then delegate a shared-memory region (as a
|
||||
capability, over kernel-ipc) and move its bulk traffic there. The display
|
||||
path already does exactly this by hand; formalizing it in the channel layer
|
||||
makes it every protocol's option.
|
||||
|
||||
## The channel (L1)
|
||||
|
||||
A channel has two ends, and **the ends are peers**: each may send packets,
|
||||
each may receive, each may signal. Request/reply is a *pattern* over the
|
||||
channel — a send with a correlated receive, which the kernel-ipc transport
|
||||
happens to accelerate as a single rendezvous — not the definition of it. The
|
||||
event stream (subscribe, then pushes) and the change signal (poke, then
|
||||
re-read) are the other two patterns; all three are catalogued in
|
||||
[protocol-namespace.md](protocol-namespace.md)'s wiring section.
|
||||
|
||||
The channel layer's obligations: deliver packets whole, attach the verified
|
||||
source to every receive, expose the transport's properties, and hide the
|
||||
transport's mechanics. The client library's `Channel` type is this layer made
|
||||
concrete — a program holds channels that speak protocols and never touches a
|
||||
raw handle.
|
||||
|
||||
## The protocol (L2) and the namespace (L3)
|
||||
|
||||
A protocol defines its packets through the envelope — every packet begins
|
||||
`{operation, target}`, reserved verbs (`describe`, `enumerate`, `subscribe`,
|
||||
`unsubscribe`) mean the same thing in every protocol, and `Define` checks
|
||||
every packet against the transport floor at compile time. The full treatment,
|
||||
including how names are granted, resolved, and restricted per process, is
|
||||
[protocol-namespace.md](protocol-namespace.md).
|
||||
|
||||
Establishment points are named by contract — `/protocol/display`, never
|
||||
`/protocol/ipc-1` — because the name must outlive the mechanism: a
|
||||
transport named in the namespace could never be swapped, which would defeat
|
||||
this document's premise.
|
||||
@@ -231,8 +231,8 @@ Two consequences of neutrality bind on later work:
|
||||
|
||||
- **Cross-firmware surfaces are named by domain, not firmware.** System power is
|
||||
a [`power`](power.md) protocol, not an "ACPI events" protocol: on x86 the acpi
|
||||
service binds it, on ARM a PSCI/mailbox service binds the same
|
||||
`/protocol/power`, and subscribers never learn the difference.
|
||||
service registers it, on ARM a PSCI/mailbox service registers the same
|
||||
`ServiceId.power`, and subscribers never learn the difference.
|
||||
- **Identity must widen before the fdt service exists.** `DeviceDescriptor`'s
|
||||
8-byte `hid` holds an EISA id but cannot hold an FDT `compatible` string
|
||||
(`"brcm,bcm2835-aux-uart"`); the identity field grows before the ARM path can
|
||||
|
||||
@@ -24,9 +24,8 @@ EFI/BOOT/BOOTX64.efi <- the "removable media" default for x86-64
|
||||
```
|
||||
|
||||
The boot volume is **FHS-shaped** (see the repository-layout note in
|
||||
[README.md](../README.md)): the root `build.zig` compiles `boot/efi.zig` (built
|
||||
for the `uefi` target) and `build/images.zig` places it at
|
||||
`EFI/BOOT/BOOTX64.efi` — the one path UEFI firmware fixes — and lays
|
||||
[README.md](../README.md)): `build.zig` installs `boot/efi.zig` (built for the `uefi`
|
||||
target) at `EFI/BOOT/BOOTX64.efi` — the one path UEFI firmware fixes — and lays
|
||||
the rest out by FHS path: the kernel at `system/kernel`, init at
|
||||
`system/services/init`, the pre-packed boot capsule at `boot/system.img`
|
||||
([system-image.md](system-image.md)).
|
||||
|
||||
@@ -15,9 +15,9 @@ Where the events come from is firmware-specific — on x86 they ride the ACPI SC
|
||||
([acpi.md](acpi.md)); on a Raspberry Pi they would come from PSCI or a mailbox.
|
||||
What subscribers want is not: *the lid closed* means the same thing regardless of
|
||||
who noticed. So the surface is **domain-named**. There is a `power-protocol`
|
||||
module and a contract named `/protocol/power`; on x86 the **acpi service**
|
||||
binds it, and on ARM a PSCI/mailbox service will bind the *same* name.
|
||||
Subscribers open `/protocol/power` and never learn which firmware they
|
||||
module and a well-known `ServiceId.power = 5`; on x86 the **acpi service**
|
||||
registers it, and on ARM a PSCI/mailbox service will register the *same* id.
|
||||
Subscribers call `ipc.lookup(.power)` and never learn which firmware they
|
||||
are on — the neutrality the whole [discovery](discovery.md) migration exists to
|
||||
preserve, carried one layer up into a running-system surface.
|
||||
|
||||
@@ -28,43 +28,28 @@ unchanged.
|
||||
## The protocol
|
||||
|
||||
The `power-protocol` module ([library/protocol/power/power-protocol.zig](../../library/protocol/power/power-protocol.zig))
|
||||
is defined through the [envelope](protocol-namespace.md), so every packet begins
|
||||
with the folded `Header`. `Header.target` is unused in both directions: the
|
||||
provider is the only object either side addresses.
|
||||
follows the vfs-protocol pattern — extern-struct messages, a version, reserved
|
||||
fields. Three operations:
|
||||
|
||||
| Direction | Packet | Purpose |
|
||||
| Direction | Operation | Purpose |
|
||||
|---|---|---|
|
||||
| subscriber → service | `subscribe` (reserved verb 2) | receive published events; the subscriber's endpoint rides as the call's **capability** (the input/device-manager pattern) |
|
||||
| init → service | `shutdown` (verb 16) | orderly shutdown's last step: enter S5 (soft off) |
|
||||
| service → subscriber | one event per kind | a published `Notice`, `ipc_send`t as a buffered packet (never sent *to* the service) |
|
||||
|
||||
`subscribe` is not one of this protocol's own verbs: a synchronous call whose
|
||||
attached capability is the subscriber's endpoint is exactly what the envelope's
|
||||
reserved `subscribe` means everywhere, so power adopts it wholesale. And no
|
||||
packet carries a version — the reserved `describe` verb is the version handshake,
|
||||
asked once at connect time rather than out of every packet's budget.
|
||||
| subscriber → service | `subscribe` | receive published events; the subscriber's endpoint rides as the call's **capability** (the input/device-manager pattern) |
|
||||
| init → service | `shutdown` | orderly shutdown's last step: enter S5 (soft off) |
|
||||
| service → subscriber | `event` | a published `EventMessage`, delivered as a buffered message (never sent *to* the service) |
|
||||
|
||||
Events are published, not polled: like the input service, the service holds
|
||||
subscriber endpoints as capabilities and `ipc_send`s each event as a buffered
|
||||
packet, so a slow or dead subscriber can never wedge the source. The table, the
|
||||
reserved `subscribe`/`unsubscribe` verbs and the fan-out are the **service
|
||||
harness's** (`service.Subscribers`), shared with input and the device manager, so
|
||||
the acpi service's own code is the ACPI half only — and a subscriber that dies is
|
||||
now swept on its exit notification, where before this service had no sweep at
|
||||
all. **The kind is the packet's operation** — one declared event per named kind, exactly as the
|
||||
input service delivers one per device class — so a subscriber reads *what
|
||||
happened* out of the header rather than out of a tag inside the payload. The
|
||||
message, so a slow or dead subscriber can never wedge the source. The event
|
||||
vocabulary is hardware-neutral:
|
||||
|
||||
- `power_button` (event 16) — the button was pressed (a fixed ACPI event on x86).
|
||||
- `lid` (17), `ac` (18), `battery` (19) — the named GPE-driven events.
|
||||
- `notify` (20) — a device notification that maps to none of the above; its
|
||||
`code` (the ACPI `Notify` argument) and the notifying device's `hid` say which
|
||||
device and what happened.
|
||||
- `power_button` — the button was pressed (a fixed ACPI event on x86).
|
||||
- `lid`, `ac`, `battery` — the named GPE-driven events.
|
||||
- `notify` — a device notification that maps to none of the above; its `code`
|
||||
(the ACPI `Notify` argument) and the notifying device's `hid` say which device
|
||||
and what happened.
|
||||
|
||||
The payload every one of them carries is a `Notice`: `code` plus an 8-byte `hid`,
|
||||
so a generic `notify` is fully described without a second round trip, and the
|
||||
four named kinds leave both fields zero because the verb already said it all.
|
||||
An `EventMessage` carries the `event` tag plus `code` and an 8-byte `hid`, so a
|
||||
generic `notify` is fully described without a second round trip.
|
||||
|
||||
**`shutdown` is authority, not information.** It is the only operation that
|
||||
*does* something irreversible, so it is gated: the contract is that only init
|
||||
@@ -73,10 +58,7 @@ sequence over everything else. The acpi service implements this as a **soft
|
||||
gate** — it honors `shutdown` only from a process that is a *subscriber*, and
|
||||
init is the one subscriber. That stands in for "only the system supervisor may
|
||||
power off" without hard-coding a pid, so it still holds under tests where PID 1
|
||||
is not init. The question is asked of the harness's table now
|
||||
(`Subscribers.has(sender)`), which is why the harness exposes it: the gate is
|
||||
unchanged, including the badge being the whole of it — the badge is
|
||||
kernel-stamped, so nothing inside a packet can claim to be init.
|
||||
is not init.
|
||||
|
||||
## Orderly shutdown
|
||||
|
||||
|
||||
@@ -188,15 +188,9 @@ zombie state or privileged snooping:
|
||||
state by all along is the id the exit event carries.
|
||||
|
||||
Subscription, not broadcast-to-everyone: only processes that asked receive
|
||||
events, the kernel keeps a bounded subscriber table (sixteen — a normal boot
|
||||
already fields six, since this is what *every* provider with per-client state
|
||||
releases on), and delivery is the same non-blocking coalescing notification as
|
||||
everything else — a dying process never waits on its mourners.
|
||||
A service does not usually write the sweep itself: the shared service harness
|
||||
subscribes for it and drops a dead task's event subscriptions
|
||||
(`service.Subscribers`), and a provider adds its own handler only for state the
|
||||
harness knows nothing about — open files, layers, device tokens.
|
||||
Subscribing is ungated, like `process_enumerate`: what is
|
||||
events, the kernel keeps a bounded subscriber table, and delivery is the same
|
||||
non-blocking coalescing notification as everything else — a dying process never
|
||||
waits on its mourners. Subscribing is ungated, like `process_enumerate`: what is
|
||||
running (and dying) is not a secret between cooperating processes. Subscribers
|
||||
do not receive the exit reason — the filesystem server does not care *why*
|
||||
the client died.
|
||||
@@ -291,10 +285,6 @@ callbacks (`on_terminate`, `on_reload`) for programs that want defaults.
|
||||
|
||||
`service` owns the `replyWait` loop and folds every event source — signals,
|
||||
child exits, protocol messages — into callbacks, with the vocabulary's defaults:
|
||||
it also owns the **subscriber side** of any protocol that declares events
|
||||
(`service.Subscribers`: the table, the reserved `subscribe`/`unsubscribe` verbs,
|
||||
the fan-out, and the sweep on a subscriber's published exit), so every event
|
||||
stream in the system behaves identically.
|
||||
`terminate` returns from the loop (clean exit), the common `ping` is answered automatically,
|
||||
`reload` is ignored unless overridden. One loop, no locking, nothing reentrant. A
|
||||
service author writes domain logic; the lifecycle contract is satisfied by the
|
||||
|
||||
@@ -1,474 +0,0 @@
|
||||
# The protocol namespace
|
||||
|
||||
*Design, agreed 2026-07-31. Supersedes the `ServiceId` registry. P1–P3 of the
|
||||
migration plan at the end have landed (the envelope, the registry and the
|
||||
`ServiceId` flag-day, and restriction stage one); P4 and P5 are the remaining
|
||||
work list.*
|
||||
|
||||
How a program finds, connects to, and is restricted from the things it talks to.
|
||||
Three ideas, kept deliberately separate:
|
||||
|
||||
1. **Naming** — a path under `/protocol` names a *contract*, not a service.
|
||||
2. **Access** — resolving that path yields an endpoint *capability*; what a process
|
||||
cannot resolve, it cannot reach.
|
||||
3. **Transport** — unchanged: packets over channels, moved by whichever
|
||||
transport the channel rides (kernel-ipc first).
|
||||
This document is layers **L3** (the namespace) and **L2** (the protocol
|
||||
and its envelope) of the communication stack;
|
||||
[communication.md](communication.md) owns the model and the vocabulary
|
||||
(*protocol* the language, *channel* the conversation, *packet* the
|
||||
transmitted unit, *signal* the payload-less poke, *transport* the
|
||||
replaceable mechanism), and
|
||||
[ipc.md](../device-driver-development/ipc.md) is the first transport.
|
||||
|
||||
## Why ServiceId has to go
|
||||
|
||||
Today a service calls `ipc_register(service_id, endpoint)` and a client calls
|
||||
`ipc_lookup(service_id)`, where `ServiceId` is a compile-time enum in `abi.zig`
|
||||
backed by a flat 16-slot table in the kernel. Three defects, in rising order:
|
||||
|
||||
- **Static.** The id space is baked into the ABI at compile time. A third-party
|
||||
program can never introduce a service; the one place danos is *less* dynamic
|
||||
than its own design.
|
||||
- **Ungated.** `ipc_register` is callable by any process and *replaces* an
|
||||
existing registration. Any process can hijack `.fat` or `.display` and
|
||||
impersonate it. `ipc_lookup` is equally ambient.
|
||||
- **Unrestrictable.** Because lookup is a syscall available to everyone, there is
|
||||
no point at which "this process may not talk to the display" can be enforced.
|
||||
Any future file-access restriction would be bypassable by speaking to the FAT
|
||||
server directly.
|
||||
|
||||
## Naming: contracts, not services
|
||||
|
||||
`/protocol/<name>` names a protocol — the contract a conversation follows — and
|
||||
resolving it connects you to whatever process currently provides that contract.
|
||||
The client never cared *which* binary answers; it cares that its messages are
|
||||
understood. Naming the contract makes that explicit, and buys:
|
||||
|
||||
- **Swappable providers.** Replace the display server; `/protocol/display`
|
||||
routes to the new one; clients notice nothing.
|
||||
- **Test fakes.** Spawn a program whose namespace wires `/protocol/display` to a
|
||||
mock. The name promises the protocol; the mock speaks it.
|
||||
- **One vocabulary.** The names mirror `library/protocol/`: a program imports
|
||||
the `display-protocol` module, then opens `/protocol/display`. What you
|
||||
compiled against and what you ask the namespace for are the same word.
|
||||
|
||||
A leaf names one contract — kebab-case, full words, matching the
|
||||
`library/protocol/` module that defines its wire format — and related
|
||||
contracts group into directories: `/protocol/networking/ip`,
|
||||
`/protocol/networking/bluetooth`. Directories organize *contracts only*;
|
||||
they never encode addressing (see below), so a directory appears because a
|
||||
domain has several contracts, never because hardware multiplied. The module
|
||||
tree mirrors the namespace (`library/protocol/networking/ip` ↔
|
||||
`/protocol/networking/ip`), and registrar grants scope naturally to subtrees
|
||||
— an application installed at `/applications/foo` can be granted
|
||||
`/protocol/applications/foo/...` and nothing above it. `/protocol` is
|
||||
top level, beside `/system` and `/applications`, because the boundary it names
|
||||
is spoken on both sides: applications talk to protocols as much as the OS does
|
||||
(see [file-system-hierarchy.md](../file-system-development/file-system-hierarchy.md)).
|
||||
|
||||
**Addressing lives inside the protocol, never in the path.** Which volume, which
|
||||
layer, which input device — that is a destination field in the messages, the way
|
||||
TCP carries a destination address, and the way danos protocols already work (the
|
||||
display protocol multiplexes layer ids; the vfs protocol addresses node ids).
|
||||
The namespace answers exactly one question — *may this process speak this
|
||||
protocol at all* — so `/protocol/block` is one name no matter how many disks are
|
||||
attached. The source address is never in the message either: it is the IPC
|
||||
badge, stamped by the kernel per message, unforgeable — a property TCP's source
|
||||
address does not have.
|
||||
|
||||
`/system/devices` (the device inventory) stays purely informational: facts for
|
||||
diagnosis, never a routing mechanism. Unix conflated the two in `/dev`; danos
|
||||
does not. You *read about* hardware in `/system/devices`; you *talk to* it
|
||||
through `/protocol`.
|
||||
|
||||
## Resolution: a protocol node in the VFS
|
||||
|
||||
The kernel VFS router already does the hard part: `fs_resolve` matches a mount
|
||||
prefix and installs the backend's endpoint capability in the caller's handle
|
||||
table. The registry is just a backend mounted at `/protocol` — ring 3, like FAT.
|
||||
Connecting is a normal vfs-protocol `open` with one twist in the reply:
|
||||
|
||||
```
|
||||
client kernel router registry backend
|
||||
│ fs_resolve("/protocol/display") │
|
||||
│──────────────────────────▶│ prefix match: /protocol │
|
||||
│◀── registry endpoint ─────│ (capability installed) │
|
||||
│ vfs open("display") ──────────────────────────────────────▶│
|
||||
│◀───────────────── Reply + capability = provider endpoint ──│
|
||||
│ ipc_call(provider, display-protocol messages...) │
|
||||
```
|
||||
|
||||
Both capability moves use machinery the kernel already has: request-direction
|
||||
and reply-direction `send_cap` on `call`/`replyWait`. The vfs protocol needs two
|
||||
additions, both append-only:
|
||||
|
||||
- `NodeKind.protocol` — a node that names a contract; its `open` establishes
|
||||
a **channel** (delivered as an endpoint capability) instead of returning a
|
||||
file id. The node is the protocol, the channel is the conversation, and the
|
||||
addressing inside the packets decides where within the provider each one
|
||||
lands. `readdir` over `/protocol` lists protocol nodes like any others, so
|
||||
the tree stays browsable for diagnosis.
|
||||
- The convention that an `open` reply may carry a capability. File backends
|
||||
(FAT) never use it; synthetic backends (the registry, later the device
|
||||
inventory) do.
|
||||
|
||||
The path lookup happens once, at connect time. The hot path — `ipc_call` on the
|
||||
cached endpoint — is untouched. A provider crash turns the cached endpoint dead
|
||||
(`-EPEER`), and the client's recovery is to re-resolve: the restart story falls
|
||||
out of the naming layer for free.
|
||||
|
||||
## Registration: the registrar, held by init
|
||||
|
||||
The registry backend is **init**. It is already PID 1, already spawns every
|
||||
service from its manifest, and already holds the supervision link to each — it
|
||||
is the process that *knows* which binary is which. (If init grows
|
||||
uncomfortable, the same design lifts into a dedicated registry service that
|
||||
init spawns first and delegates to; nothing below changes.)
|
||||
|
||||
- **Binding.** A service creates its endpoint and sends the registry a `bind`
|
||||
request with the protocol name as payload and the endpoint attached as the
|
||||
call's capability.
|
||||
- **Authorization.** Init's manifest gains a column: the protocols each spawned
|
||||
binary may bind. A `bind` from any process not granted that name is refused
|
||||
(`-EPERM`) — the badge identifies the caller, the supervision records map
|
||||
badge to binary. This is the registrar authority; it never leaves init.
|
||||
- **Collision is an error.** A name already bound refuses a second bind — never
|
||||
last-writer-wins. When a provider dies, init (its supervisor) unbinds its
|
||||
names; the restarted instance binds again.
|
||||
- **Provenance.** The registry records name → task id → binary path, so a
|
||||
diagnostic listing answers "who serves this?" at a glance:
|
||||
|
||||
```
|
||||
/protocol/display pid 12 /system/services/display
|
||||
/protocol/input pid 7 /system/services/input
|
||||
```
|
||||
|
||||
`ipc_register` and `ipc_lookup` retire; the `ServiceId` enum leaves `abi.zig`.
|
||||
The kernel keeps one residual rule: `/protocol` becomes a reserved prefix like
|
||||
`/system` — `fs_mount` refuses to shadow it, and init's boot-time mount is the
|
||||
only one it will ever hold. (Full gating of `fs_mount` is a separate item on
|
||||
the security track; the reserved prefix closes the hole for this namespace
|
||||
without waiting for it.)
|
||||
|
||||
## Restriction: per-process namespaces, not ACLs
|
||||
|
||||
danos has no users and no principals, deliberately. Restriction is therefore
|
||||
**delegation**: what a process may open is decided by whoever spawned it, and
|
||||
enforcement is absence — a protocol you cannot resolve does not exist for you.
|
||||
"Permission denied" and "not found" are the same answer, which is the same
|
||||
discipline the device layer already follows: the claim is the capability; here,
|
||||
the resolvable name is the capability.
|
||||
|
||||
Two stages, deliberately ordered so the useful half lands first:
|
||||
|
||||
**Stage one — the registry filters by badge.** Init is both the spawner and the
|
||||
registry, so its manifest already knows which binary may *open* which protocols
|
||||
(a second manifest column, beside the bind grants). An `open` from a process
|
||||
whose binary is not granted that protocol is refused. No new kernel mechanism
|
||||
at all; the display driver's view can be narrowed to nothing, a future
|
||||
downloaded application's to `display` and `input`, today.
|
||||
|
||||
**Stage two — spawn passes the namespace.** `spawn` gains an initial
|
||||
capability: the child's connection to *its* registry view, chosen by the
|
||||
spawner. A newly spawned process starts with an empty handle table and this one
|
||||
handle — its world is whatever its parent wired in. This removes the last
|
||||
ambient reach (`fs_resolve` finding `/protocol` globally), lets any supervisor
|
||||
— not just init — narrow or fake a child's view (an application launcher
|
||||
granting an app only what its manifest declares; a test harness substituting
|
||||
every provider), and composes down the supervision tree. Stage one's manifest
|
||||
column becomes the *content* of the view init builds, so nothing is thrown
|
||||
away.
|
||||
|
||||
### A worked example: the microphone prompt
|
||||
|
||||
The scenario stage two exists for: an application opens
|
||||
`/protocol/audio-input`, and the user should be asked. The supervisor is an
|
||||
ordinary user process — an application launcher — and the flow needs no new
|
||||
security concepts:
|
||||
|
||||
1. The launcher spawned the app with a namespace channel that terminates at
|
||||
**the launcher itself**. The app's whole world is a conversation with its
|
||||
supervisor.
|
||||
2. The app's `open("audio-input")` packet lands in the launcher,
|
||||
badge-stamped. The launcher spawned the app, so badge → binary path
|
||||
(`/applications/foo`) is its own supervision record — "remember my choice"
|
||||
needs no identity system.
|
||||
3. Grant unknown → the launcher parks the request and shows a prompt (it is a
|
||||
user process with display access; init never does UI). Blocking an open on
|
||||
a human is architecturally fine: opens are connect-time, never hot-path.
|
||||
4. **Yes** → the launcher opens `/protocol/audio-input` in *its own*
|
||||
namespace and attaches the resulting channel to the parked reply. The app
|
||||
cannot tell a prompt happened — a consented open is indistinguishable from
|
||||
a direct one, merely slower.
|
||||
5. **No** → refuse the open, indistinguishable from "no such protocol" — or
|
||||
hand the app a **fake**: a silence-generating provider. The test-fake
|
||||
mechanism doubles as a privacy feature.
|
||||
|
||||
The capability discipline holds throughout: the launcher can only grant what
|
||||
it holds — if init never gave the launcher `audio-input`, no prompt can
|
||||
conjure it. Consent is delegation flowing down the supervision tree, never a
|
||||
global ACL edit. And the provider still sees the app's badge on every packet,
|
||||
so a coarser second check at the audio service remains possible.
|
||||
|
||||
Two mechanical requirements this scenario pins on stage two:
|
||||
|
||||
- **Parked replies.** A prompt takes seconds, and the service loop holds one
|
||||
outstanding reply today — the launcher must park request A, keep serving B
|
||||
and C, and reply to A later (by badge). The kernel already tracks owed
|
||||
replies (that is how death delivers `-EPEER`); multiple parked replies is
|
||||
the extension, in the harness and, if needed, the kernel.
|
||||
- **Granted channels are dedicated, hence revocable.** Once the app holds a
|
||||
channel capability, nobody reaches into its handle table — so a
|
||||
prompt-granted channel must be one that can be *killed*: a dedicated
|
||||
endpoint pair (or per-client session at the provider) whose death turns
|
||||
the app's capability into `-EPEER`. Revoking microphone access is then
|
||||
killing that channel, using machinery that already exists.
|
||||
|
||||
One adjacent problem, named and deferred: **trusted UI**. The prompt is only
|
||||
meaningful if the app cannot draw a convincing fake or overlay the real one —
|
||||
a display-layer question (a reserved surface for the supervisor chain), owned
|
||||
by the display track, not this one.
|
||||
|
||||
Fine-grained restriction *within* a protocol (this process may use volume A but
|
||||
not volume B) is not the namespace's job. The capability-shaped answer, when it
|
||||
is needed: the supervisor pre-opens a connection scoped to one target and passes
|
||||
that connection to the child, which never opens `/protocol/block` at all.
|
||||
Delegation again, not ACLs.
|
||||
|
||||
## The envelope: one addressing scheme for every protocol
|
||||
|
||||
Every protocol module today hand-rolls its `Request`/`Reply` with an
|
||||
`operation` first field. That convention becomes a library, so addressing is
|
||||
uniform and the rules are enforced by construction rather than by review. New
|
||||
module: **`library/protocol/envelope`** (the one protocol-layer module that is
|
||||
not itself a protocol).
|
||||
|
||||
```zig
|
||||
/// Every packet a danos protocol transmits begins with this header.
|
||||
pub const Header = extern struct {
|
||||
operation: u32, // the verb; values 0..15 are reserved universal verbs
|
||||
_padding: u32 = 0,
|
||||
/// Object addressing, never party addressing: which of the peer's
|
||||
/// objects this packet operates on — a volume, layer, node, device.
|
||||
/// 0 addresses the provider itself. Parties are addressed by the
|
||||
/// channel; the protocol defines target's meaning; the field's place
|
||||
/// and width are universal.
|
||||
target: u64 = 0,
|
||||
};
|
||||
|
||||
/// Reserved verbs, answered by every provider.
|
||||
pub const operation_describe: u32 = 0; // -> protocol name, version, target kinds
|
||||
pub const operation_enumerate: u32 = 1; // -> the current targets, one per reply page
|
||||
pub const operation_subscribe: u32 = 2; // capability = the subscriber's endpoint
|
||||
pub const operation_unsubscribe: u32 = 3;
|
||||
pub const first_protocol_operation: u32 = 16;
|
||||
|
||||
/// Every reply begins with this.
|
||||
pub const Status = extern struct {
|
||||
status: i32, // 0 or a negative errno
|
||||
_padding: u32 = 0,
|
||||
len: u32 = 0, // payload bytes following the header
|
||||
_padding2: u32 = 0,
|
||||
};
|
||||
```
|
||||
|
||||
A protocol is then *defined through* the envelope, not beside it:
|
||||
|
||||
```zig
|
||||
pub const Protocol = envelope.Define(.{
|
||||
.name = "display",
|
||||
.version = 1,
|
||||
.operations = &.{
|
||||
.{ .name = "configure_layer", .request = ConfigureLayer, .reply = void },
|
||||
.{ .name = "blit", .request = Blit, .reply = void },
|
||||
...
|
||||
},
|
||||
});
|
||||
```
|
||||
|
||||
`Define` is comptime and is where the enforcement lives:
|
||||
|
||||
- verbs are numbered automatically from `first_protocol_operation`, so no
|
||||
protocol can collide with the reserved range;
|
||||
- every packet is size-checked at compile time against the kernel-ipc floor
|
||||
— `packet_maximum` (256) for request/reply, `post_maximum` (64) for event
|
||||
packets. Ceilings are transport properties
|
||||
([communication.md](communication.md)); the floor is what every protocol
|
||||
may assume on any transport. The errors that today surface as runtime
|
||||
truncation become compile errors, and packets-never-fragment is enforced
|
||||
at the source;
|
||||
- the generated type carries encode/decode helpers and a provider-side dispatch
|
||||
table, so a provider answers `describe` automatically and unknown operations
|
||||
with `-ENOSYS` uniformly;
|
||||
- the service harness (`library/kernel/service.zig`) accepts the generated
|
||||
dispatch type, which is what makes the envelope *enforced*: a protocol that
|
||||
bypasses `Define` does not plug into the harness.
|
||||
|
||||
Universal conventions that ride on the reserved verbs:
|
||||
|
||||
- **`describe`** is the version handshake. Version lives in the handshake, not
|
||||
in every message — the 256-byte budget is too small to spend per call.
|
||||
- **`enumerate`** is how multi-target protocols expose their targets, and the
|
||||
standard `targets_changed` notification (a notify bit) tells subscribers to
|
||||
re-enumerate — arrival and removal of volumes, layers, devices all take the
|
||||
same shape. Hotplug fits the notification ring far better than a filesystem
|
||||
tree ever did.
|
||||
- **Source is the badge.** No protocol defines a "sender" field; the kernel's
|
||||
per-message badge is the only source identity, and providers key per-client
|
||||
state on it.
|
||||
|
||||
### Paths resolve once; integers do the work
|
||||
|
||||
A rule the envelope makes official: **a path appears in a conversation at most
|
||||
once — at resolve or open — and everything after it addresses integers.** The
|
||||
namespace resolves `/protocol/display` to an endpoint; a backend's `open`
|
||||
resolves a path payload to a node id; from then on every packet carries the
|
||||
integer in `target`. Integers compare in one instruction and fit the fixed
|
||||
header, and the 256-byte message budget never re-carries path strings on the
|
||||
hot path. This is already the system's shape — vfs node ids, display layer ids
|
||||
— and the envelope pins it as the required shape for every protocol.
|
||||
|
||||
Two integer identities, not to be confused:
|
||||
|
||||
- **An open handle** — what vfs `open` returns today: transient, meaningful
|
||||
only within one client's session with one provider, swept when the client
|
||||
exits. Cheap, and all a protocol usually needs. Handles must be **scoped per
|
||||
client** — validated against the badge, or drawn from a per-client id
|
||||
namespace. (Today the FAT server's node ids are guessable small integers
|
||||
honoured across clients; that hole closes with this rule.)
|
||||
- **A persistent node identity** — a unix inode number, stable across opens
|
||||
and renames. danos deliberately does not promise this, because FAT cannot
|
||||
deliver it: a FAT file's identity is its directory entry, and rename or
|
||||
truncation moves every candidate anchor. If a future filesystem or a cache
|
||||
layer needs stable identity, that is the backend's promise to make, never
|
||||
the protocol's assumption.
|
||||
|
||||
The five existing protocol modules (`vfs`, `display`, `input`, `power`,
|
||||
`block`, plus `scanout`, `usb-transfer`, `device-manager`) rebase onto the
|
||||
envelope during the migration flag-day. `input-protocol`'s subscribe/publish
|
||||
split and `vfs-protocol`'s node addressing both map cleanly (`node` and layer
|
||||
ids become `target`).
|
||||
|
||||
## Wiring: how conversations flow
|
||||
|
||||
The patterns below are channel-layer (L1) shapes; the delivery mechanics are
|
||||
the kernel-ipc transport's, described here because it is the transport every
|
||||
channel starts on. Kernel-ipc provides exactly three delivery shapes, and
|
||||
every one is unicast. An endpoint is a mailbox owned by one process — its
|
||||
creator receives; anyone holding its capability sends into it. That direction
|
||||
never reverses:
|
||||
|
||||
1. **Synchronous call** — request/reply. The kernel parks the caller and
|
||||
`replyWait` delivers the reply straight back, so the provider answers
|
||||
without holding any capability to the client. Badge-stamped, blocking, and
|
||||
the *only* shape that carries capabilities (in the request, and in the
|
||||
reply — which is how a reverse path is bootstrapped).
|
||||
2. **Asynchronous send** — an event packet pushed into the receiver's post
|
||||
ring, at most `post_maximum` (64) bytes, no reply owed, never blocks the
|
||||
sender. Strictly one-way: to be pushed to, you must first hand the pusher
|
||||
your endpoint.
|
||||
3. **Signals** — payload-less notification bits, below the packet layer,
|
||||
coalescing: "something changed, come look."
|
||||
|
||||
A bidirectional link is therefore always **a pair of endpoints**, one per
|
||||
direction, each delivered by cap-passing. Three conversation patterns are
|
||||
built from these, and the envelope names all three:
|
||||
|
||||
- **Request/response** — the synchronous call. The default, and the only
|
||||
place capabilities move.
|
||||
- **Event stream** — `subscribe` (a synchronous call whose attached
|
||||
capability is the subscriber's own endpoint), after which the provider
|
||||
pushes events asynchronously; `unsubscribe` or subscriber exit ends it.
|
||||
Listened-to, not blocked-on.
|
||||
- **Change signal** — a signal plus re-read: `targets_changed` →
|
||||
`enumerate`. For state whose truth lives with the provider.
|
||||
|
||||
**Broadcast is a provider pattern, never a kernel primitive.** The kernel
|
||||
does not know subscriber sets — a service does. The input service is the
|
||||
model: sources *publish* (a unicast call to the service), the service
|
||||
*broadcasts* (a fan-out loop of asynchronous sends over its subscriber list,
|
||||
so one dead subscriber can never stall the rest). One fan-out point per event
|
||||
domain, owned by the service that defines the event.
|
||||
|
||||
The harness owns the machinery: the subscriber table, the dead-subscriber
|
||||
sweep (via process-exit notifications), and the fan-out loop — all written by
|
||||
hand in `input.zig` today, lifted into the service harness so every protocol
|
||||
gets identical semantics. `Define` declares a protocol's events (`.events`),
|
||||
and each event type is checked against `post_maximum` at compile time,
|
||||
generalizing the assert `input-protocol` already carries.
|
||||
|
||||
**Event packets are droppable.** A slow subscriber's ring fills, and the
|
||||
provider must not block on it — so an event stream is a hint or a coalescing
|
||||
signal, never a ledger. Anything that must not be lost is either re-readable
|
||||
state (the change-signal pattern) or bulk data in shared memory with a
|
||||
packet as the doorbell, which is how the display path already works — the
|
||||
packets-never-fragment rule and this one are the same rule seen from two
|
||||
sides.
|
||||
|
||||
**Source direction (open point).** Today event sources are *clients*: an
|
||||
input driver resolves `/protocol/input` and delivers each event as a
|
||||
synchronous `publish` call — one capability, obtained by resolution, covers
|
||||
everything, and the badge tells the service exactly who each event came from.
|
||||
The inversion — the service subscribing to each driver — would require every
|
||||
driver to be individually discoverable and its endpoint ferried to the
|
||||
service, machinery whose payoff (the service choosing its sources) the
|
||||
namespace already provides more cheaply: only a process granted open on
|
||||
`/protocol/input` can publish into it. Sources stay clients for now;
|
||||
revisited at restriction stage two, when a supervisor can wire capabilities
|
||||
at spawn time.
|
||||
|
||||
## What this deliberately does not solve
|
||||
|
||||
The wider security track, for which this namespace is the foundation, not the
|
||||
whole:
|
||||
|
||||
- **File access restriction** — the point of the exercise. The same stage-two
|
||||
namespace mechanism extends from protocol names to file paths: the spawner
|
||||
decides which subtrees resolve. Designed separately once this lands.
|
||||
- `fs_mount` gating beyond the reserved prefixes; `system_spawn` gating;
|
||||
`klog_read` being world-readable; backends checking the badge on per-node
|
||||
operations (the FAT server honours node ids across clients today).
|
||||
- Kernel hardening items already noted in-tree: SMEP/SMAP and SYSRET
|
||||
canonical-RIP, now designed in [smep-smap.md](smep-smap.md).
|
||||
- Pipes/FIFOs for the POSIX layer — a byte-stream object *beside* message IPC,
|
||||
wanted by the Python track, unrelated to naming.
|
||||
- **Trusted UI** — a permission prompt an application cannot fake or overlay
|
||||
(see the microphone example). A display-track concern: the supervisor chain
|
||||
needs a reserved surface.
|
||||
|
||||
## Migration plan
|
||||
|
||||
Flag-day per phase, in the style of the DMA-capability conversion — no
|
||||
dual-stack periods, the QEMU suite green at each phase boundary.
|
||||
|
||||
**P1 — mechanics, no behavior change.** The `envelope` module with its comptime
|
||||
`Define`, unit tests; `NodeKind.protocol` and the open-reply-capability
|
||||
convention in `vfs-protocol`; existing protocols untouched.
|
||||
|
||||
**P2 — the registry.** Init serves `/protocol` (bind with manifest
|
||||
authorization, collision refusal, unbind on provider death, provenance);
|
||||
kernel reserves the `/protocol` prefix; every service converts from
|
||||
`ipc_register` to `bind`, every client from `ipc_lookup` to resolve-and-open;
|
||||
`ServiceId`, `ipc_register`, `ipc_lookup` deleted. Tests: unauthorized bind
|
||||
refused, collision refused, provider restart re-binds and a client re-resolves.
|
||||
|
||||
**P3 — restriction, stage one.** The open-grant column in init's manifest;
|
||||
registry refuses ungranted opens. Test: a fixture process denied a protocol its
|
||||
neighbour is granted.
|
||||
|
||||
**P4 — protocol rebase.** Existing protocol modules re-expressed through
|
||||
`Define`; providers move onto the generated dispatch; `describe`/`enumerate`
|
||||
answered everywhere; the conformance test fixture exercises the reserved verbs
|
||||
against every registered provider.
|
||||
|
||||
**P5 — restriction, stage two.** Spawn's initial capability; namespace views
|
||||
built by the spawner; ambient resolution of `/protocol` retired. Includes the
|
||||
two requirements the microphone example pins: **parked replies** (a
|
||||
supervisor parks an open, keeps serving, replies later by badge) and
|
||||
**dedicated, killable granted channels** (revocation = channel death →
|
||||
`-EPEER`). Scoped separately — it touches `spawn`, the loader contract, and
|
||||
every supervisor — and lands together with the file-path half of namespacing.
|
||||
|
||||
The unix-path migration ([file-system-hierarchy.md](../file-system-development/file-system-hierarchy.md#migration))
|
||||
is independent of P1–P5 and can land before or after.
|
||||
@@ -1,149 +0,0 @@
|
||||
# SMEP and SMAP — supervisor-mode hardening
|
||||
|
||||
*Design, 2026-07-31. H1 (the copy layer), H2 (SMEP), HS (the SYSRET guard) and
|
||||
H3 (SMAP) have all landed; the track is complete. Companion to
|
||||
[protocol-namespace.md](protocol-namespace.md) on the security track — this is
|
||||
the hardware half; that is the namespace half.*
|
||||
|
||||
Two CR4 bits that make the CPU refuse the two things a kernel should never do
|
||||
with user memory:
|
||||
|
||||
- **SMEP** (Supervisor Mode Execution Prevention, CR4 bit 20): instruction
|
||||
fetch in ring 0 from a page whose U/S bit says *user* → #PF. Kills the
|
||||
classic ret2usr exploit shape — a kernel bug that redirects control flow
|
||||
can no longer land in attacker-prepared user code.
|
||||
- **SMAP** (Supervisor Mode Access Prevention, CR4 bit 21): data read/write
|
||||
in ring 0 to a user page → #PF, unless `EFLAGS.AC` is set. `stac`/`clac`
|
||||
open and close deliberate access windows; danos's design needs no windows
|
||||
at all (below).
|
||||
|
||||
Detection is CPUID leaf 7, subleaf 0, EBX bit 7 (SMEP) and bit 20 (SMAP).
|
||||
Both bits are per-core state: the BSP and every AP must set them.
|
||||
|
||||
## Why, in danos terms
|
||||
|
||||
Every syscall argument is an attacker-controlled integer, and several take
|
||||
pointers. A kernel bug that dereferences a crafted pointer reads, writes, or
|
||||
executes memory of the attacker's choosing — the exact bug class the
|
||||
isolation tracks exist to prevent. SMEP/SMAP turn that class from "silent
|
||||
compromise" into "immediate, attributable #PF with a kernel RIP in the log."
|
||||
|
||||
The second benefit matters as much as the first: **SMAP is a permanent
|
||||
tripwire.** Once it is on, any *future* syscall that touches user memory
|
||||
directly — instead of going through the checked copy layer — faults the
|
||||
first time the QEMU suite runs it. The discipline stops depending on review.
|
||||
|
||||
## Where danos already stands
|
||||
|
||||
The design is closer than it looks, because the IPC layer was built right:
|
||||
|
||||
- **The copy layer is already SMAP-proof.** `copyAcross` and `copyFromUser`
|
||||
(`system/kernel/ipc-synchronous.zig:305,333`) never dereference a user
|
||||
virtual address: they walk the page tables and move bytes through the
|
||||
physmap — kernel mappings throughout. SMAP cannot object.
|
||||
- **Syscall entry already clears AC.** `SFMASK = 0x4_0700` clears IF, TF,
|
||||
DF, **AC** on every `syscall`
|
||||
(`system/kernel/architecture/x86_64/per-cpu.zig:76`). The syscall path is
|
||||
SMAP-clean from day one.
|
||||
- **The interrupt path is not.** Hardware does *not* clear AC on IDT
|
||||
delivery, and ring 3 can set AC with `popfq` — so a hostile process could
|
||||
take an interrupt with AC=1 and have the handler run with SMAP suspended.
|
||||
`isr_common` (`system/kernel/architecture/x86_64/isr.s:366`) needs a
|
||||
`clac` beside its `swapgs`.
|
||||
- **CR4 today:** the BSP inherits firmware CR4 (no kernel write anywhere);
|
||||
APs set PAE/OSFXSR/OSXMMEXCPT in `trampoline.s:62-68`. Neither path sets
|
||||
SMEP/SMAP yet, and both must.
|
||||
- **The stragglers.** Nine syscalls still dereference user pointers raw
|
||||
after a bounds check — every one is a SMAP #PF waiting to happen, and
|
||||
every one is *already* a latent kernel fault today (an unmapped-but-in-
|
||||
range user page oopses the kernel instead of failing the call). The
|
||||
verified sweep of `system/kernel/process.zig` (2026-07-31; a
|
||||
whole-kernel `@ptrFromInt` audit found no user-address dereference
|
||||
outside this file):
|
||||
|
||||
| Syscall | Raw access | Direction |
|
||||
|---|---|---|
|
||||
| `system_spawn` | name + argument blob (`:972`, `:980`) | read |
|
||||
| `fs_resolve` | path in (`:1780`), result out (`:1797`) | read + write |
|
||||
| `fs_mount` | prefix + rewrite strings (`:1864`, `:1865`) | read |
|
||||
| `fs_unmount` | prefix string (`:1883`) | read |
|
||||
| `fs_node` | read buffer out (`:1820`) | write |
|
||||
| `debug_write` | message bytes (`:1700`; read twice — memcpy `:1710` and `log.append` `:1717`) | read |
|
||||
| `klog_read` | log bytes out (`:1741`) | write |
|
||||
| `klog_status` | status struct out (`:1758`) | write |
|
||||
| `process_enumerate` | descriptor array out (`:1132`) | write |
|
||||
| `device_enumerate` | descriptor array out (`:388`) | write |
|
||||
|
||||
For the write-direction rows the `@ptrFromInt` is in process.zig but the
|
||||
stores happen in callees (`scheduler.enumerate`
|
||||
`system/kernel/scheduler.zig:1209`, `devices_broker.enumerate`
|
||||
`devices-broker.zig:136`, `log.readAt` `log.zig:209`, the vfs node calls
|
||||
`vfs.zig:257/269/289`) — converting them means bounce buffers plus
|
||||
`copyToUser` around those calls, not just editing the process.zig lines.
|
||||
(Some paths already do it right — the futex word and the device-register
|
||||
descriptor go through `copyFromUser` (`:1087`, `:924`). The write
|
||||
direction has no public helper yet, but the mechanism exists:
|
||||
`copyAcross` with a kernel source is exactly how IPC replies reach user
|
||||
buffers, so `copyToUser` is a mechanical mirror.)
|
||||
|
||||
- **One known gap inside the copy layer itself:** the walk checks presence,
|
||||
not the leaf U/S and writable bits (`ipc-synchronous.zig:20-22` flags
|
||||
this). Today that is nearly moot — the user half contains only mappings
|
||||
the kernel itself created for that process — but it must close before
|
||||
shared or copy-on-write mappings exist, and closing it is part of making
|
||||
the copy layer the single trusted door.
|
||||
|
||||
## The plan
|
||||
|
||||
**H1 — copy discipline (the real work).** A `user-memory` kernel module:
|
||||
`copyFromUser` / `copyToUser` (the missing write direction) via the physmap
|
||||
walk, with U/S and writable leaf checks closing the in-tree TODO. Convert
|
||||
the nine stragglers. This fixes the latent unmapped-page kernel fault on
|
||||
its own — it is worth doing even if SMEP/SMAP never shipped. QEMU suite
|
||||
green; no behavior change visible to correct programs.
|
||||
|
||||
**H2 — SMEP.** A leaf-7 feature probe (the kernel has per-leaf `cpuid`
|
||||
helpers in `apic.zig` to generalize); set CR4.SMEP during per-CPU bring-up
|
||||
on BSP and APs — prefer the Zig-side per-CPU init over the trampoline
|
||||
assembly, so one code path covers every core and the trampoline stays
|
||||
minimal. Audit first that ring 0 never executes user-mapped pages: kernel
|
||||
text lives in the kernel half, `jump_to_user` is kernel code, and the AP
|
||||
trampoline page is kernel-mapped — expected clean, verify before flipping.
|
||||
|
||||
**H3 — SMAP.** Add `clac` at `isr_common` entry. `clac` is #UD on CPUs
|
||||
without SMAP, so the instruction is a 3-byte NOP in the image, patched to
|
||||
`clac` at boot when CPUID advertises SMAP (one-time patch beats a
|
||||
conditional branch in the hottest path in the kernel). Then set CR4.SMAP in
|
||||
the same per-CPU init. From this point the whole QEMU suite doubles as the
|
||||
enforcement test: any missed raw dereference is a vector-14 with a kernel
|
||||
RIP and a user CR2 — loud and attributable.
|
||||
|
||||
**H4 — keep it honest.** A line in the coding standards: kernel code
|
||||
touches user memory only through `user-memory`; there is no `stac` anywhere
|
||||
in the tree, and a PR that adds one is wrong by definition. SMAP enforces
|
||||
the rule mechanically at test time.
|
||||
|
||||
Feature-gating follows the timekeeping rule (work on any VM, real Intel,
|
||||
real AMD): both bits are probed, absence is logged and tolerated — like the
|
||||
IOMMU's fail-open, the machine still boots, just unhardened. QEMU: TCG
|
||||
implements both; KVM inherits the host (Intel Ivy Bridge+ for SMEP,
|
||||
Broadwell+ for SMAP; AMD Zen+ for both). The test images should run with
|
||||
`-cpu max` so the suite always exercises the enabled paths.
|
||||
|
||||
## Adjacent, deliberately separate
|
||||
|
||||
- **SYSRET canonical-RIP hardening** (`isr.s:192-194` documents it): a
|
||||
non-canonical return RIP makes `sysretq` #GP *in ring 0* on Intel. Same
|
||||
hardening bucket, independent fix (validate RCX before `sysretq`, fall
|
||||
back to `iretq`), should ride the same branch as H2/H3 but is not
|
||||
SMEP/SMAP.
|
||||
**Status — landed 2026-08-01 (HS).** The syscall exit sign-extends the
|
||||
return RIP from bit 47 (danos is 4-level only; nothing sets CR4.LA57) and
|
||||
falls back to `iretq` when that changes it, counting each refusal for the
|
||||
`sysret-canonical` case. Ring 3 could reach it: `syscall` as the last two
|
||||
bytes of the last canonical page returns to `user_half_end`.
|
||||
- **KPTI / Meltdown-class leaks are out of scope.** SMEP/SMAP police
|
||||
architectural accesses, not speculative ones. danos runs one kernel
|
||||
mapping in every address space and accepts that on affected hardware;
|
||||
revisit only if the threat model ever includes hostile native code on
|
||||
shared machines.
|
||||
@@ -11,7 +11,7 @@ sequential pass and hands the bytes to the kernel unmodified.
|
||||
|
||||
The capsule is a *performance artifact*, not a source of truth. The boot
|
||||
volume's `/system` and `/test` file trees remain the canonical layout (see
|
||||
[file-system-hierarchy.md](../file-system-development/file-system-hierarchy.md));
|
||||
[danos-file-system-hierarchy-FSH.md](../file-system-development/danos-file-system-hierarchy-FSH.md));
|
||||
the capsule is a pre-baked snapshot of the same binaries, derived from the same
|
||||
build graph, so the running system is identical whether the loader read the
|
||||
capsule or walked the tree.
|
||||
@@ -36,11 +36,11 @@ so it need be no fancier. Little-endian throughout:
|
||||
|
||||
```
|
||||
Header magic: u32 = "DNR2" (0x32524E44), count: u32
|
||||
Entry × count name: [64]u8 (NUL-padded hierarchy path), offset: u64, len: u64
|
||||
Entry × count name: [64]u8 (NUL-padded FHS path), offset: u64, len: u64
|
||||
blobs... each entry's file bytes, at its offset within the image
|
||||
```
|
||||
|
||||
- **Names are full hierarchy paths** (`/system/services/init`), not basenames — that
|
||||
- **Names are full FHS paths** (`/system/services/init`), not basenames — that
|
||||
is what "v2" means. The 64-byte capacity matches `abi.maximum_process_name`,
|
||||
so a task named after its binary path is never truncated. Paths longer than
|
||||
63 bytes are a build error (`pack-system-image.py` rejects them).
|
||||
@@ -54,14 +54,14 @@ blobs... each entry's file bytes, at its offset within the image
|
||||
|
||||
## How it is built
|
||||
|
||||
`build.zig` maintains one `bundled` list — every user binary and its hierarchy home.
|
||||
`build.zig` maintains one `bundled` list — every user binary and its FHS home.
|
||||
Three artifacts are derived from that same list, in the same build graph, so
|
||||
they cannot drift apart:
|
||||
|
||||
1. **The tree**: each binary installed at its hierarchy path (`zig-out/system/...`
|
||||
1. **The tree**: each binary installed at its FHS path (`zig-out/system/...`
|
||||
and `zig-out/test/...`, mirrored onto the FAT boot volume by
|
||||
`tools/make-fat-image.py`).
|
||||
2. **The manifest** (`system/manifest`): the hierarchy path of every bundled binary,
|
||||
2. **The manifest** (`system/manifest`): the FHS path of every bundled binary,
|
||||
one per line — the loader's per-file fallback input.
|
||||
3. **The capsule**: `tools/pack-system-image.py` packs the same binaries into
|
||||
the v2 container, installed at `zig-out/boot/system.img` and placed on the
|
||||
@@ -103,7 +103,7 @@ the kernel (`kernel.zig`) then publishes the same bytes twice, to two
|
||||
consumers:
|
||||
|
||||
- **The process layer** (`process.zig`): `system_spawn` looks binaries up in
|
||||
the ramdisk via `Reader.find` — exact hierarchy path, or unique basename for
|
||||
the ramdisk via `Reader.find` — exact FHS path, or unique basename for
|
||||
pre-path callers — and loads them as fresh ring-3 processes. The stored path
|
||||
becomes the task's name.
|
||||
- **The VFS root** (`vfs.zig`, `setInitialRamdisk`): the image is mounted as
|
||||
@@ -111,7 +111,7 @@ consumers:
|
||||
paths, so `/system` and, when the fixtures are bundled, `/test`. Directory
|
||||
nodes are derived from the entry paths (the unique parents), so the trees
|
||||
are listable and their files readable over the normal VFS protocol — the
|
||||
boot tree every process sees comes straight out of the capsule bytes.
|
||||
FHS boot tree every process sees comes straight out of the capsule bytes.
|
||||
|
||||
The image is never copied after the handoff and never mutated: the initrd is
|
||||
immutable, which is what makes the VFS's node serving lock-free.
|
||||
|
||||
@@ -22,12 +22,11 @@ lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run,
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym abbreviations,
|
||||
kebab-case file names, no `Co-Authored-By` trailers. New user binaries are
|
||||
packages whose build.zig calls `build_support.userBinary` (with `.threaded =
|
||||
true` where a binary spawns threads) and get packed into the initial-ramdisk;
|
||||
new syscalls extend [abi.zig](../../system/abi.zig) `SystemCall` + a
|
||||
`library/kernel` wrapper; test services live beside the code they exercise and
|
||||
bind a `/protocol/test/...` name if they must be reachable.
|
||||
kebab-case file names, no `Co-Authored-By` trailers. New user binaries go through
|
||||
`addUserBinary` (with the new `threaded` flag where a binary spawns threads) and get
|
||||
packed into the initial-ramdisk; new syscalls extend [abi.zig](../../system/abi.zig)
|
||||
`SystemCall` + a `library/runtime` wrapper; test services live beside the code they
|
||||
exercise and register a `ServiceId` if they must be looked up.
|
||||
|
||||
## How to verify along the way
|
||||
|
||||
|
||||
@@ -61,10 +61,9 @@ runtime — rebuilt in lockstep — knows the mapping.
|
||||
backend would either bake danos syscall numbers into std (breaking ABI privacy and
|
||||
renumbering) or fork std to route back through the runtime — a permanent rebase
|
||||
cost that buys nothing the native type doesn't.
|
||||
2. **Our user binaries are built `single_threaded = true`** (the shared recipe in
|
||||
[build-support/build.zig](../../build-support/build.zig)), which compiles threading
|
||||
out entirely and makes atomics and TLS single-threaded. Threads need this flipped
|
||||
per binary regardless.
|
||||
2. **Our user binaries are built `single_threaded = true`** ([build.zig](../../build.zig)
|
||||
`addUserBinary`), which compiles threading out entirely and makes atomics and TLS
|
||||
single-threaded. Threads need this flipped per binary regardless.
|
||||
|
||||
So we take the *shape* of `std.Thread`, not the *type*. The cost of replicating the
|
||||
surface (spawn/join/Mutex/Condition) is small; the cost of the std type is the ABI
|
||||
@@ -237,10 +236,9 @@ see the intro). Two scoped pieces, as built:
|
||||
|
||||
### Build: multi-threaded codegen, opt-in
|
||||
|
||||
A binary opts in with `.threaded = true` in its package's
|
||||
`build_support.userBinary` call — the shared recipe in build-support then builds it
|
||||
`single_threaded = false` — so atomics
|
||||
and (later) TLS are real. Threads and atomics are unsound in a `single_threaded` image,
|
||||
A binary opts in by being added with `addThreadedUserBinary` — as `addUserBinary`,
|
||||
but the shared implementation builds it `single_threaded = false` — so atomics and
|
||||
(later) TLS are real. Threads and atomics are unsound in a `single_threaded` image,
|
||||
so a binary must opt in **before** it may call `Thread.spawn`. Everyone else
|
||||
stays single-threaded and lean.
|
||||
|
||||
@@ -270,15 +268,13 @@ stays single-threaded and lean.
|
||||
- *Handles do not cross threads.* The handle table lives on the `Task`
|
||||
([scheduler.zig](../../system/kernel/scheduler.zig)), so a handle number is meaningful
|
||||
only to the thread that created it — thread A's endpoint handle `3` is not thread B's.
|
||||
A thread that needs to reach an endpoint another thread owns opens the name
|
||||
(`channel.openEndpoint("display")`) to install its **own** handle to the same
|
||||
underlying endpoint — an ordinary client open, with no special mechanism for the
|
||||
fact that the provider happens to be this process. This is how the display's
|
||||
mouse-listener thread reaches the compositor loop's endpoint to poke it awake
|
||||
(docs/display.md).
|
||||
A thread that needs to reach an endpoint another thread owns looks it up
|
||||
(`ipc.lookup(service)`) to install its **own** handle to the same underlying endpoint.
|
||||
This is how the display's mouse-listener thread reaches the compositor loop's endpoint
|
||||
to poke it awake (docs/display.md).
|
||||
- *IPC syscalls that touch shared kernel state now serialize under the big kernel lock.*
|
||||
`create_ipc_endpoint` allocates from the kernel heap and
|
||||
mutates endpoint refcounts and handle tables. Those paths
|
||||
`create_ipc_endpoint`/`ipc_register`/`ipc_lookup` allocate from the kernel heap and
|
||||
mutate the global service registry, endpoint refcounts, and handle tables. Those paths
|
||||
were unlocked because a single-threaded process could not race itself; a multi-threaded
|
||||
one can, from two cores at once. They now take `sync.enter()` like `call`/`reply_wait`/
|
||||
`send` already did — the kernel heap has no lock of its own (heap.zig: "every kernel
|
||||
|
||||
@@ -137,15 +137,15 @@ Grouped as `abi.zig` groups them:
|
||||
| process | `danos_exit`, `danos_yield`, `danos_sleep`, `danos_spawn`, `danos_process_enumerate`, `danos_process_kill`, `danos_process_exit_reason`, `danos_process_subscribe`, `danos_process_signal`, `danos_signal_bind` |
|
||||
| threads | `danos_thread_spawn`, `danos_thread_exit`, `danos_current_core`, `danos_futex_wait`, `danos_futex_wake`, `danos_thread_self`, `danos_thread_join`, `danos_set_thread_pointer` |
|
||||
| memory | `danos_mmap`, `danos_munmap`, `danos_dma_alloc`, `danos_dma_free`, `danos_shared_memory_create`, `danos_shared_memory_map`, `danos_shared_memory_physical` |
|
||||
| ipc | `danos_endpoint_create`, `danos_ipc_call`, `danos_ipc_reply_wait`, `danos_ipc_send` (naming is not a syscall: a provider binds its contract at the registry and a client resolves `/protocol/<name>` — see [protocol-namespace.md](protocol-namespace.md)) |
|
||||
| ipc | `danos_endpoint_create`, `danos_ipc_register`, `danos_ipc_lookup`, `danos_ipc_call`, `danos_ipc_reply_wait`, `danos_ipc_send` |
|
||||
| devices | `danos_device_enumerate`, `danos_device_claim`, `danos_device_register`, `danos_mmio_map`, `danos_irq_bind`, `danos_irq_ack`, `danos_msi_bind`, `danos_io_read`, `danos_io_write` |
|
||||
| time | `danos_clock`, `danos_wall_clock`, `danos_timer_bind` |
|
||||
| diagnostics | `danos_debug_write` (leveled, kernel-stamped records), `danos_klog_read`, `danos_klog_status` |
|
||||
| filesystem naming | `danos_fs_resolve`, `danos_fs_node`, `danos_fs_mount`, `danos_fs_unmount` (naming only — file DATA still crosses the vfs-protocol IPC, see below) |
|
||||
|
||||
The constants that ride alongside the calls — mmap protection bits, DMA
|
||||
flags, notification badge bits, `ExitReason`, `Signal`, `page_size`, the IPC
|
||||
message maximum — move to the public header too:
|
||||
flags, notification badge bits, `ExitReason`, `Signal`, well-known service
|
||||
ids, `page_size`, the IPC message maximum — move to the public header too:
|
||||
they are wire values a Rust program needs verbatim. What stays private in
|
||||
`abi.zig` is exactly the thing the vDSO exists to hide: the `SystemCall`
|
||||
numbers and the trap convention.
|
||||
|
||||
@@ -1,192 +0,0 @@
|
||||
# Python on danos: the milestone plan
|
||||
|
||||
The execution plan for [python-on-danos.md](python-on-danos.md). That note holds
|
||||
the *why* and the design decisions; this one slices the work into milestones with
|
||||
concrete deliverables, tests, and exit criteria. Milestones are numbered **P0–P5**
|
||||
(track-local — the global M-series stays with the driver/lifecycle tracks).
|
||||
|
||||
Dependencies at a glance:
|
||||
|
||||
```
|
||||
P0 toolchain + mini-libc ──┐
|
||||
P1 streams + console + seam ─┴─→ P2 CPython minimal ─→ P3 terminal + REPL
|
||||
│ │
|
||||
└─→ P4 danos module │
|
||||
+ Python service│
|
||||
P5 process control + shell ←─────────────────────────────────┘
|
||||
```
|
||||
|
||||
P0 and P1 are independent of each other and can proceed in parallel. P1 is shared
|
||||
work — it is also Zig self-hosting Phase 1 and the first three slices of
|
||||
[character-devices-and-tty.md](character-devices-and-tty.md).
|
||||
|
||||
## P0 — Toolchain + the C library compatibility layer
|
||||
|
||||
**Goal:** a C hello-world, cross-compiled on the host with `zig cc`, runs on danos.
|
||||
|
||||
Design and slicing live in
|
||||
[c-library-compatibility.md](c-library-compatibility.md): the **libdanos-c**
|
||||
sysroot (hand-written danos-native headers + `libc.a`) as a `library/c/` build
|
||||
package — pure computation (string, libm, `strtod`, the printf/scanf engines)
|
||||
lifted from a vendored, pinned musl subtree; the OS plumbing written in Zig over
|
||||
the `runtime` surface (re-targeting `runtime.os` when the Zig track authors it);
|
||||
`malloc` over danos `mmap`; a `crt0` bridging the danos entry shim to C `main`.
|
||||
Driven by `zig cc -target x86_64-freestanding-none -isystem` (the triple becomes
|
||||
`x86_64-danos` if the Zig fork lands first; nothing else changes).
|
||||
|
||||
Its five slices (sysroot-skeleton, fd-plumbing, malloc, stdio,
|
||||
mathematics-and-time) carry their own tests — host-side oracle suites for the
|
||||
computation layer, QEMU cases (`c-hello`, `c-file-io`, `c-stdio`, `c-time`) for
|
||||
the plumbing.
|
||||
|
||||
**Exit:** `c-hello` and `c-file-io` green in the QEMU suite; host computation
|
||||
tests green.
|
||||
|
||||
## P1 — Stream nodes, console, and the seam pieces
|
||||
|
||||
**Goal:** the shared Phase-1 surface exists: byte-stream stdio, cwd, environment,
|
||||
entropy. Design and slicing live in
|
||||
[character-devices-and-tty.md](character-devices-and-tty.md); this milestone is
|
||||
its slices 1–3 plus three small seam pieces:
|
||||
|
||||
- **cwd/chdir** — per-process current directory used by path resolution (the
|
||||
kernel already anchors a VFS root per `fs_resolve`; the cwd is the same idea,
|
||||
process-scoped, with `getcwd`/`chdir` exposed through `runtime` and the libc).
|
||||
- **Environment** — spawn carries an environment block; the SysV entry stack's
|
||||
`envp` slot ([sysv.md](os-development/sysv.md)) stops being empty; `getenv`
|
||||
reads it. An empty block stays valid.
|
||||
- **Entropy** — a kernel `entropy` syscall (RDSEED/RDRAND with a jitter fallback,
|
||||
mirroring the TSC-reliability posture of not trusting one CPU feature blindly);
|
||||
the libc exposes `getentropy`.
|
||||
|
||||
- **Tests.** QEMU: the character-device tests from the tty note (offsetless
|
||||
read/write, blocking read, cooked/raw control round-trip), plus `cwd-basics`
|
||||
(chdir + relative open), `env-roundtrip` (spawn with env, child reads it),
|
||||
`entropy-sane` (nonzero, changing, correct length).
|
||||
|
||||
**Exit:** a C program reads a cooked line from fd 0 and echoes it to fd 1 —
|
||||
injected key events in, bytes read back through the console's in-memory sink,
|
||||
all under QEMU with no hardware involved — and `getcwd`/`getenv`/`getentropy`
|
||||
return real answers.
|
||||
|
||||
## P2 — CPython, minimal configuration
|
||||
|
||||
**Goal:** `python -c 'print(2**100)'` runs on danos under QEMU.
|
||||
|
||||
- Pin **CPython 3.13.x**; vendor as `third-party/cpython/` or fetch via the build
|
||||
(decide with the build-packages conventions).
|
||||
- Host build-Python of the same version (`--with-build-python`).
|
||||
- `config.site` cache for the cross answers; `config.sub` patch so
|
||||
`x86_64-unknown-danos` parses; a small `configure`/`pyconfig` patch set kept as
|
||||
rebasable diffs, WASI-style.
|
||||
- `--disable-shared`; static `Modules/Setup`: `posix errno _io _codecs _weakref
|
||||
time math _stat _collections itertools _functools _locale _sre` plus what the
|
||||
interpreter core insists on; threadless build (WASI precedent).
|
||||
- `Lib/` on the FAT image under the hierarchy (e.g. `/system/python/lib`);
|
||||
`PYTHONHOME` set accordingly; `.pyc` written with **checked-hash
|
||||
invalidation** (FAT's 2-second mtime granularity makes mtime-based validation
|
||||
lie during fast edit-run cycles).
|
||||
- `PYTHONHASHSEED` pinned only if P1's entropy slipped — otherwise real
|
||||
hash randomization from day one.
|
||||
- **Tests.** QEMU: `python-expr` (the exit criterion), `python-file` (run a
|
||||
script from FAT, write a file, read it back), then a curated slice of CPython's
|
||||
own suite (`test_int`, `test_float`, `test_io`, `test_dict`) as a
|
||||
longer-running target — the suite is the porting harness.
|
||||
|
||||
**Exit:** the four QEMU cases green; the CPython test slice green or with a
|
||||
short, documented skip list.
|
||||
|
||||
## P3 — Terminal + REPL: the first real application
|
||||
|
||||
**Goal:** an interactive `python` REPL in a graphical danos terminal — the
|
||||
milestone demo for the OS.
|
||||
|
||||
- Depends on the display track's font rendering (its stated next step) — until
|
||||
that lands, the REPL is exercised end-to-end through the pseudo-device
|
||||
harness from P1+P2 (scripted input in, output read back), so P2's exit is
|
||||
never blocked on graphics; the graphical terminal is the *interactive* debut.
|
||||
- The terminal application: draws with the UI toolkit / display client, consumes
|
||||
keyboard `InputEvent`s, and — per the tty note's load-bearing decision —
|
||||
**serves the VFS stream protocol itself** to its children, reusing the console's
|
||||
line-discipline library. Spawns `python` with its endpoints as fd 0/1/2.
|
||||
- Raw mode + the control set give the REPL line editing; window-size control
|
||||
gives it wrapping.
|
||||
- **Tests.** QEMU: scripted terminal session (inject key events, assert rendered
|
||||
or captured output). Real-hardware smoke on the Intel box joins the existing
|
||||
checklist.
|
||||
|
||||
**Exit:** typing `2+2` into the terminal on the QEMU GPU target prints `4`.
|
||||
|
||||
## P4 — The `danos` extension module + a Python service
|
||||
|
||||
**Goal:** Python can speak danos: IPC, capabilities, spawn.
|
||||
|
||||
- The `danos` module, **written in Zig against `Python.h`**, statically linked
|
||||
via `Modules/Setup`: endpoints (create/send/receive), capability passing,
|
||||
spawn + exit-notification, and the service bootstrap (announce, supervision
|
||||
handshake) — the same surface Zig services use, re-exposed.
|
||||
- UI-toolkit bindings as a second module once the toolkit's API settles.
|
||||
- Prototype **one real service in Python** — policy-shaped, not data-plane
|
||||
(candidates: hot-plug policy, a settings service) — speaking an existing wire
|
||||
protocol, supervised by the device manager like any service.
|
||||
- **Tests.** QEMU: `python-ipc-echo` (Python service echoes over an endpoint, a
|
||||
Zig client asserts), plus the prototype service's own protocol test.
|
||||
|
||||
**Exit:** a Python process runs as a supervised danos service exchanging IPC
|
||||
with Zig peers.
|
||||
|
||||
## P5 — Process control, then the shell
|
||||
|
||||
**Goal:** danos can spawn arbitrary programs with arguments and pipes; a small
|
||||
Python shell uses it.
|
||||
|
||||
The kernel/VFS cluster a shell forces (any shell, any language):
|
||||
|
||||
- **exec-of-path** — spawn an arbitrary VFS path, not a named ramdisk binary;
|
||||
- **argv/envp** — carried through spawn onto the child's entry stack (env from
|
||||
P1, argv new);
|
||||
- **numeric exit status** — extend the exit record beyond the categorical
|
||||
`ExitReason` (the gotcha the Zig roadmap flagged: `WEXITSTATUS` must be real);
|
||||
- **fd inheritance + pipes** — a kernel or service pipe (a character device by
|
||||
the tty note's definition) and spawn-time fd mapping.
|
||||
|
||||
Then, in order: `subprocess` enabled in CPython (maps onto spawn + the
|
||||
exit-notification endpoint — no fork, Windows-style); a **small Python shell** (a
|
||||
few hundred lines over `subprocess` + the console: prompt, argv parsing, pipes,
|
||||
cwd) as the forcing function that reveals what job control actually needs.
|
||||
|
||||
**Explicitly deferred past P5:** the pthread subset over `thread_spawn`/futex,
|
||||
signals-in-libc via M17, termios job control (Ctrl-C to foreground child), and
|
||||
**xonsh** — which wants all three and is the arc's endpoint, not a milestone.
|
||||
|
||||
**Tests.** QEMU: `spawn-argv-exit` (child echoes argv, exits 42, parent sees
|
||||
42), `pipe-through` (parent → child → parent), `python-subprocess`, and a
|
||||
scripted shell session.
|
||||
|
||||
**Exit:** the Python shell runs `program | program` typed at the terminal and
|
||||
reports the exit status.
|
||||
|
||||
## Post-P5 outlook
|
||||
|
||||
Two tracks continue past this plan, each with its own design doc rather than a
|
||||
P-number here:
|
||||
|
||||
- **Dynamic libraries** ([dynamic-libraries.md](dynamic-libraries.md), D1–D4) —
|
||||
an application-layer facility (the OS stays static and lean): `dlopen` in the
|
||||
libc, then libffi + `ctypes` + loadable extension modules, then shared
|
||||
read-only mappings so N Python services hold one physical `libpython`.
|
||||
- **The full C compatibility layer**
|
||||
([c-library-compatibility.md](c-library-compatibility.md), stages 2–3) — the
|
||||
standing rule that every system capability ships with its C spelling, draining
|
||||
the absence table toward "portable C builds on danos"; `fork` is the one
|
||||
permanent exception.
|
||||
|
||||
## Related
|
||||
|
||||
- [python-on-danos.md](python-on-danos.md) — the design note this executes.
|
||||
- [c-library-compatibility.md](c-library-compatibility.md) — P0's design.
|
||||
- [character-devices-and-tty.md](character-devices-and-tty.md) — P1's design.
|
||||
- [zig-self-hosting.md](zig-self-hosting.md) — shares P1; its fork makes P0's
|
||||
triple prettier but gates nothing here.
|
||||
- [os-development/process-management.md](os-development/process-management.md) —
|
||||
the spawn/exit surface P5 extends.
|
||||
@@ -1,261 +0,0 @@
|
||||
# Python on danos: the CPython milestone
|
||||
|
||||
A design note (not built yet) on bringing **CPython** to danos, compiled with the Zig
|
||||
toolchain (`zig cc`). Like [zig-self-hosting.md](zig-self-hosting.md), it is
|
||||
forward-looking: it sets a direction and the decisions that follow from it.
|
||||
|
||||
## Why Python, and why now
|
||||
|
||||
The Zig self-hosting road is gated on a compiler fork and a long std-library seam.
|
||||
Python is the **stop-gap that removes the wait**: a working CPython gives danos a way
|
||||
to write programs — services, tools, application prototypes — *without* the Zig
|
||||
compiler being self-hosted, and it brings the pure-Python package ecosystem along as
|
||||
a bonus. The intended division of labour:
|
||||
|
||||
- **Zig** — the kernel, drivers, and anything on a data plane (interrupt paths,
|
||||
DMA rings, block I/O). Unchanged.
|
||||
- **Python** — the control plane and the prototyping surface: services that are
|
||||
event loops over IPC, policy logic that changes often, application experiments,
|
||||
and eventually the shell.
|
||||
|
||||
Python is also the scripting language for the terminal-and-shell arc: the first
|
||||
real danos application is planned as a terminal, a terminal wants a shell, a shell
|
||||
wants a scripting language — and [xonsh](https://xon.sh) (a shell written in
|
||||
Python) marks where that road can end.
|
||||
|
||||
### Non-goals
|
||||
|
||||
- **No drivers in Python.** Interrupt handling, ring management, and DMA stay in
|
||||
Zig. Python may *supervise and configure* drivers; it does not sit in their hot
|
||||
paths (interpreter overhead and garbage-collection pauses in an interrupt path
|
||||
are disqualifying).
|
||||
- **No dynamic loading during bring-up, no `pip`.** The whole arc here ships
|
||||
statically linked. Dynamic libraries are a real *later* milestone
|
||||
([dynamic-libraries.md](dynamic-libraries.md)) — an application-layer
|
||||
facility that unlocks `ctypes` and loadable extension modules; the operating
|
||||
system itself stays static and lean regardless (the size doctrine below).
|
||||
`pip` stays out either way until a networking track exists.
|
||||
- **No fork.** `os.fork` will not exist. This costs almost nothing (see "The
|
||||
spawn model fits").
|
||||
|
||||
## The realization that shapes everything: the compiler is not the obstacle
|
||||
|
||||
`zig cc` is a full Clang-based C cross-compiler, and CPython is portable C with
|
||||
official precedent for stranger targets than danos — the WASI port is upstream
|
||||
tier-2, and it runs **without fork, without dynamic loading, and without working
|
||||
threads**. Every "CPython can't possibly run there" objection has already been
|
||||
answered upstream by a target *more* constrained than danos.
|
||||
|
||||
What CPython actually needs is a **C environment**: headers and a `libc.a`. danos
|
||||
has neither — and that is the whole project. In the language of the Zig roadmap's
|
||||
three doors, this is the **door-2-shaped work** (the deferred "musl door"), not the
|
||||
`std.os.danos` seam: CPython never touches Zig's std.
|
||||
|
||||
### The same surface, a third time
|
||||
|
||||
The Zig roadmap observed that door 1 (`std.os.danos`) and door 2 (a libc) implement
|
||||
the *same* ~30 danos-facing operations at different layers. CPython consumes that
|
||||
identical surface through C spellings. So nothing here is throwaway: the
|
||||
danos-native operations backing `runtime.os` are the same ones the libc bottoms out
|
||||
in, and the gaps this track must close (stdio byte streams, cwd, environment,
|
||||
entropy) are **exactly the Phase-1 gaps the Zig roadmap already lists**. The two
|
||||
tracks share a road until Python forks off at "build the libc."
|
||||
|
||||
## Where danos stands: coverage vs. the gaps
|
||||
|
||||
Judged against the minimal CPython configuration (static, WASI-like):
|
||||
|
||||
| CPython need | danos today | Gap |
|
||||
|--------------|-------------|-----|
|
||||
| open/read/write/close/lseek, readdir | VFS + FAT via `runtime.fs` | none — wrap in C |
|
||||
| mkdir / unlink / rename / truncate | done (self-hosting Phase 2) | none |
|
||||
| stat with mtime | done (`wall_clock` + FAT mtime) | none |
|
||||
| mmap/munmap (object allocator) | native syscalls | none |
|
||||
| monotonic + wall clock | `clock` + `wall_clock` syscalls | none |
|
||||
| a place for `Lib/` | FAT boot image | none — better than WASI has it |
|
||||
| fork / exec | not needed (subprocess disabled at first) | — |
|
||||
| dynamic loading | not needed (static extension modules) | — |
|
||||
| getcwd / chdir | — | **missing** (shared with Zig Phase 1) |
|
||||
| environment variables | `Init` has no env | **missing** (can start empty) |
|
||||
| entropy | — | **missing** (hash seed; `PYTHONHASHSEED` pins it meanwhile) |
|
||||
| byte-stream stdin/stdout (fd 0/1/2) | `debug_write` out; structured `InputEvent` in | **missing** (shared with Zig Phase 1; the REPL needs it) |
|
||||
| signals | — | stubs suffice (WASI precedent); M17 signals-over-IPC maps on later |
|
||||
| threads | native `thread_spawn`/futex | build threadless first; a pthread subset later (xonsh needs it) |
|
||||
|
||||
The clustering repeats the Zig roadmap's: **files, memory, and time are done; the
|
||||
work is the C packaging plus the small seam pieces** (tty bytes, cwd, env, entropy).
|
||||
|
||||
## The libc decision: hand-rolled in Zig, computation lifted from musl
|
||||
|
||||
Two viable shapes were considered:
|
||||
|
||||
| Option | What it is | Verdict |
|
||||
|--------|-----------|---------|
|
||||
| **Mini-libc in Zig** | C-ABI-exporting Zig library over `runtime.os`/`runtime.fs`, shipped as headers + `libc.a`. | **Take this.** Reuses the danos-native surface directly; no Linux assumptions to fight. |
|
||||
| **Port musl** | Full musl with a danos syscall backend. | Defer, again. musl assumes Linux syscall semantics in places; heavier than the need. |
|
||||
|
||||
The trick that makes the mini-libc tractable: musl's `string/`, `math/` (libm —
|
||||
CPython needs essentially all of it), and number-conversion layers are **pure
|
||||
computation with no syscalls**. Lift those wholesale (MIT-licensed, designed to
|
||||
compile standalone) and hand-write only:
|
||||
|
||||
- the OS-facing bottom: fds, `mmap`, clocks, `exit`, `getcwd` — thin C-ABI wrappers
|
||||
over `runtime.os`;
|
||||
- a `FILE*` stdio layer (buffered, over the fd layer);
|
||||
- `malloc` over danos `mmap` (a simple allocator is fine; CPython does its own
|
||||
small-object arena management above it);
|
||||
- the headers (`stdio.h`, `stdlib.h`, `string.h`, `math.h`, `errno.h`, …).
|
||||
|
||||
Estimate: **100–150 functions**, of which the hard 40% (libm, string, printf/strtod
|
||||
cores) are lifted, not written. Correctness hot spots are `strtod`/`dtoa` (Python's
|
||||
float repr round-trips through them) — another reason to lift musl's, not improvise.
|
||||
|
||||
## C interop: static extension modules, not ctypes
|
||||
|
||||
"Python can interface with C libraries" is true on danos with one important
|
||||
correction: **`ctypes` does not work at first** — it is built on `dlopen` + libffi,
|
||||
both of which arrive only with the [dynamic-libraries](dynamic-libraries.md)
|
||||
milestone (D2). Until then the interop story is the other, older one:
|
||||
|
||||
- **Extension modules statically linked into the interpreter** via CPython's
|
||||
`Modules/Setup` mechanism (the standard route for embedded/static builds).
|
||||
- **Zig speaks C ABI natively**, so danos extension modules are written in Zig
|
||||
against `Python.h` — no C required. Two modules are planned from the start:
|
||||
- **`danos`** — the system module: endpoints, send/receive, capability passing,
|
||||
spawn, exit notification. This is what makes a Python *service* possible: an
|
||||
event loop over IPC, speaking the same wire protocols as Zig services.
|
||||
- **UI toolkit bindings** — the in-progress danos UI toolkit exposed to Python,
|
||||
so application prototypes drive real windows.
|
||||
|
||||
The package story follows: **pure-Python packages work** (unpack into
|
||||
`Lib/site-packages` on the FAT image); packages with C extensions must be
|
||||
cross-compiled and baked into the interpreter — a curated set chosen per image,
|
||||
not `pip install`. That is the honest shape of the stop-gap.
|
||||
|
||||
## The roadmap
|
||||
|
||||
### Phase 0 — Toolchain + libc bring-up
|
||||
|
||||
`zig cc -target x86_64-freestanding-none` plus `-isystem` the danos headers and the
|
||||
mini-libc archive. No compiler fork required — this track deliberately avoids the
|
||||
Zig roadmap's Phase-0 gate (if the fork lands first, the triple becomes a clean
|
||||
`x86_64-danos`; nothing else changes). Exit criterion: a **hello-world C program**
|
||||
compiles on the host and runs on danos, printing via the libc's `write`.
|
||||
|
||||
### Phase 1 — The shared seam pieces
|
||||
|
||||
The same list as Zig self-hosting Phase 1, closed once for both tracks:
|
||||
|
||||
- fd 0/1/2 as console **byte** streams (output exists as `debug_write`; input is a
|
||||
new small thing — cooked line input first, raw mode when the REPL wants editing);
|
||||
- `getcwd`/`chdir`;
|
||||
- environment variables (an empty block is a valid start);
|
||||
- an entropy syscall or service (until then, builds pin `PYTHONHASHSEED`).
|
||||
|
||||
### Phase 2 — Cross-compile CPython, minimal configuration
|
||||
|
||||
Pin one CPython release (3.13 — strongest WASI-era cross-compile support). The
|
||||
mechanics are well-trodden upstream since 3.11:
|
||||
|
||||
- a same-version **build-Python on the host** (`--with-build-python`);
|
||||
- a `config.site` cache answering what configure cannot probe cross
|
||||
(`ac_cv_file__dev_ptmx=no` and friends);
|
||||
- a `config.sub` patch so `x86_64-unknown-danos` parses;
|
||||
- `--disable-shared`, static `Modules/Setup` with a minimal module set
|
||||
(`posix`, `errno`, `_io`, `_codecs`, `time`, `math`, …);
|
||||
- `Lib/` shipped on the FAT image; `PYTHONHOME` pointed at it.
|
||||
|
||||
Exit criterion: `python -c 'print(2**100)'` runs on danos under QEMU.
|
||||
|
||||
### Phase 3 — Terminal + REPL: the first real application
|
||||
|
||||
Depends on the display track's font rendering (already its stated next step) and
|
||||
Phase 1's tty. A terminal emulator drawing a `python` REPL is the milestone demo:
|
||||
interactive, self-evidently real, and it needs **zero** process-control machinery.
|
||||
|
||||
### Phase 4 — The `danos` module and Python services
|
||||
|
||||
Write the `danos` extension module and the UI-toolkit bindings; prototype one real
|
||||
service in Python (a policy-shaped one — e.g. hot-plug policy or a settings
|
||||
service) speaking the existing IPC protocols. This is the payoff phase for
|
||||
"prototyping a service or application."
|
||||
|
||||
### Phase 5 — Process control, then the shell
|
||||
|
||||
The shell — any shell, in any language — forces the surface danos has deferred so
|
||||
far: **exec-of-path, argv/envp passing, numeric exit status (`WEXITSTATUS`, not the
|
||||
categorical `ExitReason`), fd inheritance, and pipes.** That is a kernel/VFS
|
||||
milestone cluster of its own. Then, in order:
|
||||
|
||||
1. `subprocess` enabled in CPython (maps onto danos spawn — see below);
|
||||
2. a **small Python shell** (a few hundred lines over `subprocess` + line input, no
|
||||
job control) — the forcing function that reveals which process-control pieces
|
||||
actually matter;
|
||||
3. **explicitly deferred:** a pthread subset over `thread_spawn`/futex
|
||||
(create/join/mutex/condition/thread-locals), signals via M17 signals-over-IPC,
|
||||
termios job control — and then **xonsh**, which wants all three.
|
||||
|
||||
### The spawn model fits
|
||||
|
||||
One genuinely good alignment: **CPython does not need fork.** `subprocess` maps
|
||||
cleanly onto a posix_spawn-style model — exactly what danos has — and the existing
|
||||
exit-notification-via-endpoint is a *better* fit for `Popen.wait` than Unix's
|
||||
`wait` semantics. `os.fork` simply won't exist, as on Windows, and almost nothing
|
||||
in practice cares.
|
||||
|
||||
## Risks and gotchas
|
||||
|
||||
- **Binary size — and the size doctrine that makes it acceptable.** danos's
|
||||
leanness mandate applies to the **operating system**: the kernel and the system
|
||||
services stay small (the kernel is measured in kilobytes, not megabytes), and
|
||||
nothing in this track changes that — Python never enters the OS layer. An
|
||||
**application** budget is different: a statically-linked CPython with its
|
||||
module set will be tens of megabytes in ReleaseSafe (the measured ~2×
|
||||
safety-check factor compounds it), and that is *allowed* — applications live
|
||||
on the FAT image, not in the kernel's world. It still shapes the image, and it
|
||||
means every Python service shares one interpreter binary + per-service
|
||||
scripts, so the spawn model needs **argv** before "run this .py" works at all.
|
||||
- **FAT mtime granularity is 2 seconds.** CPython's `.pyc` cache validation is
|
||||
mtime-based by default; a rapid edit-run cycle can see stale bytecode. Use
|
||||
hash-based `.pyc` invalidation (PEP 552, `--invalidation-mode checked-hash` at
|
||||
freeze time) or accept the quirk during bring-up.
|
||||
- **FAT name lookups are case-insensitive.** Long file names preserve case but
|
||||
match insensitively — the same world Python inhabits on Windows/macOS, so
|
||||
importlib copes, but two modules differing only by case cannot coexist on the
|
||||
image.
|
||||
- **`strtod`/float repr correctness.** Python's float round-tripping is exacting;
|
||||
lift musl's conversions rather than writing them, and run CPython's float tests
|
||||
early.
|
||||
- **Threadless build is load-bearing, initially.** Like WASI, the first builds have
|
||||
no working `threading`. The escape hatch is real (danos has native threads and
|
||||
futexes; a pthread subset is Phase-5 work) but keep the configuration honestly
|
||||
single-threaded until then.
|
||||
- **The test suite is the porting harness.** CPython ships its own conformance
|
||||
suite; getting `test_builtin`, `test_int`, `test_float`, `test_io` running on
|
||||
danos early converts "it seems to work" into a checklist. Budget image space for
|
||||
the test `Lib/` tree during bring-up.
|
||||
- **Entropy before exposure.** `PYTHONHASHSEED=0` is fine for bring-up and wrong
|
||||
forever; hash randomization exists because attacker-controlled dict keys are a
|
||||
denial-of-service vector. Land the entropy source before any Python service
|
||||
parses external input.
|
||||
|
||||
## Related
|
||||
|
||||
- [python-on-danos-milestones.md](python-on-danos-milestones.md) — the execution
|
||||
plan (P0–P5) for this note.
|
||||
- [c-library-compatibility.md](c-library-compatibility.md) — the mini-libc
|
||||
(libdanos-c) design behind Phase 0.
|
||||
- [character-devices-and-tty.md](character-devices-and-tty.md) — the stream-node /
|
||||
console / no-pty design behind Phase 1.
|
||||
- [zig-self-hosting.md](zig-self-hosting.md) — the sibling track; shares Phase 1,
|
||||
diverges at the libc.
|
||||
- [os-development/syscall.md](os-development/syscall.md) — the kernel ABI the
|
||||
mini-libc bottoms out in.
|
||||
- [os-development/vdso.md](os-development/vdso.md) — the public ABI boundary the
|
||||
`danos` extension module wraps.
|
||||
- [os-development/sysv.md](os-development/sysv.md) — the entry stack (argv/envp)
|
||||
the spawn-argv work extends.
|
||||
- [device-driver-development/ipc.md](device-driver-development/ipc.md) — the IPC
|
||||
surface Python services speak.
|
||||
- [file-system-development/file-system-hierarchy.md](file-system-development/file-system-hierarchy.md)
|
||||
— where `Lib/` and `site-packages` land on the image.
|
||||
@@ -1,584 +0,0 @@
|
||||
# Security track execution plan: paths, protocol namespace, SMEP/SMAP
|
||||
|
||||
The design is settled in
|
||||
[communication.md](os-development/communication.md),
|
||||
[protocol-namespace.md](os-development/protocol-namespace.md),
|
||||
[file-system-hierarchy.md](file-system-development/file-system-hierarchy.md),
|
||||
and [smep-smap.md](os-development/smep-smap.md). This file is the build order
|
||||
— one phase at a time, each phase green before the next starts. Delete or
|
||||
archive this file when the last milestone lands.
|
||||
|
||||
**Context a fresh session should read first:** the four design docs above,
|
||||
then this plan's *Settled decisions* section — those decisions came out of a
|
||||
full-code grounding pass (2026-07-31) and must not be re-derived or reopened.
|
||||
|
||||
**Definition of green, every phase:** `zig build` clean, `zig build test`
|
||||
clean, `python3 test/qemu_test.py` passes (existing scenarios plus the
|
||||
phase's new ones — record the suite count in the checkbox), and the relevant
|
||||
design doc's status/known-gap lines updated in the same commit. Commit per
|
||||
green phase, style `area: lower-case declarative summary`, **no co-author
|
||||
trailers**. On a suite failure, read
|
||||
`zig-out/qemu-test/<case>-failed-serial.log` before changing anything.
|
||||
|
||||
**Workflow:** work in a dedicated git worktree on feature branches cut from
|
||||
`main` (one branch per milestone group as marked below); when a group's
|
||||
phases are all green, merge to `main` and push. The loop marks a phase `[x]`
|
||||
in the same commit that lands it.
|
||||
|
||||
**Numbering note:** milestones use the design docs' own names (PM, H1–H3,
|
||||
HS, P1–P4) — the M-number sequence is left alone (M19–M22 are reserved by
|
||||
the logging/USB-lifecycle track).
|
||||
|
||||
## Status
|
||||
|
||||
**Live state — updated on `main` after every phase, so this file read from a
|
||||
plain `main` checkout always tells the truth about where the work is.**
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| Working on | nothing — the track is complete |
|
||||
| Branch carrying it | — |
|
||||
| On `main` | every phase — groups 1, 2, 3 and 4 merged |
|
||||
| Awaiting merge | nothing |
|
||||
| Suite | 114 cases, all passing |
|
||||
| Last updated | 2026-08-01 |
|
||||
|
||||
A checkbox below means the phase met its definition of green and was
|
||||
committed — on the branch named above, which reaches `main` at the next
|
||||
group boundary.
|
||||
|
||||
- [x] **Phase 0** — baseline: suite green on `main` (106/106, 2026-07-31; `zig build` + `zig build test` clean at 9a32380), plan committed
|
||||
- [x] **PM** — path-migration flag-day (`/etc`→`/system/configuration`, `/var/log`→`/system/logs`, `/mnt/usb`→`/volumes/usb`; vfs carve-out for the two writable `/system` subtrees, FAT's `/var` mount split in two; suite 106/106)
|
||||
- [x] **H1** — the `user-memory` module; nine stragglers converted; leaf U/S+W checks (plus physmap-coverage confirmation, so an `mmio_map`'d buffer cannot fault ring 0 — this also closes the same hazard on the IPC path; `fs_resolve`'s out-capacity bound made overflow-safe; suite 107/107)
|
||||
- [x] **merge** group 1 → main, push (f3bc23c, 2026-07-31)
|
||||
- [x] **P1** — envelope module + `Define`; vfs `NodeKind.protocol` + open-reply-capability; client `Channel` (mechanics only, nothing converted; suite unchanged at 107)
|
||||
- [x] **P2** — registry in init; `/protocol` reserved; ServiceId flag-day (11 binds, 17 lookups; `protocol.csv` grants, chain-attested identity, dead-owner rebind; the kernel's endpoint-death sweep generalized off the retired registry; suite 108/108). Three adversarial review rounds closed six defects a green suite had missed: a forged power event could shut the machine down; the ping path leaked a capability per call, first in init and then in the shared harness; supervisor attestation by name was defeated by a laundering deputy; and the kernel let any handle-holder bind signals, timers, exits and IRQs to an endpoint it did not own.
|
||||
- [x] **P3** — open grants: `protocol.csv` enforcement, denial test. `onOpen`
|
||||
consults the manifest with the same chain-attested identity a bind uses, and a
|
||||
refused caller gets the *same* answer as one naming a contract nobody bound —
|
||||
`-ENOENT`, no capability, the same reply bytes, no log line, and both questions
|
||||
asked on every open so there is nothing to time. Twenty-seven `open` rows cover
|
||||
the whole live client set. One wrinkle the plan had not foreseen: the driver
|
||||
tree is three deep (device manager → PS/2 bus → keyboard/mouse) and attestation
|
||||
is one hop, so a legitimate grandchild read exactly like a laundering deputy;
|
||||
the manifest gained a third permission, `supervise`, which names an authorized
|
||||
supervising task per contract and is deliberately **open-only**, leaving P2's
|
||||
bind attestation and every refusal it makes untouched (suite 109/109)
|
||||
- [x] **merge** group 2 → main, push
|
||||
- [x] **P4a** — clean protocols rebased onto `Define` (vfs, block, display, scanout, input; display's one overloaded request split per-operation and its field abuse ended, scanout's bogus 64-byte maximum deleted, directory EOF re-spelled as a nameless entry, input moved onto the service harness; new `protocol-conformance` case asks every reachable provider for `describe` and requires `-ENOSYS` for an undefined verb; suite 110/110)
|
||||
- [x] **P4b** — misfit protocols rebased (device-manager, power, usb-transfer; every leading operation byte folded into the header, and with it the `device_id`/`device_token` that followed it — `Header.target` now carries the device in all three. device-manager's own `enumerate`/`subscribe` became the reserved verbs and its three `{status, reserved}` reply structs the envelope's `Status`; `ChildAdded` is one struct under two numbers, a call and an event, landing exactly on the 64-byte push floor. power's kinds became one declared event each, the input protocol's shape, so init reads *what happened* from the header; usb-transfer's control data stage moved to the packet tail in both directions, which made `Status.len` the transferred length and `actual_length` redundant. The two silent-breakage sites — init's byte-offset power parse and acpi's `message[0]` dispatch — are gone, the shutdown badge gate unchanged; three more rows in the conformance table. Suite 110/110)
|
||||
- [x] **P4c** — harness subscriber lift + badge-scoped per-client integers (the
|
||||
subscriber table, the reserved subscribe/unsubscribe verbs, the fan-out and the
|
||||
dead-subscriber sweep are `service.Subscribers` now; input, acpi and
|
||||
device-manager deleted three hand-rolled variants and their three different
|
||||
ideas of when a subscriber goes away, standardizing on published exit
|
||||
notifications — acpi had no sweep at all and input polled the process list on
|
||||
every subscribe. The three guessable-id namespaces are scoped to the opening
|
||||
badge: FAT node ids on every verb that names one, xHCI device tokens on open,
|
||||
control, bulk and interrupt_subscribe, display layers on configure, fill, blit,
|
||||
damage and destroy — each refusing a wrong owner with the *same* answer as an id
|
||||
nobody holds. New `badge-scope` case, two processes of one fixture, every
|
||||
refusal paired with a control; suite 111/111)
|
||||
- [x] **merge** group 3 → main, push
|
||||
- [x] **H2** — SMEP on every core (shared CPUID helper; CR4 bit 20 set in the per-CPU bring-up both the BSP and every AP run, asserted per core by the smp case; ring-0-executes-user-pages audit clean incl. the pre-paging window on the loader's tables; fail-open with a posture line; `-cpu max` added to the harness since QEMU's default model has neither bit; new `fault-smep` case; suite 112/112)
|
||||
- [x] **HS** — SYSRET canonical-RIP guard (the syscall exit sign-extends the
|
||||
return RIP from bit 47 and returns through `iretq` when that changes it —
|
||||
four register ALU ops and a never-taken branch on the hot path, no load; the
|
||||
fallback un-pops the rip slot so `iretq` consumes the frame entry already
|
||||
built, reloads R11 from the rflags slot, and keeps the `swapgs` in the same
|
||||
place relative to the ring change. 4-level is not an assumption but a fact:
|
||||
nothing sets CR4.LA57, and the comment says what a 5-level port must change.
|
||||
Ring 3 **can** reach the hazard — `syscall` as the last two bytes of the last
|
||||
canonical page returns to `user_half_end` — so the new `sysret-canonical`
|
||||
case is a real ring-3 probe doing exactly that, and dies of a ring-3 #GP at
|
||||
`0x0000800000000000` while the kernel runs on. Its teeth are the refusal
|
||||
counter, not the outcome: measured with the guard's branch removed, TCG does
|
||||
not model Intel's ring-0 #GP and every outcome check still passed. Suite
|
||||
113/113)
|
||||
- [x] **H3** — SMAP on every core, and the boot-patched `clac` that makes it
|
||||
hold. Hardware does not clear `EFLAGS.AC` on interrupt delivery and ring 3
|
||||
sets it with `popfq`, so without the patch a hostile process suspends SMAP
|
||||
for the length of any handler it can provoke; `clac` is #UD without the
|
||||
feature, so the image ships the 3-byte canonical NOP at the first byte of
|
||||
`isr_common` — ahead of the CPL test, because a fault nested in ring 0
|
||||
inherits AC just as readily — and the boot processor overwrites it through
|
||||
the physmap, the same door `smp.arm` and `process.run` already use to write
|
||||
a page the executing mapping holds read-only. The two halves are wired
|
||||
together rather than merely ordered: `initHardening` refuses CR4 bit 21
|
||||
until the patch has been written *and* read back through the text address it
|
||||
will be fetched from, so no core can turn SMAP on ahead of it and a
|
||||
translation that lied leaves the machine unhardened and saying so instead of
|
||||
enforcing over an entry path that cannot clear AC. The `smp` case now
|
||||
requires SMAP on every core it lands on, which is the ordering checked from
|
||||
the far end. New `fault-smap` case reads a mapped user page from ring 0 and
|
||||
requires the fault: error code `0x1` — present, and nothing else, since a
|
||||
SMAP violation has no bit of its own and U/S reports the ring of the access.
|
||||
The suite is now the standing enforcement test, and it found nothing left to
|
||||
find: H1's conversion of the nine stragglers was complete. Suite 114/114
|
||||
- [x] **merge** group 4 → main, push
|
||||
|
||||
---
|
||||
|
||||
## Settled decisions (grounding pass, 2026-07-31 — do not reopen)
|
||||
|
||||
These resolve every open wrinkle the code inventory surfaced. Where one
|
||||
amends a design doc, the amendment lands in the same commit as the phase
|
||||
that implements it.
|
||||
|
||||
1. **Every packet — request, reply, and event — begins with the envelope
|
||||
`Header`, exactly as the design says; the header is FOLDED, never
|
||||
stacked.** It absorbs each protocol's existing operation/id fields
|
||||
rather than sitting on top of them, so the two apparent 64-byte-limit
|
||||
offenders fit: `ChildAdded` re-lays to 60 bytes (its packed operation
|
||||
byte and `device_id` become `Header.operation`/`.target`);
|
||||
`InterruptReport` puts `device_token` in `Header.target` and trims
|
||||
inline data 48 → 40 bytes (largest real report today is 8). A
|
||||
headerless-events variant was considered and REJECTED (2026-07-31): it
|
||||
re-invents per-protocol mini-headers and breaks uniform tooling. No
|
||||
design-doc amendment; `Define`'s event check stays ≤ 64 *including*
|
||||
the header.
|
||||
2. **Bind/open authorization is chain-attested identity: the
|
||||
kernel-stamped binary name PLUS the supervision chain**, both read from
|
||||
the kernel's process records (`ProcessDescriptor` carries `name` and
|
||||
`supervisor`; init walks the chain with `process_enumerate` — no new
|
||||
protocol). A grant row names the binary *and* the supervisor expected
|
||||
in its chain, so a malicious process re-spawning a granted binary
|
||||
(ungated `spawn`, hostile argv — the confused deputy) is refused: its
|
||||
chain roots at the attacker, not at init or device-manager. Name alone
|
||||
is NOT sufficient — that was considered and rejected (2026-07-31).
|
||||
Pure delegation (device-manager forwarding driver binds as
|
||||
capabilities — "option B") is deliberately deferred to P5, whose
|
||||
spawner-wired namespaces subsume it. Amends protocol-namespace.md's
|
||||
"Authorization" bullet in P2.
|
||||
3. **Grants live in a new manifest, `/system/configuration/protocol.csv`**
|
||||
(rows: `binary-path, supervisor, bind|open, protocol-name`, where
|
||||
`supervisor` is the binary expected in the caller's supervision chain —
|
||||
`init` for init's own children, `kernel` for harness-spawned fixtures),
|
||||
not in extra init.csv columns — today every post-path init.csv field is
|
||||
argv, and overloading that is ambiguous. init parses both files.
|
||||
*(P2 spelling: the supervisor column carries the binary exactly as the
|
||||
kernel stamped it, so init's own children say `/system/services/init` and
|
||||
the drivers say `/system/services/device-manager`; `kernel` stays a bare
|
||||
word because a kernel task has no binary. A trailing `*` on any field
|
||||
matches a subtree, which is how decision 4's `/test/` rule is expressed.)*
|
||||
*(Clarification, 2026-08-01: the supervisor column names **the authorized
|
||||
supervising task, matched by identity** — the binary is how the row spells
|
||||
it, but init checks the task id. `kernel` is satisfied only by supervisor
|
||||
id 0 (which only the kernel confers — user `system_spawn` always stamps the
|
||||
caller); init's own path only by this init's task id; any other path only by
|
||||
a task init spawned itself or one the kernel spawned. Matching the supervisor
|
||||
by *name* alone is defeated by a laundering deputy — an attacker runs its own
|
||||
instance of `/system/services/init`, has that spawn `/system/services/input`,
|
||||
and both stamped names satisfy the row while the chain is entirely the
|
||||
attacker's. Walking to the root of the chain does not fix it either, since
|
||||
the laundered chain still roots at the real PID 1.)*
|
||||
*(P3 amendment: a third permission, `supervise`, joins `bind|open`. One-hop
|
||||
attestation cannot express the one three-deep chain in the tree — the device
|
||||
manager starts the PS/2 bus, and the bus starts the keyboard and mouse
|
||||
drivers — and nothing structural tells that chain apart from the laundering
|
||||
deputy, since both are a granted binary spawned by a granted binary. Only
|
||||
policy can: a `supervise` row names the authorized supervising task the way
|
||||
every other row names a claimant (binary, its own supervisor, the contract it
|
||||
concerns), and an `open` row may then name that task in its supervisor
|
||||
column. The delegate is itself attested the ordinary strict way, so the chain
|
||||
still anchors in init or the kernel one hop above it and the recursion stops
|
||||
there. It is **open-only** on purpose — a delegate may vouch for what its
|
||||
children *reach*, never for what they *claim* — so the bind path is
|
||||
byte-for-byte P2's and the laundering-deputy refusal is untouched.)*
|
||||
4. **Test fixtures bind under `/protocol/test/...`**, granted to any
|
||||
binary whose path starts `/test/` — the subtree-scoping rule from the
|
||||
design doc, dogfooded. `shared_memory_test` (the borrowed-ServiceId
|
||||
hack) becomes `/protocol/test/shared-memory`; process-test's child gets
|
||||
`/protocol/test/process`.
|
||||
5. **Rebind after provider death:** a `bind` hitting an existing binding
|
||||
succeeds only if the current owner process is dead (init checks
|
||||
liveness); otherwise `-EBUSY`. Init also unbinds in `restartChild`
|
||||
before respawning its own children. This preserves collision-refusal
|
||||
while making restart work for providers init does not supervise.
|
||||
6. **Cross-thread service access** (the display mouse-listener's
|
||||
per-thread self-lookup, `display.zig:512`): threads resolve and open
|
||||
`/protocol/<name>` like any client — once, at thread startup. No
|
||||
special mechanism.
|
||||
7. **The envelope module is `library/protocol/envelope/envelope.zig`**
|
||||
(module name `envelope`) — the one protocol-package module not ending
|
||||
in `-protocol`, because it is not a protocol. Wired as a new
|
||||
`addModule` row in `library/protocol/build.zig` with its host tests in
|
||||
that package's test step.
|
||||
8. **The QEMU harness gains `-cpu max`** (in `qemu_args`,
|
||||
`test/qemu_test.py:66-83`) so TCG exposes SMEP/SMAP — without it the
|
||||
enabled paths never execute in CI. Landed in H2 so the flag soaks
|
||||
before H3 depends on it.
|
||||
9. **Scenario fixtures that need the registry are init-driven.** Kernel
|
||||
test cases that today spawn providers directly (shared-memory,
|
||||
process-test) either spawn init first or move to init.csv-driven
|
||||
scenario boots — resolved per-case in P2 with the suite as the
|
||||
arbiter.
|
||||
*(P2 resolution: init gained a `registry` argv role — it mounts
|
||||
`/protocol`, reads the grants, and starts no services — and each affected
|
||||
case calls `spawnRegistry(rd)` before its own providers. Every case keeps
|
||||
its own spawn set, so no scenario had to be re-shaped.)*
|
||||
10. **The capsule-staleness caveat is documented, not fixed.** On-volume
|
||||
edits to `/system/configuration/*.csv` do not reach the initrd copy
|
||||
the loader boots (capsule shadows tree). Same drift exists today with
|
||||
`/etc`; PM adds the note to file-system-hierarchy.md and moves on.
|
||||
|
||||
---
|
||||
|
||||
## PM — path-migration flag-day
|
||||
|
||||
One commit, everything moves together. The authoritative site inventory is
|
||||
the grounding pass; the checklist order:
|
||||
|
||||
1. Move repo `etc/` → `configuration/` sources; fix the three CSVs'
|
||||
self-referencing headers (`etc/init.csv:1,12`, `etc/devices.csv:1`,
|
||||
`etc/init-diagnose.csv:1`).
|
||||
2. `build.zig:309-311`: bundled entries `etc/...` →
|
||||
`system/configuration/...` (this alone re-shapes the image, manifest,
|
||||
and capsule — `tools/make-fat-image.py` and the EFI loader need
|
||||
nothing; the tree-walk fallback even starts picking the CSVs up, a
|
||||
bonus fix).
|
||||
3. `system/kernel/vfs.zig` `mountBackend` (`:332-340`): allow exactly
|
||||
`/system/configuration` and `/system/logs` as backend prefixes beneath
|
||||
the initrd `/system` mount; keep refusing everything else under
|
||||
`/system` and `/test`.
|
||||
4. `system/services/fat/fat.zig`: `mount_point` → `/volumes/usb` (`:25`);
|
||||
replace the `/var` mount (`:155`) with two `mountRewritten` calls for
|
||||
`/system/configuration` and `/system/logs`; update the mount log lines
|
||||
(the harness matches them).
|
||||
5. `system/services/init/init.zig:76` and
|
||||
`system/services/device-manager/device-manager.zig:48`: open the new
|
||||
CSV paths; update the message strings (`init.zig:77,92`,
|
||||
`device-manager.zig:49,61-63,454`).
|
||||
6. `system/services/logger/logger.zig:44`: `base = "/system/logs"`
|
||||
(buffers derive from `base.len` comptime — nothing else changes).
|
||||
7. `system/kernel/tests.zig:2808-2810`: exclude `/system/configuration/`
|
||||
from the spawn-everything sweep (the CSVs are not programs).
|
||||
8. Tests: `fat-test.zig` and `vfs-test.zig` `/mnt/usb` literals →
|
||||
`/volumes/usb`; harness regexes `test/qemu_test.py:175,211,632,717`.
|
||||
9. Comment sweep (init, device-manager, logger, fat, engine, vfs, abi,
|
||||
file-system, csv, device, protocol/device-manager, drivers, acpi,
|
||||
build.zig — full list in the grounding inventory); delete vestigial
|
||||
repo `var/`.
|
||||
|
||||
**Test:** no new case — the existing 106 are the test, since fat/logger/
|
||||
init/device-manager scenarios all assert the new paths through their
|
||||
regexes. Suite stays 106.
|
||||
|
||||
## H1 — user-memory copy discipline
|
||||
|
||||
New kernel module `system/kernel/user-memory.zig`:
|
||||
|
||||
- `copyFromUser` moves from ipc-synchronous.zig (which re-exports or
|
||||
imports it); new `copyToUser(user_as, user_va, source) bool` — the
|
||||
mechanical mirror (kernel-source `copyAcross` already does this for IPC
|
||||
replies at `ipc-synchronous.zig:431,460`).
|
||||
- The page walk gains leaf U/S and writable checks: `paging.translateIn`
|
||||
(`architecture/x86_64/paging.zig:513-525`) tests only `present` today —
|
||||
add a flags-accumulating variant (2 MiB leaves included); reads require
|
||||
U/S, writes require U/S+W. Closes the TODO at
|
||||
`ipc-synchronous.zig:20-22`.
|
||||
- Convert the nine stragglers (table in smep-smap.md). Read direction is
|
||||
local to `process.zig`; the write direction restructures callees with
|
||||
kernel bounce buffers: `scheduler.enumerate` (`scheduler.zig:1209`),
|
||||
`devices_broker.enumerate` (`devices-broker.zig:136`), `log.readAt`
|
||||
(`log.zig:209`), and the `fs_node` flows through
|
||||
`vfs.nodeRead/nodeStatus/nodeReaddir` (`vfs.zig:257/269/289`).
|
||||
|
||||
**Test:** kernel unit coverage in `system/kernel/tests.zig` for
|
||||
`copyToUser` bounds/permission refusals; one new QEMU case `user-memory` —
|
||||
a fixture passes an unmapped-but-in-range buffer to `klog_read`,
|
||||
`process_enumerate`, and `fs_resolve` and asserts `-EFAULT` returns with
|
||||
the system still alive (today each would oops the kernel). Suite 107.
|
||||
|
||||
## P1 — envelope, vfs additions, Channel
|
||||
|
||||
- `library/protocol/envelope/envelope.zig`: `Header` {operation:u32, pad,
|
||||
target:u64}, `Status`, reserved verbs (describe=0, enumerate=1,
|
||||
subscribe=2, unsubscribe=3, protocol verbs from 16), `packet_maximum`
|
||||
= 256 / `post_maximum` = 64 (the floor constants protocols compile
|
||||
against — nothing exports them today), and comptime
|
||||
`Define(.{name, version, operations, events})` generating request/reply
|
||||
types, encode/decode, a provider dispatch table (automatic `describe`,
|
||||
`-ENOSYS` for unknown verbs), and compile-time size checks:
|
||||
request/reply ≤ 256, each `.events` entry ≤ 64 *including* its Header
|
||||
(decision 1). Host unit tests in the protocol package's test step.
|
||||
- `library/protocol/vfs/vfs-protocol.zig`: `NodeKind.protocol = 7`; the
|
||||
open-reply-may-carry-capability convention documented in the module.
|
||||
Rewrite the value-pinning unit test (`:108-117`) to pin the *new*
|
||||
stable values.
|
||||
- `library/kernel/file-system.zig` + a new `Channel` type in
|
||||
`library/kernel` (or `library/client`): `open("/protocol/<name>")` →
|
||||
resolve, vfs open, receive the reply capability → a `Channel` wrapping
|
||||
the handle with `call`/typed helpers. Nothing uses it yet — P2 converts
|
||||
the world.
|
||||
- Docs: vfs-protocol.md's NodeKind table gains value 7 (no
|
||||
protocol-namespace.md amendment — decision 1 conforms to it as written).
|
||||
|
||||
**Test:** host unit tests only (envelope round-trips, size-check compile
|
||||
errors via `error` tests, Channel plumbing against a mock). Suite stays
|
||||
107.
|
||||
|
||||
## P2 — the registry; ServiceId flag-day
|
||||
|
||||
The single biggest phase; one branch, may be several commits, green at the
|
||||
end of each.
|
||||
|
||||
- **init as registry backend** (`system/services/init/init.zig`): a second
|
||||
endpoint (the supervision endpoint's reply-empty loop is unsuitable for
|
||||
a vfs backend); serve vfs `open`/`readdir` over `/protocol` plus the
|
||||
`bind` operation (name payload + capability). Mount `/protocol` before
|
||||
spawning children. Parse `/system/configuration/protocol.csv`
|
||||
(decision 3). Authorization by chain-attested identity (decision 2):
|
||||
badge → kernel process records → binary name **and** supervision chain
|
||||
(walk `supervisor` links) checked against the grant row's expected
|
||||
supervisor. Unbind on child death in `restartChild`; dead-owner rebind
|
||||
rule (decision 5).
|
||||
Provenance: readdir/diagnostics show name → pid → binary path.
|
||||
- **Kernel:** reserve `/protocol` — `mountBackend` refuses mounts at or
|
||||
under it once bound, `installMount`'s remount-replace path refuses it,
|
||||
and `fs_unmount` refuses it (`vfs.zig:164-181,332-351`,
|
||||
`process.zig:1879-1889`). First mount wins (init is PID 1).
|
||||
- **Harness:** `library/kernel/service.zig` `Callbacks.service:
|
||||
?abi.ServiceId` becomes a protocol name; the register call (`:49-51`)
|
||||
becomes bind-with-retry via the registry.
|
||||
- **Flag-day conversion** — all 11 registration sites and 17 lookup sites
|
||||
from the grounding inventory: providers (input:123, ps2-bus:223,
|
||||
device-manager:569, acpi:193, usb-xhci-bus:676, usb-storage:205,
|
||||
fat:307, display:699, virtio-gpu:550, shared-memory-server:43,
|
||||
process-test:130 → `/protocol/test/...` per decision 4); clients
|
||||
(input-client:53, display-client:28, driver.zig:173, usb.zig:139,
|
||||
block.zig:72+87, ps2-bus keyboard:35 + mouse:34, virtio-gpu:478,
|
||||
display:314+512 (decision 6), acpi:212, init:218+245 — init
|
||||
short-circuits its own registry, shared-memory-client:22,
|
||||
process-test:85, device-list:22, crash-test:32). Retry loops keep their
|
||||
cadence, wrapping resolve+open instead of lookup.
|
||||
- **Delete:** `abi.zig:36-37` (syscall ids — leave holes),
|
||||
`abi.zig:287-303` (enum), `process.zig:223-224,314-343`,
|
||||
`ipc-synchronous.zig:41-43,646-664` and the registry sweep in
|
||||
`:121-140`; the wrappers `library/kernel/ipc.zig:33-35,47-50`; comment
|
||||
sweep (irq.zig:50, tests.zig:3744, vdso.md's syscall table, the docs
|
||||
list in the inventory).
|
||||
- Kernel-spawned test scenarios made init-driven where they need the
|
||||
registry (decision 9).
|
||||
|
||||
**Test:** new QEMU case `protocol-registry`: a fixture asserts (a) bind of
|
||||
an ungranted name → `-EPERM`, (b) bind collision with a live owner →
|
||||
`-EBUSY`, (c) provider kill → re-resolve reaches the restarted instance.
|
||||
Every existing scenario doubles as conversion proof. Suite 108.
|
||||
|
||||
## P3 — open grants (restriction stage one)
|
||||
|
||||
- `protocol.csv` `open` rows enforced in the registry's `open` handler,
|
||||
same name-based identity as bind. Default rows grant what today's
|
||||
clients need (from the P2 conversion table); a deliberate hole for the
|
||||
test fixture.
|
||||
- Docs: protocol-namespace.md stage-one section gets its "landed" line.
|
||||
|
||||
**Test:** new QEMU case `protocol-denied`: a fixture granted
|
||||
`/protocol/test/shared-memory` but not `/protocol/display` asserts open of
|
||||
the first succeeds and the second fails identically to not-found. Suite
|
||||
109.
|
||||
|
||||
*Landed. Four things the plan did not foresee, recorded because P4 and P5
|
||||
inherit them:*
|
||||
|
||||
- *`supervise` — decision 3's amendment. The PS/2 keyboard and mouse drivers
|
||||
are started by the PS/2 bus driver, which the device manager started: the
|
||||
tree's one three-deep chain, and one hop deeper than attestation reaches.
|
||||
Nothing structural separates it from the laundering deputy, so the manifest
|
||||
says which delegate is authorized, per contract. Open-only, so P2's bind
|
||||
attestation is unchanged.*
|
||||
- *Indistinguishability is a claim about work, not only about bytes. `onOpen`
|
||||
refreshes the process table, identifies the caller, scans the grants and
|
||||
scans the bindings on **every** open and forms one verdict at the end; and
|
||||
it logs nothing on any branch, because `klog_read` is ungated (a line
|
||||
written on one branch is a line the refused caller can read) and a serial
|
||||
line is milliseconds it could time. The operator's diagnosis is the pair the
|
||||
namespace publishes anyway: `readdir /protocol` for what is bound, the
|
||||
manifest for who may reach it.*
|
||||
- *The fixture is `protocol-denied-test`, and its scenario boots the **input
|
||||
service** so the forbidden name is genuinely bound — the fixture reads the
|
||||
namespace listing to prove it before asking for it. Without a live provider
|
||||
the case would be comparing two boot races and asserting nothing.*
|
||||
- *Two channels stay open by design, named rather than papered over: `readdir`
|
||||
over `/protocol` lists every bound name to anyone (deliberate — the tree is
|
||||
diagnosable), and `/system/configuration/protocol.csv` is world-readable on
|
||||
the `/system` mount. Stage one hides neither the set of contracts nor the
|
||||
policy; what it removes is the **oracle in the reply**, which is what stage
|
||||
two's parked and faked opens depend on.*
|
||||
|
||||
## P4a — clean protocols onto Define
|
||||
|
||||
vfs, block, display, scanout, input — the modules whose shapes map
|
||||
directly (grounding inventory §1,3,4,6,8):
|
||||
|
||||
- vfs: `node` → `target`; `Reply.node` (open's result) moves to reply
|
||||
payload — `library/kernel/file-system.zig` decoders change; readdir
|
||||
stays a protocol verb.
|
||||
- block: pure renumber; `attach`'s DMA cap rides the call as today.
|
||||
- display: the overloaded 40-byte `Request` becomes per-operation structs
|
||||
(attach_scanout's field abuse dies); `layer` → `target`; blit payload
|
||||
grows to 224 bytes.
|
||||
- scanout: renumber; drop its bogus `message_maximum=64` (sync floor is
|
||||
256); fix virtio-gpu's hard-coded `service.run(256, …)` to the
|
||||
generated constant.
|
||||
- input: subscribe merges into reserved subscribe; publish renumbers;
|
||||
the event re-lays onto the Header folded (operation = event kind,
|
||||
target = 0; 16 + 28-byte payload = 44 ≤ 64); **input moves onto the
|
||||
service harness** (it is the last hand-rolled loop, no ping/terminate
|
||||
compliance today).
|
||||
|
||||
**Test:** new QEMU case `protocol-conformance`: a fixture opens every
|
||||
registered protocol and asserts `describe` answers (name, version) and an
|
||||
unknown verb returns `-ENOSYS`. Existing input/display/fat scenarios prove
|
||||
the rebase. Suite 110.
|
||||
|
||||
## P4b — misfit protocols onto Define
|
||||
|
||||
device-manager, power, usb-transfer (inventory §2,5,7 — the u8-operation
|
||||
re-layouts and raw-offset readers):
|
||||
|
||||
- device-manager: u8 operations → Header; its enumerate=4/subscribe=5
|
||||
merge into the reserved verbs; `ChildAdded` splits its dual role —
|
||||
request struct and event, both Header-first (folded to 60 B ≤ 64);
|
||||
`ChildRemoved`'s (parent, bus_address) addressing stays payload.
|
||||
- power: u8 operations → Header; subscribe merges; **init's raw
|
||||
byte-offset event parsing (`init.zig:171-173`) and acpi's
|
||||
`message[0]` dispatch (`acpi.zig:435-467`) are rewritten against the
|
||||
generated types** — the two silent-breakage sites, called out so the
|
||||
loop treats them as first-class conversions, not collateral.
|
||||
- usb-transfer: `device_token` → `target` (already layout-identical);
|
||||
`InterruptReport` re-lays onto the Header (`device_token` → `target`,
|
||||
inline data trimmed 48 → 40 — largest real report is 8); control/bulk
|
||||
budgets re-verified by `Define` (Status absorbs `actual_length`).
|
||||
|
||||
**Test:** existing scenarios are the proof (device hot-add, power button,
|
||||
USB storage/HID all exercise these wires); the conformance case now covers
|
||||
three more providers. Suite 110.
|
||||
|
||||
*Landed. Three judgment calls the plan left open, recorded because a reader of
|
||||
the wire formats will want them:*
|
||||
|
||||
- *`ChildAdded` is 48 bytes, not the 44 the "60 B" estimate assumed: three `u64`s
|
||||
give the struct eight-byte alignment, so 41 bytes of content round up whatever
|
||||
order the fields sit in. The packet is therefore **exactly** 64 — on the push
|
||||
floor, not under it — which `Define` accepts and the module pins in a test. The
|
||||
fields are ordered small-tail-last deliberately, so the slack the rounding pays
|
||||
for is where the small ones live.*
|
||||
- *power's events are declared **per kind** (`power_button`, `lid`, `ac`,
|
||||
`battery`, `notify`), not one `event` with the kind in the payload. That is the
|
||||
shape P4a gave input — "the class is the header's operation, so a subscriber
|
||||
reads the kind from the packet rather than from a tag inside the payload" — and
|
||||
it is what makes `Header.operation` carry information here at all. It also kept
|
||||
every call site's spelling: `Protocol.Event` is re-exported as the protocol's
|
||||
own `Event`, with the members it always had.*
|
||||
- *the conformance case still checks two providers, and the three new rows report
|
||||
as unbound. All three P4b contracts arrive with the device manager — it is the
|
||||
first, it spawns the discovery service that binds the second, and the xHCI
|
||||
driver that binds the third — so booting one means booting the driver tree, and
|
||||
the fixture takes **one snapshot** of `/protocol`: a scenario whose bound set
|
||||
depends on how far that tree got would make the case's own summary line a boot
|
||||
race. The rows still earn their place — a future scenario that binds one gets it
|
||||
checked with no edit here, and `/test/*` already holds the `open` grant for
|
||||
`device-manager`.*
|
||||
|
||||
## P4c — harness subscriber lift + badge scoping
|
||||
|
||||
- `library/kernel/service.zig` grows the subscriber table, exit-
|
||||
notification sweep, and fan-out loop declared via `Define(.events)`;
|
||||
input (:33-116), acpi (:67-68,393-406), and device-manager (:155-166)
|
||||
delete their hand-rolled variants. One sweep idiom: exit notifications
|
||||
(fat's pattern), replacing input's process-list polling and acpi's
|
||||
none-at-all.
|
||||
- Badge-scoped per-client integers (the guessable-id holes): fat node ids
|
||||
gain owner checks on every operation (`fat.zig:72-76`), xhci device
|
||||
tokens validate sender and sweep on exit (`usb-xhci-bus.zig:66-88,479`),
|
||||
display layers gain an owner field.
|
||||
|
||||
**Test:** extend the fat scenario: a second fixture guesses the first's
|
||||
node id and asserts refusal; kernel-side unit test for the harness sweep.
|
||||
Suite 111.
|
||||
|
||||
*Landed. Four things the plan had not foreseen:*
|
||||
|
||||
- *One sweep idiom means one more kernel subscriber per provider, and the kernel's
|
||||
published-exit table held **eight**. A normal boot now fields six (fat, input,
|
||||
power, device-manager, display, and one per xHCI controller), so the table grew
|
||||
to sixteen. It is not a table anyone notices until a service silently loses its
|
||||
sweep, which is exactly the failure the old ceiling was two subscriptions away
|
||||
from.*
|
||||
- *The device manager hears each of its drivers die **twice** now — it is both the
|
||||
supervisor its spawn named and, through the harness, a subscriber to published
|
||||
exits — and the notify ring delivers the two badges separately. Untreated, one
|
||||
death counted as two: the restart backoff doubled and the crash-loop cap fired
|
||||
at half the deaths it names. `onDriverExit` therefore retires the dead process
|
||||
id before it decides anything, and the second notification finds nothing to act
|
||||
on. (The `driver-restart` and `pci-scan` drills are what would have caught it.)*
|
||||
- *Refusal-equals-absence has a corollary for the verbs that **release**: FAT's
|
||||
`close` used to answer 0 for an unknown node, so scoping it had to change that
|
||||
too — a foreign node and a free one both answer `-ENOENT`, or the pair would
|
||||
have been an oracle for which ids are live. The same applies to the harness's
|
||||
`unsubscribe`.*
|
||||
- *Ownership is per **task**, not per process, because the badge is: the kernel
|
||||
stamps the sending thread's id, which is already the granularity of the exit
|
||||
sweep that releases the state (a worker thread's death releases the handles that
|
||||
worker opened). Nothing in the tree shares an id across its own threads today;
|
||||
a per-process notion would need the kernel to stamp the leader, and belongs with
|
||||
P5's spawner-wired namespaces if it is ever wanted.*
|
||||
|
||||
## H2 — SMEP
|
||||
|
||||
- Generalize the cpuid helper (`apic.zig:351-365`, private, subleaf-0) to
|
||||
a shared probe; gate on `cpuid(0).eax >= 7`.
|
||||
- Set CR4 bit 20 in `per-cpu.zig:initSystemCall` (or a sibling called
|
||||
from both `cpu.zig:148` and `smp.zig:181` — the one path both BSP and
|
||||
every AP already execute). Log enabled/absent (fail-open, IOMMU style).
|
||||
- Harness: add `-cpu max` to `qemu_args` (decision 8).
|
||||
|
||||
**Test:** new QEMU case `fault-smep` mirroring the `fault-*` injector
|
||||
pattern (`tests.zig:3906-3938`): ring-0 call through a pointer into a
|
||||
user-mapped page; expect `page fault (vector 14)` + `error code : 0x11` +
|
||||
kernel-half IP, machine reports the exception (deliberate-exception cases
|
||||
put the text in `expect`, per `qemu_test.py:189`). Suite 112.
|
||||
|
||||
## HS — SYSRET canonical-RIP guard
|
||||
|
||||
- `isr.s` syscall exit (`:256`): validate RCX canonicality before
|
||||
`sysretq`; non-canonical → `iretq` fallback (or kill), per the hazard
|
||||
note at `isr.s:192-194`.
|
||||
|
||||
**Test:** kernel unit case driving a thread whose return RIP is forged
|
||||
non-canonical via the syscall path if constructible cheaply; otherwise the
|
||||
review-level proof plus the existing fault cases regression. Suite 112.
|
||||
*(Landed: it was constructible, and from ring 3 rather than by forgery — the
|
||||
`sysret-canonical` case, suite 113.)*
|
||||
|
||||
## H3 — SMAP
|
||||
|
||||
- `clac` patch site at `isr_common` (`isr.s:367`, before the CPL test —
|
||||
ring-0 nesting inherits AC too): assemble a 3-byte NOP, patch to `clac`
|
||||
at boot through the physmap (the `process.zig:1990-1995` /
|
||||
`smp.zig:79-111` precedent), BSP-only before AP bring-up.
|
||||
- Set CR4 bit 21 in the same per-CPU init as SMEP.
|
||||
- Coding standards: kernel code touches user memory only through
|
||||
`user-memory`; no `stac` anywhere, ever.
|
||||
|
||||
**Test:** new QEMU case `fault-smap`: ring-0 deliberate read of a mapped
|
||||
user page; expect vector 14 + `error code : 0x1` + kernel IP. And the
|
||||
whole suite becomes the tripwire — any missed straggler now fails loudly.
|
||||
*(Landed: the patch site sits at the very first byte of `isr_common`, and
|
||||
CR4.SMAP is refused on every core until the patch has been read back through
|
||||
the text mapping it will execute from — so the ordering is enforced, not
|
||||
merely documented. Error code observed: `0x1` exactly, present and nothing
|
||||
else. The tripwire found no missed straggler: H1 had converted them all.)*
|
||||
Suite 114 (HS added one).
|
||||
|
||||
---
|
||||
|
||||
**Explicitly out of scope** (own tracks, after this plan): P5 restriction
|
||||
stage two (spawn's initial capability, namespace views, parked replies,
|
||||
dedicated killable channels — needs a design session on the spawn
|
||||
contract), file-path namespacing, trusted UI (display track), pipes/FIFOs
|
||||
(Python track), `/applications` and its storage, `fs_mount`/`spawn`/
|
||||
`klog_read` gating beyond the `/protocol` reserved prefix, KPTI, IPC
|
||||
priority inheritance.
|
||||
@@ -95,9 +95,9 @@ hypervisor configured for UEFI firmware and an xHCI USB controller.
|
||||
|
||||
| Requirement | Detail | Source |
|
||||
|---|---|---|
|
||||
| **x86-64, 64-bit only** | Kernel and loader are built exclusively for `x86_64`; the loader rejects any non-x86-64 kernel ELF (`error.WrongArchitecture`). | `build-support/build.zig` (`freestandingTarget`), `boot/efi.zig:622` |
|
||||
| **x86-64, 64-bit only** | Kernel and loader are built exclusively for `x86_64`; the loader rejects any non-x86-64 kernel ELF (`error.WrongArchitecture`). | `build.zig:481`, `boot/efi.zig:622` |
|
||||
| **Long mode + PAE + NX** | AP trampoline sets `CR4.PAE`, `EFER.LME`, `EFER.NXE`; NX is used in kernel page-table entries. | `system/kernel/architecture/x86_64/trampoline.s:62` |
|
||||
| **SSE / SSE2** | Baseline: the compiler emits SSE for ordinary struct copies. Trampoline enables `CR4.OSFXSR` + `OSXMMEXCPT` and clears `CR0.EM`. | `build-support/build.zig` (`freestandingTarget`), `trampoline.s:62` |
|
||||
| **SSE / SSE2** | Baseline: the compiler emits SSE for ordinary struct copies. Trampoline enables `CR4.OSFXSR` + `OSXMMEXCPT` and clears `CR0.EM`. | `build.zig:477`, `trampoline.s:62` |
|
||||
| **`syscall` / `sysret`** | Primary user↔kernel entry path. `EFER.SCE` enabled; `STAR`/`LSTAR`/`SFMASK` programmed per core. (`int 0x80` exists as a parallel gate.) | `architecture/x86_64/per-cpu.zig:71`, `isr.s:196` |
|
||||
| **Local APIC (xAPIC)** | LAPIC accessed via MMIO at `0xFEE00000`. LAPIC ID read as a `u8` — classic xAPIC. **x2APIC is not supported** (no MSR path). | `apic.zig:67`, `apic.zig:646` |
|
||||
| **CPUID + RDTSC** | CPUID leaf `0x15` for TSC frequency; RDTSC is the monotonic clock. | `apic.zig:333`, `apic.zig:113` |
|
||||
@@ -108,7 +108,7 @@ hypervisor configured for UEFI firmware and an xHCI USB controller.
|
||||
- **UEFI only.** A custom UEFI application loader is installed to
|
||||
`\EFI\BOOT\BOOTX64.efi`. There is **no BIOS, multiboot, or limine** path. The
|
||||
loader tolerates UEFI Class-3 machines with no legacy PIC/PIT.
|
||||
(`build/images.zig` — the EFI/BOOT install — and `boot/efi.zig`)
|
||||
(`build.zig:246`, `boot/efi.zig`)
|
||||
- **ACPI is the hardware-discovery mechanism.** The RSDP is taken from the UEFI
|
||||
configuration table (ACPI 2.0 GUID preferred, 1.0 fallback). Without a valid
|
||||
RSDP there is **no device discovery** — no SMP, no IOAPIC routing, no PCI/USB.
|
||||
|
||||
+1
-3
@@ -12,9 +12,7 @@ There are two layers:
|
||||
`system/abi.zig`, `library/device/model/device-abi.zig`) now spans ~26 modules:
|
||||
protocol and on-wire definitions (VFS, USB, virtio-gpu), the FAT engine, the
|
||||
display compositor, PS/2 and HID decoding, the kernel log ring, and the
|
||||
runtime's `time`/`thread` — the list is distributed across the library-domain
|
||||
and binary packages' own `test` steps, which the root `zig build test`
|
||||
aggregates (docs/build-packages-plan.md).
|
||||
runtime's `time`/`thread` — the full list is the test step in `build.zig`.
|
||||
These compile for the host and run natively.
|
||||
- **QEMU integration tests** (`python3 test/qemu_test.py`) — boot the real kernel
|
||||
and check its behaviour. This is the interesting part.
|
||||
|
||||
@@ -347,7 +347,7 @@ Two current decisions fall out of this roadmap:
|
||||
- [syscall.md](os-development/syscall.md) — the kernel↔runtime ABI `runtime.os` is built on.
|
||||
- [sysv.md](os-development/sysv.md) — the entry stack (`argc/argv/envp/auxv`) danos already constructs.
|
||||
- [ipc.md](device-driver-development/ipc.md) — the IPC the VFS/FAT operations travel over.
|
||||
- [file-system-hierarchy.md](file-system-development/file-system-hierarchy.md) — the
|
||||
- [danos-file-system-hierarchy-FSH.md](file-system-development/danos-file-system-hierarchy-FSH.md) — the
|
||||
filesystem layout the file surface serves.
|
||||
- [coding-standards.md](coding-standards.md) — danos naming (why the compat spellings
|
||||
are confined, and now retired).
|
||||
|
||||
@@ -0,0 +1,339 @@
|
||||
Vendor ID,Device ID,Graphics Family,GPU Name
|
||||
0x8086,0x0152,HD Graphics,Xeon E3-1200 v2/3rd Gen Core GT1
|
||||
0x8086,0x0155,HD Graphics,Xeon E3-1200 v2/3rd Gen Core
|
||||
0x8086,0x0156,HD Graphics,Ivy Bridge mobile GT1
|
||||
0x8086,0x0157,HD Graphics,Ivy Bridge mobile GT1
|
||||
0x8086,0x015a,HD Graphics,Xeon E3-1200 v2/Ivy Bridge
|
||||
0x8086,0x0162,HD Graphics,Ivy Bridge GT2 (HD Graphics 4000)
|
||||
0x8086,0x0166,HD Graphics,Ivy Bridge mobile GT2 (HD Graphics 4000)
|
||||
0x8086,0x016a,HD Graphics,Xeon E3-1200 v2/3rd Gen Core GT3
|
||||
0x8086,0x0402,HD Graphics,Xeon E3-1200 v3/4th Gen Core GT1
|
||||
0x8086,0x0406,HD Graphics,Haswell GT1
|
||||
0x8086,0x040a,HD Graphics,Xeon E3-1200 v3 GT1
|
||||
0x8086,0x040b,HD Graphics,Haswell GT1
|
||||
0x8086,0x040e,HD Graphics,Haswell GT1
|
||||
0x8086,0x0412,HD Graphics,Xeon E3-1200 v3/4th Gen Core GT2
|
||||
0x8086,0x0416,HD Graphics,4th Gen Core GT2
|
||||
0x8086,0x041a,HD Graphics,Xeon E3-1200 v3 GT2
|
||||
0x8086,0x041b,HD Graphics,Haswell GT2
|
||||
0x8086,0x041e,HD Graphics,4th Gen Core Family GT2
|
||||
0x8086,0x0422,HD Graphics,Haswell GT3
|
||||
0x8086,0x0426,HD Graphics,Haswell GT3
|
||||
0x8086,0x042a,HD Graphics,Haswell GT3
|
||||
0x8086,0x042b,HD Graphics,Haswell GT3
|
||||
0x8086,0x042e,HD Graphics,Haswell GT3
|
||||
0x8086,0x0a02,HD Graphics,Haswell-ULT GT1
|
||||
0x8086,0x0a06,HD Graphics,Haswell-ULT GT1
|
||||
0x8086,0x0a0a,HD Graphics,Haswell-ULT GT1
|
||||
0x8086,0x0a0b,HD Graphics,Haswell-ULT GT1
|
||||
0x8086,0x0a0e,HD Graphics,Haswell-ULT GT1
|
||||
0x8086,0x0a12,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a16,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a1a,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a1b,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a1e,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a22,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a26,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a2a,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a2b,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0a2e,HD Graphics,Haswell-ULT GT2
|
||||
0x8086,0x0d02,Iris Pro Graphics,Crystal Well GT1
|
||||
0x8086,0x0d06,Iris Pro Graphics,Crystal Well GT2
|
||||
0x8086,0x0d0a,Iris Pro Graphics,Crystal Well GT2
|
||||
0x8086,0x0d0b,Iris Pro Graphics,Crystal Well GT2
|
||||
0x8086,0x0d0e,Iris Pro Graphics,Crystal Well GT2
|
||||
0x8086,0x0d12,Iris Pro Graphics,Crystal Well GT3 (Iris Pro 5200)
|
||||
0x8086,0x0d16,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d1a,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d1b,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d1e,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d22,Iris Pro Graphics,Crystal Well (Iris Pro 5200)
|
||||
0x8086,0x0d26,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d2a,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d2b,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d2e,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d32,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d36,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x0d3a,Iris Pro Graphics,Crystal Well GT3
|
||||
0x8086,0x1602,HD Graphics,Broadwell-U GT1
|
||||
0x8086,0x1606,HD Graphics,Broadwell-U GT1
|
||||
0x8086,0x160a,HD Graphics,Broadwell-U GT1
|
||||
0x8086,0x160b,HD Graphics,Broadwell-U GT1
|
||||
0x8086,0x160d,HD Graphics,Broadwell-U GT1
|
||||
0x8086,0x160e,HD Graphics,Broadwell-U GT1
|
||||
0x8086,0x1612,HD Graphics,Broadwell-H GT2 (HD Graphics 5600)
|
||||
0x8086,0x1616,HD Graphics,Broadwell-U GT2 (HD Graphics 5500)
|
||||
0x8086,0x161a,HD Graphics,Broadwell-U GT2
|
||||
0x8086,0x161b,HD Graphics,Broadwell-U GT2
|
||||
0x8086,0x161d,HD Graphics,Broadwell-U GT2
|
||||
0x8086,0x161e,HD Graphics,Broadwell-Y GT2 (HD Graphics 5300)
|
||||
0x8086,0x1622,Iris Pro Graphics,Broadwell-DT/H GT3 (Iris Pro 6200)
|
||||
0x8086,0x1626,HD Graphics,Broadwell-U GT3 (HD Graphics 6000)
|
||||
0x8086,0x162a,Iris Pro Graphics,Broadwell-DT GT3 (Iris Pro P6300)
|
||||
0x8086,0x162b,Iris Graphics,Broadwell-U GT3 (Iris 6100)
|
||||
0x8086,0x162d,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x162e,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x1632,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x1636,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x163a,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x163b,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x163d,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x163e,HD Graphics,Broadwell-U GT3
|
||||
0x8086,0x1902,HD Graphics,Skylake-S GT1 (HD Graphics 510)
|
||||
0x8086,0x1906,HD Graphics,Skylake-U GT1 (HD Graphics 510)
|
||||
0x8086,0x190a,HD Graphics,Skylake-DT GT1
|
||||
0x8086,0x190b,HD Graphics,Skylake GT1 (HD Graphics 510)
|
||||
0x8086,0x190e,HD Graphics,Skylake GT1
|
||||
0x8086,0x1912,HD Graphics,Skylake-S GT2 (HD Graphics 530)
|
||||
0x8086,0x1913,HD Graphics,Skylake GT2
|
||||
0x8086,0x1915,HD Graphics,Skylake GT2
|
||||
0x8086,0x1916,HD Graphics,Skylake-U GT2 (HD Graphics 520)
|
||||
0x8086,0x1917,HD Graphics,Skylake GT2
|
||||
0x8086,0x191a,HD Graphics,Skylake GT2
|
||||
0x8086,0x191b,HD Graphics,Skylake-H GT2 (HD Graphics 530)
|
||||
0x8086,0x191d,HD Graphics,Skylake-DT/H GT2 (HD Graphics P530)
|
||||
0x8086,0x191e,HD Graphics,Skylake-Y GT2 (HD Graphics 515)
|
||||
0x8086,0x1921,HD Graphics,Skylake GT2 (HD Graphics 520)
|
||||
0x8086,0x1923,HD Graphics,Skylake GT2 (HD Graphics 535)
|
||||
0x8086,0x1926,Iris Graphics,Skylake-U GT3 (Iris Graphics 540)
|
||||
0x8086,0x1927,Iris Graphics,Skylake-U GT3 (Iris Graphics 550)
|
||||
0x8086,0x192a,Iris Graphics,Skylake GT3
|
||||
0x8086,0x192b,Iris Graphics,Skylake GT3 (Iris Graphics 555)
|
||||
0x8086,0x192d,Iris Graphics,Skylake-H GT3 (Iris Graphics P555)
|
||||
0x8086,0x1932,Iris Pro Graphics,Skylake GT4 (Iris Pro 580)
|
||||
0x8086,0x193a,Iris Pro Graphics,Skylake-H GT4 (Iris Pro P580)
|
||||
0x8086,0x193b,Iris Pro Graphics,Skylake-H GT4 (Iris Pro 580)
|
||||
0x8086,0x193d,Iris Pro Graphics,Skylake-H GT4 (Iris Pro P580)
|
||||
0x8086,0x1a84,UHD Graphics,Skylake-DT
|
||||
0x8086,0x1a85,UHD Graphics,Skylake-DT
|
||||
0x8086,0x5902,HD Graphics,Kaby Lake-S GT1 (HD Graphics 610)
|
||||
0x8086,0x5906,HD Graphics,Kaby Lake-U GT1 (HD Graphics 610)
|
||||
0x8086,0x5908,HD Graphics,Kaby Lake GT1
|
||||
0x8086,0x590a,HD Graphics,Kaby Lake GT1
|
||||
0x8086,0x590b,HD Graphics,Kaby Lake GT1 (HD Graphics 610)
|
||||
0x8086,0x590e,HD Graphics,Kaby Lake GT1
|
||||
0x8086,0x5912,HD Graphics,Kaby Lake-S GT2 (HD Graphics 630)
|
||||
0x8086,0x5913,HD Graphics,Kaby Lake GT2
|
||||
0x8086,0x5915,HD Graphics,Kaby Lake GT2
|
||||
0x8086,0x5916,HD Graphics,Kaby Lake-U GT2 (HD Graphics 620)
|
||||
0x8086,0x5917,UHD Graphics,Kaby Lake-R GT2 (UHD Graphics 620)
|
||||
0x8086,0x591a,HD Graphics,Kaby Lake GT2
|
||||
0x8086,0x591b,HD Graphics,Kaby Lake-H GT2 (HD Graphics 630)
|
||||
0x8086,0x591c,UHD Graphics,Kaby Lake GT2 (UHD Graphics 615)
|
||||
0x8086,0x591d,HD Graphics,Kaby Lake-DT GT2 (HD Graphics P630)
|
||||
0x8086,0x591e,HD Graphics,Kaby Lake-Y GT2 (HD Graphics 615)
|
||||
0x8086,0x5921,HD Graphics,Kaby Lake GT2 (HD Graphics 620)
|
||||
0x8086,0x5923,HD Graphics,Kaby Lake GT2 (HD Graphics 635)
|
||||
0x8086,0x5926,Iris Plus Graphics,Kaby Lake-U GT3 (Iris Plus 640)
|
||||
0x8086,0x5927,Iris Plus Graphics,Kaby Lake-U GT3 (Iris Plus 650)
|
||||
0x8086,0x592a,Iris Plus Graphics,Kaby Lake GT3
|
||||
0x8086,0x592b,Iris Plus Graphics,Kaby Lake GT3
|
||||
0x8086,0x5932,Iris Plus Graphics,Kaby Lake GT3
|
||||
0x8086,0x593a,Iris Plus Graphics,Kaby Lake GT3
|
||||
0x8086,0x593b,Iris Plus Graphics,Kaby Lake GT3
|
||||
0x8086,0x593d,Iris Plus Graphics,Kaby Lake GT3
|
||||
0x8086,0x5a40,Intel Graphics,Apollolake
|
||||
0x8086,0x5a41,Intel Graphics,Apollolake
|
||||
0x8086,0x5a42,Intel Graphics,Apollolake
|
||||
0x8086,0x5a44,Intel Graphics,Apollolake
|
||||
0x8086,0x5a49,Intel Graphics,Apollolake
|
||||
0x8086,0x5a4a,Intel Graphics,Apollolake
|
||||
0x8086,0x5a4c,Intel Graphics,Apollolake
|
||||
0x8086,0x5a50,Intel Graphics,Apollolake
|
||||
0x8086,0x5a51,Intel Graphics,Apollolake
|
||||
0x8086,0x5a52,Intel Graphics,Apollolake
|
||||
0x8086,0x5a54,Intel Graphics,Apollolake
|
||||
0x8086,0x5a59,Intel Graphics,Apollolake
|
||||
0x8086,0x5a5a,Intel Graphics,Apollolake
|
||||
0x8086,0x5a5c,Intel Graphics,Apollolake
|
||||
0x8086,0x5a71,Intel Graphics,Apollolake
|
||||
0x8086,0x5a79,Intel Graphics,Apollolake
|
||||
0x8086,0x5a84,HD Graphics,Apollo Lake GT1 (HD Graphics 505)
|
||||
0x8086,0x5a85,HD Graphics,Apollo Lake GT1 (HD Graphics 500)
|
||||
0x8086,0x3184,UHD Graphics,GeminiLake (UHD Graphics 605)
|
||||
0x8086,0x3185,UHD Graphics,GeminiLake (UHD Graphics 600)
|
||||
0x8086,0x3e90,UHD Graphics,Coffee Lake-S GT1 (UHD Graphics 610)
|
||||
0x8086,0x3e91,UHD Graphics,Coffee Lake-S GT2 (UHD Graphics 630)
|
||||
0x8086,0x3e92,UHD Graphics,Coffee Lake-S GT2 (UHD Graphics 630)
|
||||
0x8086,0x3e93,UHD Graphics,Coffee Lake-S GT1 (UHD Graphics 610)
|
||||
0x8086,0x3e94,UHD Graphics,Coffee Lake-S GT2 (UHD Graphics P630)
|
||||
0x8086,0x3e96,UHD Graphics,Coffee Lake-S GT2 (UHD Graphics P630)
|
||||
0x8086,0x3e98,UHD Graphics,Coffee Lake-S GT2 (UHD Graphics 630)
|
||||
0x8086,0x3e99,UHD Graphics,Coffee Lake GT2
|
||||
0x8086,0x3e9a,UHD Graphics,Coffee Lake-S GT2 (UHD Graphics P630)
|
||||
0x8086,0x3e9b,UHD Graphics,Coffee Lake-H GT2 (UHD Graphics 630)
|
||||
0x8086,0x3e9c,UHD Graphics,Coffee Lake-S GT1 (UHD Graphics 610)
|
||||
0x8086,0x3ea0,UHD Graphics,Whiskey Lake-U GT2 (UHD Graphics 620)
|
||||
0x8086,0x3ea1,UHD Graphics,Whiskey Lake-U GT1 (UHD Graphics 610)
|
||||
0x8086,0x3ea2,UHD Graphics,Whiskey Lake GT1
|
||||
0x8086,0x3ea3,UHD Graphics,Whiskey Lake GT1
|
||||
0x8086,0x3ea4,UHD Graphics,Whiskey Lake GT1
|
||||
0x8086,0x3ea5,Iris Plus Graphics,Coffee Lake-U GT3e (Iris Plus 655)
|
||||
0x8086,0x3ea6,Iris Plus Graphics,Coffee Lake-U GT3 (Iris Plus 645)
|
||||
0x8086,0x3ea7,Iris Plus Graphics,Whiskey Lake GT3
|
||||
0x8086,0x3ea8,Iris Plus Graphics,Coffee Lake-U GT3 (Iris Plus 655)
|
||||
0x8086,0x3ea9,UHD Graphics,Coffee Lake-U GT2 (UHD Graphics 620)
|
||||
0x8086,0x87c0,UHD Graphics,9th Gen Core (UHD Graphics 617)
|
||||
0x8086,0x87ca,UHD Graphics,9th Gen Core (UHD Graphics 617)
|
||||
0x8086,0x8a50,Iris Plus Graphics,Ice Lake Gen1
|
||||
0x8086,0x8a51,Iris Plus Graphics,Ice Lake GT2 (Iris Plus G7)
|
||||
0x8086,0x8a52,Iris Plus Graphics,Ice Lake GT2 (Iris Plus G7)
|
||||
0x8086,0x8a53,Iris Plus Graphics,Ice Lake GT2 (Iris Plus G7)
|
||||
0x8086,0x8a54,Iris Plus Graphics,Ice Lake GT2
|
||||
0x8086,0x8a56,Iris Plus Graphics,Ice Lake GT1 (Iris Plus G1)
|
||||
0x8086,0x8a57,Iris Plus Graphics,Ice Lake GT1
|
||||
0x8086,0x8a58,UHD Graphics,Ice Lake-Y GT1 (UHD Graphics G1)
|
||||
0x8086,0x8a59,UHD Graphics,Ice Lake GT1
|
||||
0x8086,0x8a5a,Iris Plus Graphics,Ice Lake GT4 (Iris Plus G4)
|
||||
0x8086,0x8a5b,Iris Plus Graphics,Ice Lake GT4
|
||||
0x8086,0x8a5c,Iris Plus Graphics,Ice Lake GT4 (Iris Plus G4)
|
||||
0x8086,0x8a5d,Iris Plus Graphics,Ice Lake GT4
|
||||
0x8086,0x8a70,Iris Plus Graphics,Ice Lake
|
||||
0x8086,0x8a71,Iris Plus Graphics,Ice Lake
|
||||
0x8086,0x9a40,Iris Xe Graphics,Tiger Lake-UP4 GT2
|
||||
0x8086,0x9a49,Iris Xe Graphics,Tiger Lake-LP GT2
|
||||
0x8086,0x9a59,Iris Xe Graphics,Tiger Lake GT2
|
||||
0x8086,0x9a60,UHD Graphics,Tiger Lake-H GT1
|
||||
0x8086,0x9a68,UHD Graphics,Tiger Lake-H GT1
|
||||
0x8086,0x9a70,UHD Graphics,Tiger Lake-H GT1
|
||||
0x8086,0x9a78,UHD Graphics,Tiger Lake-LP GT2 (UHD Graphics G4)
|
||||
0x8086,0x9ac0,Iris Xe Graphics,Tiger Lake
|
||||
0x8086,0x9ac9,Iris Xe Graphics,Tiger Lake
|
||||
0x8086,0x9ad9,Iris Xe Graphics,Tiger Lake
|
||||
0x8086,0x9af8,Iris Xe Graphics,Tiger Lake
|
||||
0x8086,0x9b21,UHD Graphics,Comet Lake-U GT2 (UHD Graphics 620)
|
||||
0x8086,0x9b41,UHD Graphics,Comet Lake-U GT2
|
||||
0x8086,0x9ba0,UHD Graphics,Comet Lake GT1
|
||||
0x8086,0x9ba2,UHD Graphics,Comet Lake GT1
|
||||
0x8086,0x9ba4,UHD Graphics,Comet Lake-H GT1 (UHD Graphics 610)
|
||||
0x8086,0x9ba5,UHD Graphics,Comet Lake GT1
|
||||
0x8086,0x9ba8,UHD Graphics,Comet Lake-S GT1 (UHD Graphics 610)
|
||||
0x8086,0x9baa,UHD Graphics,Comet Lake GT1
|
||||
0x8086,0x9bab,UHD Graphics,Comet Lake GT1
|
||||
0x8086,0x9bac,UHD Graphics,Comet Lake GT1
|
||||
0x8086,0x9bc0,UHD Graphics,Comet Lake GT2
|
||||
0x8086,0x9bc2,UHD Graphics,Comet Lake GT2
|
||||
0x8086,0x9bc4,UHD Graphics,Comet Lake-H GT2
|
||||
0x8086,0x9bc5,UHD Graphics,Comet Lake-S GT2 (UHD Graphics 630)
|
||||
0x8086,0x9bc6,UHD Graphics,Comet Lake-S GT2 (UHD Graphics P630)
|
||||
0x8086,0x9bc8,UHD Graphics,Comet Lake-S GT2 (UHD Graphics 630)
|
||||
0x8086,0x9bca,UHD Graphics,Comet Lake GT2
|
||||
0x8086,0x9bcb,UHD Graphics,Comet Lake GT2
|
||||
0x8086,0x9bcc,UHD Graphics,Comet Lake GT2
|
||||
0x8086,0x9be6,UHD Graphics,Comet Lake-S GT2 (UHD Graphics P630)
|
||||
0x8086,0x9bf6,UHD Graphics,Coffee Lake-S GT2 (UHD Graphics P630)
|
||||
0x8086,0x4555,UHD Graphics,Elkhart Lake GT2 (UHD Graphics Gen11 16EU)
|
||||
0x8086,0x4571,UHD Graphics,Elkhart Lake GT2 (UHD Graphics Gen11 32EU)
|
||||
0x8086,0x4500,Intel Graphics,Elkhart Lake GT1
|
||||
0x8086,0x4541,Intel Graphics,Elkhart Lake GT1
|
||||
0x8086,0x4551,Intel Graphics,Elkhart Lake GT1
|
||||
0x8086,0x4557,Intel Graphics,Elkhart Lake GT1
|
||||
0x8086,0x4570,Intel Graphics,Elkhart Lake GT1
|
||||
0x8086,0x4c80,Intel Graphics,Rocket Lake
|
||||
0x8086,0x4c8a,UHD Graphics,Rocket Lake-S GT1 (UHD Graphics 750)
|
||||
0x8086,0x4c8b,UHD Graphics,Rocket Lake-S GT1 (UHD Graphics 730)
|
||||
0x8086,0x4c8c,Intel Graphics,Rocket Lake GT1
|
||||
0x8086,0x4c90,UHD Graphics,Rocket Lake-S GT1 (UHD Graphics P750)
|
||||
0x8086,0x4c9a,UHD Graphics,Rocket Lake-S
|
||||
0x8086,0x4e51,Intel Graphics,Jasper Lake GT1
|
||||
0x8086,0x4e55,UHD Graphics,Jasper Lake
|
||||
0x8086,0x4e57,Intel Graphics,Jasper Lake GT1
|
||||
0x8086,0x4e61,UHD Graphics,Jasper Lake
|
||||
0x8086,0x4e71,UHD Graphics,Jasper Lake
|
||||
0x8086,0x4626,Intel Graphics,Alder Lake-P GT1
|
||||
0x8086,0x4628,UHD Graphics,Alder Lake-UP3 GT2
|
||||
0x8086,0x462a,Intel Graphics,Alder Lake-P GT1
|
||||
0x8086,0x4680,UHD Graphics,Alder Lake-S GT1 (UHD Graphics 770)
|
||||
0x8086,0x4681,Intel Graphics,Alder Lake-S GT1
|
||||
0x8086,0x4682,UHD Graphics,Alder Lake-S GT1 (UHD Graphics 730)
|
||||
0x8086,0x4683,Intel Graphics,Alder Lake-S GT1
|
||||
0x8086,0x4688,UHD Graphics,Alder Lake-HX GT1 (UHD Graphics 770)
|
||||
0x8086,0x4689,Intel Graphics,Alder Lake-HX GT1
|
||||
0x8086,0x468a,Intel Graphics,Alder Lake-S
|
||||
0x8086,0x468b,Intel Graphics,Alder Lake-S
|
||||
0x8086,0x4690,UHD Graphics,Alder Lake-S GT1 (UHD Graphics 770)
|
||||
0x8086,0x4691,Intel Graphics,Alder Lake-S GT1
|
||||
0x8086,0x4692,UHD Graphics,Alder Lake-S GT1 (UHD Graphics 730)
|
||||
0x8086,0x4693,UHD Graphics,Alder Lake-S GT1 (UHD Graphics 710)
|
||||
0x8086,0x46a0,Intel Graphics,Alder Lake-P GT2
|
||||
0x8086,0x46a1,UHD Graphics,Alder Lake-P GT1
|
||||
0x8086,0x46a2,Intel Graphics,Alder Lake-P GT1
|
||||
0x8086,0x46a3,UHD Graphics,Alder Lake-P GT1
|
||||
0x8086,0x46a6,Iris Xe Graphics,Alder Lake-P GT2
|
||||
0x8086,0x46a8,Iris Xe Graphics,Alder Lake-UP3 GT2
|
||||
0x8086,0x46aa,Iris Xe Graphics,Alder Lake-UP4 GT2
|
||||
0x8086,0x46b0,Iris Xe Graphics,Alder Lake-P
|
||||
0x8086,0x46b1,Iris Xe Graphics,Alder Lake-P
|
||||
0x8086,0x46b2,Intel Graphics,Alder Lake-P GT1
|
||||
0x8086,0x46b3,UHD Graphics,Alder Lake-UP3 GT1
|
||||
0x8086,0x46c0,Intel Graphics,Alder Lake-M GT1
|
||||
0x8086,0x46c1,Iris Xe Graphics,Alder Lake-M
|
||||
0x8086,0x46c2,Intel Graphics,Alder Lake-M GT1
|
||||
0x8086,0x46c3,UHD Graphics,Alder Lake-UP4 GT1
|
||||
0x8086,0x46d0,UHD Graphics,Alder Lake-N
|
||||
0x8086,0x46d1,UHD Graphics,Alder Lake-N
|
||||
0x8086,0x46d2,UHD Graphics,Alder Lake-N
|
||||
0x8086,0x46d3,Intel Graphics,Alder Lake-N
|
||||
0x8086,0x46d4,Intel Graphics,Alder Lake-N
|
||||
0x8086,0x7d40,Intel Graphics,Meteor Lake-M
|
||||
0x8086,0x7d41,Intel Graphics,Arrow Lake-U
|
||||
0x8086,0x7d45,Intel Graphics,Meteor Lake-P
|
||||
0x8086,0x7d51,Arc Pro Graphics,Arrow Lake-P (Arc Pro 130T/140T)
|
||||
0x8086,0x7d55,Intel Arc Graphics,Meteor Lake-P
|
||||
0x8086,0x7d60,Intel Graphics,Meteor Lake-M
|
||||
0x8086,0x7d67,Intel Graphics,Arrow Lake-S
|
||||
0x8086,0x7dd1,Intel Graphics,Arrow Lake-P
|
||||
0x8086,0x7dd5,Intel Graphics,Meteor Lake-P
|
||||
0x8086,0xa720,UHD Graphics,Raptor Lake-P
|
||||
0x8086,0xa721,UHD Graphics,Raptor Lake-P
|
||||
0x8086,0xa780,UHD Graphics,Raptor Lake-S GT1 (UHD Graphics 770)
|
||||
0x8086,0xa781,UHD Graphics,Raptor Lake-S
|
||||
0x8086,0xa782,UHD Graphics,Raptor Lake-S
|
||||
0x8086,0xa783,UHD Graphics,Raptor Lake-S
|
||||
0x8086,0xa788,UHD Graphics,Raptor Lake-S
|
||||
0x8086,0xa789,UHD Graphics,Raptor Lake-S
|
||||
0x8086,0xa78a,UHD Graphics,Raptor Lake-S
|
||||
0x8086,0xa78b,UHD Graphics,Raptor Lake-S
|
||||
0x8086,0xa7a0,Iris Xe Graphics,Raptor Lake-P
|
||||
0x8086,0xa7a1,Iris Xe Graphics,Raptor Lake-P
|
||||
0x8086,0xa7a8,UHD Graphics,Raptor Lake-P
|
||||
0x8086,0xa7a9,UHD Graphics,Raptor Lake-P
|
||||
0x8086,0xa7aa,Intel Graphics,Raptor Lake-P
|
||||
0x8086,0xa7ab,Intel Graphics,Raptor Lake-P
|
||||
0x8086,0xa7ac,Intel Graphics,Raptor Lake-U
|
||||
0x8086,0xa7ad,Intel Graphics,Raptor Lake-U
|
||||
0x8086,0xb640,Intel Graphics,Arrow Lake-H
|
||||
0x8086,0x4905,Iris Xe MAX Graphics,DG1
|
||||
0x8086,0x4906,Iris Xe Graphics,DG1
|
||||
0x8086,0x4907,Intel Graphics,DG1 Server
|
||||
0x8086,0x4908,Iris Xe Graphics,DG1
|
||||
0x8086,0x4909,Iris Xe MAX Graphics,DG1 (Iris Xe MAX 100)
|
||||
0x8086,0x5690,Arc Graphics,DG2 (Arc A770M)
|
||||
0x8086,0x5691,Arc Graphics,DG2 (Arc A730M)
|
||||
0x8086,0x5692,Arc Graphics,DG2 (Arc A550M)
|
||||
0x8086,0x5693,Arc Graphics,DG2 (Arc A370M)
|
||||
0x8086,0x5694,Arc Graphics,DG2 (Arc A350M)
|
||||
0x8086,0x5695,Iris Xe MAX Graphics,DG2 (Iris Xe MAX A200M)
|
||||
0x8086,0x5696,Arc Graphics,DG2 (Arc A570M)
|
||||
0x8086,0x5697,Arc Graphics,DG2 (Arc A530M)
|
||||
0x8086,0x5698,Arc Graphics,DG2 (Arc Xe Graphics)
|
||||
0x8086,0x56a0,Arc Graphics,DG2 (Arc A770)
|
||||
0x8086,0x56a1,Arc Graphics,DG2 (Arc A750)
|
||||
0x8086,0x56a2,Arc Graphics,DG2 (Arc A580)
|
||||
0x8086,0x56a3,Arc Graphics,DG2 (Arc Xe Graphics)
|
||||
0x8086,0x56a4,Arc Graphics,DG2 (Arc Xe Graphics)
|
||||
0x8086,0x56a5,Arc Graphics,DG2 (Arc A380)
|
||||
0x8086,0x56a6,Arc Graphics,DG2 (Arc A310)
|
||||
0x8086,0x56b0,Arc Pro Graphics,DG2 (Arc Pro A30M)
|
||||
0x8086,0x56b1,Arc Pro Graphics,DG2 (Arc Pro A40/A50)
|
||||
0x8086,0x56b2,Arc Pro Graphics,DG2 (Arc Pro A60M)
|
||||
0x8086,0x56b3,Arc Pro Graphics,DG2 (Arc Pro A60)
|
||||
0x8086,0x56ba,Arc Graphics,DG2 (Arc A380E)
|
||||
0x8086,0x56bb,Arc Graphics,DG2 (Arc A310E)
|
||||
0x8086,0x56bc,Arc Graphics,DG2 (Arc A370E)
|
||||
0x8086,0x56bd,Arc Graphics,DG2 (Arc A350E)
|
||||
0x8086,0x56be,Arc Graphics,DG2 (Arc A750E)
|
||||
0x8086,0x56bf,Arc Graphics,DG2 (Arc A580E)
|
||||
0x8086,0x56c0,Arc Graphics,DG2 (Data Center GPU Flex 170)
|
||||
0x8086,0x56c1,Arc Graphics,DG2 (Data Center GPU Flex 140)
|
||||
0x8086,0x56c2,Arc Graphics,DG2 (Data Center GPU Flex 170V)
|
||||
|
@@ -1,4 +1,4 @@
|
||||
# /system/configuration/devices.csv — the device→driver registry.
|
||||
# /etc/devices.csv — the device→driver registry.
|
||||
#
|
||||
# The device manager reads this at boot and binds each device a bus driver
|
||||
# reports to the driver named here. It is AUTHORITATIVE: a device that no row
|
||||
@@ -25,6 +25,7 @@
|
||||
#
|
||||
# bus base class prog_if vendor device subsystem hid driver
|
||||
pci, 0C, 03, 30, *, *, *, *, /system/drivers/usb-xhci-bus
|
||||
pci, 03, 00, 00, 8086, 4C8A, *, *, /system/drivers/intel-uhd-graphics-750
|
||||
pci, 03, 80, *, 1AF4, 1050, *, *, /system/drivers/virtio-gpu
|
||||
usb, 03, 01, 01, *, *, *, *, /system/drivers/usb-hid-keyboard
|
||||
usb, 03, 01, 02, *, *, *, *, /system/drivers/usb-hid-mouse
|
||||
|
@@ -1,10 +1,9 @@
|
||||
# /system/configuration/init.csv — diagnose variant (-Ddiagnose), bundled at
|
||||
# /system/configuration/init.csv.
|
||||
# /etc/init.csv — diagnose variant (-Ddiagnose), bundled at /etc/init.csv.
|
||||
#
|
||||
# The display stack (display, display-demo) is omitted so the kernel's timestamped
|
||||
# on-screen boot transcript is never suppressed — the bring-up timeline (USB,
|
||||
# storage, logger) stays readable on real hardware with no serial. See
|
||||
# system/configuration/init.csv for the format; this file must otherwise track it.
|
||||
# storage, logger) stays readable on real hardware with no serial. See etc/init.csv
|
||||
# for the format; this file must otherwise track it.
|
||||
#
|
||||
# service args...
|
||||
/system/services/input
|
||||
|
@@ -1,4 +1,4 @@
|
||||
# /system/configuration/init.csv — the services init (PID 1) starts at boot, in order.
|
||||
# /etc/init.csv — the services init (PID 1) starts at boot, in order.
|
||||
#
|
||||
# init reads this at startup and spawns each service supervised (restarting it on
|
||||
# a crash, up to a cap). Startup order is top->bottom; shutdown is the reverse, so
|
||||
@@ -9,7 +9,7 @@
|
||||
# '#' starts a comment (whole-line or trailing); blank lines are ignored. The
|
||||
# first field is the service binary path; any fields after it are the service's
|
||||
# argv. Drivers are absent on purpose — the device manager discovers hardware and
|
||||
# spawns those (see /system/configuration/devices.csv).
|
||||
# spawns those (see /etc/devices.csv).
|
||||
#
|
||||
# service args...
|
||||
/system/services/input
|
||||
|
@@ -1,50 +0,0 @@
|
||||
//! The "client" library domain (library/client): userspace-service clients —
|
||||
//! they talk to services over IPC, not to the kernel. Client modules end in
|
||||
//! `-client` the way wire protocols end in `-protocol`, so a service, its
|
||||
//! protocol, and its client never share a name (`display` the service,
|
||||
//! `display-protocol` the wire contract, `display-client` a program's view).
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const kernel = b.dependency("kernel", .{});
|
||||
const protocol = b.dependency("protocol", .{});
|
||||
|
||||
const ipc = kernel.module("ipc");
|
||||
const time = kernel.module("time");
|
||||
// Every client reaches its service by name now: resolve `/protocol/<name>`,
|
||||
// open it, and take the provider's endpoint out of the reply
|
||||
// (docs/os-development/protocol-namespace.md).
|
||||
const channel = kernel.module("channel");
|
||||
|
||||
// A client frames its own packets, so it needs the envelope alongside the
|
||||
// protocol whose verbs it speaks.
|
||||
const envelope = protocol.module("envelope");
|
||||
|
||||
_ = b.addModule("display-client", .{
|
||||
.root_source_file = b.path("display/display-client.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "channel", .module = channel },
|
||||
.{ .name = "envelope", .module = envelope },
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "time", .module = time },
|
||||
.{ .name = "display-protocol", .module = protocol.module("display-protocol") },
|
||||
},
|
||||
});
|
||||
_ = b.addModule("input-client", .{
|
||||
.root_source_file = b.path("input/input-client.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "channel", .module = channel },
|
||||
.{ .name = "envelope", .module = envelope },
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "time", .module = time },
|
||||
.{ .name = "input-protocol", .module = protocol.module("input-protocol") },
|
||||
},
|
||||
});
|
||||
|
||||
// Standalone `zig build test`, kept for uniformity across the domains (the
|
||||
// root aggregate depends on every domain's test step). The clients have no
|
||||
// host-runnable unit tests yet — they are thin IPC conversation wrappers —
|
||||
// so the step is empty until one grows some.
|
||||
_ = b.step("test", "Run the client unit tests (none yet)");
|
||||
}
|
||||
@@ -1,13 +0,0 @@
|
||||
.{
|
||||
.name = .client,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0xc74404553e73d4ff, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// The clients converse over ipc with time-bounded waits.
|
||||
.kernel = .{ .path = "../kernel" },
|
||||
// Each client speaks its service's wire protocol.
|
||||
.protocol = .{ .path = "../protocol" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -1,165 +0,0 @@
|
||||
//! User-space display client: talk to the display service (query the mode, and — from D3
|
||||
//! — create layers, draw, and present) without hand-rolling the IPC. The `runtime.block`
|
||||
//! shape: a cached `/protocol/display` open with a boot-race retry, then extern-struct request/
|
||||
//! reply marshalling. See system/services/display/ and docs/display.md.
|
||||
|
||||
const std = @import("std");
|
||||
const channel = @import("channel");
|
||||
const envelope = @import("envelope");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const display_protocol = @import("display-protocol");
|
||||
|
||||
const Protocol = display_protocol.Protocol;
|
||||
|
||||
/// The display's current mode, as `info()` reports it.
|
||||
pub const Info = struct {
|
||||
width: u32,
|
||||
height: u32,
|
||||
pitch: u32, // bytes per row (may exceed width*4; see docs/framebuffer.md)
|
||||
format: u32, // a device-abi DisplayFormat value (0 = rgbx, 1 = bgrx)
|
||||
};
|
||||
|
||||
/// The service endpoint, looked up once and cached.
|
||||
var handle: ?ipc.Handle = null;
|
||||
|
||||
/// Open `/protocol/display`, retrying while it comes up (a client races the
|
||||
/// service's bind at boot). Returns the endpoint, or null if it never appears.
|
||||
fn service() ?ipc.Handle {
|
||||
if (handle) |h| return h;
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 100) : (attempts += 1) {
|
||||
if (channel.openEndpoint("display")) |h| {
|
||||
handle = h;
|
||||
return h;
|
||||
}
|
||||
time.sleepMillis(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// A reply the compositor answered with, kept whole so the caller can decode the
|
||||
/// verb's own fixed part out of it.
|
||||
const Answered = struct {
|
||||
packet: [display_protocol.message_maximum]u8,
|
||||
len: usize,
|
||||
|
||||
fn bytes(self: *const Answered) []const u8 {
|
||||
return self.packet[0..self.len];
|
||||
}
|
||||
};
|
||||
|
||||
/// Send one request (`target` addresses a layer, or 0 for the compositor itself)
|
||||
/// and keep the reply. Null when the transport failed or the compositor refused.
|
||||
fn transact(
|
||||
comptime operation: Protocol.Operation,
|
||||
target: u64,
|
||||
request: Protocol.RequestOf(operation),
|
||||
tail: []const u8,
|
||||
) ?Answered {
|
||||
const h = service() orelse return null;
|
||||
var packet: [display_protocol.message_maximum]u8 = undefined;
|
||||
const framed = Protocol.encodeRequest(operation, target, request, tail, &packet) orelse return null;
|
||||
var answered: Answered = .{ .packet = undefined, .len = 0 };
|
||||
answered.len = ipc.call(h, framed, &answered.packet) catch return null;
|
||||
const status = envelope.statusOf(answered.bytes()) orelse return null;
|
||||
if (status.status != 0) return null;
|
||||
return answered;
|
||||
}
|
||||
|
||||
/// The display's current mode, or null if the service never came up.
|
||||
pub fn info() ?Info {
|
||||
const answered = transact(.info, 0, {}, &.{}) orelse return null;
|
||||
const reply = Protocol.decodeReply(.info, answered.bytes()) orelse return null;
|
||||
return .{ .width = reply.width, .height = reply.height, .pitch = reply.pitch, .format = reply.format };
|
||||
}
|
||||
|
||||
/// Composite the dirty layers and flush the frame to the screen.
|
||||
pub fn present() bool {
|
||||
return transact(.present, 0, {}, &.{}) != null;
|
||||
}
|
||||
|
||||
/// One selectable display mode.
|
||||
pub const Mode = display_protocol.Mode;
|
||||
|
||||
/// Fill `out` with the resolutions the display can switch to; returns how many were written
|
||||
/// (zero on the GOP floor, or if the service never came up).
|
||||
pub fn modes(out: []Mode) usize {
|
||||
const answered = transact(.get_modes, 0, {}, &.{}) orelse return 0;
|
||||
const offered = Protocol.decodeReply(.get_modes, answered.bytes()) orelse return 0;
|
||||
const count = @min(@min(offered.count, display_protocol.max_modes), out.len);
|
||||
for (0..count) |i| out[i] = offered.modes[i];
|
||||
return count;
|
||||
}
|
||||
|
||||
/// Change the display resolution. Only a native backend that supports mode-setting honours it
|
||||
/// (on the GOP floor it returns false); on success the display's `info()` reports the new mode.
|
||||
pub fn setMode(width: u32, height: u32) bool {
|
||||
const changed = transact(.set_mode, 0, .{ .width = width, .height = height }, &.{}) != null;
|
||||
if (changed) mode = null; // the cached mode is stale now
|
||||
return changed;
|
||||
}
|
||||
|
||||
/// The mode, cached after the first `info()` so `color()` doesn't round-trip per pixel.
|
||||
var mode: ?Info = null;
|
||||
|
||||
fn cachedInfo() ?Info {
|
||||
if (mode) |m| return m;
|
||||
const i = info() orelse return null;
|
||||
mode = i;
|
||||
return i;
|
||||
}
|
||||
|
||||
/// The native pixel value for an 8-bit-per-channel colour, in the display's format. A
|
||||
/// client packs colours through this so it never has to know the byte order itself.
|
||||
pub fn color(r: u8, g: u8, b: u8) u32 {
|
||||
const format = if (cachedInfo()) |i| i.format else 0;
|
||||
return display_protocol.pack(format, r, g, b);
|
||||
}
|
||||
|
||||
/// A handle to a server-owned layer: a positioned, z-ordered surface the client draws
|
||||
/// into by command. Create with `createLayer`; drawing and moves take effect on the next
|
||||
/// `present`. Coordinates are signed (a layer may sit partly off-screen).
|
||||
///
|
||||
/// The id is the packet header's `target` on every call below, so it is named once
|
||||
/// per request rather than repeated inside one.
|
||||
pub const Layer = struct {
|
||||
id: u32,
|
||||
|
||||
/// Fill a rectangle of this layer (layer-local coordinates) with a native `colour`.
|
||||
pub fn fill(self: Layer, x: i32, y: i32, w: u32, h: u32, colour: u32) bool {
|
||||
return transact(.fill_rect, self.id, .{ .x = x, .y = y, .width = w, .height = h, .colour = colour }, &.{}) != null;
|
||||
}
|
||||
|
||||
/// Copy a `w`×`h` tile of native pixels (row-major, little-endian bytes) into this
|
||||
/// layer at (`x`, `y`). The tile rides inline as the request's tail, so `w*h*4` must
|
||||
/// fit `display_protocol.maximum_payload` — the bound the protocol derives from this
|
||||
/// verb's own fixed part, so the check here can never drift from what fits.
|
||||
pub fn blitTile(self: Layer, x: i32, y: i32, w: u32, h: u32, pixels: []const u8) bool {
|
||||
if (pixels.len > display_protocol.maximum_payload) return false;
|
||||
return transact(.blit_tile, self.id, .{ .x = x, .y = y, .width = w, .height = h }, pixels) != null;
|
||||
}
|
||||
|
||||
/// Move / restack / show or hide the layer.
|
||||
pub fn configure(self: Layer, x: i32, y: i32, z: u32, visible: bool) bool {
|
||||
return transact(.configure_layer, self.id, .{ .x = x, .y = y, .z = z, .visible = if (visible) 1 else 0 }, &.{}) != null;
|
||||
}
|
||||
|
||||
/// Mark a rectangle of this layer (layer-local) dirty for the next present — for when
|
||||
/// the layer's pixels changed without a drawing call the compositor already tracked.
|
||||
pub fn damage(self: Layer, x: i32, y: i32, w: u32, h: u32) bool {
|
||||
return transact(.damage, self.id, .{ .x = x, .y = y, .width = w, .height = h }, &.{}) != null;
|
||||
}
|
||||
|
||||
/// Release the layer and its surface.
|
||||
pub fn destroy(self: Layer) bool {
|
||||
return transact(.destroy_layer, self.id, {}, &.{}) != null;
|
||||
}
|
||||
};
|
||||
|
||||
/// Create a server-owned layer of `w`×`h` pixels at screen (`x`, `y`) with stacking order
|
||||
/// `z` (higher is nearer the front), initially visible. Returns a handle, or null.
|
||||
pub fn createLayer(x: i32, y: i32, w: u32, h: u32, z: u32) ?Layer {
|
||||
const answered = transact(.create_layer, 0, .{ .x = x, .y = y, .width = w, .height = h, .z = z, .visible = 1 }, &.{}) orelse return null;
|
||||
return .{ .id = (Protocol.decodeReply(.create_layer, answered.bytes()) orelse return null).layer };
|
||||
}
|
||||
@@ -0,0 +1,198 @@
|
||||
//! User-space display client: talk to the display service (query the mode, and — from D3
|
||||
//! — create layers, draw, and present) without hand-rolling the IPC. The `runtime.block`
|
||||
//! shape: a cached `.display` lookup with a boot-race retry, then extern-struct request/
|
||||
//! reply marshalling. See system/services/display/ and docs/display.md.
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const display_protocol = @import("display-protocol");
|
||||
|
||||
/// The display's current mode, as `info()` reports it.
|
||||
pub const Info = struct {
|
||||
width: u32,
|
||||
height: u32,
|
||||
pitch: u32, // bytes per row (may exceed width*4; see docs/framebuffer.md)
|
||||
format: u32, // a device-abi DisplayFormat value (0 = rgbx, 1 = bgrx)
|
||||
};
|
||||
|
||||
/// The service endpoint, looked up once and cached.
|
||||
var handle: ?ipc.Handle = null;
|
||||
|
||||
/// Look up the display service, retrying while it comes up (a client races its
|
||||
/// registration at boot). Returns the endpoint, or null if it never appears.
|
||||
fn service() ?ipc.Handle {
|
||||
if (handle) |h| return h;
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 100) : (attempts += 1) {
|
||||
if (ipc.lookup(.display)) |h| {
|
||||
handle = h;
|
||||
return h;
|
||||
}
|
||||
time.sleepMillis(50);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Send one request, receive its reply; true on a zero status. `out` receives the reply
|
||||
/// so callers can read `info`/`layer` fields on success.
|
||||
fn transact(request: display_protocol.Request, out: *display_protocol.Reply) bool {
|
||||
const h = service() orelse return false;
|
||||
var req = request;
|
||||
var reply: [display_protocol.reply_size]u8 = undefined;
|
||||
const len = ipc.call(h, std.mem.asBytes(&req), &reply) catch return false;
|
||||
if (len < display_protocol.reply_size) return false;
|
||||
out.* = std.mem.bytesToValue(display_protocol.Reply, reply[0..display_protocol.reply_size]);
|
||||
return out.status == 0;
|
||||
}
|
||||
|
||||
/// The display's current mode, or null if the service never came up.
|
||||
pub fn info() ?Info {
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
if (!transact(.{ .operation = @intFromEnum(display_protocol.Operation.info) }, &reply)) return null;
|
||||
return .{ .width = reply.width, .height = reply.height, .pitch = reply.pitch, .format = reply.format };
|
||||
}
|
||||
|
||||
/// Composite the dirty layers and flush the frame to the screen.
|
||||
pub fn present() bool {
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
return transact(.{ .operation = @intFromEnum(display_protocol.Operation.present) }, &reply);
|
||||
}
|
||||
|
||||
/// One selectable display mode.
|
||||
pub const Mode = display_protocol.Mode;
|
||||
|
||||
/// Fill `out` with the resolutions the display can switch to; returns how many were written
|
||||
/// (zero on the GOP floor, or if the service never came up).
|
||||
pub fn modes(out: []Mode) usize {
|
||||
const h = service() orelse return 0;
|
||||
var request = display_protocol.Request{ .operation = @intFromEnum(display_protocol.Operation.get_modes) };
|
||||
var reply: [display_protocol.modes_reply_size]u8 = undefined;
|
||||
const len = ipc.call(h, std.mem.asBytes(&request), &reply) catch return 0;
|
||||
if (len < display_protocol.modes_reply_size) return 0;
|
||||
const answer = std.mem.bytesToValue(display_protocol.ModesReply, reply[0..display_protocol.modes_reply_size]);
|
||||
if (answer.status != 0) return 0;
|
||||
const count = @min(@min(answer.count, display_protocol.max_modes), out.len);
|
||||
for (0..count) |i| out[i] = answer.modes[i];
|
||||
return count;
|
||||
}
|
||||
|
||||
/// Change the display resolution. Only a native backend that supports mode-setting honours it
|
||||
/// (on the GOP floor it returns false); on success the display's `info()` reports the new mode.
|
||||
pub fn setMode(width: u32, height: u32) bool {
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
const changed = transact(.{ .operation = @intFromEnum(display_protocol.Operation.set_mode), .width = width, .height = height }, &reply);
|
||||
if (changed) mode = null; // the cached mode is stale now
|
||||
return changed;
|
||||
}
|
||||
|
||||
/// The mode, cached after the first `info()` so `color()` doesn't round-trip per pixel.
|
||||
var mode: ?Info = null;
|
||||
|
||||
fn cachedInfo() ?Info {
|
||||
if (mode) |m| return m;
|
||||
const i = info() orelse return null;
|
||||
mode = i;
|
||||
return i;
|
||||
}
|
||||
|
||||
/// The native pixel value for an 8-bit-per-channel colour, in the display's format. A
|
||||
/// client packs colours through this so it never has to know the byte order itself.
|
||||
pub fn color(r: u8, g: u8, b: u8) u32 {
|
||||
const format = if (cachedInfo()) |i| i.format else 0;
|
||||
return display_protocol.pack(format, r, g, b);
|
||||
}
|
||||
|
||||
/// A handle to a server-owned layer: a positioned, z-ordered surface the client draws
|
||||
/// into by command. Create with `createLayer`; drawing and moves take effect on the next
|
||||
/// `present`. Coordinates are signed (a layer may sit partly off-screen).
|
||||
pub const Layer = struct {
|
||||
id: u32,
|
||||
|
||||
/// Fill a rectangle of this layer (layer-local coordinates) with a native `colour`.
|
||||
pub fn fill(self: Layer, x: i32, y: i32, w: u32, h: u32, colour: u32) bool {
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
return transact(.{
|
||||
.operation = @intFromEnum(display_protocol.Operation.fill_rect),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
.width = w,
|
||||
.height = h,
|
||||
.colour = colour,
|
||||
}, &reply);
|
||||
}
|
||||
|
||||
/// Copy a `w`×`h` tile of native pixels (row-major, little-endian bytes) into this
|
||||
/// layer at (`x`, `y`). The tile rides inline in the request, so `w*h*4` must fit
|
||||
/// `display_protocol.maximum_payload`.
|
||||
pub fn blitTile(self: Layer, x: i32, y: i32, w: u32, h: u32, pixels: []const u8) bool {
|
||||
var request = display_protocol.Request{
|
||||
.operation = @intFromEnum(display_protocol.Operation.blit_tile),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
.width = w,
|
||||
.height = h,
|
||||
};
|
||||
const header = std.mem.asBytes(&request);
|
||||
if (header.len + pixels.len > display_protocol.message_maximum) return false;
|
||||
var buffer: [display_protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(buffer[0..header.len], header);
|
||||
@memcpy(buffer[header.len..][0..pixels.len], pixels);
|
||||
const h_svc = service() orelse return false;
|
||||
var reply: [display_protocol.reply_size]u8 = undefined;
|
||||
const len = ipc.call(h_svc, buffer[0 .. header.len + pixels.len], &reply) catch return false;
|
||||
if (len < display_protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(display_protocol.Reply, reply[0..display_protocol.reply_size]).status == 0;
|
||||
}
|
||||
|
||||
/// Move / restack / show or hide the layer.
|
||||
pub fn configure(self: Layer, x: i32, y: i32, z: u32, visible: bool) bool {
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
return transact(.{
|
||||
.operation = @intFromEnum(display_protocol.Operation.configure_layer),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
.z = z,
|
||||
.visible = if (visible) 1 else 0,
|
||||
}, &reply);
|
||||
}
|
||||
|
||||
/// Mark a rectangle of this layer (layer-local) dirty for the next present — for when
|
||||
/// the layer's pixels changed without a drawing call the compositor already tracked.
|
||||
pub fn damage(self: Layer, x: i32, y: i32, w: u32, h: u32) bool {
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
return transact(.{
|
||||
.operation = @intFromEnum(display_protocol.Operation.damage),
|
||||
.layer = self.id,
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
.width = w,
|
||||
.height = h,
|
||||
}, &reply);
|
||||
}
|
||||
|
||||
/// Release the layer and its surface.
|
||||
pub fn destroy(self: Layer) bool {
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
return transact(.{ .operation = @intFromEnum(display_protocol.Operation.destroy_layer), .layer = self.id }, &reply);
|
||||
}
|
||||
};
|
||||
|
||||
/// Create a server-owned layer of `w`×`h` pixels at screen (`x`, `y`) with stacking order
|
||||
/// `z` (higher is nearer the front), initially visible. Returns a handle, or null.
|
||||
pub fn createLayer(x: i32, y: i32, w: u32, h: u32, z: u32) ?Layer {
|
||||
var reply: display_protocol.Reply = undefined;
|
||||
if (!transact(.{
|
||||
.operation = @intFromEnum(display_protocol.Operation.create_layer),
|
||||
.x = @bitCast(x),
|
||||
.y = @bitCast(y),
|
||||
.width = w,
|
||||
.height = h,
|
||||
.z = z,
|
||||
.visible = 1,
|
||||
}, &reply)) return null;
|
||||
return .{ .id = reply.layer };
|
||||
}
|
||||
@@ -21,14 +21,12 @@
|
||||
//! if (event.asKeyboard()) |k| { ... } else if (event.asMouse()) |m| { ... }
|
||||
//! }
|
||||
|
||||
const channel = @import("channel");
|
||||
const envelope = @import("envelope");
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const input_protocol = @import("input-protocol");
|
||||
|
||||
const Protocol = input_protocol.Protocol;
|
||||
|
||||
pub const DeviceKind = input_protocol.DeviceKind;
|
||||
pub const InputEvent = input_protocol.InputEvent;
|
||||
pub const KeyEvent = input_protocol.KeyEvent;
|
||||
@@ -46,13 +44,13 @@ pub const device_mouse = input_protocol.device_mouse;
|
||||
pub const device_joystick = input_protocol.device_joystick;
|
||||
pub const device_all = input_protocol.device_all;
|
||||
|
||||
/// Open `/protocol/input`, retrying while it is still coming up. Both a subscriber and
|
||||
/// a source race the service's bind at boot, so both wait for it here rather than
|
||||
/// failing. Returns the provider's endpoint handle, or null if it never appears.
|
||||
/// Look up the input service, retrying while it is still coming up. Both a subscriber and
|
||||
/// a source race the service's registration at boot, so both wait for it here rather than
|
||||
/// failing. Returns the service endpoint handle, or null if it never appears.
|
||||
fn lookupService() ?ipc.Handle {
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 100) : (attempts += 1) {
|
||||
if (channel.openEndpoint("input")) |handle| return handle;
|
||||
if (ipc.lookup(.input)) |handle| return handle;
|
||||
time.sleepMillis(50);
|
||||
}
|
||||
return null;
|
||||
@@ -68,44 +66,31 @@ pub const Subscriber = struct {
|
||||
/// The endpoint the service delivers events to (created and owned by us; its handle
|
||||
/// was handed to the service as a capability at subscribe time).
|
||||
endpoint: ipc.Handle,
|
||||
/// A pushed packet is the folded header plus one typed event, so the buffer is
|
||||
/// the push floor rather than any one event's size.
|
||||
receive: [envelope.post_maximum]u8 = undefined,
|
||||
receive: [input_protocol.event_size]u8 = undefined,
|
||||
|
||||
/// Block until the next event is pushed, and return it. Events arrive as asynchronous
|
||||
/// buffered messages (`ipc_send` from the service), so nothing is owed in reply — the
|
||||
/// empty reply this issues is a harmless no-op. Returns null for any non-event wake-up
|
||||
/// (there should be none), so callers can loop.
|
||||
///
|
||||
/// The device class is the packet's operation, so it is read from the header and
|
||||
/// re-tagged into an `InputEvent` here — one decoded type for a caller that took
|
||||
/// several classes on one stream.
|
||||
pub fn next(self: *Subscriber) ?InputEvent {
|
||||
const got = ipc.replyWait(self.endpoint, &.{}, &self.receive, null);
|
||||
if (!got.isMessage()) return null;
|
||||
const packet = self.receive[0..@min(got.len, self.receive.len)];
|
||||
return switch (Protocol.eventOf(packet) orelse return null) {
|
||||
.keyboard => InputEvent.fromKeyboard(Protocol.decodeEvent(.keyboard, packet) orelse return null),
|
||||
.mouse => InputEvent.fromMouse(Protocol.decodeEvent(.mouse, packet) orelse return null),
|
||||
.joystick => InputEvent.fromJoystick(Protocol.decodeEvent(.joystick, packet) orelse return null),
|
||||
};
|
||||
if (!got.isMessage() or got.len < input_protocol.event_size) return null;
|
||||
return std.mem.bytesToValue(InputEvent, self.receive[0..input_protocol.event_size]);
|
||||
}
|
||||
};
|
||||
|
||||
/// Subscribe to the input classes named in `device_mask` (an OR of `device_*`, or
|
||||
/// `device_all`). Creates an endpoint for the service to push to and hands it over as a
|
||||
/// capability — the envelope's reserved `subscribe`, whose shape this is exactly. Returns
|
||||
/// a `Subscriber` to loop `next` on, or null on failure.
|
||||
/// capability. Returns a `Subscriber` to loop `next` on, or null on failure.
|
||||
pub fn subscribe(device_mask: u32) ?Subscriber {
|
||||
const service = lookupService() orelse return null;
|
||||
const endpoint = ipc.createIpcEndpoint() orelse return null;
|
||||
|
||||
var packet: [input_protocol.message_maximum]u8 = undefined;
|
||||
const framed = input_protocol.encodeSubscribe(device_mask, &packet) orelse return null;
|
||||
var reply: [input_protocol.message_maximum]u8 = undefined;
|
||||
const result = ipc.callCap(service, framed, &reply, endpoint) catch return null;
|
||||
const status = envelope.statusOf(reply[0..result.len]) orelse return null;
|
||||
if (status.status != 0) return null;
|
||||
var request = input_protocol.Request{ .operation = @intFromEnum(input_protocol.Operation.subscribe), .device_mask = device_mask };
|
||||
var reply: [input_protocol.reply_size]u8 = undefined;
|
||||
const result = ipc.callCap(service, std.mem.asBytes(&request), &reply, endpoint) catch return null;
|
||||
if (result.len < input_protocol.reply_size) return null;
|
||||
if (std.mem.bytesToValue(input_protocol.Reply, reply[0..input_protocol.reply_size]).status != 0) return null;
|
||||
return .{ .endpoint = endpoint };
|
||||
}
|
||||
|
||||
@@ -164,12 +149,11 @@ pub const Publisher = struct {
|
||||
service: ipc.Handle,
|
||||
|
||||
fn publish(self: Publisher, event: InputEvent) bool {
|
||||
var packet: [input_protocol.message_maximum]u8 = undefined;
|
||||
const framed = Protocol.encodeRequest(.publish, 0, event, &.{}, &packet) orelse return false;
|
||||
var reply: [input_protocol.message_maximum]u8 = undefined;
|
||||
const len = ipc.call(self.service, framed, &reply) catch return false;
|
||||
const status = envelope.statusOf(reply[0..len]) orelse return false;
|
||||
return status.status == 0;
|
||||
var request = input_protocol.Request{ .operation = @intFromEnum(input_protocol.Operation.publish), .event = event };
|
||||
var reply: [input_protocol.reply_size]u8 = undefined;
|
||||
const len = ipc.call(self.service, std.mem.asBytes(&request), &reply) catch return false;
|
||||
if (len < input_protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(input_protocol.Reply, reply[0..input_protocol.reply_size]).status == 0;
|
||||
}
|
||||
|
||||
/// Broadcast a keyboard event to every subscriber that took keyboard events.
|
||||
@@ -1,20 +0,0 @@
|
||||
//! The "csv" library domain: shared CSV helpers (comment stripping, field
|
||||
//! iteration) for the /system/configuration/*.csv config files — the device registry and the
|
||||
//! init service list both parse them.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
_ = b.addModule("csv", .{ .root_source_file = b.path("csv.zig") });
|
||||
|
||||
// Standalone `zig build test` for this domain alone; the root build keeps
|
||||
// its aggregate test step.
|
||||
const test_step = b.step("test", "Run the csv unit tests");
|
||||
const csv_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("csv.zig"),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(csv_tests).step);
|
||||
}
|
||||
@@ -1,8 +0,0 @@
|
||||
.{
|
||||
.name = .csv,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x8a4525791f4e5b6, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{},
|
||||
.paths = .{""},
|
||||
}
|
||||
+2
-2
@@ -1,5 +1,5 @@
|
||||
//! Minimal CSV helpers shared by the `/system/configuration/*.csv` config files — the device
|
||||
//! registry (`/system/configuration/devices.csv`) and the init service list (`/system/configuration/init.csv`).
|
||||
//! Minimal CSV helpers shared by the `/etc/*.csv` config files — the device
|
||||
//! registry (`/etc/devices.csv`) and the init service list (`/etc/init.csv`).
|
||||
//! Freestanding, no allocator: returned fields are slices into the source line,
|
||||
//! so the source must outlive them. `#` starts a comment (whole-line or trailing);
|
||||
//! whitespace around a field is trimmed, so columns may be padded for alignment.
|
||||
|
||||
@@ -7,14 +7,11 @@
|
||||
//! `runtime.dma.alloc`), so whole sectors move without crossing the IPC size
|
||||
//! limit — the same handoff usb-storage uses toward the controller.
|
||||
|
||||
const channel = @import("channel");
|
||||
const envelope = @import("envelope");
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const block_protocol = @import("block-protocol");
|
||||
|
||||
const Protocol = block_protocol.Protocol;
|
||||
|
||||
pub const Geometry = struct { block_size: u32, block_count: u64 };
|
||||
|
||||
pub const Device = struct {
|
||||
@@ -22,9 +19,12 @@ pub const Device = struct {
|
||||
|
||||
/// The device's block size and total block count.
|
||||
pub fn geometry(self: Device) ?Geometry {
|
||||
var reply: [block_protocol.message_maximum]u8 = undefined;
|
||||
const answered = self.call(.geometry, {}, null, &reply) orelse return null;
|
||||
const result = Protocol.decodeReply(.geometry, answered) orelse return null;
|
||||
var request = block_protocol.Request{ .operation = @intFromEnum(block_protocol.Operation.geometry), .lba = 0, .count = 0, .physical = 0 };
|
||||
var reply: [block_protocol.reply_size]u8 = undefined;
|
||||
const n = ipc.call(self.endpoint, std.mem.asBytes(&request), &reply) catch return null;
|
||||
if (n < block_protocol.reply_size) return null;
|
||||
const result = std.mem.bytesToValue(block_protocol.Reply, reply[0..block_protocol.reply_size]);
|
||||
if (result.status != 0) return null;
|
||||
return .{ .block_size = result.block_size, .block_count = result.block_count };
|
||||
}
|
||||
|
||||
@@ -33,56 +33,47 @@ pub const Device = struct {
|
||||
/// addresses become reachable by the device. Call once per buffer before naming it
|
||||
/// in `read`/`write`. Harmless success when no IOMMU is enforcing.
|
||||
pub fn attach(self: Device, handle: ipc.Handle) bool {
|
||||
var reply: [block_protocol.message_maximum]u8 = undefined;
|
||||
return self.call(.attach, {}, handle, &reply) != null;
|
||||
var request = block_protocol.Request{ .operation = @intFromEnum(block_protocol.Operation.attach), .lba = 0, .count = 0, .physical = 0 };
|
||||
var reply: [block_protocol.reply_size]u8 = undefined;
|
||||
const result = ipc.callCap(self.endpoint, std.mem.asBytes(&request), &reply, handle) catch return false;
|
||||
if (result.len < block_protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(block_protocol.Reply, reply[0..block_protocol.reply_size]).status == 0;
|
||||
}
|
||||
|
||||
/// Read `count` blocks starting at `lba` into the DMA buffer at `physical`.
|
||||
pub fn read(self: Device, lba: u64, count: u32, physical: u64) bool {
|
||||
var reply: [block_protocol.message_maximum]u8 = undefined;
|
||||
return self.call(.read, .{ .lba = lba, .count = count, .physical = physical }, null, &reply) != null;
|
||||
return self.transfer(.read, lba, count, physical);
|
||||
}
|
||||
|
||||
/// Write `count` blocks starting at `lba` from the DMA buffer at `physical`.
|
||||
pub fn write(self: Device, lba: u64, count: u32, physical: u64) bool {
|
||||
var reply: [block_protocol.message_maximum]u8 = undefined;
|
||||
return self.call(.write, .{ .lba = lba, .count = count, .physical = physical }, null, &reply) != null;
|
||||
return self.transfer(.write, lba, count, physical);
|
||||
}
|
||||
|
||||
/// Commit any device write cache to stable media (SCSI SYNCHRONIZE CACHE), so
|
||||
/// prior writes survive a power-off. A filesystem calls this before the machine
|
||||
/// goes down; no data transfer, so the buffer arguments are unused.
|
||||
pub fn flush(self: Device) bool {
|
||||
var reply: [block_protocol.message_maximum]u8 = undefined;
|
||||
return self.call(.flush, {}, null, &reply) != null;
|
||||
return self.transfer(.flush, 0, 0, 0);
|
||||
}
|
||||
|
||||
/// One request at the driver. `target` is always 0: one endpoint per device, so
|
||||
/// there is no object within the peer to address.
|
||||
fn call(
|
||||
self: Device,
|
||||
comptime operation: Protocol.Operation,
|
||||
request: Protocol.RequestOf(operation),
|
||||
capability: ?ipc.Handle,
|
||||
reply: []u8,
|
||||
) ?[]u8 {
|
||||
var packet: [block_protocol.message_maximum]u8 = undefined;
|
||||
const framed = Protocol.encodeRequest(operation, 0, request, &.{}, &packet) orelse return null;
|
||||
const answer = ipc.callCap(self.endpoint, framed, reply, capability) catch return null;
|
||||
const status = envelope.statusOf(reply[0..answer.len]) orelse return null;
|
||||
if (status.status != 0) return null;
|
||||
return reply[0..answer.len];
|
||||
fn transfer(self: Device, operation: block_protocol.Operation, lba: u64, count: u32, physical: u64) bool {
|
||||
var request = block_protocol.Request{ .operation = @intFromEnum(operation), .lba = lba, .count = count, .physical = physical };
|
||||
var reply: [block_protocol.reply_size]u8 = undefined;
|
||||
const n = ipc.call(self.endpoint, std.mem.asBytes(&request), &reply) catch return false;
|
||||
if (n < block_protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(block_protocol.Reply, reply[0..block_protocol.reply_size]).status == 0;
|
||||
}
|
||||
};
|
||||
|
||||
/// One open attempt, no waiting — for a server that retries on its own
|
||||
/// One lookup attempt, no waiting — for a server that retries on its own
|
||||
/// timer (the fat service) instead of blocking its harness in here.
|
||||
pub fn tryOpen() ?Device {
|
||||
if (channel.openEndpoint("block")) |handle| return .{ .endpoint = handle };
|
||||
if (ipc.lookup(.block)) |handle| return .{ .endpoint = handle };
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Open `/protocol/block`, retrying generously while the USB storage chain
|
||||
/// Look up the block device, retrying generously while the USB storage chain
|
||||
/// (controller reset, enumeration, mass-storage bring-up) comes up.
|
||||
pub fn open() ?Device {
|
||||
// Patient: the whole USB storage chain (firmware discovery, xHCI reset and
|
||||
@@ -93,7 +84,7 @@ pub fn open() ?Device {
|
||||
// completed at ~24 s); a machine whose stick genuinely failed setup should
|
||||
// not sit a further minute pretending otherwise.
|
||||
while (attempts < 600) : (attempts += 1) {
|
||||
if (channel.openEndpoint("block")) |handle| return .{ .endpoint = handle };
|
||||
if (ipc.lookup(.block)) |handle| return .{ .endpoint = handle };
|
||||
time.sleepMillis(50);
|
||||
}
|
||||
return null;
|
||||
|
||||
@@ -1,147 +0,0 @@
|
||||
//! The "device" library domain (library/device): what a driver author imports.
|
||||
//! The flat reference data (device-abi, pci-class, acpi-ids, usb-abi, usb-ids),
|
||||
//! typed MMIO access, the driver-side client libraries (driver, pci, usb,
|
||||
//! block), the AML interpreter, and the data-driven device registry.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const kernel = b.dependency("kernel", .{});
|
||||
const protocol = b.dependency("protocol", .{});
|
||||
const csv = b.dependency("csv", .{});
|
||||
|
||||
const abi = kernel.module("abi");
|
||||
const system_call = kernel.module("system-call");
|
||||
const ipc = kernel.module("ipc");
|
||||
const time = kernel.module("time");
|
||||
// A driver finds the bus it attaches to by name — `/protocol/device-manager`,
|
||||
// `/protocol/usb-transfer`, `/protocol/block`
|
||||
// (docs/os-development/protocol-namespace.md).
|
||||
const channel = kernel.module("channel");
|
||||
|
||||
// The devices sub-project's public interface (the flat wire types),
|
||||
// importable by user space, unlike the kernel-internal device model it
|
||||
// also feeds (system/kernel/device-model.zig).
|
||||
const device_abi = b.addModule("device-abi", .{
|
||||
.root_source_file = b.path("model/device-abi.zig"),
|
||||
});
|
||||
// PCI class-code decoding (class/subclass/prog-IF -> names). Pure reference
|
||||
// data, shared by kernel discovery and any user-space PCI tool.
|
||||
const pci_class = b.addModule("pci-class", .{
|
||||
.root_source_file = b.path("pci/pci-class.zig"),
|
||||
});
|
||||
// ACPI/PnP hardware-ID (_HID) names — the flat analog of pci-class.
|
||||
_ = b.addModule("acpi-ids", .{
|
||||
.root_source_file = b.path("acpi/acpi-ids.zig"),
|
||||
});
|
||||
// The AML interpreter, a build module so the ring-3 acpi service can run
|
||||
// the same parser the kernel does (docs/discovery.md). Pure Zig, no kernel
|
||||
// imports — one source, two builds.
|
||||
_ = b.addModule("aml", .{
|
||||
.root_source_file = b.path("acpi/aml/aml.zig"),
|
||||
});
|
||||
// The USB device-framework wire ABI (chapter-9 set-up packets, standard +
|
||||
// class requests, descriptors) and the USB class-code taxonomy.
|
||||
const usb_abi = b.addModule("usb-abi", .{
|
||||
.root_source_file = b.path("usb/usb-abi.zig"),
|
||||
});
|
||||
const usb_ids = b.addModule("usb-ids", .{
|
||||
.root_source_file = b.path("usb/usb-ids.zig"),
|
||||
});
|
||||
// Typed volatile MMIO register access + memory-ordering barriers, for
|
||||
// drivers on top of an mmio_map grant. Depends only on `builtin`.
|
||||
const mmio = b.addModule("mmio", .{
|
||||
.root_source_file = b.path("mmio/mmio.zig"),
|
||||
});
|
||||
// The driver author's interface: device access + the device-manager hello
|
||||
// handshake, folded together.
|
||||
const driver = b.addModule("driver", .{
|
||||
.root_source_file = b.path("driver/driver.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi },
|
||||
.{ .name = "channel", .module = channel },
|
||||
.{ .name = "device-abi", .module = device_abi },
|
||||
.{ .name = "envelope", .module = protocol.module("envelope") },
|
||||
.{ .name = "system-call", .module = system_call },
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "time", .module = time },
|
||||
.{ .name = "device-manager-protocol", .module = protocol.module("device-manager-protocol") },
|
||||
},
|
||||
});
|
||||
// A device driver's view of its claimed PCI function: config-space header
|
||||
// fields, BAR decode + map, capability walks (legacy + extended), MSI/MSI-X
|
||||
// programming, power state, and function-level reset — the generic PCI
|
||||
// mechanics every leaf PCI driver used to re-derive inline.
|
||||
_ = b.addModule("pci", .{
|
||||
.root_source_file = b.path("pci/pci.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "driver", .module = driver },
|
||||
.{ .name = "mmio", .module = mmio },
|
||||
.{ .name = "pci-class", .module = pci_class },
|
||||
.{ .name = "time", .module = time },
|
||||
},
|
||||
});
|
||||
// The USB class-driver transfer client: open a device on the xHCI bus and
|
||||
// drive it (control / interrupt / bulk). Re-exports usb-abi / usb-ids as
|
||||
// usb.abi / usb.ids for a single USB import.
|
||||
_ = b.addModule("usb", .{
|
||||
.root_source_file = b.path("usb/usb.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "channel", .module = channel },
|
||||
.{ .name = "envelope", .module = protocol.module("envelope") },
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "time", .module = time },
|
||||
.{ .name = "usb-transfer-protocol", .module = protocol.module("usb-transfer-protocol") },
|
||||
.{ .name = "usb-abi", .module = usb_abi },
|
||||
.{ .name = "usb-ids", .module = usb_ids },
|
||||
},
|
||||
});
|
||||
// The block-device client — a device type, so it lives here.
|
||||
_ = b.addModule("block", .{
|
||||
.root_source_file = b.path("block/block.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "channel", .module = channel },
|
||||
.{ .name = "envelope", .module = protocol.module("envelope") },
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "time", .module = time },
|
||||
.{ .name = "block-protocol", .module = protocol.module("block-protocol") },
|
||||
},
|
||||
});
|
||||
// The device registry: parse /system/configuration/devices.csv into match rules and bind a
|
||||
// reported device to a driver. Pure logic (no hardware, no syscalls), so it
|
||||
// unit-tests on the host; the device manager imports it.
|
||||
_ = b.addModule("device-registry", .{
|
||||
.root_source_file = b.path("registry/device-registry.zig"),
|
||||
.imports = &.{.{ .name = "csv", .module = csv.module("csv") }},
|
||||
});
|
||||
|
||||
// Standalone `zig build test` for this domain alone; the root build keeps
|
||||
// its aggregate test step.
|
||||
const test_step = b.step("test", "Run the device library unit tests");
|
||||
for ([_][]const u8{
|
||||
"model/device-abi.zig", // wire-type sizes
|
||||
"pci/pci-class.zig", // class/subclass/prog-IF name decoding
|
||||
"acpi/acpi-ids.zig", // _HID name decoding
|
||||
"acpi/aml/aml.zig", // AML parse + interpret, incl. Notify dispatch
|
||||
"usb/usb-abi.zig", // wire sizes + bit packings + set-up packet encodings
|
||||
"usb/usb-ids.zig", // class/subclass/protocol code assignments
|
||||
"mmio/mmio.zig", // barriers assemble + registers round-trip
|
||||
}) |root| {
|
||||
const device_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(root),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(device_tests).step);
|
||||
}
|
||||
// The registry needs its csv import wired, so it doesn't fit the loop.
|
||||
const registry_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("registry/device-registry.zig"),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
.imports = &.{.{ .name = "csv", .module = csv.module("csv") }},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(registry_tests).step);
|
||||
}
|
||||
@@ -1,15 +0,0 @@
|
||||
.{
|
||||
.name = .device,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x92fb68eace23a4f, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// driver/block/usb/pci build on the kernel library's concern modules.
|
||||
.kernel = .{ .path = "../kernel" },
|
||||
// driver speaks device-manager-protocol; block/usb their transfer protocols.
|
||||
.protocol = .{ .path = "../protocol" },
|
||||
// device-registry parses /system/configuration/devices.csv with the shared csv helpers.
|
||||
.csv = .{ .path = "../csv" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -7,8 +7,6 @@ const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const device_abi = @import("device-abi");
|
||||
const sc = @import("system-call");
|
||||
const channel = @import("channel");
|
||||
const envelope = @import("envelope");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const device_manager_protocol = @import("device-manager-protocol");
|
||||
@@ -169,38 +167,25 @@ const lookup_pause_ms: u64 = 20;
|
||||
/// (best-effort standalone bring-up) or it refused the handshake. Bus drivers keep the handle
|
||||
/// to report children through; a driver that runs fine unsupervised discards it with `_ =`,
|
||||
/// and one that requires supervision bails on null. Logs the outcome itself.
|
||||
///
|
||||
/// The device this driver was assigned is the packet's `Header.target` — the manager's
|
||||
/// object addressing, so `no_device` here is a driver that serves none.
|
||||
pub fn hello(role: Role, device_id: u64) ?ipc.Handle {
|
||||
var attempts: u32 = 0;
|
||||
const manager = while (attempts < lookup_attempts) : (attempts += 1) {
|
||||
if (channel.openEndpoint("device-manager")) |handle| break handle;
|
||||
if (ipc.lookup(.device_manager)) |handle| break handle;
|
||||
time.sleepMillis(lookup_pause_ms);
|
||||
} else {
|
||||
std.log.info("no device manager to hello", .{});
|
||||
return null;
|
||||
};
|
||||
|
||||
var packet: [device_manager_protocol.message_maximum]u8 = undefined;
|
||||
const framed = device_manager_protocol.Protocol.encodeRequest(
|
||||
.hello,
|
||||
device_id,
|
||||
.{ .role = @intFromEnum(role) },
|
||||
&.{},
|
||||
&packet,
|
||||
) orelse return null;
|
||||
|
||||
var reply: [device_manager_protocol.message_maximum]u8 = undefined;
|
||||
const length = ipc.call(manager, framed, &reply) catch {
|
||||
const message = device_manager_protocol.Hello{ .role = @intFromEnum(role), .device_id = device_id };
|
||||
var reply: [device_manager_protocol.reply_size]u8 = undefined;
|
||||
const length = ipc.call(manager, std.mem.asBytes(&message), &reply) catch {
|
||||
std.log.info("hello call failed", .{});
|
||||
return null;
|
||||
};
|
||||
const status = envelope.statusOf(reply[0..length]) orelse {
|
||||
std.log.info("hello answered nothing readable", .{});
|
||||
return null;
|
||||
};
|
||||
if (status.status != 0) {
|
||||
if (length < device_manager_protocol.reply_size or
|
||||
std.mem.bytesToValue(device_manager_protocol.HelloReply, reply[0..device_manager_protocol.reply_size]).status != 0)
|
||||
{
|
||||
std.log.info("hello refused", .{});
|
||||
return null;
|
||||
}
|
||||
|
||||
@@ -126,7 +126,7 @@ pub const DeviceDescriptor = extern struct {
|
||||
// names with the pci-class module.
|
||||
pci_class: u64,
|
||||
// Numeric identity beyond the class triple, mirrored in the bus report's
|
||||
// ChildAdded so /system/configuration/devices.csv can bind on it: `vendor`/`device` are the PCI
|
||||
// ChildAdded so /etc/devices.csv can bind on it: `vendor`/`device` are the PCI
|
||||
// vendor/device (or USB idVendor/idProduct), `subsystem` is the PCI subsystem id
|
||||
// packed `(subsystem_vendor << 16) | subsystem_device`. Zero where the bus has no
|
||||
// such concept. Defaulted so existing descriptor literals keep compiling and lay
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
//! The device registry: parse `/system/configuration/devices.csv` into match rules and bind a
|
||||
//! The device registry: parse `/etc/devices.csv` into match rules and bind a
|
||||
//! reported device to a driver. This is the data-driven replacement for the
|
||||
//! device manager's three hand-written `switch` tables (`pciDriverForIdentity`,
|
||||
//! `hidDriverFor`, `usbDriverForIdentity`); the registry is now **authoritative**
|
||||
@@ -11,7 +11,7 @@
|
||||
//! That keeps this module freestanding and unit-testable with plain `zig test`.
|
||||
//!
|
||||
//! The file format (docs/device-driver-development/device-manager.md, and the
|
||||
//! `/system/configuration/devices.csv` header itself): one rule per line, nine comma-separated
|
||||
//! `/etc/devices.csv` header itself): one rule per line, nine comma-separated
|
||||
//! fields, `#` starts a comment (whole-line or trailing), blank lines ignored.
|
||||
//!
|
||||
//! bus, base, class, prog_if, vendor, device, subsystem, hid, driver
|
||||
@@ -226,7 +226,7 @@ fn parseLine(line: []const u8) Line {
|
||||
} };
|
||||
}
|
||||
|
||||
/// Parse a whole `/system/configuration/devices.csv` into `out_rules`. The string fields of the
|
||||
/// Parse a whole `/etc/devices.csv` into `out_rules`. The string fields of the
|
||||
/// returned rules point into `source`, which must outlive them.
|
||||
pub fn parse(source: []const u8, out_rules: []Rule) ParseResult {
|
||||
var result: ParseResult = .{ .count = 0, .malformed = 0, .truncated = false };
|
||||
|
||||
+45
-79
@@ -11,24 +11,15 @@
|
||||
//! _ = device.subscribeInterrupt(address, length); // reports arrive asynchronously
|
||||
//! while (true) { ... ipc.replyWait(device.endpoint, ...) ... } // its own loop
|
||||
//!
|
||||
//! Reports are delivered to `device.endpoint` as asynchronous `interrupt_report`
|
||||
//! event packets, decoded with `reportOf` (the class driver runs a bare `replyWait`
|
||||
//! loop to read them, because the service harness drops buffered-message payloads
|
||||
//! — see service.zig).
|
||||
//!
|
||||
//! Every packet this file lays down is an envelope packet: the verb and the
|
||||
//! device token in the folded `Header`, the transfer's own fields after it, and
|
||||
//! a control transfer's data stage in the tail.
|
||||
//! Reports are delivered to `device.endpoint` as asynchronous `InterruptReport`
|
||||
//! messages (the class driver runs a bare `replyWait` loop to read them, because
|
||||
//! the service harness drops buffered-message payloads — see service.zig).
|
||||
|
||||
const std = @import("std");
|
||||
const channel = @import("channel");
|
||||
const envelope = @import("envelope");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const usb_transfer_protocol = @import("usb-transfer-protocol");
|
||||
|
||||
const Protocol = usb_transfer_protocol.Protocol;
|
||||
|
||||
/// The USB chapter-9 wire ABI and the class taxonomy, re-exported so a class driver reaches
|
||||
/// the whole USB domain through its one `usb` import (`usb.abi.getDescriptor`, `usb.ids.Class`).
|
||||
pub const abi = @import("usb-abi");
|
||||
@@ -65,41 +56,21 @@ pub const Device = struct {
|
||||
return null;
|
||||
}
|
||||
|
||||
/// One request at the bus driver, addressing this device by its token — the
|
||||
/// packet's `Header.target`, so no request body ever names the device again.
|
||||
/// Null covers both a failed transport and a refusal: a class driver has the
|
||||
/// same recourse either way.
|
||||
fn call(
|
||||
self: *Device,
|
||||
comptime operation: Protocol.Operation,
|
||||
request: Protocol.RequestOf(operation),
|
||||
tail: []const u8,
|
||||
capability: ?ipc.Handle,
|
||||
reply: []u8,
|
||||
) ?[]u8 {
|
||||
var packet: [usb_transfer_protocol.message_maximum]u8 = undefined;
|
||||
const framed = Protocol.encodeRequest(operation, self.token, request, tail, &packet) orelse return null;
|
||||
const answer = ipc.callCap(self.bus, framed, reply, capability) catch return null;
|
||||
const status = envelope.statusOf(reply[0..answer.len]) orelse return null;
|
||||
if (status.status != 0) return null;
|
||||
return reply[0..answer.len];
|
||||
}
|
||||
|
||||
/// The data stage rides the tail in both directions, so the answer's length
|
||||
/// *is* the transferred length — `Status.len`, which the envelope stamps.
|
||||
fn controlTransfer(self: *Device, setup: [8]u8, direction_in: bool, data: []u8) ?usize {
|
||||
if (data.len > usb_transfer_protocol.max_inline_data) return null;
|
||||
const outgoing: []const u8 = if (direction_in) &.{} else data;
|
||||
var reply: [usb_transfer_protocol.message_maximum]u8 = undefined;
|
||||
const answered = self.call(.control, .{
|
||||
var request = usb_transfer_protocol.ControlRequest{
|
||||
.device_token = self.token,
|
||||
.setup = setup,
|
||||
.direction_in = @intFromBool(direction_in),
|
||||
.data_length = @intCast(data.len),
|
||||
}, outgoing, null, &reply) orelse return null;
|
||||
|
||||
const returned = Protocol.replyTail(.control, answered);
|
||||
const actual = @min(returned.len, data.len);
|
||||
if (direction_in and actual > 0) @memcpy(data[0..actual], returned[0..actual]);
|
||||
};
|
||||
if (!direction_in and data.len > 0) @memcpy(request.data[0..data.len], data);
|
||||
var reply: [@sizeOf(usb_transfer_protocol.ControlReply)]u8 = undefined;
|
||||
const length = ipc.call(self.bus, std.mem.asBytes(&request), &reply) catch return null;
|
||||
if (length < @sizeOf(usb_transfer_protocol.ControlReply)) return null;
|
||||
const control_reply = std.mem.bytesToValue(usb_transfer_protocol.ControlReply, reply[0..@sizeOf(usb_transfer_protocol.ControlReply)]);
|
||||
if (control_reply.status != 0) return null;
|
||||
const actual = @min(control_reply.actual_length, data.len);
|
||||
if (direction_in and actual > 0) @memcpy(data[0..actual], control_reply.data[0..actual]);
|
||||
return actual;
|
||||
}
|
||||
|
||||
@@ -117,11 +88,15 @@ pub const Device = struct {
|
||||
/// Begin periodic IN polling of an interrupt endpoint; reports flow back to
|
||||
/// `self.endpoint` as asynchronous `InterruptReport` messages.
|
||||
pub fn subscribeInterrupt(self: *Device, endpoint_address: u8, max_length: u16) bool {
|
||||
var reply: [usb_transfer_protocol.message_maximum]u8 = undefined;
|
||||
return self.call(.interrupt_subscribe, .{
|
||||
var request = usb_transfer_protocol.InterruptSubscribeRequest{
|
||||
.device_token = self.token,
|
||||
.endpoint_address = endpoint_address,
|
||||
.max_length = max_length,
|
||||
}, &.{}, null, &reply) != null;
|
||||
};
|
||||
var reply: [@sizeOf(usb_transfer_protocol.InterruptSubscribeReply)]u8 = undefined;
|
||||
const length = ipc.call(self.bus, std.mem.asBytes(&request), &reply) catch return false;
|
||||
if (length < @sizeOf(usb_transfer_protocol.InterruptSubscribeReply)) return false;
|
||||
return std.mem.bytesToValue(usb_transfer_protocol.InterruptSubscribeReply, reply[0..@sizeOf(usb_transfer_protocol.InterruptSubscribeReply)]).status == 0;
|
||||
}
|
||||
|
||||
/// Hand the controller a DMA-region capability (`handle` — from a `shareable`
|
||||
@@ -130,57 +105,48 @@ pub const Device = struct {
|
||||
/// will name in a `bulk` transfer, before the transfer. Harmless (and a no-op
|
||||
/// success) when no IOMMU is enforcing. Returns false on failure.
|
||||
pub fn attachDma(self: *Device, handle: ipc.Handle) bool {
|
||||
var reply: [usb_transfer_protocol.message_maximum]u8 = undefined;
|
||||
return self.call(.dma_attach, {}, &.{}, handle, &reply) != null;
|
||||
var request = usb_transfer_protocol.DmaAttachRequest{ .device_token = self.token };
|
||||
var reply: [@sizeOf(usb_transfer_protocol.DmaAttachReply)]u8 = undefined;
|
||||
const result = ipc.callCap(self.bus, std.mem.asBytes(&request), &reply, handle) catch return false;
|
||||
if (result.len < @sizeOf(usb_transfer_protocol.DmaAttachReply)) return false;
|
||||
return std.mem.bytesToValue(usb_transfer_protocol.DmaAttachReply, reply[0..@sizeOf(usb_transfer_protocol.DmaAttachReply)]).status == 0;
|
||||
}
|
||||
|
||||
/// One bulk transfer (IN or OUT per `endpoint_address`'s direction bit) to or
|
||||
/// from the caller's own DMA buffer at `physical`. Returns the bytes moved.
|
||||
pub fn bulk(self: *Device, endpoint_address: u8, physical: u64, length: u32) ?u32 {
|
||||
var reply: [usb_transfer_protocol.message_maximum]u8 = undefined;
|
||||
const answered = self.call(.bulk, .{
|
||||
var request = usb_transfer_protocol.BulkRequest{
|
||||
.device_token = self.token,
|
||||
.physical_address = physical,
|
||||
.length = length,
|
||||
.endpoint_address = endpoint_address,
|
||||
}, &.{}, null, &reply) orelse return null;
|
||||
return (Protocol.decodeReply(.bulk, answered) orelse return null).actual_length;
|
||||
};
|
||||
var reply: [@sizeOf(usb_transfer_protocol.BulkReply)]u8 = undefined;
|
||||
const replied = ipc.call(self.bus, std.mem.asBytes(&request), &reply) catch return null;
|
||||
if (replied < @sizeOf(usb_transfer_protocol.BulkReply)) return null;
|
||||
const bulk_reply = std.mem.bytesToValue(usb_transfer_protocol.BulkReply, reply[0..@sizeOf(usb_transfer_protocol.BulkReply)]);
|
||||
if (bulk_reply.status != 0) return null;
|
||||
return bulk_reply.actual_length;
|
||||
}
|
||||
};
|
||||
|
||||
/// Decode one asynchronous interrupt report out of a packet that arrived on the
|
||||
/// class driver's own endpoint. Null when it is not one — a stray message, or a
|
||||
/// packet too short to carry the report it names. The device it came from is the
|
||||
/// packet's `Header.target`, which a single-device class driver never has to read.
|
||||
pub fn reportOf(packet: []const u8) ?InterruptReport {
|
||||
const event = Protocol.eventOf(packet) orelse return null;
|
||||
if (event != .interrupt_report) return null;
|
||||
return Protocol.decodeEvent(.interrupt_report, packet);
|
||||
}
|
||||
|
||||
/// Open `/protocol/usb-transfer` and, on that channel, open the device with the
|
||||
/// assigned id, handing over a freshly created endpoint for asynchronous interrupt
|
||||
/// reports. Retries while the bus is still coming up (a class driver races the bus
|
||||
/// driver at boot). Two opens, deliberately: the first names the contract, the
|
||||
/// second names an object within it.
|
||||
/// Look up the USB bus and open the device with the assigned id, handing over a
|
||||
/// freshly created endpoint for asynchronous interrupt reports. Retries while the
|
||||
/// bus is still coming up (a class driver races the bus driver at boot).
|
||||
pub fn open(device_id: u64) ?Device {
|
||||
var attempts: usize = 0;
|
||||
const bus = while (attempts < 100) : (attempts += 1) {
|
||||
if (channel.openEndpoint("usb-transfer")) |handle| break handle;
|
||||
if (ipc.lookup(.usb_bus)) |handle| break handle;
|
||||
time.sleepMillis(20);
|
||||
} else return null;
|
||||
|
||||
const endpoint = ipc.createIpcEndpoint() orelse return null;
|
||||
// The assigned device id is the target: it is what the caller has before a
|
||||
// token exists, and the token the reply hands back addresses every packet
|
||||
// after this one.
|
||||
var packet: [usb_transfer_protocol.message_maximum]u8 = undefined;
|
||||
const framed = Protocol.encodeRequest(.open, device_id, {}, &.{}, &packet) orelse return null;
|
||||
var reply: [usb_transfer_protocol.message_maximum]u8 = undefined;
|
||||
const result = ipc.callCap(bus, framed, &reply, endpoint) catch return null;
|
||||
const answered = reply[0..result.len];
|
||||
const status = envelope.statusOf(answered) orelse return null;
|
||||
if (status.status != 0) return null;
|
||||
const open_reply = Protocol.decodeReply(.open, answered) orelse return null;
|
||||
var request = usb_transfer_protocol.OpenRequest{ .device_id = device_id };
|
||||
var reply: [@sizeOf(usb_transfer_protocol.OpenReply)]u8 = undefined;
|
||||
const result = ipc.callCap(bus, std.mem.asBytes(&request), &reply, endpoint) catch return null;
|
||||
if (result.len < @sizeOf(usb_transfer_protocol.OpenReply)) return null;
|
||||
const open_reply = std.mem.bytesToValue(usb_transfer_protocol.OpenReply, reply[0..@sizeOf(usb_transfer_protocol.OpenReply)]);
|
||||
if (open_reply.status != 0) return null;
|
||||
|
||||
var device = Device{
|
||||
.bus = bus,
|
||||
|
||||
@@ -1,143 +0,0 @@
|
||||
//! The "kernel" library domain (library/kernel): the userspace private-ABI
|
||||
//! library (kernel32-style), split by concern into directly-importable
|
||||
//! modules. The graph is a DAG: memory depends on thread (heap needs
|
||||
//! Thread.Mutex), and thread does its own raw mmap so there is no cycle.
|
||||
//!
|
||||
//! This package also exports `abi` — the kernel <-> user contract (SystemCall
|
||||
//! numbers, mmap prot flags, page_size). Its source lives with the kernel in
|
||||
//! system/abi.zig, outside this directory, but userspace's one view of it is
|
||||
//! exported here so every consumer names the same module instance. Reaching
|
||||
//! outside the package root means this package is valid only as an in-repo
|
||||
//! path dependency (never fetchable by hash) — fine, since path dependencies
|
||||
//! are the only way danos packages are consumed.
|
||||
//!
|
||||
//! The root shim (root.zig) and the user link script (user.ld) are plain
|
||||
//! files, not modules; build-support reaches them through this package's
|
||||
//! directory (Dependency.path).
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const protocol = b.dependency("protocol", .{});
|
||||
|
||||
const abi = b.addModule("abi", .{
|
||||
.root_source_file = b.path("../../system/abi.zig"),
|
||||
});
|
||||
const system_call = b.addModule("system-call", .{
|
||||
.root_source_file = b.path("system-call.zig"),
|
||||
.imports = &.{.{ .name = "abi", .module = abi }},
|
||||
});
|
||||
const ipc = b.addModule("ipc", .{
|
||||
.root_source_file = b.path("ipc.zig"),
|
||||
.imports = &.{ .{ .name = "abi", .module = abi }, .{ .name = "system-call", .module = system_call } },
|
||||
});
|
||||
const time = b.addModule("time", .{
|
||||
.root_source_file = b.path("time.zig"),
|
||||
.imports = &.{.{ .name = "system-call", .module = system_call }},
|
||||
});
|
||||
const thread = b.addModule("thread", .{
|
||||
.root_source_file = b.path("thread.zig"),
|
||||
.imports = &.{ .{ .name = "abi", .module = abi }, .{ .name = "system-call", .module = system_call } },
|
||||
});
|
||||
const logging = b.addModule("logging", .{
|
||||
.root_source_file = b.path("logging.zig"),
|
||||
.imports = &.{ .{ .name = "abi", .module = abi }, .{ .name = "system-call", .module = system_call } },
|
||||
});
|
||||
const process = b.addModule("process", .{
|
||||
.root_source_file = b.path("process.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi },
|
||||
.{ .name = "system-call", .module = system_call },
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "time", .module = time },
|
||||
},
|
||||
});
|
||||
const file_system = b.addModule("file-system", .{
|
||||
.root_source_file = b.path("file-system.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi },
|
||||
.{ .name = "system-call", .module = system_call },
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "vfs-protocol", .module = protocol.module("vfs-protocol") },
|
||||
.{ .name = "envelope", .module = protocol.module("envelope") },
|
||||
},
|
||||
});
|
||||
// The channel is the L1 concept made concrete (docs/os-development/communication.md):
|
||||
// it needs the namespace (file-system, to resolve a /protocol name) and the
|
||||
// transport (ipc) both, which is why it lives here rather than in a protocol
|
||||
// module — those import nothing.
|
||||
const channel = b.addModule("channel", .{
|
||||
.root_source_file = b.path("channel.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "time", .module = time },
|
||||
.{ .name = "file-system", .module = file_system },
|
||||
.{ .name = "vfs-protocol", .module = protocol.module("vfs-protocol") },
|
||||
.{ .name = "envelope", .module = protocol.module("envelope") },
|
||||
},
|
||||
});
|
||||
_ = b.addModule("memory", .{
|
||||
.root_source_file = b.path("memory/memory.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi },
|
||||
.{ .name = "system-call", .module = system_call },
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "thread", .module = thread },
|
||||
},
|
||||
});
|
||||
// The harness binds the service's contract name at startup, which is a
|
||||
// conversation with the registry — hence channel (and time, for the patience
|
||||
// a provider that beat init to the mount needs). It also owns the subscriber
|
||||
// table and the fan-out, which are expressed in the envelope's vocabulary
|
||||
// (the reserved subscribe verb, the push floor) — hence envelope.
|
||||
_ = b.addModule("service", .{
|
||||
.root_source_file = b.path("service.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "channel", .module = channel },
|
||||
.{ .name = "envelope", .module = protocol.module("envelope") },
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "process", .module = process },
|
||||
},
|
||||
});
|
||||
_ = b.addModule("start", .{
|
||||
.root_source_file = b.path("start.zig"),
|
||||
.imports = &.{ .{ .name = "process", .module = process }, .{ .name = "logging", .module = logging } },
|
||||
});
|
||||
|
||||
// Standalone `zig build test` for this domain alone; the root build keeps
|
||||
// its aggregate test step. time and thread pull in the syscall wrappers,
|
||||
// which need the `abi` module; their danos seams fall back to host
|
||||
// primitives off the danos target, so they run with real host threads.
|
||||
const test_step = b.step("test", "Run the kernel library unit tests");
|
||||
for ([_][]const u8{
|
||||
"time.zig", // Instant/Duration arithmetic
|
||||
"thread.zig", // Mutex/Condition/RwLock/WaitGroup state machines
|
||||
}) |root| {
|
||||
const kernel_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(root),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
.imports = &.{.{ .name = "abi", .module = abi }},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(kernel_tests).step);
|
||||
}
|
||||
|
||||
// channel needs its whole import set to compile at all; only its framing is
|
||||
// host-runnable (the syscall seams are x86_64-only, and unreferenced from
|
||||
// the tests), so that is what it tests.
|
||||
const channel_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("channel.zig"),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
.imports = &.{
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "time", .module = time },
|
||||
.{ .name = "file-system", .module = file_system },
|
||||
.{ .name = "vfs-protocol", .module = protocol.module("vfs-protocol") },
|
||||
.{ .name = "envelope", .module = protocol.module("envelope") },
|
||||
},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(channel_tests).step);
|
||||
}
|
||||
@@ -1,11 +0,0 @@
|
||||
.{
|
||||
.name = .kernel,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x5dd29aab36503453, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// file-system speaks the VFS wire protocol.
|
||||
.protocol = .{ .path = "../protocol" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -1,339 +0,0 @@
|
||||
//! `Channel` — layer L1 of the communication stack
|
||||
//! (docs/os-development/communication.md) made concrete. A program holds a
|
||||
//! channel that speaks a protocol; it does not hold a raw handle and marshal
|
||||
//! bytes at one. The channel is the answer to "who am I talking to", decided
|
||||
//! once at establishment, so nothing after that ever routes a party again:
|
||||
//! every packet's `target` addresses an *object* within the peer already chosen.
|
||||
//!
|
||||
//! **Possession of the Channel is the connection.** There is no connect step, no
|
||||
//! session id, no reconnect handshake — the endpoint capability inside is the
|
||||
//! whole of the relationship, and it cannot be forged, only handed over. Which
|
||||
//! also means a channel is a resource: `close` it, or it occupies a handle-table
|
||||
//! slot for the life of the process.
|
||||
//!
|
||||
//! **A dead provider surfaces as `-EPEER`, and the recovery is to re-open.**
|
||||
//! When the process on the other end exits, the kernel fails calls on its
|
||||
//! endpoint rather than blocking forever; `call` returns null. The client does
|
||||
//! not repair the channel — it discards it and opens the name again, which
|
||||
//! reaches whatever instance the registry now points at. The restart story
|
||||
//! falls out of the naming layer for free; no protocol needs a reconnect verb.
|
||||
//!
|
||||
//! `open` resolves a `/protocol/<name>` path through the kernel VFS router and
|
||||
//! takes the provider's endpoint from the open reply's capability. The registry
|
||||
//! answering it is init, PID 1, which mounts `/protocol` before it spawns anyone
|
||||
//! (docs/os-development/protocol-namespace.md); `bind` below is the other half —
|
||||
//! how a provider claims the name in the first place.
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc");
|
||||
const time = @import("time");
|
||||
const file_system = @import("file-system");
|
||||
const vfs_protocol = @import("vfs-protocol");
|
||||
const envelope = @import("envelope");
|
||||
|
||||
/// Longest `/protocol/...` path this client marshals. The registry's names are
|
||||
/// short by construction (a contract leaf, not a file path), and the buffer is
|
||||
/// on the stack of whoever opens.
|
||||
pub const path_maximum: usize = 224;
|
||||
|
||||
/// Where the protocol namespace is rooted — the one path prefix in the system
|
||||
/// that names contracts rather than files. Spelled once, here, so no caller
|
||||
/// builds it by hand (docs/file-system-development/file-system-hierarchy.md).
|
||||
pub const root: []const u8 = "/protocol";
|
||||
|
||||
/// Longest contract name — the part after `/protocol/`. Short by construction:
|
||||
/// a leaf like `display`, or a subtree leaf like `test/shared-memory`.
|
||||
pub const name_maximum: usize = 64;
|
||||
|
||||
/// What a `call` came back with: the provider's status, the reply payload (the
|
||||
/// bytes after the `Status`, in the caller's own buffer), and any capability the
|
||||
/// reply carried.
|
||||
pub const Response = struct {
|
||||
status: envelope.Status,
|
||||
payload: []u8,
|
||||
capability: ?ipc.Handle,
|
||||
|
||||
/// Whether the provider answered success. A negative status is its refusal
|
||||
/// (`-ENOSYS` for a verb it does not implement, and so on).
|
||||
pub fn succeeded(self: Response) bool {
|
||||
return self.status.status == 0;
|
||||
}
|
||||
};
|
||||
|
||||
/// An open conversation with one provider, speaking one protocol.
|
||||
pub const Channel = struct {
|
||||
/// The provider's endpoint. Sending into it is the only thing this handle
|
||||
/// can do — an endpoint is a mailbox owned by its creator, and that
|
||||
/// direction never reverses.
|
||||
endpoint: ipc.Handle,
|
||||
|
||||
/// Adopt an endpoint that arrived some other way — a capability delivered
|
||||
/// in a reply, or one a supervisor wired in at spawn time (P5). The channel
|
||||
/// takes ownership of the handle.
|
||||
pub fn adopt(endpoint: ipc.Handle) Channel {
|
||||
return .{ .endpoint = endpoint };
|
||||
}
|
||||
|
||||
/// Establish a channel by name: resolve `/protocol/<name>` to the registry
|
||||
/// backend, `open` the contract there, and take the provider's endpoint from
|
||||
/// the reply's capability. Null if the path does not resolve, the registry
|
||||
/// refuses (an ungranted name is refused *as* not-found), or the reply
|
||||
/// carries no capability.
|
||||
///
|
||||
/// The path is spoken exactly once, here. Everything afterwards is integers
|
||||
/// in the packet header.
|
||||
pub fn open(path: []const u8) ?Channel {
|
||||
return .{ .endpoint = openPath(path) orelse return null };
|
||||
}
|
||||
|
||||
/// Establish a channel by contract name — `open` with `/protocol/` supplied,
|
||||
/// which is how every caller in the system spells it.
|
||||
pub fn connect(name: []const u8) ?Channel {
|
||||
return .{ .endpoint = openEndpoint(name) orelse return null };
|
||||
}
|
||||
|
||||
/// Send one request packet and block for the reply: `[Header][request]` out,
|
||||
/// `[Status][reply]` back. `request` is the bytes *after* the header — the
|
||||
/// protocol's fixed part plus any tail — because the header is this call's
|
||||
/// to lay down. The reply's payload lands in `into`.
|
||||
///
|
||||
/// Null means the transport failed, which today means one of: a dead
|
||||
/// provider (`-EPEER` — discard this channel and `open` the name again), an
|
||||
/// oversized packet, or a bad handle. A provider that answered *and refused*
|
||||
/// is not a failure here: it comes back with a negative `Response.status`.
|
||||
pub fn call(self: Channel, header: envelope.Header, request: []const u8, into: []u8) ?Response {
|
||||
return self.callCapability(header, request, into, null);
|
||||
}
|
||||
|
||||
/// As `call`, handing the provider a capability with the request — the only
|
||||
/// direction-crossing move kernel-ipc offers, and how `subscribe` delivers
|
||||
/// the subscriber's own endpoint.
|
||||
pub fn callCapability(
|
||||
self: Channel,
|
||||
header: envelope.Header,
|
||||
request: []const u8,
|
||||
into: []u8,
|
||||
capability: ?ipc.Handle,
|
||||
) ?Response {
|
||||
var packet: [envelope.packet_maximum]u8 = undefined;
|
||||
const framed = frame(header, request, &packet) orelse return null;
|
||||
|
||||
var reply: [envelope.packet_maximum]u8 = undefined;
|
||||
const answer = ipc.callCap(self.endpoint, framed, &reply, capability) catch return null;
|
||||
const status = envelope.statusOf(reply[0..answer.len]) orelse return null;
|
||||
const available = @min(answer.len - envelope.prefix_size, @as(usize, status.len));
|
||||
const taken = @min(available, into.len);
|
||||
@memcpy(into[0..taken], reply[envelope.prefix_size..][0..taken]);
|
||||
return .{ .status = status, .payload = into[0..taken], .capability = answer.cap };
|
||||
}
|
||||
|
||||
/// Push one event packet and return immediately — no reply owed, and a slow
|
||||
/// or dead peer can never stall the sender. Bounded by `post_maximum`: an
|
||||
/// event that does not fit is refused here rather than split, because a
|
||||
/// packet is never fragmented.
|
||||
pub fn send(self: Channel, header: envelope.Header, payload: []const u8) bool {
|
||||
var packet: [envelope.post_maximum]u8 = undefined;
|
||||
const framed = frame(header, payload, &packet) orelse return false;
|
||||
return ipc.send(self.endpoint, framed);
|
||||
}
|
||||
|
||||
/// Ask the provider what it is: the reserved `describe` verb, answered by
|
||||
/// every protocol built through `envelope.Define`. The name and version come
|
||||
/// back in `into`, which the returned `Described` borrows.
|
||||
pub fn describe(self: Channel, into: []u8) ?envelope.Described {
|
||||
var request: [envelope.packet_maximum]u8 = undefined;
|
||||
const packet = envelope.encodeDescribe(&request) orelse return null;
|
||||
|
||||
var reply: [envelope.packet_maximum]u8 = undefined;
|
||||
const answer = ipc.callCap(self.endpoint, packet, &reply, null) catch return null;
|
||||
const taken = @min(answer.len, into.len);
|
||||
@memcpy(into[0..taken], reply[0..taken]);
|
||||
return envelope.decodeDescribe(into[0..taken]);
|
||||
}
|
||||
|
||||
/// Drop the provider's endpoint and free the handle-table slot. The
|
||||
/// conversation is over the moment the capability is gone — there is nothing
|
||||
/// else holding it open.
|
||||
pub fn close(self: Channel) void {
|
||||
_ = ipc.close(self.endpoint);
|
||||
}
|
||||
};
|
||||
|
||||
// --- the namespace: resolving, opening, and claiming a contract name ---------
|
||||
|
||||
/// Where a `/protocol/...` path routed: the registry's endpoint, plus the path
|
||||
/// rewritten mount-relative (`/display` for `/protocol/display`). The handle is
|
||||
/// deduplicated by the kernel across resolves and shared with every other user
|
||||
/// of that mount, so it is never ours to close.
|
||||
const Registry = struct {
|
||||
handle: ipc.Handle,
|
||||
relative: [path_maximum]u8,
|
||||
relative_len: usize,
|
||||
|
||||
fn path(self: *const Registry) []const u8 {
|
||||
return self.relative[0..self.relative_len];
|
||||
}
|
||||
};
|
||||
|
||||
/// Route `path` to whatever backend serves it. Null when nothing is mounted
|
||||
/// there — under `/protocol` that means the registry is not up yet, which is a
|
||||
/// *retry*, not a refusal. A kernel-served route (the read-only `/system` tree)
|
||||
/// is the wrong path, not a channel, and is refused here.
|
||||
fn reach(path: []const u8) ?Registry {
|
||||
var out: Registry = .{ .handle = 0, .relative = undefined, .relative_len = 0 };
|
||||
const route = file_system.fsResolve(path, 0, &out.relative) orelse return null;
|
||||
switch (route) {
|
||||
.kernel => return null,
|
||||
.backend => |b| {
|
||||
out.handle = b.handle;
|
||||
out.relative_len = b.path_len;
|
||||
return out;
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// One vfs-protocol round trip at a backend: the folded header, the verb's own
|
||||
/// fixed part, the name as the packet's tail, and an optional capability in each
|
||||
/// direction. Both verbs this file sends address the backend itself (target 0) —
|
||||
/// the name in the tail is what they are about.
|
||||
fn transact(
|
||||
comptime operation: vfs_protocol.Operation,
|
||||
handle: ipc.Handle,
|
||||
request: vfs_protocol.Protocol.RequestOf(operation),
|
||||
name: []const u8,
|
||||
send_capability: ?ipc.Handle,
|
||||
) ?struct { status: envelope.Status, capability: ?ipc.Handle } {
|
||||
var packet: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
const framed = vfs_protocol.Protocol.encodeRequest(operation, 0, request, name, &packet) orelse return null;
|
||||
|
||||
var reply: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
const answer = ipc.callCap(handle, framed, &reply, send_capability) catch return null;
|
||||
const status = envelope.statusOf(reply[0..answer.len]) orelse return null;
|
||||
return .{ .status = status, .capability = answer.cap };
|
||||
}
|
||||
|
||||
/// Resolve an absolute `/protocol/...` path and take the provider's endpoint out
|
||||
/// of the open reply's capability.
|
||||
fn openPath(path: []const u8) ?ipc.Handle {
|
||||
const registry = reach(path) orelse return null;
|
||||
const answered = transact(.open, registry.handle, .{ .flags = 0 }, registry.path(), null) orelse return null;
|
||||
if (answered.status.status != 0) {
|
||||
// A refusal carries no channel; anything that arrived anyway would be a
|
||||
// handle-table slot spent for nothing.
|
||||
if (answered.capability) |handle| _ = ipc.close(handle);
|
||||
return null;
|
||||
}
|
||||
// The capability *is* the channel — an open that succeeds without one was
|
||||
// answered by a file backend, which does not speak protocols.
|
||||
return answered.capability;
|
||||
}
|
||||
|
||||
/// The provider's raw endpoint behind `/protocol/<name>`. The transitional form,
|
||||
/// for the clients that still marshal their protocol's bytes by hand; P4 moves
|
||||
/// them onto `Channel` proper and this shrinks back to `connect`.
|
||||
///
|
||||
/// Null covers both "no such contract" and "you may not have it" — deliberately
|
||||
/// the same answer (protocol-namespace.md: enforcement is absence), and also
|
||||
/// "the registry is not mounted yet", which is why every caller retries.
|
||||
pub fn openEndpoint(name: []const u8) ?ipc.Handle {
|
||||
var path: [path_maximum]u8 = undefined;
|
||||
const full = join(name, &path) orelse return null;
|
||||
return openPath(full);
|
||||
}
|
||||
|
||||
/// Claim `/protocol/<name>` for `endpoint`: the registry records the name
|
||||
/// against this process and hands the endpoint to whoever opens it afterwards.
|
||||
/// The endpoint rides the call as its capability, the one direction-crossing
|
||||
/// move kernel-ipc offers.
|
||||
///
|
||||
/// Three-valued on purpose. **Null** is "the registry could not be reached" —
|
||||
/// it is not mounted yet, which happens when a provider starts before init has
|
||||
/// finished coming up, and the answer is to retry. A **value** is the registry's
|
||||
/// verdict and is final: 0 bound, `-EPERM` this binary is not granted that name,
|
||||
/// `-EBUSY` a live provider already holds it.
|
||||
pub fn bind(name: []const u8, endpoint: ipc.Handle) ?i32 {
|
||||
const registry = reach(root) orelse return null;
|
||||
const answered = transact(.bind, registry.handle, {}, name, endpoint) orelse return null;
|
||||
return answered.status.status;
|
||||
}
|
||||
|
||||
/// How long a provider keeps offering itself before giving up. The registry is
|
||||
/// init, which mounts `/protocol` before it spawns anyone, so in a normal boot
|
||||
/// the first try lands; a provider the kernel test harness starts may well beat
|
||||
/// init to the mount, which is what the patience is for. Four seconds of 20 ms
|
||||
/// tries — the same cadence every client in the tree spends finding a service.
|
||||
const bind_attempts: u32 = 200;
|
||||
const bind_retry_ms: u64 = 20;
|
||||
|
||||
/// `bind`, waiting out a registry that is not mounted yet. Only unreachability
|
||||
/// is retried: a registry that *answered* has decided, and asking again cannot
|
||||
/// change its mind. True when the name is ours.
|
||||
pub fn bindPatiently(name: []const u8, endpoint: ipc.Handle) bool {
|
||||
var attempt: u32 = 0;
|
||||
while (attempt < bind_attempts) : (attempt += 1) {
|
||||
if (bind(name, endpoint)) |status| return status == 0;
|
||||
time.sleepMillis(bind_retry_ms);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// `/protocol/` + `name`, in the caller's buffer. Null if the name is empty or
|
||||
/// longer than the namespace admits.
|
||||
fn join(name: []const u8, buffer: []u8) ?[]u8 {
|
||||
if (name.len == 0 or name.len > name_maximum) return null;
|
||||
const total = root.len + 1 + name.len;
|
||||
if (total > buffer.len) return null;
|
||||
@memcpy(buffer[0..root.len], root);
|
||||
buffer[root.len] = '/';
|
||||
@memcpy(buffer[root.len + 1 ..][0..name.len], name);
|
||||
return buffer[0..total];
|
||||
}
|
||||
|
||||
/// Lay a packet down: the folded header first, then the protocol's bytes. Null
|
||||
/// when it would not fit the buffer — the same rule as `envelope`'s framing,
|
||||
/// applied where the buffer is the transport's, not the protocol's.
|
||||
fn frame(header: envelope.Header, body: []const u8, buffer: []u8) ?[]u8 {
|
||||
const total = envelope.prefix_size + body.len;
|
||||
if (total > buffer.len) return null;
|
||||
@memcpy(buffer[0..envelope.prefix_size], std.mem.asBytes(&header));
|
||||
@memcpy(buffer[envelope.prefix_size..][0..body.len], body);
|
||||
return buffer[0..total];
|
||||
}
|
||||
|
||||
// --- tests ------------------------------------------------------------------
|
||||
//
|
||||
// The syscall half cannot run on the host, and there is no registry to reach
|
||||
// until P2 — so what is testable here is the framing, which is the part with
|
||||
// arithmetic in it.
|
||||
|
||||
const testing = std.testing;
|
||||
|
||||
test "a framed packet is the header followed by the protocol's bytes" {
|
||||
var buffer: [envelope.packet_maximum]u8 = undefined;
|
||||
const header = envelope.Header{ .operation = envelope.first_protocol_operation, .target = 9 };
|
||||
const packet = frame(header, "body", &buffer).?;
|
||||
|
||||
try testing.expectEqual(envelope.prefix_size + "body".len, packet.len);
|
||||
const decoded = envelope.headerOf(packet).?;
|
||||
try testing.expectEqual(envelope.first_protocol_operation, decoded.operation);
|
||||
try testing.expectEqual(@as(u64, 9), decoded.target);
|
||||
try testing.expectEqualStrings("body", packet[envelope.prefix_size..]);
|
||||
}
|
||||
|
||||
test "a contract name joins the namespace root exactly once" {
|
||||
var buffer: [path_maximum]u8 = undefined;
|
||||
try testing.expectEqualStrings("/protocol/display", join("display", &buffer).?);
|
||||
try testing.expectEqualStrings("/protocol/test/shared-memory", join("test/shared-memory", &buffer).?);
|
||||
try testing.expect(join("", &buffer) == null);
|
||||
try testing.expect(join("x" ** (name_maximum + 1), &buffer) == null);
|
||||
}
|
||||
|
||||
test "framing refuses a packet that would not fit rather than truncating it" {
|
||||
var post: [envelope.post_maximum]u8 = undefined;
|
||||
const header = envelope.Header{ .operation = envelope.first_protocol_operation };
|
||||
const body = [_]u8{0} ** (envelope.post_maximum - envelope.prefix_size);
|
||||
const one_too_many = body ++ [_]u8{0};
|
||||
|
||||
try testing.expect(frame(header, &body, &post) != null);
|
||||
try testing.expect(frame(header, &one_too_many, &post) == null);
|
||||
}
|
||||
@@ -14,13 +14,8 @@ const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const sc = @import("system-call");
|
||||
const ipc = @import("ipc");
|
||||
const envelope = @import("envelope");
|
||||
const vfs_protocol = @import("vfs-protocol");
|
||||
|
||||
/// The generated vfs contract: encode/decode for every verb, with the node id
|
||||
/// carried in the packet header's `target`.
|
||||
const Protocol = vfs_protocol.Protocol;
|
||||
|
||||
/// The kind of a filesystem node — re-exported so a caller need not import the
|
||||
/// wire protocol.
|
||||
pub const Kind = vfs_protocol.NodeKind;
|
||||
@@ -44,7 +39,6 @@ fn kindFromWire(value: u32) Kind {
|
||||
@intFromEnum(Kind.symbolic_link) => .symbolic_link,
|
||||
@intFromEnum(Kind.fifo) => .fifo,
|
||||
@intFromEnum(Kind.socket) => .socket,
|
||||
@intFromEnum(Kind.protocol) => .protocol,
|
||||
else => .regular,
|
||||
};
|
||||
}
|
||||
@@ -94,24 +88,23 @@ fn resolve(path: []const u8, flags: usize) ?Route {
|
||||
}
|
||||
}
|
||||
|
||||
// One request/reply round trip: frame `[Header][request][tail]`, send it, and
|
||||
// hand back the whole reply packet for the caller to decode with the generated
|
||||
// helpers. A backend that refused (a negative status) reads as null, which is
|
||||
// what every caller here did with it anyway.
|
||||
fn transact(
|
||||
comptime operation: Protocol.Operation,
|
||||
handle: ipc.Handle,
|
||||
target: u64,
|
||||
request: Protocol.RequestOf(operation),
|
||||
tail: []const u8,
|
||||
reply: []u8,
|
||||
) ?[]u8 {
|
||||
var packet: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
const framed = Protocol.encodeRequest(operation, target, request, tail, &packet) orelse return null;
|
||||
const n = ipc.call(handle, framed, reply) catch return null;
|
||||
const status = envelope.statusOf(reply[0..n]) orelse return null;
|
||||
if (status.status != 0) return null;
|
||||
return reply[0..n];
|
||||
const Result = struct { reply: vfs_protocol.Reply, payload: []u8 };
|
||||
|
||||
// One request/reply round trip: [Request header][send payload] -> backend ->
|
||||
// [Reply header][receive payload]. The receive payload lands in `out`.
|
||||
fn transact(h: ipc.Handle, request: vfs_protocol.Request, send: []const u8, out: []u8) ?Result {
|
||||
var message: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
@memcpy(message[0..vfs_protocol.request_size], std.mem.asBytes(&request));
|
||||
const slen = @min(send.len, vfs_protocol.maximum_payload);
|
||||
@memcpy(message[vfs_protocol.request_size..][0..slen], send[0..slen]);
|
||||
|
||||
var rbuf: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
const n = ipc.call(h, message[0 .. vfs_protocol.request_size + slen], &rbuf) catch return null;
|
||||
if (n < vfs_protocol.reply_size) return null;
|
||||
const reply = std.mem.bytesToValue(vfs_protocol.Reply, rbuf[0..vfs_protocol.reply_size]);
|
||||
const rpl = @min(n - vfs_protocol.reply_size, out.len);
|
||||
@memcpy(out[0..rpl], rbuf[vfs_protocol.reply_size..][0..rpl]);
|
||||
return .{ .reply = reply, .payload = out[0..rpl] };
|
||||
}
|
||||
|
||||
/// An open file: a VFS node plus a byte cursor. Read and write advance the cursor.
|
||||
@@ -131,13 +124,11 @@ pub const File = struct {
|
||||
return n;
|
||||
};
|
||||
const want: u32 = @intCast(@min(buffer.len, vfs_protocol.maximum_payload));
|
||||
var reply: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
const answered = transact(.read, h, self.node, .{ .offset = self.offset, .len = want }, &.{}, &reply) orelse return null;
|
||||
const bytes = Protocol.replyTail(.read, answered);
|
||||
const n = @min(bytes.len, buffer.len);
|
||||
@memcpy(buffer[0..n], bytes[0..n]);
|
||||
self.offset += n;
|
||||
return n;
|
||||
const request = vfs_protocol.Request{ .operation = .read, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
|
||||
const r = transact(h, request, &.{}, buffer) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
self.offset += r.reply.len;
|
||||
return r.reply.len;
|
||||
}
|
||||
|
||||
/// Write `data` at the current offset; returns the count written. A single
|
||||
@@ -147,11 +138,11 @@ pub const File = struct {
|
||||
pub fn write(self: *File, data: []const u8) ?usize {
|
||||
const h = self.backend orelse return null;
|
||||
const want: u32 = @intCast(@min(data.len, vfs_protocol.maximum_payload));
|
||||
var reply: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
const answered = transact(.write, h, self.node, .{ .offset = self.offset, .len = want }, data[0..want], &reply) orelse return null;
|
||||
const written = Protocol.decodeReply(.write, answered) orelse return null;
|
||||
self.offset += written.count;
|
||||
return written.count;
|
||||
const request = vfs_protocol.Request{ .operation = .write, .node = self.node, .offset = self.offset, .len = want, .flags = 0 };
|
||||
const r = transact(h, request, data[0..want], &.{}) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
self.offset += r.reply.len;
|
||||
return r.reply.len;
|
||||
}
|
||||
|
||||
/// Write all of `data`, looping past the per-call payload cap. Returns the
|
||||
@@ -177,9 +168,11 @@ pub const File = struct {
|
||||
const a = fsNodeStatus(self.node) orelse return null;
|
||||
return .{ .size = a.size, .kind = if (a.kind == file_kind_directory) .directory else .regular, .mtime = a.mtime };
|
||||
};
|
||||
var reply: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
const answered = transact(.status, h, self.node, {}, &.{}, &reply) orelse return null;
|
||||
const status = Protocol.decodeReply(.status, answered) orelse return null;
|
||||
const request = vfs_protocol.Request{ .operation = .status, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
var buffer: [@sizeOf(vfs_protocol.FileStatus)]u8 = undefined;
|
||||
const r = transact(h, request, &.{}, &buffer) orelse return null;
|
||||
if (r.reply.status != 0 or r.payload.len < @sizeOf(vfs_protocol.FileStatus)) return null;
|
||||
const status = std.mem.bytesToValue(vfs_protocol.FileStatus, buffer[0..@sizeOf(vfs_protocol.FileStatus)]);
|
||||
return .{ .size = status.size, .kind = kindFromWire(status.kind), .mtime = status.mtime };
|
||||
}
|
||||
|
||||
@@ -187,8 +180,8 @@ pub const File = struct {
|
||||
/// tokens are permanent — nothing to release.
|
||||
pub fn close(self: *File) void {
|
||||
const h = self.backend orelse return;
|
||||
var reply: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
_ = transact(.close, h, self.node, {}, &.{}, &reply);
|
||||
const request = vfs_protocol.Request{ .operation = .close, .node = self.node, .offset = 0, .len = 0, .flags = 0 };
|
||||
_ = transact(h, request, &.{}, &.{});
|
||||
}
|
||||
};
|
||||
|
||||
@@ -199,10 +192,10 @@ pub fn open(path: []const u8, options: OpenOptions) ?File {
|
||||
.kernel => |token| return .{ .node = token, .backend = null },
|
||||
.backend => |b| {
|
||||
const relative = route.backendPath();
|
||||
var reply: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
const answered = transact(.open, b.handle, 0, .{ .flags = options.wireFlags() }, relative, &reply) orelse return null;
|
||||
const opened = Protocol.decodeReply(.open, answered) orelse return null;
|
||||
return .{ .node = opened.node, .backend = b.handle };
|
||||
const request = vfs_protocol.Request{ .operation = .open, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = options.wireFlags() };
|
||||
const r = transact(b.handle, request, relative, &.{}) orelse return null;
|
||||
if (r.reply.status != 0) return null;
|
||||
return .{ .node = r.reply.node, .backend = b.handle };
|
||||
},
|
||||
}
|
||||
}
|
||||
@@ -254,13 +247,15 @@ pub const Directory = struct {
|
||||
self.cursor += 1;
|
||||
return true;
|
||||
};
|
||||
var reply: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
const answered = transact(.readdir, h, self.node, .{ .cursor = self.cursor }, &.{}, &reply) orelse return false;
|
||||
const header = Protocol.decodeReply(.readdir, answered) orelse return false;
|
||||
if (header.name_len == 0) return false; // end of directory
|
||||
const request = vfs_protocol.Request{ .operation = .readdir, .node = self.node, .offset = self.cursor, .len = 0, .flags = 0 };
|
||||
var buffer: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
const r = transact(h, request, &.{}, &buffer) orelse return false;
|
||||
if (r.reply.status != 0 or r.reply.len == 0) return false; // error or EOF
|
||||
if (r.payload.len < vfs_protocol.directory_entry_size) return false;
|
||||
const header = std.mem.bytesToValue(vfs_protocol.DirectoryEntry, r.payload[0..vfs_protocol.directory_entry_size]);
|
||||
entry.kind = kindFromWire(header.kind);
|
||||
entry.size = header.size;
|
||||
const source = Protocol.replyTail(.readdir, answered);
|
||||
const source = r.payload[vfs_protocol.directory_entry_size..];
|
||||
const nlen = @min(@min(@as(usize, header.name_len), source.len), entry.name_buffer.len);
|
||||
@memcpy(entry.name_buffer[0..nlen], source[0..nlen]);
|
||||
entry.name_len = nlen;
|
||||
@@ -284,11 +279,13 @@ pub fn openDirectory(path: []const u8) ?Directory {
|
||||
// A path-based request that returns only a status (mkdir, unlink). Kernel-served
|
||||
// paths (the read-only /system) refuse mutation by construction: the resolve
|
||||
// must land on a backend.
|
||||
fn pathOperation(comptime operation: Protocol.Operation, path: []const u8) bool {
|
||||
fn pathOperation(operation: vfs_protocol.Operation, path: []const u8) bool {
|
||||
const route = resolve(path, 0) orelse return false;
|
||||
if (route != .backend) return false;
|
||||
var reply: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
return transact(operation, route.backend.handle, 0, {}, route.backendPath(), &reply) != null;
|
||||
const relative = route.backendPath();
|
||||
const request = vfs_protocol.Request{ .operation = operation, .node = 0, .offset = 0, .len = @intCast(relative.len), .flags = 0 };
|
||||
const r = transact(route.backend.handle, request, relative, &.{}) orelse return false;
|
||||
return r.reply.status == 0;
|
||||
}
|
||||
|
||||
/// Create a directory at `path` (its parent must already exist). Returns true on
|
||||
@@ -308,7 +305,7 @@ pub fn makePath(path: []const u8) bool {
|
||||
while (end < path.len and path[end] != '/') end += 1;
|
||||
const prefix = path[0..end];
|
||||
if (prefix.len == 0 or (prefix.len == 1 and prefix[0] == '/')) continue;
|
||||
// Best-effort per prefix: components at or above a mount point ("/volumes")
|
||||
// Best-effort per prefix: components at or above a mount point ("/mnt")
|
||||
// are router names, not filesystem nodes — they neither exist as nodes
|
||||
// nor accept mkdir, and that is fine. Only the final verdict counts.
|
||||
if (!exists(prefix)) _ = makeDirectory(prefix);
|
||||
@@ -338,8 +335,9 @@ pub fn rename(old_path: []const u8, new_path: []const u8) bool {
|
||||
@memcpy(payload[0..old_relative.len], old_relative);
|
||||
payload[old_relative.len] = 0;
|
||||
@memcpy(payload[old_relative.len + 1 ..][0..new_relative.len], new_relative);
|
||||
var reply: [vfs_protocol.message_maximum]u8 = undefined;
|
||||
return transact(.rename, old_route.backend.handle, 0, {}, payload[0..total], &reply) != null;
|
||||
const request = vfs_protocol.Request{ .operation = .rename, .node = 0, .offset = 0, .len = @intCast(total), .flags = 0 };
|
||||
const r = transact(old_route.backend.handle, request, payload[0..total], &.{}) orelse return false;
|
||||
return r.reply.status == 0;
|
||||
}
|
||||
|
||||
/// Mount a filesystem backend (its server endpoint) at absolute path `target`;
|
||||
@@ -350,9 +348,8 @@ pub fn mount(target: []const u8, backend: ipc.Handle) bool {
|
||||
}
|
||||
|
||||
/// As `mount`, with a backend-side rewrite prefix: a path under `target` reaches
|
||||
/// the backend as `rewrite` + the mount-relative tail. How one volume serves
|
||||
/// several mounts ("/volumes/usb" from its root, "/system/logs" from its
|
||||
/// /system/logs subtree).
|
||||
/// the backend as `rewrite` + the mount-relative tail. How one volume serves two
|
||||
/// mounts ("/mnt/usb" from its root, "/var" from its /var subtree).
|
||||
pub fn mountRewritten(target: []const u8, backend: ipc.Handle, rewrite: []const u8) bool {
|
||||
return fsMount(target, backend, rewrite);
|
||||
}
|
||||
|
||||
+11
-54
@@ -29,10 +29,10 @@ pub fn createIpcEndpoint() ?Handle {
|
||||
return if (failed(r)) null else r;
|
||||
}
|
||||
|
||||
// `register`/`lookup` lived here — the two wrappers over the flat ServiceId
|
||||
// registry. Naming is not a system call any more: a provider binds its contract
|
||||
// name at the registry and a client resolves and opens `/protocol/<name>`, both
|
||||
// through `channel` (docs/os-development/protocol-namespace.md).
|
||||
/// Publish endpoint `h` under a well-known service id so other processes find it.
|
||||
pub fn register(id: abi.ServiceId, h: Handle) bool {
|
||||
return !failed(sc.systemCall2(.ipc_register, @intFromEnum(id), h));
|
||||
}
|
||||
|
||||
/// Drop a capability handle (endpoint, shared-memory, or DMA-region) and free its table
|
||||
/// slot. A forwarding hop closes a cap it passed on; a binder closes a DMA-region cap
|
||||
@@ -42,6 +42,13 @@ pub fn close(h: Handle) bool {
|
||||
return !failed(sc.systemCall1(.handle_close, h));
|
||||
}
|
||||
|
||||
/// Find the endpoint published under `id`, installing a handle to it in this
|
||||
/// process.
|
||||
pub fn lookup(id: abi.ServiceId) ?Handle {
|
||||
const r = sc.systemCall1(.ipc_lookup, @intFromEnum(id));
|
||||
return if (failed(r)) null else r;
|
||||
}
|
||||
|
||||
pub const CallError = error{Failed};
|
||||
|
||||
/// The result of a capability-passing `callCap`: the reply length, and the handle of
|
||||
@@ -171,56 +178,6 @@ pub const Received = struct {
|
||||
}
|
||||
};
|
||||
|
||||
/// A capability that arrived with one turn of a receive loop, and the ownership
|
||||
/// rule for it: **the turn owns it until a handler takes it, and closes whatever
|
||||
/// is left.**
|
||||
///
|
||||
/// The kernel installs a sent capability in the receiver's handle table whenever
|
||||
/// the caller attached one, *independent of the message's length or kind*
|
||||
/// (system/kernel/ipc-synchronous.zig `replyWait`), so every path out of a loop
|
||||
/// has to dispose of one — including the paths that never look at the message.
|
||||
/// The table is thirty-two slots, and `ipc_call` does not dedupe, so a client
|
||||
/// looping on `callCap(server, &.{}, endpoint)` spends one slot per call: about
|
||||
/// thirty-two zero-length pings and the service can never accept another
|
||||
/// capability, which means no subscribe and no shared-memory handover, for the
|
||||
/// rest of the boot. It is unauthenticated and it is two lines to write.
|
||||
///
|
||||
/// So ownership is structural rather than a close per branch — the per-branch
|
||||
/// version has already failed twice in this tree, in PID 1's ping path and in
|
||||
/// every `service.run` callback that simply ignored its capability argument.
|
||||
/// Written this way, forgetting **closes**, and *keeping* a capability is the
|
||||
/// thing a handler has to say out loud:
|
||||
///
|
||||
/// ```zig
|
||||
/// var arrived: ipc.Arrival = .{ .handle = got.cap };
|
||||
/// defer arrived.release(); // every exit path, including `continue`
|
||||
/// ...
|
||||
/// const kept = arrived.take().?; // claimed: mine to hold or close
|
||||
/// ```
|
||||
pub const Arrival = struct {
|
||||
handle: ?Handle = null,
|
||||
|
||||
/// Look without claiming — a handler that may still refuse wants no close of
|
||||
/// its own on the refusal paths.
|
||||
pub fn peek(self: *const Arrival) ?Handle {
|
||||
return self.handle;
|
||||
}
|
||||
|
||||
/// Claim ownership: from here the capability is the taker's to keep or close,
|
||||
/// and the turn will not touch it.
|
||||
pub fn take(self: *Arrival) ?Handle {
|
||||
defer self.handle = null;
|
||||
return self.handle;
|
||||
}
|
||||
|
||||
/// Close whatever nobody claimed. Idempotent, so it is safe as a `defer` next
|
||||
/// to any number of `take`s.
|
||||
pub fn release(self: *Arrival) void {
|
||||
if (self.handle) |handle| _ = close(handle);
|
||||
self.handle = null;
|
||||
}
|
||||
};
|
||||
|
||||
/// Server side of IPC_ReplyWait: deliver `reply` to the client last received (if any,
|
||||
/// optionally handing it `send_cap`), then block until the next request arrives in
|
||||
/// `receive`. Returns its length, the sender badge, and any capability the request
|
||||
|
||||
@@ -142,18 +142,6 @@ pub fn subscribeExits(endpoint: usize) bool {
|
||||
/// snapshot buffer without importing `abi` itself.
|
||||
pub const ProcessDescriptor = abi.ProcessDescriptor;
|
||||
|
||||
/// The calling task's own kernel id — its row in the process table, and the value
|
||||
/// every other process sees as this one's `supervisor` after it spawns them. For a
|
||||
/// single-threaded program that is its process id; in a threaded one it is the
|
||||
/// calling thread's id (`Thread.getCurrentId` is the same system call, named for
|
||||
/// the threading vocabulary). Ids are monotonic and never reused
|
||||
/// (system/kernel/process.zig), which is what makes comparing one an identity
|
||||
/// test where comparing a *name* is only a resemblance test — the registrar in
|
||||
/// init leans on exactly that.
|
||||
pub fn taskId() u32 {
|
||||
return @intCast(sc.systemCall0(.thread_self));
|
||||
}
|
||||
|
||||
/// Give up the rest of this quantum.
|
||||
pub fn yield() void {
|
||||
_ = sc.systemCall0(.yield);
|
||||
|
||||
+16
-288
@@ -6,64 +6,26 @@
|
||||
//! loop chose, never on a hijacked stack — the whole reason signals are
|
||||
//! messages.
|
||||
//!
|
||||
//! One rule a service author does have to know, and it is stated on
|
||||
//! `Callbacks.on_message`: **a capability that arrives belongs to the turn** —
|
||||
//! the loop closes it unless the callback claims it with `take()`. Forgetting is
|
||||
//! therefore safe, and keeping is explicit; the opposite arrangement quietly
|
||||
//! spends a handle-table slot per request.
|
||||
//!
|
||||
//! The harness also owns the **subscriber side** of a protocol that declares
|
||||
//! `.events` — see `Subscribers`. The table, the reserved subscribe/unsubscribe
|
||||
//! verbs, the fan-out, and the dead-subscriber sweep live here rather than in
|
||||
//! each provider, so every event stream in the system has identical semantics
|
||||
//! (docs/os-development/protocol-namespace.md, "Wiring").
|
||||
//!
|
||||
//! The liveness probe: a **zero-length request is the universal ping**, answered
|
||||
//! with a zero-length reply by the harness itself. No protocol's requests start
|
||||
//! at length zero, so the encoding cannot collide, and there is nothing for a
|
||||
//! service author to implement — a wedged service simply fails to answer, which
|
||||
//! is the diagnosis (see docs/ipc.md).
|
||||
|
||||
const channel = @import("channel");
|
||||
const envelope = @import("envelope");
|
||||
const abi = @import("abi");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
|
||||
/// The harness's handle on a provider's subscriber table, type-erased because
|
||||
/// `run` is not generic over the protocol while `Subscribers` is. A service names
|
||||
/// its table once, as `Callbacks.subscribers`, and the loop does the rest: it
|
||||
/// subscribes to published process exits at startup and drops a dead task's
|
||||
/// subscriptions before the service's own notification callback ever sees the
|
||||
/// badge.
|
||||
pub const SubscriberHooks = struct {
|
||||
/// Ask the kernel for published exit events on this service's endpoint.
|
||||
watch: *const fn (endpoint: ipc.Handle) void,
|
||||
/// Drop everything task `dead` had subscribed.
|
||||
forget: *const fn (dead: u32) void,
|
||||
};
|
||||
|
||||
pub const Callbacks = struct {
|
||||
/// Called once with the service's endpoint before the loop starts — the
|
||||
/// place to subscribe to exit events, bind IRQs, or announce readiness.
|
||||
/// Return false to abort startup (the process exits).
|
||||
init: ?*const fn (endpoint: ipc.Handle) bool = null,
|
||||
/// One protocol request from `sender` (a task id): write the reply into
|
||||
/// `reply`, return its length. The zero-length ping never reaches this.
|
||||
///
|
||||
/// `arrived` is the capability the request carried (M13 cap passing — how a
|
||||
/// subscriber hands over its endpoint), and it comes with **an ownership
|
||||
/// rule: the turn owns it, and a handler that wants to keep it must say so
|
||||
/// with `take()`.** Whatever is left when this returns, the loop closes.
|
||||
/// `peek()` reads it without claiming, which is what a handler that may
|
||||
/// still refuse wants — no close of its own on the refusal paths.
|
||||
///
|
||||
/// The rule is stated here, in the contract, because the alternative has
|
||||
/// failed in practice: an implementation that simply ignored a `?ipc.Handle`
|
||||
/// argument leaked a handle table slot per request, and every operation
|
||||
/// except a subscribe ignores it. Thirty-two such requests — zero-length
|
||||
/// pings will do, and they need no authorization — and the service can never
|
||||
/// accept another capability for the rest of the boot. See `ipc.Arrival`.
|
||||
on_message: *const fn (message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize,
|
||||
/// `reply`, return its length. `capability` is the handle the request
|
||||
/// carried, if any (M13 cap passing — how a subscriber hands over its
|
||||
/// endpoint). The zero-length ping never reaches this.
|
||||
on_message: *const fn (message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize,
|
||||
/// A notification that is not a signal — a subscribed exit event, a bound
|
||||
/// IRQ, a timer landing. The raw badge; decode with the ipc helpers.
|
||||
on_notification: ?*const fn (badge: u64) void = null,
|
||||
@@ -73,238 +35,21 @@ pub const Callbacks = struct {
|
||||
/// the return itself — never put *necessary* work here (iron rule 1: a kill
|
||||
/// arrives with no warning; this is for graceful extras only).
|
||||
on_terminate: ?*const fn () void = null,
|
||||
/// The contract this service provides: a name under `/protocol`, mirroring
|
||||
/// the `library/protocol/` module that defines the wire format — a program
|
||||
/// imports `display-protocol` and the provider binds `"display"`
|
||||
/// (docs/os-development/protocol-namespace.md). Bound at startup, before
|
||||
/// `init` runs, so the service is reachable the moment it serves. A refusal
|
||||
/// (not granted, or a live provider already holds the name) aborts startup.
|
||||
service: ?[]const u8 = null,
|
||||
/// This provider's subscriber table — `Subscribers(Protocol, Context).hooks`
|
||||
/// — for a protocol that declares `.events`. Naming it here is what buys the
|
||||
/// exit-notification sweep: the loop subscribes to published deaths at
|
||||
/// startup and releases a dead subscriber's slot (and the endpoint capability
|
||||
/// in it) when one lands.
|
||||
subscribers: ?SubscriberHooks = null,
|
||||
/// Publish the endpoint under a well-known service id at startup.
|
||||
service: ?abi.ServiceId = null,
|
||||
};
|
||||
|
||||
/// How many subscribers one provider fans out to. Bounded like every table in
|
||||
/// this system; a subscribe past the end is refused with `-ENOSPC` rather than
|
||||
/// silently forgetting an earlier one.
|
||||
pub const subscriber_capacity = 8;
|
||||
|
||||
/// The interest mask that means "every event of this protocol" — what a
|
||||
/// subscriber which named no class gets, and what a provider passes when the
|
||||
/// event it is publishing belongs to no class.
|
||||
pub const every_event: u32 = 0;
|
||||
|
||||
/// The subscriber side of a protocol, for a provider whose contract declares
|
||||
/// `.events` (docs/os-development/protocol-namespace.md: *the harness owns the
|
||||
/// machinery — the subscriber table, the dead-subscriber sweep, and the fan-out
|
||||
/// loop*). Three services hand-rolled this, with three different ideas of when a
|
||||
/// dead subscriber goes away — a poll of the process list on subscribe, a drop on
|
||||
/// a failed send, and nothing at all. This is the one idiom.
|
||||
///
|
||||
/// ```zig
|
||||
/// const Subscriptions = service.Subscribers(power_protocol.Protocol, void);
|
||||
/// ...
|
||||
/// fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||
/// return Subscriptions.dispatch({}, handlers, message, sender, arrived, reply);
|
||||
/// }
|
||||
/// pub fn main() void {
|
||||
/// service.run(power_protocol.message_maximum, .{
|
||||
/// .service = "power",
|
||||
/// .on_message = onMessage,
|
||||
/// .subscribers = Subscriptions.hooks,
|
||||
/// });
|
||||
/// }
|
||||
/// ```
|
||||
///
|
||||
/// What the provider still writes is its own events — `publish(.power_button, 0,
|
||||
/// .{})`. Everything else happens here: registering the caller's endpoint on the
|
||||
/// reserved `subscribe` verb, taking that capability out of the turn, dropping it
|
||||
/// on `unsubscribe` or on the subscriber's death, and framing one packet for the
|
||||
/// whole fan-out.
|
||||
///
|
||||
/// The table is per instantiation (a container-level `var` inside the generic
|
||||
/// type), so a process providing two contracts gets two tables and neither can
|
||||
/// see the other's subscribers.
|
||||
pub fn Subscribers(comptime Protocol: type, comptime Context: type) type {
|
||||
return struct {
|
||||
/// The generated dispatch this provider answers with.
|
||||
pub const Provider = Protocol.Provider(Context);
|
||||
pub const Handlers = Provider.Handlers;
|
||||
|
||||
/// One registered subscriber: the endpoint events are pushed to (the
|
||||
/// capability it handed over at subscribe time, which this slot owns),
|
||||
/// the task that handed it over — the kernel-stamped badge, the only
|
||||
/// source identity there is — and which classes of event it asked for.
|
||||
const Slot = struct {
|
||||
used: bool = false,
|
||||
endpoint: ipc.Handle = 0,
|
||||
task: u32 = 0,
|
||||
interest: u32 = every_event,
|
||||
};
|
||||
|
||||
var slots: [subscriber_capacity]Slot = .{Slot{}} ** subscriber_capacity;
|
||||
|
||||
/// Set when a slot has taken the capability the turn carried, and read
|
||||
/// back in `dispatch`, which is where the turn's `Arrival` lives. The
|
||||
/// generated dispatch hands a handler the raw handle rather than the
|
||||
/// `Arrival` — deliberately, since a handler has no business closing the
|
||||
/// turn's property — so the *claim* has to travel back out this way. One
|
||||
/// turn, one handler, one thread: there is nothing here to race.
|
||||
var claimed = false;
|
||||
|
||||
/// What `Callbacks.subscribers` is given.
|
||||
pub const hooks: SubscriberHooks = .{ .watch = watchExits, .forget = forget };
|
||||
|
||||
fn watchExits(endpoint: ipc.Handle) void {
|
||||
// Published exits, not a poll of the process list: a service must
|
||||
// never depend on clients cleaning up after themselves, and it must
|
||||
// not have to walk the whole table on every subscribe to find out
|
||||
// either (docs/process-lifecycle.md, "Who learns of a death").
|
||||
_ = process.subscribeExits(endpoint);
|
||||
}
|
||||
|
||||
/// Release everything task `dead` had subscribed. The slot owns the
|
||||
/// endpoint capability, so reclaiming the slot closes it — otherwise a
|
||||
/// process that subscribes and dies costs a handle-table slot that never
|
||||
/// comes back.
|
||||
pub fn forget(dead: u32) void {
|
||||
for (&slots) |*slot| {
|
||||
if (slot.used and slot.task == dead) {
|
||||
_ = ipc.close(slot.endpoint);
|
||||
slot.* = .{};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether `task` is a subscriber — the gate for an operation a provider
|
||||
/// honours from its subscribers and nobody else. The power service's
|
||||
/// shutdown is the one: the badge is kernel-stamped, so nothing in a
|
||||
/// packet can claim to be the subscriber that already ran the stop
|
||||
/// sequence.
|
||||
pub fn has(task: u32) bool {
|
||||
for (&slots) |*slot| {
|
||||
if (slot.used and slot.task == task) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Answer one received packet, with the reserved `subscribe` and
|
||||
/// `unsubscribe` verbs already wired — a provider that leaves those two
|
||||
/// handlers null (every provider should) gets the harness's. The turn's
|
||||
/// capability is peeked, never taken, unless a slot actually kept it.
|
||||
pub fn dispatch(
|
||||
context: Context,
|
||||
handlers: Handlers,
|
||||
packet: []const u8,
|
||||
sender: u32,
|
||||
arrived: *ipc.Arrival,
|
||||
reply: []u8,
|
||||
) usize {
|
||||
var wired = handlers;
|
||||
if (wired.subscribe == null) wired.subscribe = onSubscribe;
|
||||
if (wired.unsubscribe == null) wired.unsubscribe = onUnsubscribe;
|
||||
claimed = false;
|
||||
const written = Provider.dispatch(context, wired, packet, sender, arrived.peek(), reply);
|
||||
if (claimed) _ = arrived.take();
|
||||
return written;
|
||||
}
|
||||
|
||||
/// Push one event to every subscriber.
|
||||
pub fn publish(
|
||||
comptime event: Protocol.Event,
|
||||
target: u64,
|
||||
payload: Protocol.PayloadOf(event),
|
||||
) void {
|
||||
publishClass(event, target, payload, every_event);
|
||||
}
|
||||
|
||||
/// Push one event to the subscribers whose interest mask includes
|
||||
/// `class` (a subscriber that named no class takes everything). The
|
||||
/// packet is framed **once**, outside the loop, so every subscriber of a
|
||||
/// class receives identical bytes; and delivery is `ipc.send`, which
|
||||
/// never blocks, so one slow or dead subscriber can never stall the rest
|
||||
/// — the whole reason broadcast is a provider pattern and not a kernel
|
||||
/// primitive.
|
||||
pub fn publishClass(
|
||||
comptime event: Protocol.Event,
|
||||
target: u64,
|
||||
payload: Protocol.PayloadOf(event),
|
||||
class: u32,
|
||||
) void {
|
||||
var packet: [envelope.post_maximum]u8 = undefined;
|
||||
const framed = Protocol.encodeEvent(event, target, payload, &packet) orelse return;
|
||||
for (&slots) |*slot| {
|
||||
if (!slot.used) continue;
|
||||
if (!wants(slot.*, class)) continue;
|
||||
// The sweep is what normally reclaims a dead subscriber, promptly
|
||||
// and with its capability closed. This is the backstop for a
|
||||
// notification that never arrived: an endpoint's notify ring is
|
||||
// bounded, so a burst of deaths can drop one, and a send to an
|
||||
// endpoint whose owner is gone fails rather than blocking.
|
||||
if (!ipc.send(slot.endpoint, framed)) {
|
||||
_ = ipc.close(slot.endpoint);
|
||||
slot.* = .{};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn wants(slot: Slot, class: u32) bool {
|
||||
if (class == every_event) return true; // the event belongs to no class
|
||||
if (slot.interest == every_event) return true; // the subscriber named none
|
||||
return slot.interest & class != 0;
|
||||
}
|
||||
|
||||
/// The reserved `subscribe` verb: register the caller's endpoint (the
|
||||
/// call's capability) for the classes its tail names. A refusal simply
|
||||
/// returns and the turn closes what arrived — the harness's ownership
|
||||
/// rule (`ipc.Arrival`), which is why a subscribe storm against a full
|
||||
/// table cannot spend the handle table.
|
||||
fn onSubscribe(_: Context, invocation: envelope.Invocation(void), _: envelope.Answer(void)) isize {
|
||||
const endpoint = invocation.capability orelse return -envelope.EPROTO; // no endpoint passed
|
||||
const interest = envelope.decodeSubscribe(invocation.tail).interest;
|
||||
for (&slots) |*slot| {
|
||||
if (slot.used) continue;
|
||||
// Appended, not replaced: one task may hold several subscriptions
|
||||
// on different endpoints (a client taking keyboard and mouse as
|
||||
// two streams), and each is its own conversation.
|
||||
slot.* = .{ .used = true, .endpoint = endpoint, .task = invocation.sender, .interest = interest };
|
||||
claimed = true; // the table holds it until that task dies
|
||||
return 0;
|
||||
}
|
||||
return -envelope.ENOSPC; // table full
|
||||
}
|
||||
|
||||
/// The reserved `unsubscribe` verb: every subscription the calling task
|
||||
/// holds here goes, which is exactly what its death would do. It names no
|
||||
/// endpoint because the badge already names the only subscriber a caller
|
||||
/// can speak for — its own.
|
||||
fn onUnsubscribe(_: Context, invocation: envelope.Invocation(void), _: envelope.Answer(void)) isize {
|
||||
if (!has(invocation.sender)) return -envelope.ENOENT;
|
||||
forget(invocation.sender);
|
||||
return 0;
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
/// Run the service: create the endpoint, bind it under the service's contract
|
||||
/// name (if it has one), bind signals to it, call `init`, then serve until
|
||||
/// `terminate` arrives — at which point the loop returns and main's return is
|
||||
/// the clean exit the supervisor reads as `ExitReason.exited`.
|
||||
/// `maximum_message` sizes the receive and reply buffers (a service passes its
|
||||
/// protocol's message maximum).
|
||||
/// Run the service: create and (optionally) register the endpoint, bind signals
|
||||
/// to it, call `init`, then serve until `terminate` arrives — at which point the
|
||||
/// loop returns and main's return is the clean exit the supervisor reads as
|
||||
/// `ExitReason.exited`. `maximum_message` sizes the receive and reply buffers
|
||||
/// (a service passes its protocol's message maximum).
|
||||
pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
|
||||
const endpoint = ipc.createIpcEndpoint() orelse return;
|
||||
if (callbacks.service) |name| {
|
||||
if (!channel.bindPatiently(name, endpoint)) return;
|
||||
if (callbacks.service) |id| {
|
||||
if (!ipc.register(id, endpoint)) return;
|
||||
}
|
||||
_ = process.bindSignals(endpoint);
|
||||
// Before `init`, so a subscriber that arrives the instant the name is bound
|
||||
// is already covered by the sweep that will release it.
|
||||
if (callbacks.subscribers) |subscribers| subscribers.watch(endpoint);
|
||||
if (callbacks.init) |initialise| {
|
||||
if (!initialise(endpoint)) return;
|
||||
}
|
||||
@@ -314,16 +59,6 @@ pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
|
||||
var receive: [maximum_message]u8 = undefined;
|
||||
while (true) {
|
||||
const got = ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive, null);
|
||||
// Whatever capability came with this turn is the turn's, and the turn
|
||||
// closes it unless a callback claims it (`ipc.Arrival`). Structural
|
||||
// rather than a close per branch, because the branches are exactly what
|
||||
// gets forgotten: the ping's `continue` below, and every `on_message`
|
||||
// that has no use for a capability — which is every operation but a
|
||||
// subscribe. A `defer` in a loop body runs on `continue` and on the
|
||||
// `return` that ends the loop, so this covers all four exits.
|
||||
var arrived: ipc.Arrival = .{ .handle = got.cap };
|
||||
defer arrived.release();
|
||||
|
||||
if (got.isNotification()) {
|
||||
reply_len = 0; // nothing owed for a notification
|
||||
if (process.signalsFrom(got.badge)) |signals| {
|
||||
@@ -336,20 +71,13 @@ pub fn run(comptime maximum_message: usize, callbacks: Callbacks) void {
|
||||
}
|
||||
continue;
|
||||
}
|
||||
// A death sweeps the subscriber table first, then still reaches the
|
||||
// service: a provider often has its own per-client state to release
|
||||
// (open file handles, device tokens, layers) and the same badge is
|
||||
// the notice for both.
|
||||
if (got.isChildExit()) {
|
||||
if (callbacks.subscribers) |subscribers| subscribers.forget(got.childProcessId());
|
||||
}
|
||||
if (callbacks.on_notification) |onNotification| onNotification(got.badge);
|
||||
continue;
|
||||
}
|
||||
if (got.len == 0) {
|
||||
reply_len = 0; // the universal ping: a zero-length reply, from the harness
|
||||
continue; // any capability it carried goes out through the turn's `defer`
|
||||
continue;
|
||||
}
|
||||
reply_len = callbacks.on_message(receive[0..got.len], &reply_buffer, got.senderTaskId(), &arrived);
|
||||
reply_len = callbacks.on_message(receive[0..got.len], &reply_buffer, got.senderTaskId(), got.cap);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,61 +1,51 @@
|
||||
//! The block-device wire protocol — what a filesystem (the FAT server) says to a
|
||||
//! block driver (usb-storage) over `/protocol/block`. Defined through the
|
||||
//! envelope, so every packet begins with the folded `Header`.
|
||||
//!
|
||||
//! **`Header.target` is always 0 here**: a block driver instance serves exactly
|
||||
//! one device over its own endpoint, so there is no object within the peer to
|
||||
//! address. A driver that later fronts several volumes gives them target ids and
|
||||
//! `enumerate` lists them; nothing else about the protocol changes.
|
||||
//! block driver (usb-storage) over its well-known `.block` endpoint. A protocol
|
||||
//! module like vfs-protocol / usb-transfer-protocol: extern-struct messages, an
|
||||
//! `Operation` tag, everything in one IPC message.
|
||||
//!
|
||||
//! Data path: read and write move whole blocks to or from a **caller-owned DMA
|
||||
//! buffer**, named by its physical address — the same physical-address handoff
|
||||
//! usb-storage already uses toward the controller, one layer up. So a 512-byte
|
||||
//! sector never has to cross the packet floor; only the small request / reply
|
||||
//! parts do. Under an enforcing IOMMU the buffer's physical addresses are only
|
||||
//! reachable by the device once the filesystem has `attach`ed the buffer's
|
||||
//! capability (the block server forwards it to the controller); see
|
||||
//! docs/driver-model.md.
|
||||
//! sector never has to cross the 256-byte IPC boundary; only the small request /
|
||||
//! reply headers do. Under an enforcing IOMMU the buffer's physical addresses are
|
||||
//! only reachable by the device once the filesystem has `attach`ed the buffer's
|
||||
//! capability (the block server forwards it to the controller); see docs/driver-model.md.
|
||||
|
||||
const envelope = @import("envelope");
|
||||
|
||||
/// The answer to `geometry()`.
|
||||
pub const Geometry = extern struct {
|
||||
block_size: u32, // bytes per block (512)
|
||||
_padding: u32 = 0,
|
||||
block_count: u64, // total blocks
|
||||
pub const Operation = enum(u32) {
|
||||
/// geometry() -> { block_size, block_count }
|
||||
geometry = 0,
|
||||
/// read(lba, count, physical): read `count` blocks from `lba` into the buffer
|
||||
read = 1,
|
||||
/// write(lba, count, physical): write `count` blocks at `lba` from the buffer
|
||||
write = 2,
|
||||
/// flush(): commit any device write cache to stable media (no data transfer).
|
||||
/// A filesystem calls this to make prior writes durable — e.g. before power-off,
|
||||
/// so a shutdown-time write isn't lost in the USB flash controller's cache.
|
||||
flush = 3,
|
||||
/// attach(): the caller's DMA-region capability rides the call's cap slot; the
|
||||
/// block server forwards it to the controller so the buffer's physical addresses
|
||||
/// (named in later read/write) are reachable by the device under an enforcing
|
||||
/// IOMMU. Call once per buffer before using it in a transfer.
|
||||
attach = 4,
|
||||
};
|
||||
|
||||
/// `read(lba, count, physical)` / `write(...)`: move `count` blocks between the
|
||||
/// device and the caller's DMA buffer at `physical`.
|
||||
pub const Transfer = extern struct {
|
||||
pub const Request = extern struct {
|
||||
operation: u32,
|
||||
reserved: u32 = 0,
|
||||
lba: u64,
|
||||
count: u32,
|
||||
_padding: u32 = 0,
|
||||
physical: u64, // caller's DMA buffer physical address
|
||||
count: u32, // number of blocks (read/write)
|
||||
reserved2: u32 = 0,
|
||||
physical: u64, // caller's DMA buffer physical address (read/write)
|
||||
};
|
||||
|
||||
/// How many blocks a transfer actually moved.
|
||||
pub const Transferred = extern struct { count: u32 };
|
||||
pub const Reply = extern struct {
|
||||
status: i32, // 0 on success, negative on failure
|
||||
reserved: u32 = 0,
|
||||
block_size: u32, // geometry: bytes per block (512)
|
||||
reserved2: u32 = 0,
|
||||
block_count: u64, // geometry: total blocks; read/write: blocks moved
|
||||
};
|
||||
|
||||
pub const Protocol = envelope.Define(.{
|
||||
.name = "block",
|
||||
.version = 1,
|
||||
.operations = &.{
|
||||
.{ .name = "geometry", .reply = Geometry },
|
||||
.{ .name = "read", .request = Transfer, .reply = Transferred },
|
||||
.{ .name = "write", .request = Transfer, .reply = Transferred },
|
||||
// flush(): commit any device write cache to stable media (no data
|
||||
// transfer). A filesystem calls this to make prior writes durable —
|
||||
// before power-off, so a shutdown-time write isn't lost in the USB flash
|
||||
// controller's cache.
|
||||
.{ .name = "flush" },
|
||||
// attach(): the caller's DMA-region capability rides the call's cap
|
||||
// slot; the block server forwards it to the controller so the buffer's
|
||||
// physical addresses (named in later read/write) are reachable by the
|
||||
// device under an enforcing IOMMU. Call once per buffer before using it.
|
||||
.{ .name = "attach" },
|
||||
},
|
||||
});
|
||||
|
||||
pub const Operation = Protocol.Operation;
|
||||
pub const message_maximum: usize = Protocol.message_maximum;
|
||||
pub const message_maximum: usize = 256;
|
||||
pub const request_size: usize = @sizeOf(Request);
|
||||
pub const reply_size: usize = @sizeOf(Reply);
|
||||
|
||||
@@ -1,74 +0,0 @@
|
||||
//! The "protocol" library domain: the wire protocols — each service's public
|
||||
//! interface, exposed as its own module (docs/driver-model.md). Both sides of
|
||||
//! every conversation depend on the contract by name; neither reaches into the
|
||||
//! other's files. Pure flat wire types: no protocol module imports anything.
|
||||
//!
|
||||
//! One module here is not a protocol but the shape the others are written in,
|
||||
//! and therefore the one module every other one imports:
|
||||
//!
|
||||
//! envelope : the packet prefix + comptime Define (docs/os-development/protocol-namespace.md)
|
||||
//!
|
||||
//! vfs-protocol : the VFS server <-> the file layer (unistd/stdio)
|
||||
//! input-protocol : the input fan-out service <-> sources + subscribers
|
||||
//! block-protocol : a filesystem <-> a block driver (usb-storage)
|
||||
//! usb-transfer-protocol : a USB class driver <-> the xHCI bus driver
|
||||
//! device-manager-protocol : the device manager <-> drivers + discovery
|
||||
//! display-protocol : the compositor's client-facing surface
|
||||
//! scanout-protocol : the compositor -> a native scanout driver (docs/display-v2.md)
|
||||
//! power-protocol : system power's domain-named surface (docs/power.md)
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
// Not a protocol, hence not `-protocol`: the envelope is what a protocol is
|
||||
// defined *through*, so it is built first and handed to every protocol
|
||||
// below as their one import.
|
||||
const envelope = b.addModule("envelope", .{ .root_source_file = b.path("envelope/envelope.zig") });
|
||||
|
||||
for ([_]struct { name: []const u8, root: []const u8 }{
|
||||
.{ .name = "vfs-protocol", .root = "vfs/vfs-protocol.zig" },
|
||||
.{ .name = "input-protocol", .root = "input/input-protocol.zig" },
|
||||
.{ .name = "block-protocol", .root = "block/block-protocol.zig" },
|
||||
.{ .name = "usb-transfer-protocol", .root = "usb-transfer/usb-transfer-protocol.zig" },
|
||||
.{ .name = "device-manager-protocol", .root = "device-manager/device-manager-protocol.zig" },
|
||||
.{ .name = "display-protocol", .root = "display/display-protocol.zig" },
|
||||
.{ .name = "scanout-protocol", .root = "scanout/scanout-protocol.zig" },
|
||||
.{ .name = "power-protocol", .root = "power/power-protocol.zig" },
|
||||
}) |protocol| {
|
||||
_ = b.addModule(protocol.name, .{
|
||||
.root_source_file = b.path(protocol.root),
|
||||
.imports = &.{.{ .name = "envelope", .module = envelope }},
|
||||
});
|
||||
}
|
||||
|
||||
// Standalone `zig build test` for this domain alone; the root build keeps
|
||||
// its aggregate test step.
|
||||
const test_step = b.step("test", "Run the protocol unit tests");
|
||||
// The envelope tests itself with no import of its own — everything else
|
||||
// imports it, so it is built separately rather than importing itself.
|
||||
const envelope_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("envelope/envelope.zig"), // framing round trips, verb numbering, dispatch, the floors
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(envelope_tests).step);
|
||||
|
||||
for ([_][]const u8{
|
||||
"vfs/vfs-protocol.zig", // NodeKind / DirectoryEntry sizes + op values
|
||||
"input/input-protocol.zig", // event numbering + the push-floor budget
|
||||
"display/display-protocol.zig", // pack(): native pixel encoding per format
|
||||
"device-manager/device-manager-protocol.zig", // the dual-use report, exactly on the push floor
|
||||
"power/power-protocol.zig", // the event kind as the packet's verb
|
||||
"usb-transfer/usb-transfer-protocol.zig", // the tail-carried control stage + the trimmed report
|
||||
}) |root| {
|
||||
const protocol_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(root),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
.imports = &.{.{ .name = "envelope", .module = envelope }},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(protocol_tests).step);
|
||||
}
|
||||
}
|
||||
@@ -1,8 +0,0 @@
|
||||
.{
|
||||
.name = .protocol,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0xc8c0bc4c4d551283, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -1,51 +1,18 @@
|
||||
//! The device-manager protocol (docs/device-driver-development/device-manager.md):
|
||||
//! what drivers and applications say to the device manager over
|
||||
//! `/protocol/device-manager`. Defined through the envelope
|
||||
//! (docs/os-development/protocol-namespace.md), so every packet — request, reply,
|
||||
//! and pushed event alike — begins with the folded `Header`.
|
||||
//!
|
||||
//! **`Header.target` is the device id.** It was the `device_id` field of three
|
||||
//! different messages; folding it into the header is what made the packed
|
||||
//! leading operation byte disappear along with it. `no_device` addresses a
|
||||
//! driver that serves no enumerated device.
|
||||
//!
|
||||
//! Two of the manager's four old operations were the reserved verbs under
|
||||
//! another name and are gone from this protocol's own numbering: `enumerate`
|
||||
//! (the tree, one `ChildEntry` per record in the reply tail) and `subscribe`
|
||||
//! (the watcher's endpoint rides as the call's capability). What is left is the
|
||||
//! driver-facing half — the handshake and the two tree reports.
|
||||
//!
|
||||
//! **`ChildAdded` travels in both directions, and says so twice.** A bus driver
|
||||
//! *calls* `child_added` to report a device; the manager then *pushes* the same
|
||||
//! struct to every subscriber as the `child_added` event. Operations and events
|
||||
//! are numbered in separate spaces, so one struct under two numbers is exactly
|
||||
//! how the envelope spells "one encoding, both directions" — and the direction
|
||||
//! (call vs. send) already tells them apart.
|
||||
//!
|
||||
//! Deliberately contains nothing lifecycle-shaped: stopping, liveness (the
|
||||
//! zero-length ping), and exit reasons are the universal vocabulary of
|
||||
//! The device-manager protocol (docs/device-manager.md): what drivers and
|
||||
//! applications say to the device manager over its well-known endpoint. The
|
||||
//! vfs-protocol pattern — extern-struct messages, a version in the handshake,
|
||||
//! reserved fields — so both sides depend on the contract by name. Deliberately
|
||||
//! contains nothing lifecycle-shaped: stopping, liveness (the zero-length ping),
|
||||
//! and exit reasons are the universal vocabulary of
|
||||
//! docs/process-lifecycle.md, not this protocol.
|
||||
|
||||
const std = @import("std");
|
||||
const envelope = @import("envelope");
|
||||
|
||||
/// The protocol version a driver states in its hello, and the version this
|
||||
/// contract answers `describe` with. A manager that cannot serve a driver's
|
||||
/// version refuses the hello, and the mismatch is loud at startup instead of
|
||||
/// quiet corruption later.
|
||||
///
|
||||
/// `describe` publishes the same number, but it cannot replace this: it tells a
|
||||
/// *client* what the provider is, and here it is the **provider** that has to
|
||||
/// learn what the client was built against in order to refuse it.
|
||||
pub const version = 1;
|
||||
|
||||
/// `Header.target` for a driver that serves no enumerated device (a test
|
||||
/// fixture, a synthetic source), and `ChildAdded`'s answer for a leaf that was
|
||||
/// never `device_register`ed.
|
||||
pub const no_device: u64 = ~@as(u64, 0);
|
||||
/// The protocol version a driver states in its hello. A manager that cannot
|
||||
/// serve a driver's version refuses the hello, and the mismatch is loud at
|
||||
/// startup instead of quiet corruption later.
|
||||
pub const version: u16 = 1;
|
||||
|
||||
/// Which bus a `child_added` came from — stated by the reporting bus driver so
|
||||
/// the manager's /system/configuration/devices.csv matcher knows how to read the report's identity
|
||||
/// the manager's /etc/devices.csv matcher knows how to read the report's identity
|
||||
/// (a PCI class triple vs a USB class triple are the same 24 bits but different
|
||||
/// namespaces) and which `bus` column a rule must name to bind it. `unknown` is
|
||||
/// the zero default, so an un-upgraded reporter fails to match rather than
|
||||
@@ -57,7 +24,7 @@ pub const BusKind = enum(u8) {
|
||||
acpi = 3,
|
||||
};
|
||||
|
||||
/// What kind of driver is talking (docs/device-driver-development/driver-model.md's shapes).
|
||||
/// What kind of driver is talking (docs/driver-model.md's shapes).
|
||||
pub const Role = enum(u8) {
|
||||
/// Owns a controller and reports the devices behind it (`child_added`).
|
||||
bus = 1,
|
||||
@@ -65,45 +32,58 @@ pub const Role = enum(u8) {
|
||||
device = 2,
|
||||
};
|
||||
|
||||
// --- the per-operation request parts ----------------------------------------
|
||||
//
|
||||
// Each names the bytes AFTER the prefix. Nothing here carries an operation or a
|
||||
// device id: those are the packet header's, folded in once. No reply part
|
||||
// carries a status either — that is the `Status` every reply already begins
|
||||
// with, so the manager's old three `{status, reserved}` reply structs are gone.
|
||||
/// The message kinds.
|
||||
pub const Operation = enum(u8) {
|
||||
hello = 1,
|
||||
child_added = 2,
|
||||
child_removed = 3,
|
||||
enumerate = 4,
|
||||
subscribe = 5,
|
||||
};
|
||||
|
||||
/// `Hello.device_id` for a driver that serves no enumerated device (a test
|
||||
/// fixture, a synthetic source).
|
||||
pub const no_device: u64 = ~@as(u64, 0);
|
||||
|
||||
/// The handshake, sent once by every driver the manager spawns — the manager's
|
||||
/// one self-enforced deadline: spawned and silent past it means wrong binary,
|
||||
/// wrong version, or wedged before main, and the stop sequence follows. The
|
||||
/// device this driver was assigned (its argv[1]) is `Header.target`.
|
||||
/// wrong version, or wedged before main, and the stop sequence follows.
|
||||
pub const Hello = extern struct {
|
||||
/// A `Role` value.
|
||||
operation: u8 = @intFromEnum(Operation.hello),
|
||||
/// A Role value.
|
||||
role: u8,
|
||||
_padding: u8 = 0,
|
||||
/// The protocol version this driver was built against (`version`).
|
||||
version: u16 = version,
|
||||
reserved: u32 = 0,
|
||||
/// The device this driver was assigned (its argv[1]), or `no_device`.
|
||||
device_id: u64,
|
||||
};
|
||||
|
||||
pub const hello_size = @sizeOf(Hello);
|
||||
|
||||
/// The manager's answer to a hello. Nonzero status = refused (version mismatch,
|
||||
/// unknown sender); a refused driver should exit cleanly.
|
||||
pub const HelloReply = extern struct {
|
||||
status: i32,
|
||||
reserved: u32 = 0,
|
||||
};
|
||||
|
||||
pub const reply_size = @sizeOf(HelloReply);
|
||||
|
||||
/// A bus driver reporting one device it discovered behind its controller
|
||||
/// (docs/device-driver-development/device-manager.md "the tree"), and the payload
|
||||
/// the manager pushes to its subscribers for the same event. Identity is the
|
||||
/// bus's native language — for USB a port-speed class, for PCI the class triple.
|
||||
/// The manager mirrors the child into its tree; when the reporting driver dies,
|
||||
/// the manager prunes everything it reported (the children describe protocol
|
||||
/// state that died with it) and the restarted instance rediscovers and
|
||||
/// re-reports.
|
||||
///
|
||||
/// `Header.target` is the kernel device id this child was `device_register`ed
|
||||
/// as — what the manager hands a matched driver as its argv assignment — or
|
||||
/// `no_device` for an unregistered leaf (a USB port before the descriptor
|
||||
/// track). That is the field that used to sit at the end of this struct.
|
||||
///
|
||||
/// **The field order is the size budget.** An event packet is the header plus
|
||||
/// this, within 64 bytes, and three `u64`s round the whole struct up to a
|
||||
/// multiple of eight whatever order they sit in — so the small fields are
|
||||
/// packed tail-first into the space the rounding pays for anyway. `Define`
|
||||
/// checks the result; this comment is why there is no slack in it.
|
||||
/// (docs/device-manager.md "the tree"). Identity is the bus's native language —
|
||||
/// for USB a port-speed class; the (class, subclass, protocol) triple joins it
|
||||
/// once control transfers exist (the USB track). The manager mirrors the child
|
||||
/// into its tree; when the reporting driver dies, the manager prunes everything
|
||||
/// it reported (the children describe protocol state that died with it) and the
|
||||
/// restarted instance rediscovers and re-reports.
|
||||
pub const ChildAdded = extern struct {
|
||||
operation: u8 = @intFromEnum(Operation.child_added),
|
||||
/// A `BusKind` value: which bus reported this child, so the manager reads the
|
||||
/// identity in the right namespace and matches against the right `bus` column.
|
||||
bus: u8 = @intFromEnum(BusKind.unknown),
|
||||
reserved1: u16 = 0,
|
||||
reserved2: u32 = 0,
|
||||
/// The reporting driver's own device (the controller) — the child's parent.
|
||||
parent: u64,
|
||||
/// Where on the bus (for USB: the root port number, 1-based).
|
||||
@@ -111,130 +91,85 @@ pub const ChildAdded = extern struct {
|
||||
/// Bus-specific identity (for USB: the PORTSC port-speed class; for PCI:
|
||||
/// the class triple; for ACPI devices, 0 — identity is the hid below).
|
||||
identity: u64,
|
||||
/// The PCI subsystem id, packed `(subsystem_vendor << 16) | subsystem_device`
|
||||
/// (so it reads vendor-first, matching the CSV's `ssvid:ssid`), or 0 when the
|
||||
/// device has no subsystem id (a bridge, or a non-PCI bus).
|
||||
subsystem: u32 = 0,
|
||||
/// The kernel device id this child was `device_register`ed as — what the
|
||||
/// manager hands a matched driver as its argv assignment — or `no_device`
|
||||
/// for an unregistered leaf (a USB port before the descriptor track).
|
||||
device_id: u64 = no_device,
|
||||
/// The vendor id (PCI vendor / USB idVendor), or 0 when the bus has no such
|
||||
/// concept (ACPI). Carried so the manager's /system/configuration/devices.csv matcher can bind
|
||||
/// concept (ACPI). Carried so the manager's /etc/devices.csv matcher can bind
|
||||
/// on vendor — a level the bus-native `identity` (a class triple) cannot express.
|
||||
vendor: u16 = 0,
|
||||
/// The device id (PCI device / USB idProduct), or 0. The most specific numeric
|
||||
/// level: this is what lets one virtio-gpu (1AF4:1050) be told from any other
|
||||
/// virtio display function without the driver re-confirming after it is spawned.
|
||||
device: u16 = 0,
|
||||
/// The PCI subsystem id, packed `(subsystem_vendor << 16) | subsystem_device`
|
||||
/// (so it reads vendor-first, matching the CSV's `ssvid:ssid`), or 0 when the
|
||||
/// device has no subsystem id (a bridge, or a non-PCI bus).
|
||||
subsystem: u32 = 0,
|
||||
/// The ACPI hardware id (`_HID`), EISA-decoded (e.g. "PNP0303"), for devices
|
||||
/// discovered by firmware string rather than a numeric bus identity. Empty
|
||||
/// (all zero) otherwise. Widens for FDT `compatible` strings later.
|
||||
hid: [8]u8 = .{0} ** 8,
|
||||
/// A `BusKind` value: which bus reported this child, so the manager reads the
|
||||
/// identity in the right namespace and matches against the right `bus` column.
|
||||
bus: u8 = @intFromEnum(BusKind.unknown),
|
||||
_padding: [7]u8 = .{0} ** 7,
|
||||
};
|
||||
|
||||
/// A bus driver reporting a device gone (hot-unplug), and the payload pushed to
|
||||
/// subscribers for it.
|
||||
///
|
||||
/// **This is the one message whose target stays 0.** A removal is addressed by
|
||||
/// the composite (parent, bus address) — the reporter knows where the device
|
||||
/// *was*, not necessarily what id it had been registered under — and a single
|
||||
/// `u64` cannot carry a pair. So the address stays in the payload, where it
|
||||
/// always was, and the header addresses the provider itself.
|
||||
pub const child_added_size = @sizeOf(ChildAdded);
|
||||
|
||||
/// A bus driver reporting a device gone (hot-unplug). Not yet sent by any
|
||||
/// driver — the port scan has no unplug interrupt — but the manager handles it;
|
||||
/// death-pruning covers removal until hotplug lands.
|
||||
pub const ChildRemoved = extern struct {
|
||||
operation: u8 = @intFromEnum(Operation.child_removed),
|
||||
reserved0: u8 = 0,
|
||||
reserved1: u16 = 0,
|
||||
reserved2: u32 = 0,
|
||||
parent: u64,
|
||||
bus_address: u64,
|
||||
};
|
||||
|
||||
/// One record of the reserved `enumerate` reply: the manager's mirror, one
|
||||
/// entry per known child, packed into the reply tail. The count is
|
||||
/// `Status.len / @sizeOf(ChildEntry)` — the envelope's reply length says how
|
||||
/// many arrived, so no count header is spent on saying it twice.
|
||||
pub const child_removed_size = @sizeOf(ChildRemoved);
|
||||
|
||||
/// The manager's answer to a tree report.
|
||||
pub const ReportReply = extern struct {
|
||||
status: i32,
|
||||
reserved: u32 = 0,
|
||||
};
|
||||
|
||||
/// An application asking for the tree (M18.3): the reply is an EnumerateReply
|
||||
/// header followed by `count` ChildEntry records.
|
||||
pub const Enumerate = extern struct {
|
||||
operation: u8 = @intFromEnum(Operation.enumerate),
|
||||
reserved0: u8 = 0,
|
||||
reserved1: u16 = 0,
|
||||
reserved2: u32 = 0,
|
||||
};
|
||||
|
||||
pub const EnumerateReply = extern struct {
|
||||
status: i32,
|
||||
/// ChildEntry records following this header.
|
||||
count: u32,
|
||||
};
|
||||
|
||||
pub const ChildEntry = extern struct {
|
||||
parent: u64,
|
||||
bus_address: u64,
|
||||
identity: u64,
|
||||
};
|
||||
|
||||
/// How many `ChildEntry` records one `enumerate` reply can carry. Paging joins
|
||||
/// the protocol if a tree ever outgrows one packet.
|
||||
pub const entries_per_reply: usize = (envelope.packet_maximum - envelope.prefix_size) / @sizeOf(ChildEntry);
|
||||
|
||||
pub const Protocol = envelope.Define(.{
|
||||
.name = "device-manager",
|
||||
.version = version,
|
||||
.operations = &.{
|
||||
// The driver-facing half. `enumerate` and `subscribe` are not here: they
|
||||
// are the reserved verbs, which mean the same thing at every provider.
|
||||
.{ .name = "hello", .request = Hello },
|
||||
.{ .name = "child_added", .request = ChildAdded },
|
||||
.{ .name = "child_removed", .request = ChildRemoved },
|
||||
},
|
||||
.events = &.{
|
||||
// The watcher-facing half — the same two structs, pushed rather than
|
||||
// called, in the events' own numbering space.
|
||||
.{ .name = "child_added", .payload = ChildAdded },
|
||||
.{ .name = "child_removed", .payload = ChildRemoved },
|
||||
},
|
||||
});
|
||||
|
||||
pub const Operation = Protocol.Operation;
|
||||
pub const Event = Protocol.Event;
|
||||
|
||||
/// What the manager sizes its buffers to — the call floor, as every protocol does.
|
||||
pub const message_maximum: usize = Protocol.message_maximum;
|
||||
|
||||
test "a tree report fits the push floor with the header folded in" {
|
||||
// The dual-use struct is the tight one: `child_added` is both a call and an
|
||||
// event, and the event floor is 64 bytes *including* the header. Forty-one
|
||||
// bytes of content, rounded to 48 by the three u64s' alignment, plus the
|
||||
// 16-byte header — exactly on the floor, which is what folding the operation
|
||||
// byte and the device id out of the payload bought.
|
||||
try std.testing.expectEqual(@as(usize, 48), @sizeOf(ChildAdded));
|
||||
try std.testing.expectEqual(envelope.post_maximum, Protocol.event_maximum);
|
||||
try std.testing.expect(Protocol.event_maximum <= envelope.post_maximum);
|
||||
// Ten records per enumerate reply — what the old count-header layout carried.
|
||||
try std.testing.expectEqual(@as(usize, 10), entries_per_reply);
|
||||
}
|
||||
|
||||
test "the verb and event numbering, and the device id in the header" {
|
||||
try std.testing.expectEqual(@as(u32, 16), @intFromEnum(Operation.hello));
|
||||
try std.testing.expectEqual(@as(u32, 17), @intFromEnum(Operation.child_added));
|
||||
try std.testing.expectEqual(@as(u32, 18), @intFromEnum(Operation.child_removed));
|
||||
// Events number in their own space, so the same two reports start at 16 too.
|
||||
try std.testing.expectEqual(@as(u32, 16), @intFromEnum(Event.child_added));
|
||||
try std.testing.expectEqual(@as(u32, 17), @intFromEnum(Event.child_removed));
|
||||
// The manager's own enumerate/subscribe became the RESERVED verbs, below the
|
||||
// protocol range entirely.
|
||||
try std.testing.expectEqual(@as(u32, 1), envelope.operation_enumerate);
|
||||
try std.testing.expectEqual(@as(u32, 2), envelope.operation_subscribe);
|
||||
|
||||
var buffer: [message_maximum]u8 = undefined;
|
||||
const hello = Protocol.encodeRequest(.hello, 7, .{ .role = @intFromEnum(Role.bus) }, &.{}, &buffer).?;
|
||||
try std.testing.expectEqual(@as(u64, 7), envelope.headerOf(hello).?.target);
|
||||
try std.testing.expectEqual(@as(u16, 1), Protocol.decodeRequest(.hello, hello).?.version);
|
||||
}
|
||||
|
||||
test "one struct, two numbers: the report a bus calls and the event a watcher is pushed" {
|
||||
const report = ChildAdded{
|
||||
.parent = 3,
|
||||
.bus_address = 1,
|
||||
.identity = 0x030000,
|
||||
.bus = @intFromEnum(BusKind.pci),
|
||||
.vendor = 0x1AF4,
|
||||
/// An application subscribing to published add/remove events (the input-service
|
||||
/// pattern): the subscriber's endpoint rides as the call's **capability**, and
|
||||
/// events arrive on it as buffered messages whose payload is the same
|
||||
/// ChildAdded / ChildRemoved struct the bus drivers send — one encoding, both
|
||||
/// directions.
|
||||
pub const Subscribe = extern struct {
|
||||
operation: u8 = @intFromEnum(Operation.subscribe),
|
||||
reserved0: u8 = 0,
|
||||
reserved1: u16 = 0,
|
||||
reserved2: u32 = 0,
|
||||
};
|
||||
|
||||
var call: [message_maximum]u8 = undefined;
|
||||
const called = Protocol.encodeRequest(.child_added, 42, report, &.{}, &call).?;
|
||||
try std.testing.expectEqual(Operation.child_added, Protocol.operationOf(called).?);
|
||||
try std.testing.expectEqual(@as(u64, 42), envelope.headerOf(called).?.target);
|
||||
|
||||
var push: [envelope.post_maximum]u8 = undefined;
|
||||
const pushed = Protocol.encodeEvent(.child_added, 42, report, &push).?;
|
||||
try std.testing.expectEqual(envelope.post_maximum, pushed.len);
|
||||
try std.testing.expectEqual(Event.child_added, Protocol.eventOf(pushed).?);
|
||||
try std.testing.expectEqual(@as(u16, 0x1AF4), Protocol.decodeEvent(.child_added, pushed).?.vendor);
|
||||
// Same bytes after the prefix, different verb in it — the direction is what
|
||||
// tells a call from a push, and the numbering spaces never collide.
|
||||
try std.testing.expectEqualSlices(u8, called[envelope.prefix_size..], pushed[envelope.prefix_size..]);
|
||||
}
|
||||
/// Upper bound on any message in this protocol — sizes the endpoint buffers.
|
||||
/// Capped by the kernel's IPC MESSAGE_MAXIMUM (256): an EnumerateReply carries
|
||||
/// up to ten ChildEntry records per call, plenty for the mirror's current
|
||||
/// bounds; paging joins the protocol if a tree ever outgrows one message.
|
||||
pub const message_maximum = 256;
|
||||
|
||||
@@ -1,137 +1,93 @@
|
||||
//! The display wire protocol — what a client says to the display service over
|
||||
//! `/protocol/display`. The compositor owns the framebuffer and an ordered stack of
|
||||
//! **layers**; a client creates layers, draws into them with these operations, marks damage,
|
||||
//! and asks for a `present`. v1 surfaces are server-owned (a client draws by command);
|
||||
//! shared-memory surfaces are a later milestone (docs/display.md).
|
||||
//!
|
||||
//! **`Header.target` is the layer** on every verb that names one — the field that used to be
|
||||
//! `Request.layer`. `info`, `present`, `set_mode`, `get_modes` and `attach_scanout` address
|
||||
//! the compositor itself, so they leave it 0.
|
||||
//!
|
||||
//! Every verb carries its own request type. The single overloaded 40-byte request this
|
||||
//! protocol used to have is gone, and with it the field abuse it invited: `attach_scanout`
|
||||
//! spent `x` on a stride, `y` on a refresh rate and `colour` on a pixel format, which no
|
||||
//! reader could have guessed and no compiler could have caught.
|
||||
//! The display wire protocol — what a client says to the display service over its
|
||||
//! well-known `.display` endpoint. extern-struct messages with an `Operation` tag, the
|
||||
//! same shape as block/vfs/input protocols. The compositor owns the framebuffer and an
|
||||
//! ordered stack of **layers**; a client creates layers, draws into them with these
|
||||
//! operations, marks damage, and asks for a `present`. v1 surfaces are server-owned (a
|
||||
//! client draws by command); shared-memory surfaces are a later milestone (docs/display.md).
|
||||
|
||||
const envelope = @import("envelope");
|
||||
const std = @import("std");
|
||||
|
||||
/// The answer to `info()`: the display's current mode.
|
||||
pub const Info = extern struct {
|
||||
pub const Operation = enum(u32) {
|
||||
/// info() -> { width, height, pitch, format }: the display's current mode.
|
||||
info = 0,
|
||||
/// create_layer(x, y, width, height, z) -> { layer }: a new server-owned surface.
|
||||
create_layer = 1,
|
||||
/// configure_layer(layer, x, y, z, visible): move, restack, show, or hide a layer.
|
||||
configure_layer = 2,
|
||||
/// destroy_layer(layer): release a layer.
|
||||
destroy_layer = 3,
|
||||
/// fill_rect(layer, x, y, width, height, colour): fill a rectangle of a layer.
|
||||
fill_rect = 4,
|
||||
/// blit_tile(layer, x, y, width, height, <inline pixels>): copy a small pixel tile in.
|
||||
blit_tile = 5,
|
||||
/// damage(layer, x, y, width, height): mark a region dirty for the next present.
|
||||
damage = 6,
|
||||
/// present(): composite the dirty layers and flush to the screen.
|
||||
present = 7,
|
||||
/// attach_scanout(x=stride, y=refresh_hz, width, height, colour=format) + <surface
|
||||
/// capability>: a native scanout driver announces itself, handing over the shared scanout
|
||||
/// surface as an `ipc_call` send_cap. The compositor maps it, looks up the driver's
|
||||
/// `.scanout` present channel, and upgrades off the GOP floor (docs/display-v2.md V4).
|
||||
/// `x` is the surface's row stride in pixels, `y` the panel refresh rate from the
|
||||
/// driver's EDID read (0 = unknown; paces the compositor's frame clock), `colour` the
|
||||
/// DisplayFormat.
|
||||
attach_scanout = 8,
|
||||
/// set_mode(width, height): change the display resolution — only a native backend that
|
||||
/// reports `canModeSet` honours it; on the GOP floor it fails (docs/display-v2.md V5).
|
||||
set_mode = 9,
|
||||
/// get_modes() -> ModesReply: the resolutions the display can switch to (empty on GOP).
|
||||
get_modes = 10,
|
||||
};
|
||||
|
||||
/// The fixed request header. A `blit_tile`'s pixel payload (width*height 32-bit pixels)
|
||||
/// follows this header inline in the same message, up to `maximum_payload`.
|
||||
pub const Request = extern struct {
|
||||
operation: u32,
|
||||
layer: u32 = 0, // create/configure/destroy/fill/blit/damage: the target layer
|
||||
x: u32 = 0,
|
||||
y: u32 = 0,
|
||||
width: u32 = 0,
|
||||
height: u32 = 0,
|
||||
pitch: u32 = 0, // bytes per row (may exceed width*4)
|
||||
z: u32 = 0, // create_layer / configure_layer: stacking order (higher = in front)
|
||||
colour: u32 = 0, // fill_rect: the fill colour (native pixel value)
|
||||
visible: u32 = 1, // configure_layer: 0 hides the layer
|
||||
reserved: u32 = 0,
|
||||
};
|
||||
|
||||
pub const Reply = extern struct {
|
||||
status: i32, // 0 on success, negative on failure
|
||||
reserved: u32 = 0,
|
||||
// info():
|
||||
width: u32 = 0,
|
||||
height: u32 = 0,
|
||||
pitch: u32 = 0,
|
||||
format: u32 = 0, // a device-abi DisplayFormat value (0 = rgbx, 1 = bgrx)
|
||||
// create_layer():
|
||||
layer: u32 = 0,
|
||||
reserved2: u32 = 0,
|
||||
};
|
||||
|
||||
/// `create_layer(...)`: a new server-owned surface. Coordinates are signed — a layer may sit
|
||||
/// partly off-screen.
|
||||
pub const CreateLayer = extern struct {
|
||||
x: i32,
|
||||
y: i32,
|
||||
width: u32,
|
||||
height: u32,
|
||||
z: u32 = 0, // stacking order (higher = nearer the front)
|
||||
visible: u32 = 1,
|
||||
};
|
||||
|
||||
/// The layer a `create_layer` established — the integer later packets put in `Header.target`.
|
||||
pub const Created = extern struct { layer: u32 };
|
||||
|
||||
/// `configure_layer(...)` on `Header.target`: move, restack, show, or hide it.
|
||||
pub const ConfigureLayer = extern struct {
|
||||
x: i32,
|
||||
y: i32,
|
||||
z: u32 = 0,
|
||||
visible: u32 = 1, // 0 hides the layer
|
||||
};
|
||||
|
||||
/// `fill_rect(...)` on `Header.target`: fill a layer-local rectangle with a native pixel value.
|
||||
pub const FillRect = extern struct {
|
||||
x: i32,
|
||||
y: i32,
|
||||
width: u32,
|
||||
height: u32,
|
||||
colour: u32,
|
||||
};
|
||||
|
||||
/// `blit_tile(...)` on `Header.target`: copy a `width`×`height` tile of native pixels
|
||||
/// (row-major, little-endian) into the layer. The pixels ride inline as the packet's tail,
|
||||
/// up to `maximum_payload`.
|
||||
pub const BlitTile = extern struct {
|
||||
x: i32,
|
||||
y: i32,
|
||||
width: u32,
|
||||
height: u32,
|
||||
};
|
||||
|
||||
/// `damage(...)` on `Header.target`: mark a layer-local region dirty for the next present.
|
||||
pub const Damage = extern struct {
|
||||
x: i32,
|
||||
y: i32,
|
||||
width: u32,
|
||||
height: u32,
|
||||
};
|
||||
|
||||
/// `attach_scanout(...)` + the shared surface as the call's capability: a native scanout
|
||||
/// driver announces itself. The compositor maps the surface, opens the driver's
|
||||
/// `/protocol/scanout` present channel, and upgrades off the GOP floor (docs/display-v2.md
|
||||
/// V4). Each field says what it is, which the old shared request could not.
|
||||
pub const AttachScanout = extern struct {
|
||||
/// The surface's row stride in pixels (it is sized to the driver's largest mode).
|
||||
stride: u32,
|
||||
/// The active mode within that surface.
|
||||
width: u32,
|
||||
height: u32,
|
||||
/// A device-abi DisplayFormat value.
|
||||
format: u32,
|
||||
/// The panel refresh rate from the driver's EDID read (0 = unknown); it paces the
|
||||
/// compositor's frame clock.
|
||||
refresh_hz: u32 = 0,
|
||||
};
|
||||
|
||||
/// `set_mode(width, height)`: change the display resolution — only a native backend that
|
||||
/// reports `canModeSet` honours it; on the GOP floor it fails (docs/display-v2.md V5).
|
||||
pub const SetMode = extern struct { width: u32, height: u32 };
|
||||
|
||||
/// One selectable display mode.
|
||||
pub const Mode = extern struct { width: u32, height: u32 };
|
||||
pub const max_modes = 4;
|
||||
|
||||
/// The answer to `get_modes`: the resolutions the display can switch to (empty on GOP).
|
||||
pub const Modes = extern struct {
|
||||
count: u32 = 0,
|
||||
_padding: u32 = 0,
|
||||
modes: [max_modes]Mode = @splat(.{ .width = 0, .height = 0 }),
|
||||
/// The reply to `get_modes`: a small fixed list of resolutions the display can switch to.
|
||||
pub const ModesReply = extern struct {
|
||||
status: i32,
|
||||
count: u32,
|
||||
modes: [max_modes]Mode,
|
||||
};
|
||||
pub const modes_reply_size: usize = @sizeOf(ModesReply);
|
||||
|
||||
pub const Protocol = envelope.Define(.{
|
||||
.name = "display",
|
||||
.version = 1,
|
||||
.operations = &.{
|
||||
.{ .name = "info", .reply = Info },
|
||||
.{ .name = "create_layer", .request = CreateLayer, .reply = Created },
|
||||
.{ .name = "configure_layer", .request = ConfigureLayer },
|
||||
.{ .name = "destroy_layer" },
|
||||
.{ .name = "fill_rect", .request = FillRect },
|
||||
.{ .name = "blit_tile", .request = BlitTile },
|
||||
.{ .name = "damage", .request = Damage },
|
||||
.{ .name = "present" },
|
||||
.{ .name = "attach_scanout", .request = AttachScanout },
|
||||
.{ .name = "set_mode", .request = SetMode },
|
||||
.{ .name = "get_modes", .reply = Modes },
|
||||
},
|
||||
});
|
||||
|
||||
pub const Operation = Protocol.Operation;
|
||||
pub const message_maximum: usize = Protocol.message_maximum;
|
||||
|
||||
/// The largest inline pixel tile a `blit_tile` may carry: the call floor less the header and
|
||||
/// this verb's own fixed part — 224 bytes, up to 56 pixels, enough for a cursor or a small
|
||||
/// sprite. Larger bitmaps are the deferred shared-memory surface path (docs/display.md).
|
||||
/// Per-verb rather than protocol-wide, because with per-operation requests there is no
|
||||
/// single "request size" to subtract any more.
|
||||
pub const maximum_payload: usize = envelope.packet_maximum - envelope.prefix_size - @sizeOf(BlitTile);
|
||||
/// The IPC message size — the kernel caps every message at `MESSAGE_MAXIMUM` (256 bytes,
|
||||
/// system/kernel/ipc-synchronous.zig), so this matches it (a larger receive/reply buffer
|
||||
/// is rejected with -E2BIG). A `blit_tile` therefore carries only a *small* tile inline —
|
||||
/// `maximum_payload` bytes = up to 54 pixels, enough for a cursor or small sprite; larger
|
||||
/// bitmaps are the deferred shared-memory surface path (docs/display.md).
|
||||
pub const message_maximum: usize = 256;
|
||||
pub const request_size: usize = @sizeOf(Request);
|
||||
pub const reply_size: usize = @sizeOf(Reply);
|
||||
pub const maximum_payload: usize = message_maximum - request_size;
|
||||
|
||||
/// Pack an 8-bit-per-channel colour into the display's native 32-bit pixel for `format`
|
||||
/// (a device-abi `DisplayFormat`: 0 = rgbx, 1 = bgrx). Shared so a `colour` in a
|
||||
@@ -157,15 +113,3 @@ test "pack encodes native byte order for rgbx and bgrx" {
|
||||
try std.testing.expectEqual(@as(u32, 0x00AA_0000), pack(1, 0xAA, 0, 0));
|
||||
try std.testing.expectEqual(@as(u32, 0x0000_3020), pack(0, 0x20, 0x30, 0)); // green in byte 1
|
||||
}
|
||||
|
||||
test "the layer rides the header, and the blit tile grew with the split" {
|
||||
var buffer: [message_maximum]u8 = undefined;
|
||||
const pixels = [_]u8{0xFF} ** 16;
|
||||
const packet = Protocol.encodeRequest(.blit_tile, 3, .{ .x = 1, .y = 2, .width = 2, .height = 2 }, &pixels, &buffer).?;
|
||||
try std.testing.expectEqual(@as(u64, 3), envelope.headerOf(packet).?.target);
|
||||
try std.testing.expectEqual(@as(i32, 1), Protocol.decodeRequest(.blit_tile, packet).?.x);
|
||||
try std.testing.expectEqual(@as(usize, 16), Protocol.requestTail(.blit_tile, packet).len);
|
||||
// 216 bytes under the old 40-byte shared request; the header plus this
|
||||
// verb's own four fields is 32.
|
||||
try std.testing.expectEqual(@as(usize, 224), maximum_payload);
|
||||
}
|
||||
|
||||
@@ -1,974 +0,0 @@
|
||||
//! The envelope — the fixed prefix every danos packet begins with, and the
|
||||
//! comptime `Define` that builds a protocol out of it. Layer L2 of
|
||||
//! [communication.md](../../../docs/os-development/communication.md); the
|
||||
//! authoritative description is
|
||||
//! [protocol-namespace.md](../../../docs/os-development/protocol-namespace.md).
|
||||
//!
|
||||
//! This is the one module in the protocol domain that is not itself a protocol:
|
||||
//! it is the shape every protocol is expressed in. A protocol module hands
|
||||
//! `Define` its verbs and events and gets back numbered operations (never
|
||||
//! colliding with the reserved range), typed encode/decode helpers, a
|
||||
//! provider-side dispatch table that answers `describe` on its own — and, the
|
||||
//! point of the exercise, compile-time proof that none of its packets can
|
||||
//! exceed the transport floor. Errors that used to surface as runtime
|
||||
//! truncation are compile errors, and "packets never fragment" is enforced at
|
||||
//! the source rather than by review.
|
||||
//!
|
||||
//! **The prefix is folded, never stacked.** A `request`, `reply`, or `payload`
|
||||
//! type names the bytes that follow the prefix — never the whole packet. The
|
||||
//! verb and the object being addressed live in the prefix, so a protocol type
|
||||
//! carries neither an `operation` field of its own nor a nested `Header`;
|
||||
//! `rejectStacking` refuses one at compile time, and every size check adds
|
||||
//! `prefix_size` exactly once.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
// --- the prefix -------------------------------------------------------------
|
||||
|
||||
/// Every packet a danos protocol transmits begins with this header — requests
|
||||
/// on the synchronous call path, event packets on the asynchronous push path.
|
||||
/// A reply spends the same 16 bytes on `Status` instead.
|
||||
pub const Header = extern struct {
|
||||
/// The verb. Values below `first_protocol_operation` are the reserved
|
||||
/// universal verbs, which mean the same thing in every protocol.
|
||||
operation: u32,
|
||||
_padding: u32 = 0,
|
||||
/// **Object** addressing within the peer, never party addressing: which of
|
||||
/// the peer's objects this packet operates on — a volume, a layer, a node,
|
||||
/// a device. `0` addresses the provider itself, and a protocol with no
|
||||
/// objects never uses the field. *Which* party is at the other end was
|
||||
/// decided once, when the channel was opened, and *who sent this* is the
|
||||
/// kernel-stamped badge; neither is ever written here, which is what keeps
|
||||
/// the source unforgeable.
|
||||
target: u64 = 0,
|
||||
};
|
||||
|
||||
/// Every reply begins with this. `len` counts the bytes that follow: the
|
||||
/// reply's fixed part plus whatever variable tail the operation defines.
|
||||
pub const Status = extern struct {
|
||||
status: i32, // 0, or a negative errno
|
||||
_padding: u32 = 0,
|
||||
len: u32 = 0,
|
||||
_padding2: u32 = 0,
|
||||
};
|
||||
|
||||
/// The fixed prefix every packet spends — `Header` on a request or an event,
|
||||
/// `Status` on a reply. One constant, because the two are deliberately the same
|
||||
/// width: the packet budget does not depend on the direction.
|
||||
pub const prefix_size: usize = @sizeOf(Header);
|
||||
|
||||
comptime {
|
||||
if (@sizeOf(Header) != 16 or @sizeOf(Status) != 16)
|
||||
@compileError("the envelope prefix is 16 bytes in both directions");
|
||||
}
|
||||
|
||||
// --- the reserved verbs -----------------------------------------------------
|
||||
|
||||
/// Reserved verbs, answered by every provider. `Define` numbers a protocol's
|
||||
/// own verbs from `first_protocol_operation`, so no protocol can reach in here.
|
||||
pub const operation_describe: u32 = 0; // -> protocol name, version, target kinds
|
||||
pub const operation_enumerate: u32 = 1; // -> the current targets, one per reply page
|
||||
pub const operation_subscribe: u32 = 2; // capability = the subscriber's endpoint
|
||||
pub const operation_unsubscribe: u32 = 3;
|
||||
pub const first_protocol_operation: u32 = 16;
|
||||
|
||||
/// The optional body of a reserved `subscribe`: **which** of a provider's events
|
||||
/// the subscriber wants, as a bit mask whose meaning the protocol defines (the
|
||||
/// input service's device classes are the model). A reserved verb carries no
|
||||
/// typed request, so this rides the packet's tail — and zero, which is also what
|
||||
/// a subscribe that sent no body at all reads as, means *every* event.
|
||||
///
|
||||
/// The mask lives here rather than in each protocol because the subscriber
|
||||
/// machinery is the service harness's (library/kernel/service.zig): the harness
|
||||
/// records the number, the protocol decides what its bits mean, and neither has
|
||||
/// to know the other.
|
||||
pub const Subscription = extern struct { interest: u32 = 0 };
|
||||
|
||||
/// Frame a `subscribe` request. The subscriber's own endpoint travels as the
|
||||
/// call's *capability*, never in the packet — that is what makes the reverse
|
||||
/// path unforgeable.
|
||||
pub fn encodeSubscribe(interest: u32, buffer: []u8) ?[]u8 {
|
||||
const header = Header{ .operation = operation_subscribe };
|
||||
const body = Subscription{ .interest = interest };
|
||||
return frame(std.mem.asBytes(&header), std.mem.asBytes(&body), &.{}, buffer);
|
||||
}
|
||||
|
||||
/// Frame a bare `unsubscribe`: it names no event and no endpoint, because it
|
||||
/// means "every subscription this task holds here" (one task, one voice).
|
||||
pub fn encodeUnsubscribe(buffer: []u8) ?[]u8 {
|
||||
const header = Header{ .operation = operation_unsubscribe };
|
||||
return frame(std.mem.asBytes(&header), &.{}, &.{}, buffer);
|
||||
}
|
||||
|
||||
/// The interest mask out of a `subscribe` packet's tail, on the provider's side.
|
||||
/// A caller that sent no mask reads as the every-event mask.
|
||||
pub fn decodeSubscribe(tail: []const u8) Subscription {
|
||||
if (tail.len < @sizeOf(Subscription)) return .{};
|
||||
return std.mem.bytesToValue(Subscription, tail[0..@sizeOf(Subscription)]);
|
||||
}
|
||||
|
||||
/// The `describe` reply's fixed part, followed inline by `name_len` bytes of the
|
||||
/// protocol's name. This is the version handshake: the version is asked for
|
||||
/// once, at connect time, rather than re-carried by every packet out of a
|
||||
/// 256-byte budget.
|
||||
pub const Description = extern struct {
|
||||
version: u32,
|
||||
operation_count: u32,
|
||||
event_count: u32,
|
||||
name_len: u32,
|
||||
};
|
||||
|
||||
/// Longest protocol name a `describe` reply can carry.
|
||||
pub const name_maximum: usize = packet_maximum - prefix_size - @sizeOf(Description);
|
||||
|
||||
// --- the transport floor ----------------------------------------------------
|
||||
|
||||
/// The packet budget every protocol may assume on *any* transport. These are
|
||||
/// the kernel-ipc transport's limits — `MESSAGE_MAXIMUM` and `POST_MAXIMUM` in
|
||||
/// system/kernel/ipc-synchronous.zig — restated here because the kernel keeps
|
||||
/// them private and a protocol has to compile against something. A fatter
|
||||
/// transport raises its own ceiling; the floor does not move, so a protocol
|
||||
/// that fits here fits everywhere (communication.md: ceilings are transport
|
||||
/// properties, the floor is the protocol's contract).
|
||||
pub const packet_maximum: usize = 256; // one request or one reply (ipc_call)
|
||||
pub const post_maximum: usize = 64; // one event packet (ipc_send)
|
||||
|
||||
/// Whether a request or reply whose fixed part is `T` fits the call floor once
|
||||
/// the prefix is counted. The folded rule in one line: `prefix_size` is added
|
||||
/// exactly once, because `T` describes only what follows it. Exported so the
|
||||
/// rule itself is testable — `Define` enforces it as a compile error.
|
||||
pub fn fitsPacket(comptime T: type) bool {
|
||||
return prefix_size + @sizeOf(T) <= packet_maximum;
|
||||
}
|
||||
|
||||
/// The same, against the much smaller push floor an event packet lives within.
|
||||
pub fn fitsPost(comptime T: type) bool {
|
||||
return prefix_size + @sizeOf(T) <= post_maximum;
|
||||
}
|
||||
|
||||
// --- reply statuses the envelope itself produces -----------------------------
|
||||
|
||||
/// Continued from the kernel's danos-native errno numbering
|
||||
/// (system/kernel/ipc-synchronous.zig, which ends at `EPERM` = 9), so a client
|
||||
/// reads one vocabulary whether the number came from the kernel or a provider.
|
||||
/// Positive here, sent negated in `Status.status`, as the kernel spells it.
|
||||
pub const ENOSYS: i32 = 10; // this protocol has no such operation
|
||||
pub const EPROTO: i32 = 11; // malformed packet: shorter than the verb it names
|
||||
pub const EBUSY: i32 = 12; // the thing asked for is held by someone still alive
|
||||
|
||||
/// Restated from the kernel's half of the numbering, because a provider refuses
|
||||
/// too and userspace has no other place to read these from: `ENOENT` is "no such
|
||||
/// name", `EPERM` "not permitted". The protocol registry answers an ungranted
|
||||
/// bind with the second and a name a live provider already holds with `EBUSY`.
|
||||
pub const ENOENT: i32 = 4;
|
||||
pub const ENOSPC: i32 = 5;
|
||||
pub const EPERM: i32 = 9;
|
||||
|
||||
// --- framing ----------------------------------------------------------------
|
||||
|
||||
/// The header of a received packet, or null when it is too short to have one.
|
||||
pub fn headerOf(packet: []const u8) ?Header {
|
||||
if (packet.len < prefix_size) return null;
|
||||
return std.mem.bytesToValue(Header, packet[0..prefix_size]);
|
||||
}
|
||||
|
||||
/// The status of a received reply, or null when it is too short to have one.
|
||||
pub fn statusOf(packet: []const u8) ?Status {
|
||||
if (packet.len < prefix_size) return null;
|
||||
return std.mem.bytesToValue(Status, packet[0..prefix_size]);
|
||||
}
|
||||
|
||||
/// Frame a bare `describe` request. Protocol-independent: the reserved verbs
|
||||
/// are asked the same way of every provider.
|
||||
pub fn encodeDescribe(buffer: []u8) ?[]u8 {
|
||||
const header = Header{ .operation = operation_describe };
|
||||
return frame(std.mem.asBytes(&header), &.{}, &.{}, buffer);
|
||||
}
|
||||
|
||||
/// A decoded `describe` reply: the fixed part, plus the name that follows it.
|
||||
pub const Described = struct {
|
||||
description: Description,
|
||||
name: []const u8,
|
||||
};
|
||||
|
||||
/// Decode a `describe` reply packet. Null if it failed, was truncated, or is
|
||||
/// not a description at all.
|
||||
pub fn decodeDescribe(packet: []const u8) ?Described {
|
||||
const status = statusOf(packet) orelse return null;
|
||||
if (status.status != 0) return null;
|
||||
const body = packet[prefix_size..];
|
||||
if (body.len < @sizeOf(Description)) return null;
|
||||
const description = std.mem.bytesToValue(Description, body[0..@sizeOf(Description)]);
|
||||
const name = body[@sizeOf(Description)..];
|
||||
if (name.len < description.name_len) return null;
|
||||
return .{ .description = description, .name = name[0..description.name_len] };
|
||||
}
|
||||
|
||||
/// The single framing point: prefix, then the fixed part, then the variable
|
||||
/// tail, contiguous in one buffer. Null when the packet would not fit — a
|
||||
/// packet is never split, so not fitting is a failure, not a continuation.
|
||||
fn frame(prefix: []const u8, fixed: []const u8, tail: []const u8, buffer: []u8) ?[]u8 {
|
||||
const total = prefix.len + fixed.len + tail.len;
|
||||
if (total > buffer.len) return null;
|
||||
@memcpy(buffer[0..prefix.len], prefix);
|
||||
@memcpy(buffer[prefix.len..][0..fixed.len], fixed);
|
||||
@memcpy(buffer[prefix.len + fixed.len ..][0..tail.len], tail);
|
||||
return buffer[0..total];
|
||||
}
|
||||
|
||||
/// The bytes of a fixed part — empty for `void`, which is how an operation says
|
||||
/// "nothing but the verb".
|
||||
fn bytesOf(comptime T: type, value: *const T) []const u8 {
|
||||
if (@sizeOf(T) == 0) return &.{};
|
||||
return @as([*]const u8, @ptrCast(value))[0..@sizeOf(T)];
|
||||
}
|
||||
|
||||
/// Read a fixed part out of a packet body. A zero-sized part always succeeds
|
||||
/// (there is nothing to be short of); anything else needs its full width.
|
||||
fn valueOf(comptime T: type, body: []const u8) ?T {
|
||||
if (@sizeOf(T) == 0) return @as(T, undefined);
|
||||
if (body.len < @sizeOf(T)) return null;
|
||||
return std.mem.bytesToValue(T, body[0..@sizeOf(T)]);
|
||||
}
|
||||
|
||||
// --- the specification ------------------------------------------------------
|
||||
|
||||
/// One verb of a protocol. `request` and `reply` describe the bytes *after* the
|
||||
/// prefix; either may be `void`, meaning the verb (and its target) says it all.
|
||||
pub const OperationSpecification = struct {
|
||||
name: []const u8,
|
||||
request: type = void,
|
||||
reply: type = void,
|
||||
};
|
||||
|
||||
/// One event a provider pushes to its subscribers. `payload` is the bytes after
|
||||
/// the `Header`, and the whole packet must fit the push floor.
|
||||
pub const EventSpecification = struct {
|
||||
name: []const u8,
|
||||
payload: type = void,
|
||||
};
|
||||
|
||||
/// What `Define` is given: the contract, whole.
|
||||
pub const Specification = struct {
|
||||
/// The contract's name — the same word as its `/protocol/<name>` leaf and
|
||||
/// its `library/protocol/` module.
|
||||
name: []const u8,
|
||||
version: u32,
|
||||
operations: []const OperationSpecification = &.{},
|
||||
events: []const EventSpecification = &.{},
|
||||
};
|
||||
|
||||
const reserved_names = [_][]const u8{ "describe", "enumerate", "subscribe", "unsubscribe" };
|
||||
|
||||
fn isReservedName(comptime name: []const u8) bool {
|
||||
for (reserved_names) |reserved| {
|
||||
if (std.mem.eql(u8, reserved, name)) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Refuse a protocol type that carries the prefix inside itself. The header is
|
||||
/// folded into every packet, so a type that also holds one would send it twice
|
||||
/// and re-invent per-protocol addressing — the mistake the envelope exists to
|
||||
/// prevent.
|
||||
fn rejectStacking(comptime protocol: []const u8, comptime verb: []const u8, comptime T: type) void {
|
||||
switch (@typeInfo(T)) {
|
||||
.@"struct" => |info| for (info.fields) |field| {
|
||||
if (field.type == Header or field.type == Status) @compileError(std.fmt.comptimePrint(
|
||||
"protocol '{s}', verb '{s}': the envelope prefix is folded, not stacked — " ++
|
||||
"drop the {s} field '{s}' and use the packet's own Header.operation / Header.target",
|
||||
.{ protocol, verb, @typeName(field.type), field.name },
|
||||
));
|
||||
},
|
||||
else => {},
|
||||
}
|
||||
}
|
||||
|
||||
fn nullDefault(comptime T: type) *const anyopaque {
|
||||
const empty: ?T = null;
|
||||
return @ptrCast(&empty);
|
||||
}
|
||||
|
||||
// --- Define -----------------------------------------------------------------
|
||||
|
||||
/// Build a protocol from its specification. Everything below happens at compile
|
||||
/// time; the generated type is what both sides of the conversation import.
|
||||
///
|
||||
/// ```zig
|
||||
/// pub const Protocol = envelope.Define(.{
|
||||
/// .name = "display",
|
||||
/// .version = 1,
|
||||
/// .operations = &.{
|
||||
/// .{ .name = "configure_layer", .request = ConfigureLayer, .reply = void },
|
||||
/// .{ .name = "blit", .request = Blit, .reply = void },
|
||||
/// },
|
||||
/// .events = &.{
|
||||
/// .{ .name = "layer_lost", .payload = LayerLost },
|
||||
/// },
|
||||
/// });
|
||||
/// ```
|
||||
///
|
||||
/// Refused at compile time, each with the protocol, the verb, and the numbers
|
||||
/// named in the message:
|
||||
///
|
||||
/// - a request or reply that does not fit `packet_maximum` once `prefix_size`
|
||||
/// is added (`.request = extern struct { bytes: [241]u8 }` — 241 + 16 = 257);
|
||||
/// - an event payload that does not fit `post_maximum` the same way
|
||||
/// (`.payload = extern struct { bytes: [49]u8 }` — 49 + 16 = 65);
|
||||
/// - a type that stacks the prefix instead of folding it (a `Header` field);
|
||||
/// - a verb named after a reserved one, or named twice.
|
||||
///
|
||||
/// A variable tail is bounded at *run* time instead, by `encodeRequest` and its
|
||||
/// siblings, because only the caller knows how long it is.
|
||||
pub fn Define(comptime specification: Specification) type {
|
||||
comptime {
|
||||
if (specification.name.len == 0) @compileError("a protocol needs a name");
|
||||
if (specification.name.len > name_maximum) @compileError(std.fmt.comptimePrint(
|
||||
"protocol '{s}': the name is {d} bytes, and a describe reply carries at most {d}",
|
||||
.{ specification.name, specification.name.len, name_maximum },
|
||||
));
|
||||
|
||||
for (specification.operations, 0..) |operation, index| {
|
||||
if (isReservedName(operation.name)) @compileError(std.fmt.comptimePrint(
|
||||
"protocol '{s}': '{s}' is a reserved universal verb — the envelope already answers it",
|
||||
.{ specification.name, operation.name },
|
||||
));
|
||||
for (specification.operations[0..index]) |earlier| {
|
||||
if (std.mem.eql(u8, earlier.name, operation.name)) @compileError(std.fmt.comptimePrint(
|
||||
"protocol '{s}': operation '{s}' is declared twice",
|
||||
.{ specification.name, operation.name },
|
||||
));
|
||||
}
|
||||
rejectStacking(specification.name, operation.name, operation.request);
|
||||
rejectStacking(specification.name, operation.name, operation.reply);
|
||||
if (!fitsPacket(operation.request)) @compileError(std.fmt.comptimePrint(
|
||||
"protocol '{s}', operation '{s}': the request is {d} bytes and the header {d}, " ++
|
||||
"over the {d}-byte call floor — packets never fragment, so this has to shrink " ++
|
||||
"or move its bulk to shared memory",
|
||||
.{ specification.name, operation.name, @sizeOf(operation.request), prefix_size, packet_maximum },
|
||||
));
|
||||
if (!fitsPacket(operation.reply)) @compileError(std.fmt.comptimePrint(
|
||||
"protocol '{s}', operation '{s}': the reply is {d} bytes and the status header {d}, " ++
|
||||
"over the {d}-byte call floor",
|
||||
.{ specification.name, operation.name, @sizeOf(operation.reply), prefix_size, packet_maximum },
|
||||
));
|
||||
}
|
||||
|
||||
for (specification.events, 0..) |event, index| {
|
||||
for (specification.events[0..index]) |earlier| {
|
||||
if (std.mem.eql(u8, earlier.name, event.name)) @compileError(std.fmt.comptimePrint(
|
||||
"protocol '{s}': event '{s}' is declared twice",
|
||||
.{ specification.name, event.name },
|
||||
));
|
||||
}
|
||||
rejectStacking(specification.name, event.name, event.payload);
|
||||
if (!fitsPost(event.payload)) @compileError(std.fmt.comptimePrint(
|
||||
"protocol '{s}', event '{s}': the payload is {d} bytes and the header {d}, " ++
|
||||
"over the {d}-byte push floor — an event carries the header too, so it is the " ++
|
||||
"payload that has to give",
|
||||
.{ specification.name, event.name, @sizeOf(event.payload), prefix_size, post_maximum },
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
return struct {
|
||||
pub const protocol_name: []const u8 = specification.name;
|
||||
pub const version: u32 = specification.version;
|
||||
|
||||
/// What a provider sizes its receive and reply buffers to. A packet's
|
||||
/// fixed part may be far smaller, but any caller may send up to the
|
||||
/// floor and a short buffer truncates rather than refuses.
|
||||
pub const message_maximum: usize = packet_maximum;
|
||||
|
||||
/// The widest packet this protocol's fixed parts can actually produce,
|
||||
/// prefix included — a diagnostic, and what a test pins.
|
||||
pub const request_maximum: usize = widest(specification.operations, .request);
|
||||
pub const reply_maximum: usize = widest(specification.operations, .reply);
|
||||
pub const event_maximum: usize = blk: {
|
||||
var widest_event: usize = prefix_size;
|
||||
for (specification.events) |event| widest_event = @max(widest_event, prefix_size + @sizeOf(event.payload));
|
||||
break :blk widest_event;
|
||||
};
|
||||
|
||||
/// This protocol's verbs, numbered from `first_protocol_operation` in
|
||||
/// declaration order.
|
||||
pub const Operation = numbered(specification.operations, "name");
|
||||
|
||||
/// This protocol's events, numbered from `first_protocol_operation` in
|
||||
/// their **own** space. Events travel only provider → subscriber over
|
||||
/// `ipc_send` and operations only client → provider over `ipc_call`, so
|
||||
/// the direction already tells the two apart; separate spaces mean
|
||||
/// appending an operation can never renumber a shipped event.
|
||||
pub const Event = numbered(specification.events, "name");
|
||||
|
||||
/// The bytes after the `Header` on a request for `operation`.
|
||||
pub fn RequestOf(comptime operation: Operation) type {
|
||||
return specification.operations[indexOf(@intFromEnum(operation))].request;
|
||||
}
|
||||
|
||||
/// The bytes after the `Status` on the reply to `operation`.
|
||||
pub fn ReplyOf(comptime operation: Operation) type {
|
||||
return specification.operations[indexOf(@intFromEnum(operation))].reply;
|
||||
}
|
||||
|
||||
/// The bytes after the `Header` on an `event` packet.
|
||||
pub fn PayloadOf(comptime event: Event) type {
|
||||
return specification.events[indexOf(@intFromEnum(event))].payload;
|
||||
}
|
||||
|
||||
// --- client side ----------------------------------------------------
|
||||
|
||||
/// Frame `[Header][request][tail]`. `tail` is the variable part (a path,
|
||||
/// write bytes); pass `&.{}` when the verb has none. Null if the packet
|
||||
/// would exceed the buffer or the call floor.
|
||||
pub fn encodeRequest(
|
||||
comptime operation: Operation,
|
||||
target: u64,
|
||||
request: RequestOf(operation),
|
||||
tail: []const u8,
|
||||
buffer: []u8,
|
||||
) ?[]u8 {
|
||||
const header = Header{ .operation = @intFromEnum(operation), .target = target };
|
||||
const packet = frame(std.mem.asBytes(&header), bytesOf(RequestOf(operation), &request), tail, buffer) orelse return null;
|
||||
return if (packet.len > packet_maximum) null else packet;
|
||||
}
|
||||
|
||||
/// Frame `[Status][reply][tail]` — the provider's answer, for a provider
|
||||
/// that composes its own reply rather than using `Provider.dispatch`.
|
||||
pub fn encodeReply(
|
||||
comptime operation: Operation,
|
||||
status: i32,
|
||||
reply: ReplyOf(operation),
|
||||
tail: []const u8,
|
||||
buffer: []u8,
|
||||
) ?[]u8 {
|
||||
const fixed = bytesOf(ReplyOf(operation), &reply);
|
||||
const head = Status{ .status = status, .len = @intCast(fixed.len + tail.len) };
|
||||
const packet = frame(std.mem.asBytes(&head), fixed, tail, buffer) orelse return null;
|
||||
return if (packet.len > packet_maximum) null else packet;
|
||||
}
|
||||
|
||||
/// Frame `[Header][payload]` for an asynchronous push. Null if it would
|
||||
/// exceed the buffer or the push floor — an event that does not fit is
|
||||
/// dropped at the source, never split.
|
||||
pub fn encodeEvent(
|
||||
comptime event: Event,
|
||||
target: u64,
|
||||
payload: PayloadOf(event),
|
||||
buffer: []u8,
|
||||
) ?[]u8 {
|
||||
const header = Header{ .operation = @intFromEnum(event), .target = target };
|
||||
const packet = frame(std.mem.asBytes(&header), bytesOf(PayloadOf(event), &payload), &.{}, buffer) orelse return null;
|
||||
return if (packet.len > post_maximum) null else packet;
|
||||
}
|
||||
|
||||
/// Which of this protocol's verbs a packet names — null for a reserved
|
||||
/// verb, or for a number this protocol does not define.
|
||||
pub fn operationOf(packet: []const u8) ?Operation {
|
||||
const header = headerOf(packet) orelse return null;
|
||||
const index = header.operation -% first_protocol_operation;
|
||||
if (header.operation < first_protocol_operation or index >= specification.operations.len) return null;
|
||||
return @enumFromInt(header.operation);
|
||||
}
|
||||
|
||||
/// Which of this protocol's events a pushed packet carries.
|
||||
pub fn eventOf(packet: []const u8) ?Event {
|
||||
const header = headerOf(packet) orelse return null;
|
||||
const index = header.operation -% first_protocol_operation;
|
||||
if (header.operation < first_protocol_operation or index >= specification.events.len) return null;
|
||||
return @enumFromInt(header.operation);
|
||||
}
|
||||
|
||||
/// The fixed request part of a packet already known to name `operation`.
|
||||
pub fn decodeRequest(comptime operation: Operation, packet: []const u8) ?RequestOf(operation) {
|
||||
if (packet.len < prefix_size) return null;
|
||||
return valueOf(RequestOf(operation), packet[prefix_size..]);
|
||||
}
|
||||
|
||||
/// The bytes after the fixed request part — empty when there are none.
|
||||
pub fn requestTail(comptime operation: Operation, packet: []const u8) []const u8 {
|
||||
const start = prefix_size + @sizeOf(RequestOf(operation));
|
||||
return if (packet.len <= start) &.{} else packet[start..];
|
||||
}
|
||||
|
||||
/// The fixed reply part of a reply packet. Null on a short packet; the
|
||||
/// caller checks `statusOf(packet).status` for the provider's verdict.
|
||||
pub fn decodeReply(comptime operation: Operation, packet: []const u8) ?ReplyOf(operation) {
|
||||
if (packet.len < prefix_size) return null;
|
||||
return valueOf(ReplyOf(operation), packet[prefix_size..]);
|
||||
}
|
||||
|
||||
/// The bytes after the fixed reply part, clipped to what `Status.len`
|
||||
/// says actually arrived.
|
||||
pub fn replyTail(comptime operation: Operation, packet: []const u8) []const u8 {
|
||||
const status = statusOf(packet) orelse return &.{};
|
||||
const start = prefix_size + @sizeOf(ReplyOf(operation));
|
||||
const end = @min(packet.len, prefix_size + @as(usize, status.len));
|
||||
return if (end <= start) &.{} else packet[start..end];
|
||||
}
|
||||
|
||||
/// The payload of a pushed packet already known to carry `event`.
|
||||
pub fn decodeEvent(comptime event: Event, packet: []const u8) ?PayloadOf(event) {
|
||||
if (packet.len < prefix_size) return null;
|
||||
return valueOf(PayloadOf(event), packet[prefix_size..]);
|
||||
}
|
||||
|
||||
// --- provider side --------------------------------------------------
|
||||
|
||||
/// This protocol's dispatch table, bound to the provider's own state
|
||||
/// type. `describe` is answered here, from the specification; every verb
|
||||
/// this provider left null answers `-ENOSYS`, which is what makes the
|
||||
/// reserved verbs mean the same thing at every provider in the system.
|
||||
///
|
||||
/// ```zig
|
||||
/// const Serve = Protocol.Provider(*Server);
|
||||
/// const handlers = Serve.Handlers{ .blit = onBlit, .configure_layer = onConfigureLayer };
|
||||
/// const reply_len = Serve.dispatch(server, handlers, message, sender, capability, reply);
|
||||
/// ```
|
||||
///
|
||||
/// A handler returns the number of `answer.tail()` bytes it wrote, or a
|
||||
/// negative errno.
|
||||
pub fn Provider(comptime Context: type) type {
|
||||
return struct {
|
||||
/// A reserved verb a provider chooses to implement itself.
|
||||
/// `enumerate` writes its targets into the tail; `subscribe`
|
||||
/// takes the subscriber's endpoint from `invocation.capability`.
|
||||
pub const ReservedHandler = *const fn (Context, Invocation(void), Answer(void)) isize;
|
||||
|
||||
/// One optional handler per verb, named exactly as the verb,
|
||||
/// plus the reserved verbs the envelope cannot answer alone.
|
||||
pub const Handlers = handlerTable(Context);
|
||||
|
||||
/// Answer one received packet: writes `[Status][reply][tail]`
|
||||
/// into `reply` and returns its length. Zero means the reply
|
||||
/// buffer could not even hold a status, so nothing was written.
|
||||
pub fn dispatch(
|
||||
context: Context,
|
||||
handlers: Handlers,
|
||||
packet: []const u8,
|
||||
sender: u32,
|
||||
capability: ?usize,
|
||||
reply: []u8,
|
||||
) usize {
|
||||
if (reply.len < prefix_size) return 0;
|
||||
const header = headerOf(packet) orelse return refuse(reply, -EPROTO);
|
||||
const body = packet[prefix_size..];
|
||||
|
||||
if (header.operation == operation_describe) return describeInto(reply);
|
||||
|
||||
inline for (specification.operations, 0..) |operation, index| {
|
||||
if (header.operation == first_protocol_operation + index) {
|
||||
return invoke(
|
||||
Context,
|
||||
operation.request,
|
||||
operation.reply,
|
||||
@field(handlers, operation.name),
|
||||
context,
|
||||
header.target,
|
||||
body,
|
||||
sender,
|
||||
capability,
|
||||
reply,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
const reserved: ?ReservedHandler = switch (header.operation) {
|
||||
operation_enumerate => handlers.enumerate,
|
||||
operation_subscribe => handlers.subscribe,
|
||||
operation_unsubscribe => handlers.unsubscribe,
|
||||
else => null,
|
||||
};
|
||||
return invoke(Context, void, void, reserved, context, header.target, body, sender, capability, reply);
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
// Shared by the protocol verbs and the reserved ones: decode, hand the
|
||||
// handler a typed invocation, stamp the status. One place, so a reserved
|
||||
// verb and a protocol verb behave identically.
|
||||
fn invoke(
|
||||
comptime Context: type,
|
||||
comptime RequestType: type,
|
||||
comptime ReplyType: type,
|
||||
handler: ?*const fn (Context, Invocation(RequestType), Answer(ReplyType)) isize,
|
||||
context: Context,
|
||||
target: u64,
|
||||
body: []const u8,
|
||||
sender: u32,
|
||||
capability: ?usize,
|
||||
reply: []u8,
|
||||
) usize {
|
||||
const call = handler orelse return refuse(reply, -ENOSYS);
|
||||
const request = valueOf(RequestType, body) orelse return refuse(reply, -EPROTO);
|
||||
if (reply.len < prefix_size + @sizeOf(ReplyType)) return refuse(reply, -EPROTO);
|
||||
const produced = call(context, .{
|
||||
.target = target,
|
||||
.request = request,
|
||||
.tail = body[@min(@sizeOf(RequestType), body.len)..],
|
||||
.sender = sender,
|
||||
.capability = capability,
|
||||
}, .{ .buffer = reply[prefix_size..] });
|
||||
if (produced < 0) return refuse(reply, @intCast(produced));
|
||||
return succeed(reply, @sizeOf(ReplyType) + @as(usize, @intCast(produced)));
|
||||
}
|
||||
|
||||
fn describeInto(reply: []u8) usize {
|
||||
const description = Description{
|
||||
.version = specification.version,
|
||||
.operation_count = specification.operations.len,
|
||||
.event_count = specification.events.len,
|
||||
.name_len = specification.name.len,
|
||||
};
|
||||
const total = @sizeOf(Description) + specification.name.len;
|
||||
if (reply.len < prefix_size + total) return refuse(reply, -EPROTO);
|
||||
@memcpy(reply[prefix_size..][0..@sizeOf(Description)], std.mem.asBytes(&description));
|
||||
@memcpy(reply[prefix_size + @sizeOf(Description) ..][0..specification.name.len], specification.name);
|
||||
return succeed(reply, total);
|
||||
}
|
||||
|
||||
// "describe" is answered by the envelope, so it is the one reserved verb
|
||||
// with no slot in the table.
|
||||
const implementable_reserved = [_][]const u8{ "enumerate", "subscribe", "unsubscribe" };
|
||||
|
||||
fn handlerTable(comptime Context: type) type {
|
||||
const count = specification.operations.len + implementable_reserved.len;
|
||||
var names: [count][]const u8 = undefined;
|
||||
var types: [count]type = undefined;
|
||||
var attributes: [count]std.builtin.Type.StructField.Attributes = undefined;
|
||||
for (specification.operations, 0..) |operation, index| {
|
||||
const Handler = *const fn (Context, Invocation(operation.request), Answer(operation.reply)) isize;
|
||||
names[index] = operation.name;
|
||||
types[index] = ?Handler;
|
||||
attributes[index] = .{ .default_value_ptr = nullDefault(Handler) };
|
||||
}
|
||||
const Reserved = *const fn (Context, Invocation(void), Answer(void)) isize;
|
||||
for (implementable_reserved, 0..) |name, offset| {
|
||||
const slot = specification.operations.len + offset;
|
||||
names[slot] = name;
|
||||
types[slot] = ?Reserved;
|
||||
attributes[slot] = .{ .default_value_ptr = nullDefault(Reserved) };
|
||||
}
|
||||
const frozen_names = names;
|
||||
const frozen_types = types;
|
||||
const frozen_attributes = attributes;
|
||||
return @Struct(.auto, null, &frozen_names, &frozen_types, &frozen_attributes);
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
// --- the provider's view of one packet --------------------------------------
|
||||
|
||||
/// What a provider's handler is given.
|
||||
pub fn Invocation(comptime RequestType: type) type {
|
||||
return struct {
|
||||
/// Object addressing within this provider — the packet's `Header.target`.
|
||||
target: u64,
|
||||
/// The fixed request part, already decoded.
|
||||
request: RequestType,
|
||||
/// The bytes after it: a path, write data, a name.
|
||||
tail: []const u8,
|
||||
/// The kernel-stamped badge of the caller. The only source identity
|
||||
/// there is — no protocol defines a sender field — so per-client state
|
||||
/// is keyed on this.
|
||||
sender: u32,
|
||||
/// A capability the call carried: a subscriber's endpoint, a DMA
|
||||
/// region. Only the synchronous call path can move one.
|
||||
capability: ?usize,
|
||||
};
|
||||
}
|
||||
|
||||
/// Where a provider's handler writes its answer. The `Status` in front of it is
|
||||
/// the dispatcher's to stamp — a handler never writes its own.
|
||||
pub fn Answer(comptime ReplyType: type) type {
|
||||
return struct {
|
||||
buffer: []u8,
|
||||
|
||||
const fixed_size = @sizeOf(ReplyType);
|
||||
|
||||
/// Write the fixed reply part. A `void` reply writes nothing.
|
||||
pub fn set(self: @This(), reply: ReplyType) void {
|
||||
if (fixed_size == 0) return;
|
||||
@memcpy(self.buffer[0..fixed_size], bytesOf(ReplyType, &reply));
|
||||
}
|
||||
|
||||
/// Room for the variable tail; the handler returns how much of it it used.
|
||||
pub fn tail(self: @This()) []u8 {
|
||||
return self.buffer[fixed_size..];
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
fn refuse(reply: []u8, status: i32) usize {
|
||||
const head = Status{ .status = status, .len = 0 };
|
||||
@memcpy(reply[0..prefix_size], std.mem.asBytes(&head));
|
||||
return prefix_size;
|
||||
}
|
||||
|
||||
fn succeed(reply: []u8, len: usize) usize {
|
||||
const head = Status{ .status = 0, .len = @intCast(len) };
|
||||
@memcpy(reply[0..prefix_size], std.mem.asBytes(&head));
|
||||
return prefix_size + len;
|
||||
}
|
||||
|
||||
/// The widest `[prefix][fixed]` over a set of operations, on one side.
|
||||
fn widest(comptime operations: []const OperationSpecification, comptime side: enum { request, reply }) usize {
|
||||
var found: usize = prefix_size;
|
||||
for (operations) |operation| {
|
||||
const size = switch (side) {
|
||||
.request => @sizeOf(operation.request),
|
||||
.reply => @sizeOf(operation.reply),
|
||||
};
|
||||
found = @max(found, prefix_size + size);
|
||||
}
|
||||
return found;
|
||||
}
|
||||
|
||||
/// An enum over `specifications`, tagged by their `name` field, numbered from
|
||||
/// `first_protocol_operation` in declaration order.
|
||||
fn numbered(comptime specifications: anytype, comptime field: []const u8) type {
|
||||
var names: [specifications.len][]const u8 = undefined;
|
||||
var values: [specifications.len]u32 = undefined;
|
||||
for (specifications, 0..) |specification, index| {
|
||||
names[index] = @field(specification, field);
|
||||
values[index] = first_protocol_operation + index;
|
||||
}
|
||||
const frozen_names = names;
|
||||
const frozen_values = values;
|
||||
return @Enum(u32, .exhaustive, &frozen_names, &frozen_values);
|
||||
}
|
||||
|
||||
/// A verb's position in its declaration list, from its wire number.
|
||||
fn indexOf(comptime operation: u32) usize {
|
||||
return operation - first_protocol_operation;
|
||||
}
|
||||
|
||||
// --- tests ------------------------------------------------------------------
|
||||
|
||||
const testing = std.testing;
|
||||
|
||||
const Produce = extern struct { count: u32, flags: u32 = 0 };
|
||||
const Produced = extern struct { total: u64 };
|
||||
const Changed = extern struct { kind: u32, value: u32 };
|
||||
|
||||
const Sample = Define(.{
|
||||
.name = "sample",
|
||||
.version = 3,
|
||||
.operations = &.{
|
||||
.{ .name = "produce", .request = Produce, .reply = Produced },
|
||||
.{ .name = "reset" },
|
||||
},
|
||||
.events = &.{
|
||||
.{ .name = "changed", .payload = Changed },
|
||||
},
|
||||
});
|
||||
|
||||
test "the prefix is 16 bytes in both directions" {
|
||||
try testing.expectEqual(@as(usize, 16), @sizeOf(Header));
|
||||
try testing.expectEqual(@as(usize, 16), @sizeOf(Status));
|
||||
try testing.expectEqual(@as(usize, 16), prefix_size);
|
||||
try testing.expectEqual(@as(usize, 256), packet_maximum);
|
||||
try testing.expectEqual(@as(usize, 64), post_maximum);
|
||||
}
|
||||
|
||||
test "verb numbering skips the reserved range" {
|
||||
try testing.expectEqual(@as(u32, 16), first_protocol_operation);
|
||||
try testing.expectEqual(@as(u32, 16), @intFromEnum(Sample.Operation.produce));
|
||||
try testing.expectEqual(@as(u32, 17), @intFromEnum(Sample.Operation.reset));
|
||||
// Events are numbered in their own space, so appending an operation can
|
||||
// never renumber a shipped event.
|
||||
try testing.expectEqual(@as(u32, 16), @intFromEnum(Sample.Event.changed));
|
||||
for ([_]u32{ operation_describe, operation_enumerate, operation_subscribe, operation_unsubscribe }) |reserved| {
|
||||
try testing.expect(reserved < first_protocol_operation);
|
||||
}
|
||||
}
|
||||
|
||||
test "request round trip, header folded and tail carried" {
|
||||
var buffer: [packet_maximum]u8 = undefined;
|
||||
const packet = Sample.encodeRequest(.produce, 42, .{ .count = 7 }, "tail bytes", &buffer).?;
|
||||
try testing.expectEqual(prefix_size + @sizeOf(Produce) + "tail bytes".len, packet.len);
|
||||
|
||||
const header = headerOf(packet).?;
|
||||
try testing.expectEqual(@as(u32, 16), header.operation);
|
||||
try testing.expectEqual(@as(u64, 42), header.target);
|
||||
try testing.expectEqual(Sample.Operation.produce, Sample.operationOf(packet).?);
|
||||
|
||||
const request = Sample.decodeRequest(.produce, packet).?;
|
||||
try testing.expectEqual(@as(u32, 7), request.count);
|
||||
try testing.expectEqualStrings("tail bytes", Sample.requestTail(.produce, packet));
|
||||
}
|
||||
|
||||
test "reply round trip" {
|
||||
var buffer: [packet_maximum]u8 = undefined;
|
||||
const packet = Sample.encodeReply(.produce, 0, .{ .total = 99 }, "more", &buffer).?;
|
||||
const status = statusOf(packet).?;
|
||||
try testing.expectEqual(@as(i32, 0), status.status);
|
||||
try testing.expectEqual(@as(u32, @sizeOf(Produced) + "more".len), status.len);
|
||||
try testing.expectEqual(@as(u64, 99), Sample.decodeReply(.produce, packet).?.total);
|
||||
try testing.expectEqualStrings("more", Sample.replyTail(.produce, packet));
|
||||
}
|
||||
|
||||
test "a void request and reply carry nothing but the verb" {
|
||||
var buffer: [packet_maximum]u8 = undefined;
|
||||
const packet = Sample.encodeRequest(.reset, 0, {}, &.{}, &buffer).?;
|
||||
try testing.expectEqual(prefix_size, packet.len);
|
||||
try testing.expectEqual(Sample.Operation.reset, Sample.operationOf(packet).?);
|
||||
try testing.expectEqual(@as(usize, 0), Sample.requestTail(.reset, packet).len);
|
||||
}
|
||||
|
||||
test "event round trip within the push floor" {
|
||||
var buffer: [post_maximum]u8 = undefined;
|
||||
const packet = Sample.encodeEvent(.changed, 0, .{ .kind = 1, .value = 2 }, &buffer).?;
|
||||
try testing.expectEqual(prefix_size + @sizeOf(Changed), packet.len);
|
||||
try testing.expect(packet.len <= post_maximum);
|
||||
try testing.expectEqual(Sample.Event.changed, Sample.eventOf(packet).?);
|
||||
try testing.expectEqual(@as(u32, 2), Sample.decodeEvent(.changed, packet).?.value);
|
||||
}
|
||||
|
||||
// A provider over a trivial context, to drive the generated dispatch table.
|
||||
const Counter = struct {
|
||||
total: u64 = 0,
|
||||
|
||||
fn onProduce(self: *Counter, invocation: Invocation(Produce), answer: Answer(Produced)) isize {
|
||||
self.total += invocation.request.count;
|
||||
answer.set(.{ .total = self.total });
|
||||
const note = "counted";
|
||||
@memcpy(answer.tail()[0..note.len], note);
|
||||
return note.len;
|
||||
}
|
||||
};
|
||||
|
||||
const CounterProvider = Sample.Provider(*Counter);
|
||||
|
||||
test "dispatch reaches a handler and stamps the status" {
|
||||
var counter = Counter{};
|
||||
const handlers = CounterProvider.Handlers{ .produce = Counter.onProduce };
|
||||
|
||||
var request: [packet_maximum]u8 = undefined;
|
||||
const packet = Sample.encodeRequest(.produce, 0, .{ .count = 5 }, &.{}, &request).?;
|
||||
var reply: [packet_maximum]u8 = undefined;
|
||||
const len = CounterProvider.dispatch(&counter, handlers, packet, 3, null, &reply);
|
||||
|
||||
const answered = reply[0..len];
|
||||
try testing.expectEqual(@as(i32, 0), statusOf(answered).?.status);
|
||||
try testing.expectEqual(@as(u64, 5), Sample.decodeReply(.produce, answered).?.total);
|
||||
try testing.expectEqualStrings("counted", Sample.replyTail(.produce, answered));
|
||||
}
|
||||
|
||||
test "describe is answered by the envelope, not the provider" {
|
||||
var counter = Counter{};
|
||||
const handlers = CounterProvider.Handlers{ .produce = Counter.onProduce };
|
||||
|
||||
var request: [packet_maximum]u8 = undefined;
|
||||
const packet = encodeDescribe(&request).?;
|
||||
var reply: [packet_maximum]u8 = undefined;
|
||||
const len = CounterProvider.dispatch(&counter, handlers, packet, 3, null, &reply);
|
||||
|
||||
const described = decodeDescribe(reply[0..len]).?;
|
||||
try testing.expectEqualStrings("sample", described.name);
|
||||
try testing.expectEqual(@as(u32, 3), described.description.version);
|
||||
try testing.expectEqual(@as(u32, 2), described.description.operation_count);
|
||||
try testing.expectEqual(@as(u32, 1), described.description.event_count);
|
||||
}
|
||||
|
||||
test "an unimplemented or unknown verb answers -ENOSYS" {
|
||||
var counter = Counter{};
|
||||
const handlers = CounterProvider.Handlers{ .produce = Counter.onProduce };
|
||||
var reply: [packet_maximum]u8 = undefined;
|
||||
|
||||
// A verb this protocol declares but this provider left null.
|
||||
var request: [packet_maximum]u8 = undefined;
|
||||
const declared = Sample.encodeRequest(.reset, 0, {}, &.{}, &request).?;
|
||||
var len = CounterProvider.dispatch(&counter, handlers, declared, 3, null, &reply);
|
||||
try testing.expectEqual(@as(i32, -ENOSYS), statusOf(reply[0..len]).?.status);
|
||||
|
||||
// A number no verb of this protocol wears.
|
||||
const stranger = Header{ .operation = first_protocol_operation + 900 };
|
||||
len = CounterProvider.dispatch(&counter, handlers, std.mem.asBytes(&stranger), 3, null, &reply);
|
||||
try testing.expectEqual(@as(i32, -ENOSYS), statusOf(reply[0..len]).?.status);
|
||||
|
||||
// A reserved verb the provider does not implement answers the same way.
|
||||
const enumerate = Header{ .operation = operation_enumerate };
|
||||
len = CounterProvider.dispatch(&counter, handlers, std.mem.asBytes(&enumerate), 3, null, &reply);
|
||||
try testing.expectEqual(@as(i32, -ENOSYS), statusOf(reply[0..len]).?.status);
|
||||
}
|
||||
|
||||
test "a truncated packet answers -EPROTO" {
|
||||
var counter = Counter{};
|
||||
const handlers = CounterProvider.Handlers{ .produce = Counter.onProduce };
|
||||
var reply: [packet_maximum]u8 = undefined;
|
||||
|
||||
// Names `produce`, but stops before the request it promises.
|
||||
const header = Header{ .operation = @intFromEnum(Sample.Operation.produce) };
|
||||
const len = CounterProvider.dispatch(&counter, handlers, std.mem.asBytes(&header), 3, null, &reply);
|
||||
try testing.expectEqual(@as(i32, -EPROTO), statusOf(reply[0..len]).?.status);
|
||||
}
|
||||
|
||||
// The size rule, exercised directly. `Define` turns exactly these predicates
|
||||
// into compile errors, which a test cannot catch — so the predicate is what the
|
||||
// test pins, and the boundary protocol below proves the compile-time half from
|
||||
// the other side. The negative example, spelled out: giving `Define` an
|
||||
// operation with `.request = extern struct { bytes: [241]u8 }`, or an event with
|
||||
// `.payload = extern struct { bytes: [49]u8 }`, fails to compile with the
|
||||
// protocol, the verb, and the two numbers named in the message.
|
||||
test "a subscribe carries its interest mask in the reserved verb's tail" {
|
||||
var buffer: [packet_maximum]u8 = undefined;
|
||||
const packet = encodeSubscribe(0b101, &buffer).?;
|
||||
try testing.expectEqual(operation_subscribe, headerOf(packet).?.operation);
|
||||
try testing.expectEqual(@as(u32, 0b101), decodeSubscribe(packet[prefix_size..]).interest);
|
||||
// No body at all — and a body too short to be one — read as "every event",
|
||||
// which is what a subscriber that named nothing wants.
|
||||
try testing.expectEqual(@as(u32, 0), decodeSubscribe(&.{}).interest);
|
||||
try testing.expectEqual(@as(u32, 0), decodeSubscribe(&.{ 1, 2 }).interest);
|
||||
|
||||
const bare = encodeUnsubscribe(&buffer).?;
|
||||
try testing.expectEqual(operation_unsubscribe, headerOf(bare).?.operation);
|
||||
try testing.expectEqual(prefix_size, bare.len);
|
||||
}
|
||||
|
||||
test "the floor counts the header once, and the boundary is exact" {
|
||||
try testing.expect(fitsPacket(extern struct { bytes: [240]u8 }));
|
||||
try testing.expect(!fitsPacket(extern struct { bytes: [241]u8 }));
|
||||
try testing.expect(fitsPost(extern struct { bytes: [48]u8 }));
|
||||
try testing.expect(!fitsPost(extern struct { bytes: [49]u8 }));
|
||||
try testing.expect(fitsPacket(void));
|
||||
try testing.expect(fitsPost(void));
|
||||
}
|
||||
|
||||
const WidestRequest = extern struct { bytes: [packet_maximum - prefix_size]u8 };
|
||||
const WidestEvent = extern struct { bytes: [post_maximum - prefix_size]u8 };
|
||||
|
||||
// A protocol sitting exactly on both floors. That this compiles at all is the
|
||||
// positive half of the compile-time check.
|
||||
const Boundary = Define(.{
|
||||
.name = "boundary",
|
||||
.version = 1,
|
||||
.operations = &.{.{ .name = "fill", .request = WidestRequest, .reply = WidestRequest }},
|
||||
.events = &.{.{ .name = "filled", .payload = WidestEvent }},
|
||||
});
|
||||
|
||||
test "a protocol may sit exactly on the floor" {
|
||||
try testing.expectEqual(packet_maximum, Boundary.request_maximum);
|
||||
try testing.expectEqual(packet_maximum, Boundary.reply_maximum);
|
||||
try testing.expectEqual(post_maximum, Boundary.event_maximum);
|
||||
|
||||
var buffer: [packet_maximum]u8 = undefined;
|
||||
const packet = Boundary.encodeRequest(.fill, 0, .{ .bytes = @splat(0xAB) }, &.{}, &buffer).?;
|
||||
try testing.expectEqual(packet_maximum, packet.len);
|
||||
try testing.expectEqual(@as(u8, 0xAB), Boundary.decodeRequest(.fill, packet).?.bytes[239]);
|
||||
|
||||
// One byte of tail past the floor is refused at run time, not truncated.
|
||||
try testing.expect(Boundary.encodeRequest(.fill, 0, .{ .bytes = @splat(0) }, "x", &buffer) == null);
|
||||
|
||||
var post: [post_maximum]u8 = undefined;
|
||||
const event = Boundary.encodeEvent(.filled, 0, .{ .bytes = @splat(1) }, &post).?;
|
||||
try testing.expectEqual(post_maximum, event.len);
|
||||
}
|
||||
|
||||
test "a protocol's own sizes are reported prefix-included" {
|
||||
try testing.expectEqual(prefix_size + @sizeOf(Produce), Sample.request_maximum);
|
||||
try testing.expectEqual(prefix_size + @sizeOf(Produced), Sample.reply_maximum);
|
||||
try testing.expectEqual(prefix_size + @sizeOf(Changed), Sample.event_maximum);
|
||||
try testing.expectEqual(packet_maximum, Sample.message_maximum);
|
||||
try testing.expectEqualStrings("sample", Sample.protocol_name);
|
||||
}
|
||||
@@ -4,30 +4,25 @@
|
||||
//! **subscriber** (any program) that subscribes and is then pushed each event.
|
||||
//!
|
||||
//! The service handles several device classes over one endpoint. Each class has its own
|
||||
//! typed event (`KeyEvent`, `MouseEvent`, `JoystickEvent`); a subscriber declares which
|
||||
//! classes it wants with a `device_mask`, and the service routes accordingly.
|
||||
//! typed event (`KeyEvent`, `MouseEvent`, `JoystickEvent`); they all travel in a common
|
||||
//! `InputEvent` envelope tagged with a `DeviceKind`, so the fan-out path is one code path
|
||||
//! and a subscriber can take a mix of devices on a single stream. A subscriber declares
|
||||
//! which classes it wants with a `device_mask`, and the service routes accordingly.
|
||||
//!
|
||||
//! Three shapes ride over the channel, and the envelope names all three
|
||||
//! (docs/os-development/protocol-namespace.md):
|
||||
//! Two message shapes ride over the endpoint, tagged by `Operation`, like the
|
||||
//! [VFS protocol](../vfs/protocol.zig):
|
||||
//!
|
||||
//! - **subscribe** is the *reserved* verb, not one of this protocol's own: its shape — a
|
||||
//! synchronous call whose attached capability is the subscriber's endpoint — is exactly
|
||||
//! what `envelope.operation_subscribe` means everywhere. The interest mask travels as the
|
||||
//! packet's tail (`envelope.Subscription`), because a reserved verb carries no typed
|
||||
//! request; what this protocol supplies is the *meaning* of its bits — the device classes.
|
||||
//! - **publish** is this protocol's one verb: a source sends one `InputEvent` and the
|
||||
//! service answers at once, so publishing never blocks on a slow subscriber.
|
||||
//! - **delivery** is an event push: the service `ipc_send`s each event to every interested
|
||||
//! subscriber — no reply owed, so a dead subscriber can never stall the broadcast. The
|
||||
//! packet is the folded header plus the typed event, and **the device class is the
|
||||
//! header's operation**: one event per class, so a subscriber reads the kind from the
|
||||
//! packet rather than from a tag inside the payload.
|
||||
//! - **subscribe / publish**: a synchronous `ipc_call` carrying a `Request`. `subscribe`
|
||||
//! hands the service the subscriber's own endpoint as a capability (`send_cap`) and a
|
||||
//! `device_mask`; `publish` carries an `InputEvent`. The reply is a `Reply`.
|
||||
//! - **delivery**: the service pushes each `InputEvent` to every interested subscriber with
|
||||
//! the asynchronous `ipc_send` — no reply owed, and a dead subscriber can never stall the
|
||||
//! broadcast. Received in the subscriber's buffer with `Received.isMessage()` set.
|
||||
//!
|
||||
//! `Header.target` is unused (0) in both directions: the service is the only object either
|
||||
//! side addresses.
|
||||
//! This is a danos-native contract, shared by the input service, the `runtime.input`
|
||||
//! client helpers, and every source/subscriber. Everything fits one IPC message.
|
||||
|
||||
const std = @import("std");
|
||||
const envelope = @import("envelope");
|
||||
|
||||
/// The classes of input device the service fans out. Each names a typed event and a bit in
|
||||
/// the subscription mask.
|
||||
@@ -247,17 +242,14 @@ pub const JoystickEvent = extern struct {
|
||||
buttons: u32, // current pressed-button bitmask
|
||||
};
|
||||
|
||||
// --- the tagged union of the three ------------------------------------------
|
||||
// --- the common envelope ----------------------------------------------------
|
||||
|
||||
/// The largest per-device event, so `InputEvent` can hold any of them inline.
|
||||
pub const max_event_size: usize = @max(@sizeOf(KeyEvent), @max(@sizeOf(MouseEvent), @sizeOf(JoystickEvent)));
|
||||
|
||||
/// One event of any class: a `DeviceKind` plus the raw bytes of the matching per-device
|
||||
/// event. This is what a source `publish`es (one verb for all three classes) and what a
|
||||
/// subscriber's helper hands back after decoding a delivery — on the *delivery* wire the
|
||||
/// class is the packet header's operation instead, so this tag never travels there. Decode
|
||||
/// it with `asKeyboard`/`asMouse`/`asJoystick` (each returns null unless `device` matches),
|
||||
/// or build one with the `from*` constructors.
|
||||
/// The tagged envelope broadcast to subscribers: a `DeviceKind` plus the raw bytes of the
|
||||
/// matching per-device event. Decode it with `asKeyboard`/`asMouse`/`asJoystick` (each
|
||||
/// returns null unless `device` matches), or build one with the `from*` constructors.
|
||||
pub const InputEvent = extern struct {
|
||||
device: u32, // a DeviceKind
|
||||
_padding: u32 = 0,
|
||||
@@ -293,88 +285,35 @@ pub const InputEvent = extern struct {
|
||||
}
|
||||
};
|
||||
|
||||
// --- the contract -----------------------------------------------------------
|
||||
// --- request / reply --------------------------------------------------------
|
||||
|
||||
pub const Protocol = envelope.Define(.{
|
||||
.name = "input",
|
||||
.version = 1,
|
||||
.operations = &.{
|
||||
// A source submits one event; the service broadcasts it to whoever wants that class.
|
||||
.{ .name = "publish", .request = InputEvent },
|
||||
},
|
||||
.events = &.{
|
||||
// One per device class: the class is the packet's operation, the typed event its
|
||||
// payload. The push floor is 64 bytes and the header spends 16 of them, so the
|
||||
// widest of these — the 28-byte mouse event — leaves the budget with room to spare.
|
||||
.{ .name = "keyboard", .payload = KeyEvent },
|
||||
.{ .name = "mouse", .payload = MouseEvent },
|
||||
.{ .name = "joystick", .payload = JoystickEvent },
|
||||
},
|
||||
});
|
||||
/// Which side of a request this is.
|
||||
pub const Operation = enum(u32) {
|
||||
subscribe = 0, // register the caller's endpoint (send_cap) for the classes in device_mask
|
||||
publish = 1, // a source submits `event` to broadcast to interested subscribers
|
||||
};
|
||||
|
||||
pub const Operation = Protocol.Operation;
|
||||
pub const Event = Protocol.Event;
|
||||
pub const message_maximum: usize = Protocol.message_maximum;
|
||||
/// Request header. For `subscribe`, `device_mask` is the OR of `device_*` bits the caller
|
||||
/// wants (0 means all) and the caller's receive endpoint travels as the call's capability;
|
||||
/// `event` is ignored. For `publish`, `event` is the event to broadcast.
|
||||
pub const Request = extern struct {
|
||||
operation: u32, // an Operation
|
||||
device_mask: u32 = 0, // subscribe: interested device classes (0 => all)
|
||||
event: InputEvent = .{ .device = 0 },
|
||||
};
|
||||
|
||||
/// Reply header. `status` is 0 on success or a negative errno.
|
||||
pub const Reply = extern struct {
|
||||
status: i32,
|
||||
_padding: u32 = 0,
|
||||
};
|
||||
|
||||
pub const request_size: usize = @sizeOf(Request);
|
||||
pub const reply_size: usize = @sizeOf(Reply);
|
||||
pub const event_size: usize = @sizeOf(InputEvent);
|
||||
|
||||
/// The event class a `DeviceKind` value (as it appears in `InputEvent.device`) is delivered
|
||||
/// as. Null for a value no class claims, which is delivered to nobody.
|
||||
pub fn eventOfDevice(device: u32) ?Event {
|
||||
return switch (device) {
|
||||
@intFromEnum(DeviceKind.keyboard) => .keyboard,
|
||||
@intFromEnum(DeviceKind.mouse) => .mouse,
|
||||
@intFromEnum(DeviceKind.joystick) => .joystick,
|
||||
else => null,
|
||||
};
|
||||
}
|
||||
|
||||
/// Frame a `subscribe` request: the reserved verb's header, then the interest mask. Null if
|
||||
/// the buffer is too small. The mask itself is the envelope's `Subscription` — the interest
|
||||
/// a reserved subscribe carries is universal, and the *meaning* of its bits (here: the
|
||||
/// device classes above) is what each protocol supplies. Kept as a named helper because
|
||||
/// `device_mask` is what an input caller calls it.
|
||||
pub fn encodeSubscribe(device_mask: u32, buffer: []u8) ?[]u8 {
|
||||
return envelope.encodeSubscribe(device_mask, buffer);
|
||||
}
|
||||
|
||||
test "an event of every class fits the push floor, header included" {
|
||||
// What the hand-rolled comptime assert used to say about `InputEvent`, now
|
||||
// said by `Define` about each typed event — and counting the header, which
|
||||
// the old check did not.
|
||||
try std.testing.expectEqual(envelope.prefix_size + @sizeOf(MouseEvent), Protocol.event_maximum);
|
||||
try std.testing.expect(Protocol.event_maximum <= envelope.post_maximum);
|
||||
}
|
||||
|
||||
test "the verb numbering, and the class an event carries" {
|
||||
try std.testing.expectEqual(@as(u32, 16), @intFromEnum(Operation.publish));
|
||||
// Events number in their own space, so the three classes start at 16 too.
|
||||
try std.testing.expectEqual(@as(u32, 16), @intFromEnum(Event.keyboard));
|
||||
try std.testing.expectEqual(@as(u32, 17), @intFromEnum(Event.mouse));
|
||||
try std.testing.expectEqual(@as(u32, 18), @intFromEnum(Event.joystick));
|
||||
// subscribe is the RESERVED verb, below the protocol range entirely.
|
||||
try std.testing.expectEqual(@as(u32, 2), envelope.operation_subscribe);
|
||||
|
||||
var buffer: [envelope.post_maximum]u8 = undefined;
|
||||
const packet = Protocol.encodeEvent(.mouse, 0, .{
|
||||
.kind = @intFromEnum(MouseEventKind.motion),
|
||||
.button = 0,
|
||||
.dx = 3,
|
||||
.dy = -4,
|
||||
.scroll_x = 0,
|
||||
.scroll_y = 0,
|
||||
.buttons = 0,
|
||||
}, &buffer).?;
|
||||
try std.testing.expectEqual(Event.mouse, Protocol.eventOf(packet).?);
|
||||
try std.testing.expectEqual(@as(i32, -4), Protocol.decodeEvent(.mouse, packet).?.dy);
|
||||
}
|
||||
|
||||
test "a subscribe carries its mask in the tail of the reserved verb" {
|
||||
var buffer: [envelope.packet_maximum]u8 = undefined;
|
||||
const packet = encodeSubscribe(device_mouse, &buffer).?;
|
||||
try std.testing.expectEqual(envelope.operation_subscribe, envelope.headerOf(packet).?.operation);
|
||||
// The provider side of this is the service harness's, which reads the same
|
||||
// interest mask out of the tail for every protocol.
|
||||
try std.testing.expectEqual(device_mouse, envelope.decodeSubscribe(packet[envelope.prefix_size..]).interest);
|
||||
// A caller that sent nothing at all reads as the every-class mask.
|
||||
try std.testing.expectEqual(@as(u32, 0), envelope.decodeSubscribe(&.{}).interest);
|
||||
comptime {
|
||||
// The delivery path posts a bare InputEvent through ipc_send, so it must fit an
|
||||
// endpoint's async payload slot (POST_MAXIMUM is 64).
|
||||
if (event_size > 64) @compileError("InputEvent must fit the ipc_send payload (POST_MAXIMUM)");
|
||||
}
|
||||
|
||||
@@ -1,104 +1,68 @@
|
||||
//! The power protocol (docs/os-development/power.md): system power's
|
||||
//! domain-named surface, bound at `/protocol/power`. On x86 the acpi service
|
||||
//! provides it; on ARM a PSCI/mailbox service will bind the same name —
|
||||
//! subscribers never learn which firmware they are on (docs/discovery.md —
|
||||
//! firmware neutrality), which is the whole point of naming the contract rather
|
||||
//! than the provider (docs/os-development/protocol-namespace.md).
|
||||
//!
|
||||
//! Defined through the envelope, so every packet begins with the folded
|
||||
//! `Header`. Three shapes ride the channel, and the envelope names all three:
|
||||
//!
|
||||
//! - **subscribe** is the *reserved* verb, not one of this protocol's own: a
|
||||
//! synchronous call whose attached capability is the subscriber's endpoint is
|
||||
//! exactly what `envelope.operation_subscribe` means everywhere.
|
||||
//! - **shutdown** is this protocol's one verb — the only operation that *does*
|
||||
//! something irreversible, and the reason the provider gates it by badge.
|
||||
//! - **the events** are pushes: the service `ipc_send`s each one to every
|
||||
//! subscriber, no reply owed, so a slow or dead subscriber can never wedge the
|
||||
//! source. **The kind is the packet's operation** — one declared event per
|
||||
//! named kind, exactly as the input protocol delivers one per device class —
|
||||
//! so a subscriber reads *what happened* out of the header instead of a tag
|
||||
//! inside the payload. That is what the old `EventMessage`'s two leading bytes
|
||||
//! (an operation byte saying "this is an event", then the kind) fold into.
|
||||
//!
|
||||
//! `Header.target` is unused (0) in both directions: the provider is the only
|
||||
//! object either side addresses. And no packet carries a version any more — the
|
||||
//! reserved `describe` verb is the version handshake, asked once at connect time
|
||||
//! rather than re-carried out of every packet's budget.
|
||||
//! The power protocol (docs/power.md): system power's domain-named surface,
|
||||
//! registered under `ServiceId.power`. On x86 the acpi service serves it; on
|
||||
//! ARM a PSCI/mailbox service will register the same id — subscribers never
|
||||
//! learn which firmware they are on (docs/discovery.md — firmware neutrality).
|
||||
//! The vfs-protocol pattern: extern-struct messages, a version, reserved fields.
|
||||
|
||||
const std = @import("std");
|
||||
const envelope = @import("envelope");
|
||||
/// The protocol version a client states nowhere yet — reserved for the day a
|
||||
/// handshake needs it; requests carry it so a mismatch can be refused loudly.
|
||||
pub const version: u16 = 1;
|
||||
|
||||
/// What a published event carries beyond its kind. The kind is the packet's
|
||||
/// operation, so nothing here repeats it; `power_button`, `lid`, `ac` and
|
||||
/// `battery` leave both fields zero and are fully described by the verb alone.
|
||||
pub const Notice = extern struct {
|
||||
/// The device notification code (ACPI `Notify`'s second argument), or 0.
|
||||
pub const Operation = enum(u8) {
|
||||
/// Subscribe to power events: the subscriber's endpoint rides as the
|
||||
/// call's capability (the input/device-manager pattern); events arrive on
|
||||
/// it as buffered messages carrying an `EventMessage`.
|
||||
subscribe = 1,
|
||||
/// Orderly shutdown's last step: enter S5. Accepted only from PID 1
|
||||
/// (init) — the process that has already run the stop sequence over
|
||||
/// everything else.
|
||||
shutdown = 2,
|
||||
/// The published event payload (never sent *to* the service).
|
||||
event = 3,
|
||||
};
|
||||
|
||||
/// What happened. The vocabulary is hardware-neutral: a lid is a lid whether
|
||||
/// ACPI or a PSCI mailbox reported it.
|
||||
pub const Event = enum(u8) {
|
||||
power_button = 1,
|
||||
lid = 2,
|
||||
ac = 3,
|
||||
battery = 4,
|
||||
/// A device notification that maps to none of the named events — the
|
||||
/// `code` and `hid` fields say which device and what code.
|
||||
notify = 5,
|
||||
};
|
||||
|
||||
pub const Subscribe = extern struct {
|
||||
operation: u8 = @intFromEnum(Operation.subscribe),
|
||||
reserved0: u8 = 0,
|
||||
version: u16 = version,
|
||||
reserved1: u32 = 0,
|
||||
};
|
||||
|
||||
pub const Shutdown = extern struct {
|
||||
operation: u8 = @intFromEnum(Operation.shutdown),
|
||||
reserved0: u8 = 0,
|
||||
version: u16 = version,
|
||||
reserved1: u32 = 0,
|
||||
};
|
||||
|
||||
/// A published event, as the buffered-message payload subscribers receive.
|
||||
pub const EventMessage = extern struct {
|
||||
operation: u8 = @intFromEnum(Operation.event),
|
||||
/// An Event value.
|
||||
event: u8,
|
||||
reserved0: u16 = 0,
|
||||
/// The device notification code (Notify's second argument), or 0.
|
||||
code: u32 = 0,
|
||||
/// The notifying device's hardware id (EISA-decoded), or all zero.
|
||||
hid: [8]u8 = .{0} ** 8,
|
||||
};
|
||||
|
||||
pub const Protocol = envelope.Define(.{
|
||||
.name = "power",
|
||||
.version = 1,
|
||||
.operations = &.{
|
||||
// Orderly shutdown's last step: enter S5. Honored only from a
|
||||
// subscriber — init, the process that has already run the stop sequence
|
||||
// over everything else (docs/os-development/power.md, "authority, not
|
||||
// information"). Nothing to say and nothing to answer, so the verb and
|
||||
// the reply's `Status` are the whole exchange.
|
||||
.{ .name = "shutdown" },
|
||||
},
|
||||
.events = &.{
|
||||
// The vocabulary is hardware-neutral: a lid is a lid whether ACPI or a
|
||||
// PSCI mailbox reported it. One event per kind, each carrying the same
|
||||
// `Notice`, because what differs between them is which thing happened —
|
||||
// and that is the header's job now.
|
||||
.{ .name = "power_button", .payload = Notice },
|
||||
.{ .name = "lid", .payload = Notice },
|
||||
.{ .name = "ac", .payload = Notice },
|
||||
.{ .name = "battery", .payload = Notice },
|
||||
// A device notification that maps to none of the named events — the
|
||||
// `code` and `hid` say which device and what happened.
|
||||
.{ .name = "notify", .payload = Notice },
|
||||
},
|
||||
});
|
||||
pub const Reply = extern struct {
|
||||
status: i32,
|
||||
reserved: u32 = 0,
|
||||
};
|
||||
|
||||
pub const Operation = Protocol.Operation;
|
||||
|
||||
/// What happened. The event *is* the kind: this is the generated event
|
||||
/// enumeration, re-exported under the name this protocol has always called its
|
||||
/// vocabulary, with the same members it has always had.
|
||||
pub const Event = Protocol.Event;
|
||||
|
||||
/// What a provider and a subscriber size their buffers to. This module used to
|
||||
/// declare 64 — the *push* floor — which was simply wrong for a protocol whose
|
||||
/// requests ride `ipc_call`: a provider sizing its receive buffer to 64 refuses
|
||||
/// any caller that sends up to the floor it is entitled to.
|
||||
pub const message_maximum: usize = Protocol.message_maximum;
|
||||
|
||||
test "the kind is the verb, and an event fits the push floor" {
|
||||
try std.testing.expectEqual(@as(u32, 16), @intFromEnum(Operation.shutdown));
|
||||
try std.testing.expectEqual(@as(u32, 16), @intFromEnum(Event.power_button));
|
||||
try std.testing.expectEqual(@as(u32, 17), @intFromEnum(Event.lid));
|
||||
try std.testing.expectEqual(@as(u32, 18), @intFromEnum(Event.ac));
|
||||
try std.testing.expectEqual(@as(u32, 19), @intFromEnum(Event.battery));
|
||||
try std.testing.expectEqual(@as(u32, 20), @intFromEnum(Event.notify));
|
||||
// subscribe is the RESERVED verb, below the protocol range entirely.
|
||||
try std.testing.expectEqual(@as(u32, 2), envelope.operation_subscribe);
|
||||
try std.testing.expectEqual(envelope.prefix_size + @sizeOf(Notice), Protocol.event_maximum);
|
||||
try std.testing.expect(Protocol.event_maximum <= envelope.post_maximum);
|
||||
// The call floor, not the push floor: `shutdown` is a synchronous call.
|
||||
try std.testing.expectEqual(envelope.packet_maximum, message_maximum);
|
||||
}
|
||||
|
||||
test "a pushed event names its kind in the header" {
|
||||
var buffer: [envelope.post_maximum]u8 = undefined;
|
||||
const packet = Protocol.encodeEvent(.power_button, 0, .{}, &buffer).?;
|
||||
try std.testing.expectEqual(Event.power_button, Protocol.eventOf(packet).?);
|
||||
|
||||
const notified = Protocol.encodeEvent(.notify, 0, .{ .code = 0x80, .hid = "PNP0C0A\x00".* }, &buffer).?;
|
||||
try std.testing.expectEqual(Event.notify, Protocol.eventOf(notified).?);
|
||||
try std.testing.expectEqual(@as(u32, 0x80), Protocol.decodeEvent(.notify, notified).?.code);
|
||||
}
|
||||
/// Upper bound on any message in this protocol — sizes endpoint buffers.
|
||||
pub const message_maximum = 64;
|
||||
|
||||
@@ -1,55 +1,49 @@
|
||||
//! The scanout wire protocol — what the compositor says to a native scanout driver (e.g.
|
||||
//! virtio-gpu) over `/protocol/scanout` to put a composited frame on screen. The driver owns
|
||||
//! the panel and the shared scanout surface it handed the compositor (via the display
|
||||
//! service's `attach_scanout`); the compositor composites into that surface, then asks the
|
||||
//! driver to present a damaged rectangle. Tiny by design — one present request. Separate from
|
||||
//! the display protocol because the directions differ: clients call the compositor over
|
||||
//! `/protocol/display`; the compositor calls the driver over `/protocol/scanout`. See
|
||||
//! docs/display-v2.md.
|
||||
//!
|
||||
//! One scanout per driver instance, so `Header.target` is always 0.
|
||||
//! virtio-gpu) over its well-known `.scanout` endpoint to put a composited frame on screen.
|
||||
//! The driver owns the panel and the shared scanout surface it handed the compositor (via the
|
||||
//! display service's `attach_scanout`); the compositor composites into that surface, then asks
|
||||
//! the driver to present a damaged rectangle. Tiny by design — one present request. Separate
|
||||
//! from the display protocol because the directions differ: clients call the compositor over
|
||||
//! `.display`; the compositor calls the driver over `.scanout`. See docs/display-v2.md.
|
||||
|
||||
const envelope = @import("envelope");
|
||||
const std = @import("std");
|
||||
|
||||
/// `present(rect)`: put the given rectangle of the shared scanout surface on the panel (on
|
||||
/// virtio-gpu: transfer-to-host of the region, then a fenced resource flush).
|
||||
pub const Present = extern struct {
|
||||
pub const Operation = enum(u32) {
|
||||
/// present(x, y, width, height): put the given rectangle of the shared scanout surface on
|
||||
/// the panel (on virtio-gpu: transfer-to-host of the region, then a fenced resource flush).
|
||||
present = 0,
|
||||
/// get_modes() -> ModesReply: the display modes this scanout can switch to (V5).
|
||||
get_modes = 1,
|
||||
/// set_mode(width, height): change the scanout resolution — the shared surface is sized to
|
||||
/// the largest mode, so this just re-points the scanout rectangle; the surface is unchanged.
|
||||
set_mode = 2,
|
||||
};
|
||||
|
||||
pub const Request = extern struct {
|
||||
operation: u32,
|
||||
x: u32 = 0,
|
||||
y: u32 = 0,
|
||||
width: u32 = 0,
|
||||
height: u32 = 0,
|
||||
};
|
||||
|
||||
/// `set_mode(width, height)`: change the scanout resolution — the shared surface is sized to
|
||||
/// the largest mode, so this just re-points the scanout rectangle; the surface is unchanged.
|
||||
pub const SetMode = extern struct { width: u32, height: u32 };
|
||||
pub const Reply = extern struct {
|
||||
status: i32, // 0 on success, negative on failure
|
||||
reserved: u32 = 0,
|
||||
};
|
||||
|
||||
/// One offered display mode.
|
||||
pub const Mode = extern struct { width: u32, height: u32 };
|
||||
pub const max_modes = 4;
|
||||
|
||||
/// The answer to `get_modes`: a small fixed list of modes. The success/failure verdict is
|
||||
/// the reply's `Status`, so this carries only the modes.
|
||||
pub const Modes = extern struct {
|
||||
count: u32 = 0,
|
||||
_padding: u32 = 0,
|
||||
modes: [max_modes]Mode = @splat(.{ .width = 0, .height = 0 }),
|
||||
/// The reply to `get_modes`: a small fixed list of modes.
|
||||
pub const ModesReply = extern struct {
|
||||
status: i32,
|
||||
count: u32,
|
||||
modes: [max_modes]Mode,
|
||||
};
|
||||
|
||||
pub const Protocol = envelope.Define(.{
|
||||
.name = "scanout",
|
||||
.version = 1,
|
||||
.operations = &.{
|
||||
.{ .name = "present", .request = Present },
|
||||
.{ .name = "get_modes", .reply = Modes },
|
||||
.{ .name = "set_mode", .request = SetMode },
|
||||
},
|
||||
});
|
||||
|
||||
pub const Operation = Protocol.Operation;
|
||||
|
||||
/// The call floor, like every synchronous protocol. This module used to declare
|
||||
/// 64 — the *push* floor — which was simply wrong: nothing here is pushed, and a
|
||||
/// provider sizing its receive buffer to 64 refuses (`-E2BIG`) any caller that
|
||||
/// sends up to the floor it is entitled to.
|
||||
pub const message_maximum: usize = Protocol.message_maximum;
|
||||
pub const message_maximum: usize = 64;
|
||||
pub const request_size: usize = @sizeOf(Request);
|
||||
pub const reply_size: usize = @sizeOf(Reply);
|
||||
pub const modes_reply_size: usize = @sizeOf(ModesReply);
|
||||
|
||||
@@ -1,66 +1,57 @@
|
||||
//! The USB transfer protocol: what a USB class driver (a keyboard, mouse, or
|
||||
//! mass-storage driver) says to the xHCI bus driver over `/protocol/usb-transfer`
|
||||
//! to drive its device. The class driver owns no hardware — it reaches its device
|
||||
//! entirely through these packets, the way a PS/2 keyboard driver reaches the
|
||||
//! 8042 through the ps2-bus.
|
||||
//!
|
||||
//! Defined through the envelope (docs/os-development/protocol-namespace.md), so
|
||||
//! every packet begins with the folded `Header`. **`Header.target` is the device
|
||||
//! token** — the per-open handle the bus driver hands back, which every request
|
||||
//! but `open` addressed through a `device_token` field of its own before the
|
||||
//! rebase. `open` itself addresses the *assigned device id*, because that is what
|
||||
//! the caller has before there is a token.
|
||||
//! mass-storage driver) says to the xHCI bus driver over its well-known
|
||||
//! `.usb_bus` endpoint to drive its device. The class driver owns no hardware —
|
||||
//! it reaches its device entirely through these messages, the way a PS/2 keyboard
|
||||
//! driver reaches the 8042 through the ps2-bus. Extern-struct messages tagged by
|
||||
//! `Operation`, the vfs-protocol / device-manager-protocol pattern.
|
||||
//!
|
||||
//! The shape:
|
||||
//! - **open** (a capability-passing call): the class driver hands over its own
|
||||
//! endpoint (for asynchronous interrupt reports); the target is its assigned
|
||||
//! device id, and the reply carries a `device_token` plus its interface's
|
||||
//! endpoints.
|
||||
//! - **control / bulk** (synchronous calls): one transfer, answered when it
|
||||
//! completes. Control data travels **in the packet's tail** in both
|
||||
//! directions (descriptors, HID/MSC class requests are all small), so the
|
||||
//! fixed parts stay tiny and `Status.len` is the transferred length — the
|
||||
//! envelope's own field for "how many bytes follow", which is precisely what
|
||||
//! the old `actual_length` said. Bulk data travels by **physical address** —
|
||||
//! the class driver's own `dma_alloc`'d buffer — so a 512-byte sector never
|
||||
//! has to cross the packet floor.
|
||||
//! - **open** (a capability-passing `ipc.callCap`): the class driver hands over
|
||||
//! its own endpoint (for asynchronous interrupt reports) and its assigned
|
||||
//! device id, and receives a `device_token` plus its interface's endpoints.
|
||||
//! - **control / bulk** (synchronous `ipc.call`): one transfer, answered when
|
||||
//! it completes. Control data travels inline (descriptors, HID/MSC class
|
||||
//! requests are all small); bulk data travels by **physical address** — the
|
||||
//! class driver's own `dma_alloc`'d buffer — so a 512-byte sector never has
|
||||
//! to cross the 256-byte IPC boundary.
|
||||
//! - **interrupt_subscribe** (synchronous): arm periodic IN polling of an
|
||||
//! interrupt endpoint; each report the device produces is then pushed to the
|
||||
//! class driver's endpoint as an asynchronous `interrupt_report` event.
|
||||
//! It stays one of **this protocol's own verbs**, not the reserved
|
||||
//! `subscribe`: the reserved verb means "push me this provider's events" and
|
||||
//! carries the subscriber's endpoint, while this names one endpoint address
|
||||
//! on one device and a poll length, and the endpoint it pushes to was handed
|
||||
//! over at `open`. Same word, different contract.
|
||||
//! - **dma_attach**: a class driver hands the controller a DMA-region
|
||||
//! capability (riding the call's cap slot) so the controller binds that
|
||||
//! buffer into its IOMMU domain and may then DMA to the physical addresses
|
||||
//! inside it. Needed once per buffer the class driver will name in a `bulk`
|
||||
//! transfer (its own, or one forwarded to it).
|
||||
//! class driver's endpoint as an asynchronous `InterruptReport` (`ipc.send`),
|
||||
//! exactly how the input service delivers events.
|
||||
//!
|
||||
//! Single controller assumption: one provider serves QEMU's one xHCI. A
|
||||
//! multi-controller machine would need the controller in the target (or the
|
||||
//! spawner wiring each class driver its own channel); noted, not built.
|
||||
//! Single controller assumption: one `.usb_bus` singleton serves QEMU's one xHCI.
|
||||
//! A multi-controller machine would need a per-controller endpoint (the device
|
||||
//! manager handing each class driver the right one); noted, not built.
|
||||
|
||||
const std = @import("std");
|
||||
const envelope = @import("envelope");
|
||||
/// Fits one synchronous IPC message (kernel MESSAGE_MAXIMUM).
|
||||
pub const message_maximum: usize = 256;
|
||||
|
||||
/// The largest control-transfer data stage. It rides the packet's tail, so the
|
||||
/// bound is the call floor less the header and the fixed request part — derived
|
||||
/// rather than declared, which is what keeps it honest when a field moves.
|
||||
pub const max_inline_data: usize = envelope.packet_maximum - envelope.prefix_size - @sizeOf(Control);
|
||||
/// The largest inline control-transfer payload. Sized so a whole message
|
||||
/// (header + data) stays under `message_maximum`: descriptors and HID/MSC class
|
||||
/// requests are all far smaller.
|
||||
pub const max_inline_data: usize = 200;
|
||||
|
||||
/// The largest interrupt report pushed asynchronously. An event packet is the
|
||||
/// header plus the payload within 64 bytes, so this is what is left after the
|
||||
/// report's own four bytes of framing: boot keyboard reports are 8 bytes, boot
|
||||
/// mouse reports 3–4, and the whole HID boot vocabulary fits many times over.
|
||||
/// A device that produces more has its report truncated, never split.
|
||||
pub const max_report_data: usize = 40;
|
||||
/// The largest interrupt report pushed asynchronously. Sized so `InterruptReport`
|
||||
/// fits an `ipc_send` payload slot (POST_MAXIMUM = 64): boot keyboard reports are
|
||||
/// 8 bytes, boot mouse reports 3–4.
|
||||
pub const max_report_data: usize = 48;
|
||||
|
||||
/// Endpoints per interface reported back in an open reply (a boot HID interface
|
||||
/// has one interrupt endpoint, a mass-storage interface two bulk endpoints).
|
||||
pub const max_reported_endpoints: usize = 4;
|
||||
|
||||
pub const Operation = enum(u32) {
|
||||
open = 0,
|
||||
control = 1,
|
||||
interrupt_subscribe = 2,
|
||||
bulk = 3,
|
||||
/// dma_attach: a class driver hands the controller a DMA-region capability (riding
|
||||
/// the call's cap slot) so the controller binds that buffer into its IOMMU domain
|
||||
/// and may then DMA to the physical addresses inside it. Needed once per buffer the
|
||||
/// class driver will name in a `bulk` transfer (its own, or one forwarded to it).
|
||||
dma_attach = 4,
|
||||
};
|
||||
|
||||
/// The endpoint facts a class driver needs, lifted from the endpoint descriptor
|
||||
/// the bus driver already parsed during enumeration.
|
||||
pub const Endpoint = extern struct {
|
||||
@@ -73,146 +64,114 @@ pub const Endpoint = extern struct {
|
||||
reserved: [3]u8 = .{ 0, 0, 0 },
|
||||
};
|
||||
|
||||
// --- the per-operation request and reply parts ------------------------------
|
||||
//
|
||||
// Each names the bytes AFTER the prefix. Nothing here carries an operation or a
|
||||
// device token: those are the packet header's, folded in once. No reply carries
|
||||
// a status either — that is the `Status` every reply begins with.
|
||||
/// open: the class driver's receive endpoint rides as the call's capability, and
|
||||
/// `device_id` is the interface's assigned id (its argv[1]).
|
||||
pub const OpenRequest = extern struct {
|
||||
operation: u32 = @intFromEnum(Operation.open),
|
||||
reserved: u32 = 0,
|
||||
device_id: u64,
|
||||
};
|
||||
|
||||
/// The answer to `open`: the token every later packet puts in `Header.target`,
|
||||
/// the interface's class triple (a sanity check), and its endpoints.
|
||||
pub const Opened = extern struct {
|
||||
device_token: u64,
|
||||
/// The answer to open: a token scoping every later request to this device, the
|
||||
/// interface's class triple (a sanity check), and its endpoints.
|
||||
pub const OpenReply = extern struct {
|
||||
status: i32,
|
||||
endpoint_count: u32,
|
||||
device_token: u64,
|
||||
interface_class: u8,
|
||||
interface_subclass: u8,
|
||||
interface_protocol: u8,
|
||||
interface_number: u8,
|
||||
reserved2: u32 = 0,
|
||||
endpoints: [max_reported_endpoints]Endpoint = [_]Endpoint{.{ .address = 0, .transfer_type = 0, .max_packet_size = 0, .interval = 0 }} ** max_reported_endpoints,
|
||||
};
|
||||
|
||||
/// `control`: one EP0 control transfer on `Header.target`. `setup` is a bit-cast
|
||||
/// `usb_abi.Request`. For an OUT transfer the data stage is the request's tail;
|
||||
/// for an IN transfer it comes back as the reply's tail, and `Status.len` is how
|
||||
/// much of it arrived.
|
||||
pub const Control = extern struct {
|
||||
/// control: one EP0 control transfer. `setup` is a bit-cast `usb_abi.Request`.
|
||||
/// For an OUT transfer `data[0..data_length]` is sent; for an IN transfer the
|
||||
/// reply carries up to `data_length` bytes back.
|
||||
pub const ControlRequest = extern struct {
|
||||
operation: u32 = @intFromEnum(Operation.control),
|
||||
reserved: u32 = 0,
|
||||
device_token: u64,
|
||||
setup: [8]u8,
|
||||
/// 1 = device-to-host (IN), 0 = host-to-device (OUT).
|
||||
direction_in: u8,
|
||||
_padding: u8 = 0,
|
||||
/// Bytes of data stage: what an IN transfer asks for, and what an OUT
|
||||
/// transfer's tail carries.
|
||||
direction_in: u8, // 1 = device-to-host (IN), 0 = host-to-device (OUT)
|
||||
reserved2: u8 = 0,
|
||||
data_length: u16,
|
||||
_padding2: u32 = 0,
|
||||
reserved3: u32 = 0,
|
||||
data: [max_inline_data]u8 = [_]u8{0} ** max_inline_data,
|
||||
};
|
||||
|
||||
/// `interrupt_subscribe`: begin periodic IN polling of an interrupt endpoint of
|
||||
/// `Header.target`. Each report the device returns is pushed to the endpoint the
|
||||
/// caller handed over at `open`, as an `interrupt_report` event.
|
||||
pub const InterruptSubscribe = extern struct {
|
||||
pub const ControlReply = extern struct {
|
||||
status: i32, // 0 success, negative on failure/stall
|
||||
actual_length: u32,
|
||||
data: [max_inline_data]u8 = [_]u8{0} ** max_inline_data,
|
||||
};
|
||||
|
||||
/// interrupt_subscribe: begin periodic IN polling of an interrupt endpoint. Each
|
||||
/// report the device returns is pushed to the caller's endpoint (handed over at
|
||||
/// open) as an asynchronous `InterruptReport`.
|
||||
pub const InterruptSubscribeRequest = extern struct {
|
||||
operation: u32 = @intFromEnum(Operation.interrupt_subscribe),
|
||||
reserved: u32 = 0,
|
||||
device_token: u64,
|
||||
endpoint_address: u8,
|
||||
_padding: u8 = 0,
|
||||
/// Bytes to request per poll (the endpoint's max packet size).
|
||||
max_length: u16,
|
||||
reserved2: u8 = 0,
|
||||
max_length: u16, // bytes to request per poll (the endpoint's max packet size)
|
||||
};
|
||||
|
||||
/// `bulk`: one bulk IN or OUT transfer on `Header.target`. `physical_address` is
|
||||
/// the class driver's own `dma_alloc`'d buffer — the controller DMAs straight
|
||||
/// to/from it, so the bulk data never crosses IPC. `endpoint_address`'s bit 7
|
||||
/// selects IN vs OUT.
|
||||
pub const Bulk = extern struct {
|
||||
pub const InterruptSubscribeReply = extern struct {
|
||||
status: i32,
|
||||
reserved: u32 = 0,
|
||||
};
|
||||
|
||||
/// bulk: one bulk IN or OUT transfer. `physical_address` is the class driver's own
|
||||
/// `dma_alloc`'d buffer — the controller DMAs straight to/from it, so the bulk
|
||||
/// data never crosses IPC. `endpoint_address`'s bit 7 selects IN vs OUT.
|
||||
pub const BulkRequest = extern struct {
|
||||
operation: u32 = @intFromEnum(Operation.bulk),
|
||||
reserved: u32 = 0,
|
||||
device_token: u64,
|
||||
physical_address: u64,
|
||||
length: u32,
|
||||
endpoint_address: u8,
|
||||
_padding: u8 = 0,
|
||||
_padding2: u16 = 0,
|
||||
reserved2: u8 = 0,
|
||||
reserved3: u16 = 0,
|
||||
};
|
||||
|
||||
/// How many bytes a bulk transfer actually moved. It cannot ride `Status.len`
|
||||
/// the way a control transfer's does: nothing follows a bulk reply, because the
|
||||
/// data went to the caller's DMA buffer rather than into the packet.
|
||||
pub const Transferred = extern struct { actual_length: u32 };
|
||||
pub const BulkReply = extern struct {
|
||||
status: i32,
|
||||
actual_length: u32,
|
||||
};
|
||||
|
||||
/// One asynchronous interrupt report, pushed to the endpoint the class driver
|
||||
/// handed over at `open`. The device it came from is `Header.target`.
|
||||
/// dma_attach: the region capability rides the call's cap slot; the body only carries
|
||||
/// the device token (scoping) so the controller knows which caller is attaching.
|
||||
pub const DmaAttachRequest = extern struct {
|
||||
operation: u32 = @intFromEnum(Operation.dma_attach),
|
||||
reserved: u32 = 0,
|
||||
device_token: u64,
|
||||
};
|
||||
|
||||
pub const DmaAttachReply = extern struct {
|
||||
status: i32,
|
||||
reserved: u32 = 0,
|
||||
};
|
||||
|
||||
/// An asynchronous interrupt report, pushed with `ipc.send` to a subscriber's
|
||||
/// endpoint. `Received.isMessage()` is set; there is no reply owed.
|
||||
pub const InterruptReport = extern struct {
|
||||
device_token: u64,
|
||||
endpoint_address: u8,
|
||||
length: u8,
|
||||
_padding: u16 = 0,
|
||||
reserved: u16 = 0,
|
||||
data: [max_report_data]u8 = [_]u8{0} ** max_report_data,
|
||||
};
|
||||
|
||||
pub const Protocol = envelope.Define(.{
|
||||
.name = "usb-transfer",
|
||||
.version = 1,
|
||||
.operations = &.{
|
||||
// open: the target is the interface's assigned device id (its argv[1]),
|
||||
// and the class driver's receive endpoint rides as the capability.
|
||||
.{ .name = "open", .reply = Opened },
|
||||
.{ .name = "control", .request = Control },
|
||||
.{ .name = "interrupt_subscribe", .request = InterruptSubscribe },
|
||||
.{ .name = "bulk", .request = Bulk, .reply = Transferred },
|
||||
// dma_attach: the region capability rides the call's cap slot; the
|
||||
// target says which caller's device is attaching, so there is nothing
|
||||
// left for a body to carry.
|
||||
.{ .name = "dma_attach" },
|
||||
},
|
||||
.events = &.{
|
||||
.{ .name = "interrupt_report", .payload = InterruptReport },
|
||||
},
|
||||
});
|
||||
|
||||
pub const Operation = Protocol.Operation;
|
||||
pub const Event = Protocol.Event;
|
||||
|
||||
/// What both sides size their buffers to — the call floor, as every protocol does.
|
||||
pub const message_maximum: usize = Protocol.message_maximum;
|
||||
|
||||
test "the budgets, re-verified by Define rather than by hand" {
|
||||
// What the hand-rolled comptime asserts used to say, now said by `Define`
|
||||
// — and counting the header, which the old checks did not.
|
||||
try std.testing.expectEqual(@as(usize, 224), max_inline_data);
|
||||
try std.testing.expect(Protocol.request_maximum <= envelope.packet_maximum);
|
||||
try std.testing.expect(Protocol.reply_maximum <= envelope.packet_maximum);
|
||||
// The report was 48 bytes of data in a 64-byte struct that had no room left
|
||||
// for a header. Folding the device token into the target and trimming the
|
||||
// data to 40 leaves the whole packet at 60 of the 64-byte push floor.
|
||||
try std.testing.expectEqual(@as(usize, 44), @sizeOf(InterruptReport));
|
||||
try std.testing.expectEqual(@as(usize, 60), Protocol.event_maximum);
|
||||
try std.testing.expect(Protocol.event_maximum <= envelope.post_maximum);
|
||||
}
|
||||
|
||||
test "the verb numbering, and the device token in the header" {
|
||||
try std.testing.expectEqual(@as(u32, 16), @intFromEnum(Operation.open));
|
||||
try std.testing.expectEqual(@as(u32, 17), @intFromEnum(Operation.control));
|
||||
try std.testing.expectEqual(@as(u32, 18), @intFromEnum(Operation.interrupt_subscribe));
|
||||
try std.testing.expectEqual(@as(u32, 19), @intFromEnum(Operation.bulk));
|
||||
try std.testing.expectEqual(@as(u32, 20), @intFromEnum(Operation.dma_attach));
|
||||
try std.testing.expectEqual(@as(u32, 16), @intFromEnum(Event.interrupt_report));
|
||||
|
||||
var buffer: [message_maximum]u8 = undefined;
|
||||
const packet = Protocol.encodeRequest(.control, 9, .{
|
||||
.setup = .{ 0, 6, 0, 1, 0, 0, 18, 0 },
|
||||
.direction_in = 1,
|
||||
.data_length = 18,
|
||||
}, &.{}, &buffer).?;
|
||||
try std.testing.expectEqual(@as(u64, 9), envelope.headerOf(packet).?.target);
|
||||
try std.testing.expectEqual(@as(u16, 18), Protocol.decodeRequest(.control, packet).?.data_length);
|
||||
}
|
||||
|
||||
test "a control OUT carries its data stage as the packet's tail" {
|
||||
var buffer: [message_maximum]u8 = undefined;
|
||||
const payload = [_]u8{ 1, 2, 3, 4 };
|
||||
const packet = Protocol.encodeRequest(.control, 5, .{
|
||||
.setup = .{ 0x21, 11, 0, 0, 0, 0, 4, 0 },
|
||||
.direction_in = 0,
|
||||
.data_length = payload.len,
|
||||
}, &payload, &buffer).?;
|
||||
try std.testing.expectEqualSlices(u8, &payload, Protocol.requestTail(.control, packet));
|
||||
|
||||
// And the answer to an IN: the bytes follow the (empty) fixed reply part,
|
||||
// with `Status.len` counting exactly them.
|
||||
const answered = Protocol.encodeReply(.control, 0, {}, &payload, &buffer).?;
|
||||
try std.testing.expectEqual(@as(u32, payload.len), envelope.statusOf(answered).?.len);
|
||||
try std.testing.expectEqualSlices(u8, &payload, Protocol.replyTail(.control, answered));
|
||||
comptime {
|
||||
const std = @import("std");
|
||||
// Every synchronous message must fit one IPC message; the async report must
|
||||
// fit an ipc_send payload slot.
|
||||
std.debug.assert(@sizeOf(ControlRequest) <= message_maximum);
|
||||
std.debug.assert(@sizeOf(ControlReply) <= message_maximum);
|
||||
std.debug.assert(@sizeOf(OpenReply) <= message_maximum);
|
||||
std.debug.assert(@sizeOf(InterruptReport) <= 64);
|
||||
}
|
||||
|
||||
@@ -1,30 +1,40 @@
|
||||
//! The VFS wire protocol — what a client (through the file API,
|
||||
//! library/kernel/file-system.zig) says to a filesystem backend over IPC. Defined
|
||||
//! through the envelope (docs/os-development/protocol-namespace.md), so every
|
||||
//! packet begins with the folded `Header`: the verb in `Header.operation`, and
|
||||
//! **the open node id in `Header.target`** — the field that used to be
|
||||
//! `Request.node`. A path appears in the conversation once, at `open`; every
|
||||
//! packet after it addresses that integer.
|
||||
//! The VFS wire protocol — the message format spoken between a client (via the file
|
||||
//! API) and the user-space VFS server over IPC. A request is a fixed `Request` header
|
||||
//! followed by an inline payload (a path, or write bytes); a reply is a fixed `Reply`
|
||||
//! header followed by an inline payload (read bytes, or a FileStatus). Everything fits
|
||||
//! in one IPC message (<= ipc MESSAGE_MAXIMUM = 256 bytes).
|
||||
//!
|
||||
//! This is a danos-native contract, so it uses danos names throughout. It is
|
||||
//! user-space only — the kernel knows nothing of files or paths; it only routes
|
||||
//! (`fs_resolve`) and moves the bytes. The backends that serve it today are the
|
||||
//! FAT server (system/services/fat/) and the protocol registry inside PID 1
|
||||
//! (system/services/init/), which is a *synthetic* backend: `/protocol` holds
|
||||
//! contracts rather than files.
|
||||
//! This is a danos-native contract, so it uses danos names throughout. The client
|
||||
//! side is `runtime.fs` (library/runtime/fs.zig), which programs use directly.
|
||||
//!
|
||||
//! **An `open` reply may carry a capability.** The `open` request rides
|
||||
//! `ipc_call`, and the reply direction of a call can hand back an endpoint
|
||||
//! (`ipc.callCap`'s `Reply.cap`). A file backend never uses it — FAT answers with
|
||||
//! a node id and nothing else — but the registry does: opening a
|
||||
//! `NodeKind.protocol` node under `/protocol` returns the provider's endpoint,
|
||||
//! which is the channel. The convention is per-backend, not per-operation: a
|
||||
//! client that did not ask a synthetic backend simply gets no capability back.
|
||||
//! This is user-space only — the kernel knows nothing of files or paths; it only moves the bytes.
|
||||
//! Shared by library/runtime/fs.zig (the client) and the mount backends that serve it (today
|
||||
//! the fat server, system/services/fat/). The standalone user-space VFS server it was first
|
||||
//! written against has retired — path routing moved into the kernel (system/kernel/vfs.zig,
|
||||
//! fs_resolve) — but the protocol module outlived it.
|
||||
|
||||
const envelope = @import("envelope");
|
||||
pub const Operation = enum(u32) {
|
||||
open, // open(path) -> node id
|
||||
close, // close(node)
|
||||
read, // read(node, offset, len) -> bytes
|
||||
write, // write(node, offset, bytes) -> count
|
||||
status, // status(node) -> FileStatus
|
||||
// Appended for the mount router (M5). Values stay stable, so existing clients
|
||||
// and the flat-ramfs tests are unaffected.
|
||||
readdir, // readdir(dir_node, cursor=offset) -> one DirectoryEntry (len==0 => EOF)
|
||||
mount, // mount(prefix payload, capability = backend endpoint)
|
||||
unmount, // unmount(prefix payload)
|
||||
// Appended for filesystem mutation (Phase 2). Path-based (the path is the
|
||||
// payload); a mounted backend handles them, the flat ramfs refuses them.
|
||||
mkdir, // mkdir(path payload) -> status
|
||||
unlink, // unlink(path payload) -> status
|
||||
// rename: the payload is the old path, a single 0x00 separator, then the new
|
||||
// path. Same-directory rename only (the router requires both under one mount).
|
||||
rename, // rename(old\0new payload) -> status
|
||||
};
|
||||
|
||||
/// The type of a filesystem node, aligned to the node-kind table
|
||||
/// (docs/file-system-development/file-system-hierarchy.md). Fills `FileStatus.kind` and
|
||||
/// The type of a filesystem node, aligned to the FSH file-type table
|
||||
/// (docs/danos-file-system-hierarchy-FSH.md). Fills `FileStatus.kind` and
|
||||
/// `DirectoryEntry.kind`; `regular = 0` keeps the historical hardcoded value.
|
||||
pub const NodeKind = enum(u32) {
|
||||
regular = 0,
|
||||
@@ -34,26 +44,40 @@ pub const NodeKind = enum(u32) {
|
||||
symbolic_link = 4,
|
||||
fifo = 5,
|
||||
socket = 6,
|
||||
/// A node that names a *contract*, not a file: opening it establishes a
|
||||
/// channel to whatever process currently provides that protocol, delivered
|
||||
/// as an endpoint capability in the reply rather than a node id. This is
|
||||
/// what lives under `/protocol`; `readdir` lists these like any other node,
|
||||
/// so the tree stays browsable for diagnosis.
|
||||
protocol = 7,
|
||||
};
|
||||
|
||||
/// One directory entry: the fixed part of a `readdir` reply, followed inline by
|
||||
/// `name_len` bytes of name. **A zero `name_len` is end of directory** — the
|
||||
/// reply's own length cannot say so any more, because the envelope always sends
|
||||
/// the fixed part.
|
||||
/// One directory entry, returned by `readdir`: a fixed header followed inline in
|
||||
/// the reply payload by `name_len` bytes of name. A zero-length reply is EOF.
|
||||
pub const DirectoryEntry = extern struct {
|
||||
kind: u32 = 0, // a NodeKind
|
||||
name_len: u32 = 0,
|
||||
size: u64 = 0,
|
||||
kind: u32, // a NodeKind
|
||||
name_len: u32,
|
||||
size: u64,
|
||||
};
|
||||
|
||||
pub const directory_entry_size: usize = @sizeOf(DirectoryEntry);
|
||||
|
||||
/// Request header. `node` is the server-side open-file id (from a prior open);
|
||||
/// for `open` the path is the payload and `len` is its length. `offset`/`len`
|
||||
/// carry the read/write position and count.
|
||||
pub const Request = extern struct {
|
||||
operation: Operation,
|
||||
node: u64,
|
||||
offset: u64,
|
||||
len: u32,
|
||||
flags: u32,
|
||||
};
|
||||
|
||||
/// Reply header. `status` is 0 on success or a negative errno; `node` is the new
|
||||
/// open-file id (for `open`); `len` is the payload length (bytes read, or the
|
||||
/// FileStatus size).
|
||||
pub const Reply = extern struct {
|
||||
status: i32,
|
||||
_padding: u32 = 0,
|
||||
node: u64 = 0,
|
||||
len: u32 = 0,
|
||||
_padding2: u32 = 0,
|
||||
};
|
||||
|
||||
/// A file's metadata (the danos-native answer to a `status` request). The POSIX
|
||||
/// layer maps this onto `struct stat`.
|
||||
pub const FileStatus = extern struct {
|
||||
@@ -65,84 +89,13 @@ pub const FileStatus = extern struct {
|
||||
mtime: u64 = 0,
|
||||
};
|
||||
|
||||
// --- the per-operation request and reply parts ------------------------------
|
||||
//
|
||||
// Each names the bytes AFTER the prefix. Nothing here carries an operation or a
|
||||
// node id: those are the packet header's, folded in once.
|
||||
pub const message_maximum: usize = 256;
|
||||
pub const request_size: usize = @sizeOf(Request);
|
||||
pub const reply_size: usize = @sizeOf(Reply);
|
||||
/// Largest inline payload that still fits one IPC message alongside a header.
|
||||
pub const maximum_payload: usize = message_maximum - request_size;
|
||||
|
||||
/// `open(flags)` with the path as the packet's tail. The one verb that spends a
|
||||
/// path; everything after it addresses the node id this returns.
|
||||
pub const Open = extern struct { flags: u32 = 0 };
|
||||
|
||||
/// The node id an `open` established — the integer every later packet puts in
|
||||
/// `Header.target`. Meaningful only between this client and this backend.
|
||||
pub const Opened = extern struct { node: u64 };
|
||||
|
||||
/// `read(offset, len)` on `Header.target`; the bytes come back as the reply tail.
|
||||
pub const Read = extern struct {
|
||||
offset: u64,
|
||||
len: u32,
|
||||
_padding: u32 = 0,
|
||||
};
|
||||
|
||||
/// `write(offset, len)` on `Header.target`, with the data as the packet's tail.
|
||||
pub const Write = extern struct {
|
||||
offset: u64,
|
||||
len: u32,
|
||||
_padding: u32 = 0,
|
||||
};
|
||||
|
||||
/// How many bytes a `write` actually took — it may be short.
|
||||
pub const Written = extern struct { count: u32 };
|
||||
|
||||
/// `readdir(cursor)` on `Header.target`: one entry per call, cursor-advanced.
|
||||
pub const Readdir = extern struct { cursor: u64 };
|
||||
|
||||
/// The contract, whole. Verbs number from `envelope.first_protocol_operation`
|
||||
/// (16) in this order; the reserved verbs below it mean what they mean
|
||||
/// everywhere. `readdir` stays a protocol verb rather than folding into the
|
||||
/// reserved `enumerate`: it enumerates the children of one *node*, where
|
||||
/// `enumerate` names a provider's targets.
|
||||
pub const Protocol = envelope.Define(.{
|
||||
.name = "vfs",
|
||||
.version = 1,
|
||||
.operations = &.{
|
||||
.{ .name = "open", .request = Open, .reply = Opened },
|
||||
.{ .name = "close" },
|
||||
.{ .name = "read", .request = Read },
|
||||
.{ .name = "write", .request = Write, .reply = Written },
|
||||
.{ .name = "status", .reply = FileStatus },
|
||||
.{ .name = "readdir", .request = Readdir, .reply = DirectoryEntry },
|
||||
// The mount router's two verbs. Path routing lives in the kernel now
|
||||
// (system/kernel/vfs.zig), so no backend implements either; they keep
|
||||
// their numbers so the vocabulary stays the one docs/vfs-protocol.md
|
||||
// describes.
|
||||
.{ .name = "mount" }, // tail = the prefix, capability = the backend's endpoint
|
||||
.{ .name = "unmount" }, // tail = the prefix
|
||||
// Filesystem mutation, path-based: the path is the packet's tail.
|
||||
.{ .name = "mkdir" },
|
||||
.{ .name = "unlink" },
|
||||
// rename: the tail is the old path, a single 0x00 separator, then the
|
||||
// new path. Same-directory rename only.
|
||||
.{ .name = "rename" },
|
||||
// The registry's claim verb (P2): the name is the tail and the
|
||||
// provider's endpoint rides the call as its capability. A file backend
|
||||
// refuses it; only init implements it.
|
||||
.{ .name = "bind" },
|
||||
},
|
||||
});
|
||||
|
||||
pub const Operation = Protocol.Operation;
|
||||
|
||||
/// What a backend sizes its buffers to — the call floor, as every protocol does.
|
||||
pub const message_maximum: usize = Protocol.message_maximum;
|
||||
|
||||
/// The most inline payload any request may carry: the floor less the header and
|
||||
/// the widest fixed request part, so one bound serves every verb (a path, write
|
||||
/// data, a read's answer).
|
||||
pub const maximum_payload: usize = envelope.packet_maximum - Protocol.request_maximum;
|
||||
|
||||
/// Open flags (danos-native; `file_system.OpenOptions` maps its booleans onto these).
|
||||
/// Open flags (danos-native; `runtime.fs.OpenOptions` maps its booleans onto these).
|
||||
pub const create: u32 = 1;
|
||||
/// Open a directory (for readdir) rather than a file. A mounted backend uses
|
||||
/// this to open a directory node; the flat ramfs ignores it.
|
||||
@@ -152,42 +105,13 @@ pub const directory: u32 = 2;
|
||||
/// backend frees the old cluster chain; the flat ramfs ignores it.
|
||||
pub const truncate: u32 = 4;
|
||||
|
||||
test "the stable wire values: node kinds, entry layout, and the verb numbering" {
|
||||
test "protocol struct sizes and node kinds" {
|
||||
const std = @import("std");
|
||||
try std.testing.expectEqual(@as(u32, 0), @intFromEnum(NodeKind.regular));
|
||||
try std.testing.expectEqual(@as(u32, 1), @intFromEnum(NodeKind.directory));
|
||||
// Appended with the protocol namespace; every earlier value keeps its own.
|
||||
try std.testing.expectEqual(@as(u32, 6), @intFromEnum(NodeKind.socket));
|
||||
try std.testing.expectEqual(@as(u32, 7), @intFromEnum(NodeKind.protocol));
|
||||
try std.testing.expectEqual(@as(usize, 16), @sizeOf(DirectoryEntry));
|
||||
|
||||
// The numbering the envelope gives this protocol. These are NEW values: the
|
||||
// rebase moved every verb above the reserved range, so the old 0..11 are
|
||||
// gone and 16..27 are what the wire carries. Pinned because both sides of a
|
||||
// flag-day have to agree on them, not because they may never change again.
|
||||
try std.testing.expectEqual(@as(u32, 16), @intFromEnum(Operation.open));
|
||||
try std.testing.expectEqual(@as(u32, 17), @intFromEnum(Operation.close));
|
||||
try std.testing.expectEqual(@as(u32, 18), @intFromEnum(Operation.read));
|
||||
try std.testing.expectEqual(@as(u32, 19), @intFromEnum(Operation.write));
|
||||
try std.testing.expectEqual(@as(u32, 20), @intFromEnum(Operation.status));
|
||||
try std.testing.expectEqual(@as(u32, 21), @intFromEnum(Operation.readdir));
|
||||
try std.testing.expectEqual(@as(u32, 26), @intFromEnum(Operation.rename));
|
||||
try std.testing.expectEqual(@as(u32, 27), @intFromEnum(Operation.bind));
|
||||
// The payload bound is what it always was, arrived at the other way round:
|
||||
// the header plus the widest fixed request part is 32 bytes of the floor.
|
||||
try std.testing.expectEqual(@as(usize, 224), maximum_payload);
|
||||
}
|
||||
|
||||
test "the node id rides the header, and a path rides the tail" {
|
||||
const std = @import("std");
|
||||
var buffer: [message_maximum]u8 = undefined;
|
||||
|
||||
const opening = Protocol.encodeRequest(.open, 0, .{ .flags = create }, "/a/b", &buffer).?;
|
||||
try std.testing.expectEqual(@as(u32, create), Protocol.decodeRequest(.open, opening).?.flags);
|
||||
try std.testing.expectEqualStrings("/a/b", Protocol.requestTail(.open, opening));
|
||||
try std.testing.expectEqual(@as(u64, 0), envelope.headerOf(opening).?.target);
|
||||
|
||||
const reading = Protocol.encodeRequest(.read, 7, .{ .offset = 512, .len = 64 }, &.{}, &buffer).?;
|
||||
try std.testing.expectEqual(@as(u64, 7), envelope.headerOf(reading).?.target);
|
||||
try std.testing.expectEqual(@as(u64, 512), Protocol.decodeRequest(.read, reading).?.offset);
|
||||
// The appended operations keep the original values.
|
||||
try std.testing.expectEqual(@as(u32, 0), @intFromEnum(Operation.open));
|
||||
try std.testing.expectEqual(@as(u32, 4), @intFromEnum(Operation.status));
|
||||
try std.testing.expectEqual(@as(u32, 5), @intFromEnum(Operation.readdir));
|
||||
}
|
||||
|
||||
@@ -1,31 +0,0 @@
|
||||
//! The "xkeyboard-config" library domain: keyboard layouts compiled from the
|
||||
//! X11 xkeyboard-config database into native Zig (keycode + modifiers ->
|
||||
//! keysym/character). The `layouts` tables are generated by
|
||||
//! tools/make-xkeyboard-config.py; `xkeyboard-config` is the hand-written API
|
||||
//! over them.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const layouts = b.addModule("layouts", .{
|
||||
.root_source_file = b.path("generated/layouts.zig"),
|
||||
});
|
||||
_ = b.addModule("xkeyboard-config", .{
|
||||
.root_source_file = b.path("xkeyboard-config.zig"),
|
||||
.imports = &.{.{ .name = "layouts", .module = layouts }},
|
||||
});
|
||||
|
||||
// Standalone `zig build test` for this domain alone; the root build keeps
|
||||
// its aggregate test step. The keycode->character assertions are the
|
||||
// end-to-end proof that the xkb-data -> generator -> Zig-lookup pipeline
|
||||
// is correct.
|
||||
const test_step = b.step("test", "Run the xkeyboard-config unit tests");
|
||||
const xkb_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("xkeyboard-config.zig"),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
.imports = &.{.{ .name = "layouts", .module = layouts }},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(xkb_tests).step);
|
||||
}
|
||||
@@ -1,8 +0,0 @@
|
||||
.{
|
||||
.name = .xkeyboard_config,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0xea5abe82f08b6eae, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{},
|
||||
.paths = .{""},
|
||||
}
|
||||
+22
-18
@@ -1,6 +1,6 @@
|
||||
//! The **private kernel ↔ runtime** ABI: the raw system_call contract — the call
|
||||
//! numbers, `mmap` protection flags, the page size those calls work in, and the IPC
|
||||
//! notification bits. Shared by the kernel dispatcher
|
||||
//! name-registry ids and notification bit. Shared by the kernel dispatcher
|
||||
//! (system/kernel/process.zig) and the user-space runtime library (library/runtime/),
|
||||
//! so the two can never drift.
|
||||
//!
|
||||
@@ -33,13 +33,8 @@ pub const SystemCall = enum(u64) {
|
||||
mmap = 4, // mmap(len, prot) -> base: grant zeroed, page-aligned user pages
|
||||
munmap = 5, // munmap(base, len): release pages from a prior mmap
|
||||
create_ipc_endpoint = 6, // create_ipc_endpoint() -> handle: a new IPC endpoint
|
||||
// 7 and 8 were ipc_register/ipc_lookup — the flat ServiceId name registry,
|
||||
// retired with the protocol namespace (docs/os-development/protocol-namespace.md).
|
||||
// A service now binds its name at the registry (init, over /protocol) and a
|
||||
// client resolves and opens that path; neither is a system call any more. The
|
||||
// numbers stay vacant rather than being reused: every other entry is
|
||||
// position-fixed by an explicit value, so a hole costs nothing and a reused
|
||||
// number would silently mean two things across a rebuild boundary.
|
||||
ipc_register = 7, // ipc_register(service_id, handle): publish an endpoint by well-known id
|
||||
ipc_lookup = 8, // ipc_lookup(service_id) -> handle: find a published endpoint
|
||||
ipc_call = 9, // ipc_call(h, message, len, reply, cap) -> reply_len: send + block for reply
|
||||
ipc_reply_wait = 10, // ipc_reply_wait(h, reply, len, receive, cap) -> receive_len (+badge in rdx)
|
||||
device_enumerate = 11, // device_enumerate(buffer, maximum) -> count: snapshot the device table
|
||||
@@ -252,11 +247,8 @@ pub const klog_maximum_message: usize = 256;
|
||||
pub const fs_route_kernel: u64 = 0; // rdx = node token; serve via fs_node
|
||||
pub const fs_route_backend: u64 = 1; // rdx = endpoint handle; speak vfs-protocol
|
||||
|
||||
/// fs_node operations. These were once the vfs-protocol Operation numbers; the
|
||||
/// rebase onto the envelope moved every protocol verb above the reserved range
|
||||
/// (16 and up), and these did not follow — they are a *syscall* selector, not a
|
||||
/// packet's verb, and renumbering a kernel ABI to track a wire format would be
|
||||
/// coupling in the wrong direction. The two vocabularies are simply separate now.
|
||||
/// fs_node operations — the same numbers as the vfs-protocol Operation enum, so
|
||||
/// client code shares one vocabulary.
|
||||
pub const fs_node_read: u64 = 2;
|
||||
pub const fs_node_status: u64 = 4;
|
||||
pub const fs_node_readdir: u64 = 5;
|
||||
@@ -292,11 +284,23 @@ pub const KlogStatus = extern struct {
|
||||
boot_unix_seconds: u64, // wall-clock time of boot (RTC anchor)
|
||||
};
|
||||
|
||||
// The `ServiceId` enum lived here: a flat, compile-time list of well-known
|
||||
// service ids backed by a 16-slot kernel table. It is gone with the protocol
|
||||
// namespace — names are strings resolved under `/protocol` at run time, so a
|
||||
// third-party program can introduce a contract the ABI never heard of, and the
|
||||
// registrar (init) decides who may claim one.
|
||||
/// Well-known IPC service ids for the bootstrap name registry (create_ipc_endpoint +
|
||||
/// ipc_register/ipc_lookup). Small integers, so no string interning is needed
|
||||
/// during bring-up. The VFS server registers under `vfs`; clients look it up.
|
||||
pub const ServiceId = enum(u32) {
|
||||
vfs = 1, // RETIRED: the router moved into the kernel (fs_resolve); the slot stays reserved
|
||||
input = 2,
|
||||
ps2_bus = 3, // the 8042 owner; child device drivers attach here for raw bytes
|
||||
device_manager = 4, // the tree, the matcher, the supervisor (docs/device-manager.md)
|
||||
power = 5, // system power: events (button, lid, battery) + shutdown (docs/power.md; domain-named per docs/discovery.md — the acpi service registers it on x86, a PSCI service will on ARM)
|
||||
usb_bus = 6, // the xHCI host-controller driver's transfer endpoint; USB class drivers look it up and `callCap`-open their device to get a private per-device transfer channel (docs/driver-model.md)
|
||||
block = 7, // a block-device driver (USB mass storage today): read/write of fixed-size blocks, the storage a filesystem sits on
|
||||
fat = 8, // the FAT filesystem server; the VFS mounts it and forwards paths under its mount point (/mnt/usb) to it
|
||||
display = 9, // the display service: owns the framebuffer, composites a layer stack, presents frames (docs/display.md)
|
||||
shared_memory_test = 10, // the shared-memory test server (V2): a client passes it a shared-memory capability, it maps + verifies (docs/display-v2.md)
|
||||
scanout = 11, // a native scanout driver (virtio-gpu): the compositor finds it here to upgrade off the GOP framebuffer (docs/display-v2.md)
|
||||
_,
|
||||
};
|
||||
|
||||
/// Protection flags for `mmap` (matching the usual C bit values).
|
||||
pub const prot_read: u64 = 1;
|
||||
|
||||
@@ -1,175 +0,0 @@
|
||||
# /system/configuration/protocol.csv — who may claim, and who may reach, a name
|
||||
# under /protocol (docs/os-development/protocol-namespace.md).
|
||||
#
|
||||
# init is the registrar: it serves /protocol, and every bind AND every open is
|
||||
# checked against this file. It is AUTHORITATIVE — a name no row grants cannot be
|
||||
# bound or reached, and a missing file means nothing may be bound or reached at
|
||||
# all.
|
||||
#
|
||||
# A refused open is answered exactly as a name nobody bound is: -ENOENT, and no
|
||||
# capability. That is not politeness, it is the model — the namespace IS the
|
||||
# restriction, so what a process may not open simply does not exist for it, and
|
||||
# there is no "permission denied" for it to tell apart from "no such contract".
|
||||
# Which is why a missing row here shows up as a client retrying forever rather
|
||||
# than as an error: check this file first, and `readdir /protocol` second.
|
||||
#
|
||||
# '#' starts a comment (whole-line or trailing); blank lines are ignored.
|
||||
# Whitespace around a field is trimmed, so columns may be padded. Four
|
||||
# comma-separated fields per row:
|
||||
#
|
||||
# binary the claimant's binary path, exactly as the kernel stamped it at
|
||||
# spawn (argv[0]) — unforgeable, read from the process records
|
||||
# supervisor the authorized supervising TASK, written as the binary it runs —
|
||||
# the path init was started as for its own services, the device
|
||||
# manager's path for the drivers it starts. The one word that is not
|
||||
# a path is 'kernel', because a kernel task has no binary; that is
|
||||
# what the test harness's direct spawns look like.
|
||||
# Matched by IDENTITY, not by spelling. Name alone is not identity —
|
||||
# spawn is ungated, so a hostile process can start a granted binary
|
||||
# itself and inherit its grants; and it can equally start its own
|
||||
# instance of the *supervisor's* binary and have that spawn the
|
||||
# granted one, at which point both names read correctly (the
|
||||
# laundering deputy). So init also asks which task the supervisor
|
||||
# is: 'kernel' means supervisor id 0, which only the kernel can
|
||||
# confer; init's own path means this init; any other path means a
|
||||
# task init spawned itself or one the kernel spawned. Task ids are
|
||||
# monotonic and never reused, so an id cannot be borrowed.
|
||||
# permission bind (provide this contract) | open (speak to it) |
|
||||
# supervise (stand in someone else's chain — see below)
|
||||
# name the contract, relative to /protocol
|
||||
#
|
||||
# A trailing '*' on any field matches any tail — how a subtree is granted whole.
|
||||
#
|
||||
# 'supervise' exists because attestation is one hop deep and the driver tree is
|
||||
# three: the device manager starts the PS/2 bus, and the bus starts the keyboard
|
||||
# and mouse drivers. Init never met the bus, so it cannot vouch for it by
|
||||
# acquaintance — and it must not vouch for it by name, or the laundering deputy
|
||||
# walks straight in. A 'supervise' row is the manifest saying it: a task running
|
||||
# this binary, under this supervisor, may be the supervising task an 'open' row
|
||||
# names, for this contract and no other. It grants the delegate nothing itself,
|
||||
# and it is deliberately open-only — a delegate may vouch for what its children
|
||||
# REACH, never for what they CLAIM, so every bind refusal is untouched by it.
|
||||
#
|
||||
# binary supervisor permission name
|
||||
|
||||
# --- the services init spawns from init.csv ---------------------------------
|
||||
/system/services/input, /system/services/init, bind, input
|
||||
/system/services/device-manager, /system/services/init, bind, device-manager
|
||||
/system/services/fat, /system/services/init, bind, vfs
|
||||
/system/services/display, /system/services/init, bind, display
|
||||
|
||||
# The discovery service ships under one neutral name per firmware (docs/discovery.md);
|
||||
# on x86 it is the acpi service, and what it provides is the power contract.
|
||||
/system/services/discovery, /system/services/device-manager, bind, power
|
||||
|
||||
# --- the drivers, which the device manager spawns ---------------------------
|
||||
/system/drivers/ps2-bus, /system/services/device-manager, bind, ps2-bus
|
||||
/system/drivers/usb-xhci-bus, /system/services/device-manager, bind, usb-transfer
|
||||
/system/drivers/usb-storage, /system/services/device-manager, bind, block
|
||||
/system/drivers/virtio-gpu, /system/services/device-manager, bind, scanout
|
||||
|
||||
# --- the same providers when the kernel test harness starts them directly ---
|
||||
# A scenario boot spawns its own providers instead of letting init do it
|
||||
# (docs/security-track-plan.md, decision 9), so the same binaries appear with
|
||||
# 'kernel' as the supervisor. Nothing else changes: the binary must still match.
|
||||
/system/services/input, kernel, bind, input
|
||||
/system/services/device-manager, kernel, bind, device-manager
|
||||
/system/services/fat, kernel, bind, vfs
|
||||
/system/services/display, kernel, bind, display
|
||||
/system/services/discovery, kernel, bind, power
|
||||
|
||||
# --- test fixtures ----------------------------------------------------------
|
||||
# The subtree rule, dogfooded: anything installed under /test may claim anything
|
||||
# under /protocol/test, and nothing above it — whether the harness spawned it or
|
||||
# another fixture did.
|
||||
/test/*, kernel, bind, test/*
|
||||
/test/*, /test/*, bind, test/*
|
||||
|
||||
|
||||
# ============================================================================
|
||||
# open — who may REACH each contract. One row per client per contract; a client
|
||||
# with no row here simply finds the name absent, forever.
|
||||
# ============================================================================
|
||||
|
||||
# --- init's own services ----------------------------------------------------
|
||||
# fat reaches the block device behind the volume it mounts; the compositor
|
||||
# reaches the scanout its driver announced, its own endpoint (the mouse-listener
|
||||
# thread opens /protocol/display like any other client — threads share no
|
||||
# handles), and the input stream that moves the cursor.
|
||||
/system/services/fat, /system/services/init, open, block
|
||||
/system/services/display, /system/services/init, open, scanout
|
||||
/system/services/display, /system/services/init, open, display
|
||||
/system/services/display, /system/services/init, open, input
|
||||
/system/services/display-demo, /system/services/init, open, display
|
||||
|
||||
# --- the same two when the kernel test harness starts them directly ---------
|
||||
/system/services/display, kernel, open, scanout
|
||||
/system/services/display, kernel, open, display
|
||||
/system/services/display, kernel, open, input
|
||||
/system/services/display-demo, kernel, open, display
|
||||
|
||||
# --- the drivers, and the discovery service ---------------------------------
|
||||
# Every driver says hello to the manager that started it — one row for the whole
|
||||
# subtree, because that handshake is what being a driver means. The rest are per
|
||||
# driver: the storage and HID class drivers talk to their controller, the HID
|
||||
# drivers publish into the input stream, and the GPU driver announces its scanout
|
||||
# to the compositor.
|
||||
/system/drivers/*, /system/services/device-manager, open, device-manager
|
||||
/system/services/discovery, /system/services/device-manager, open, device-manager
|
||||
/system/drivers/usb-storage, /system/services/device-manager, open, usb-transfer
|
||||
/system/drivers/usb-hid-keyboard, /system/services/device-manager, open, usb-transfer
|
||||
/system/drivers/usb-hid-keyboard, /system/services/device-manager, open, input
|
||||
/system/drivers/usb-hid-mouse, /system/services/device-manager, open, usb-transfer
|
||||
/system/drivers/usb-hid-mouse, /system/services/device-manager, open, input
|
||||
/system/drivers/virtio-gpu, /system/services/device-manager, open, display
|
||||
|
||||
# --- the PS/2 child drivers, one hop further down ---------------------------
|
||||
# The keyboard and mouse drivers are started by the BUS driver, not by the
|
||||
# device manager — the one three-deep chain in the tree. Init cannot vouch for
|
||||
# the bus by acquaintance (it never started it), so the manifest authorizes it
|
||||
# explicitly, and only for the two contracts its children need.
|
||||
/system/drivers/ps2-bus, /system/services/device-manager, supervise, ps2-bus
|
||||
/system/drivers/ps2-bus, /system/services/device-manager, supervise, input
|
||||
/system/drivers/ps2-keyboard, /system/drivers/ps2-bus, open, ps2-bus
|
||||
/system/drivers/ps2-keyboard, /system/drivers/ps2-bus, open, input
|
||||
/system/drivers/ps2-mouse, /system/drivers/ps2-bus, open, ps2-bus
|
||||
/system/drivers/ps2-mouse, /system/drivers/ps2-bus, open, input
|
||||
|
||||
# --- test fixtures ----------------------------------------------------------
|
||||
# The /protocol/test subtree is theirs whole, the way the bind rows give it to
|
||||
# them. Everything ABOVE that subtree is named one fixture at a time, so a
|
||||
# fixture reaches a system contract only where a scenario needs it — which is
|
||||
# what leaves the rest genuinely absent for the rest of them (the protocol-denied
|
||||
# case asks for one it was not given, and is told there is no such thing).
|
||||
/test/*, kernel, open, test/*
|
||||
/test/*, /test/*, open, test/*
|
||||
/test/*, kernel, open, device-manager
|
||||
/test/*, /system/services/device-manager, open, device-manager
|
||||
/test/system/services/input-source, kernel, open, input
|
||||
/test/system/services/input-test, kernel, open, input
|
||||
|
||||
# The guessable-id probe (test/system/services/badge-scope-test) runs as two
|
||||
# processes of one binary: the owner, which the scenario spawns, and the intruder,
|
||||
# which the owner spawns with the ids it holds. Both reach the compositor — the
|
||||
# owner to create the layer, the intruder to be refused it — so the binary is
|
||||
# named twice, once per supervisor. The second row needs no 'supervise'
|
||||
# delegation: the owner was spawned by the KERNEL, which is a chain init can
|
||||
# vouch for on its own.
|
||||
/test/system/services/badge-scope-test, kernel, open, display
|
||||
/test/system/services/badge-scope-test, /test/*, open, display
|
||||
|
||||
# The conformance probe (test/system/services/protocol-conformance-test) asks
|
||||
# every provider its boot bound for the envelope's reserved verbs. It reaches
|
||||
# ONLY the two contracts its own scenario boots a provider for, named one at a
|
||||
# time exactly like the two rows above — no subtree, no wildcard. Everything else
|
||||
# under /protocol stays absent for it, which is the point: the fixture walks the
|
||||
# namespace listing and reports what it could not open rather than being handed
|
||||
# the tree to make the test look broad.
|
||||
/test/system/services/protocol-conformance-test, kernel, open, input
|
||||
/test/system/services/protocol-conformance-test, kernel, open, display
|
||||
|
||||
# The laundering-deputy probe (test/system/services/protocol-registry-test) runs
|
||||
# a grandchild whose supervisor is a fixture nobody authorized — that is the
|
||||
# point of it, and its bind must stay refused. It still has to report the verdict
|
||||
# it got, so its reporting channel, and nothing else, is delegated.
|
||||
/test/*, /test/*, supervise, test/verdict
|
||||
|
Can't render this file because it contains an unexpected character in line 12 and column 15.
|
@@ -0,0 +1,82 @@
|
||||
//! /system/drivers/intel-uhd-graphics-750 — spawned by the device manager with
|
||||
//! the device-tree id as argv[1]; claims that device and no other.
|
||||
|
||||
const std = @import("std");
|
||||
const device = @import("driver");
|
||||
const ipc = @import("ipc");
|
||||
const memory = @import("memory");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
|
||||
/// No protocol yet: the kernel's IPC ceiling (MESSAGE_MAXIMUM) sizes the buffers.
|
||||
const message_maximum = 256;
|
||||
|
||||
var controller_id: u64 = 0;
|
||||
var register_base: usize = 0;
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
_ = endpoint; // needed later, for irq binding and timers
|
||||
|
||||
if (!device.claim(controller_id)) {
|
||||
std.log.err("unable to claim device {d}", .{controller_id});
|
||||
return false;
|
||||
}
|
||||
|
||||
// Fetch our own descriptor back for the device's resources.
|
||||
const buffer = memory.allocator().alloc(device.DeviceDescriptor, 64) catch return false;
|
||||
defer memory.allocator().free(buffer);
|
||||
const total = device.enumerate(buffer);
|
||||
const descriptor = for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||
if (d.id == controller_id) break d;
|
||||
} else {
|
||||
std.log.err("device {d} not in the device tree", .{controller_id});
|
||||
return false;
|
||||
};
|
||||
|
||||
// Log every resource BEFORE choosing one (see step 5).
|
||||
var register_index: u64 = 0;
|
||||
for (descriptor.resources[0..@intCast(descriptor.resource_count)], 0..) |resource, index| {
|
||||
std.log.info("resource {d}: kind={d} start=0x{x} len=0x{x}", .{
|
||||
index, resource.kind, resource.start, resource.len,
|
||||
});
|
||||
// The 16 MiB window is GTTMMADR, the register BAR (this device also has
|
||||
// a 256 MiB memory BAR, GMADR — "first memory resource" would be wrong).
|
||||
if (resource.kind == @intFromEnum(device.ResourceKind.memory) and
|
||||
resource.len == 16 * 1024 * 1024) register_index = index;
|
||||
}
|
||||
if (register_index == 0) {
|
||||
std.log.err("register BAR not found", .{});
|
||||
return false;
|
||||
}
|
||||
|
||||
register_base = device.mmioMap(controller_id, register_index) orelse {
|
||||
std.log.err("mmio_map failed", .{});
|
||||
return false;
|
||||
};
|
||||
std.log.info("registers mapped at 0x{x}", .{register_base});
|
||||
return true;
|
||||
}
|
||||
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||
_ = message;
|
||||
_ = reply;
|
||||
_ = sender;
|
||||
_ = capability;
|
||||
return 0; // no protocol yet; the zero-length ping is answered by the harness
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
const argument = init.arguments.get(1) orelse {
|
||||
std.log.err("missing device id (argv[1])", .{});
|
||||
return;
|
||||
};
|
||||
controller_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
std.log.err("malformed device id '{s}'", .{argument});
|
||||
return;
|
||||
};
|
||||
service.run(message_maximum, .{
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
// .on_notification only once an IRQ or timer is bound
|
||||
});
|
||||
}
|
||||
@@ -1,18 +0,0 @@
|
||||
//! The pci-bus driver as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! build-support resolves each name from the domains this zon declares.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "pci-bus",
|
||||
.root_source_file = b.path("pci-bus.zig"),
|
||||
.imports = &.{
|
||||
"device-manager-protocol", "driver", "ipc", "logging", "memory", "pci-class",
|
||||
"process", "service",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
}
|
||||
@@ -1,16 +0,0 @@
|
||||
.{
|
||||
.name = .pci_bus,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x283fca121f0bb145, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.device = .{ .path = "../../../library/device" },
|
||||
.protocol = .{ .path = "../../../library/protocol" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -21,7 +21,7 @@ const logging = @import("logging");
|
||||
const device_manager_protocol = @import("device-manager-protocol");
|
||||
const pci_class = @import("pci-class");
|
||||
|
||||
/// Log a discovered function as its would-be /system/configuration/devices.csv columns (bus, base,
|
||||
/// Log a discovered function as its would-be /etc/devices.csv columns (bus, base,
|
||||
/// class, prog_if, vendor, device, subsystem) followed by the human-readable
|
||||
/// class/subclass/prog-IF names — so a row for a new driver reads straight off the
|
||||
/// boot log. `subsystem` prints as `*` when the function has none, matching the CSV
|
||||
@@ -154,7 +154,7 @@ fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void
|
||||
descriptor.class = @intFromEnum(device.DeviceClass.pci_device);
|
||||
descriptor.pci_class = class_triple;
|
||||
// Vendor/device from the first config dword (0x00): low half vendor, high half
|
||||
// device. These carry to the manager's /system/configuration/devices.csv matcher so a function
|
||||
// device. These carry to the manager's /etc/devices.csv matcher so a function
|
||||
// can bind on its exact 1AF4:1050 identity, not just its class triple.
|
||||
const vendor_device = configRead(bus, dev, function, 0x00);
|
||||
descriptor.vendor = @truncate(vendor_device);
|
||||
@@ -229,30 +229,27 @@ fn registerAndReport(bus: u64, dev: u64, function: u64, class_triple: u32) void
|
||||
std.log.info("register refused for {d}:{d}.{d}", .{ bus, dev, function });
|
||||
return;
|
||||
};
|
||||
// The registered device id is the packet's target — the manager's object
|
||||
// addressing — so the report body carries only where on the bus it sits and
|
||||
// what it is.
|
||||
var packet: [device_manager_protocol.message_maximum]u8 = undefined;
|
||||
const framed = device_manager_protocol.Protocol.encodeRequest(.child_added, registered, .{
|
||||
const report = device_manager_protocol.ChildAdded{
|
||||
.bus = @intFromEnum(device_manager_protocol.BusKind.pci),
|
||||
.parent = bridge_id,
|
||||
.bus_address = (bus << 8) | (dev << 3) | function,
|
||||
.identity = class_triple,
|
||||
.device_id = registered,
|
||||
.vendor = descriptor.vendor,
|
||||
.device = descriptor.device,
|
||||
.subsystem = descriptor.subsystem,
|
||||
}, &.{}, &packet) orelse return;
|
||||
};
|
||||
var reply: [device_manager_protocol.message_maximum]u8 = undefined;
|
||||
_ = ipc.call(manager_handle, framed, &reply) catch {
|
||||
_ = ipc.call(manager_handle, std.mem.asBytes(&report), &reply) catch {
|
||||
std.log.info("child report for {d}:{d}.{d} failed", .{ bus, dev, function });
|
||||
};
|
||||
}
|
||||
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||
_ = message;
|
||||
_ = reply;
|
||||
_ = sender;
|
||||
_ = arrived; // nothing here takes a capability: the harness closes what arrives
|
||||
_ = capability;
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
@@ -1,50 +0,0 @@
|
||||
//! The ps2-bus driver as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! build-support resolves each name from the domains this zon declares.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const ps2_bus_exe = build_support.userBinary(b, .{
|
||||
.name = "ps2-bus",
|
||||
.root_source_file = b.path("ps2-bus.zig"),
|
||||
.imports = &.{ "acpi-ids", "channel", "driver", "ipc", "logging", "memory", "process", "service", "time" },
|
||||
});
|
||||
b.installArtifact(ps2_bus_exe);
|
||||
|
||||
const ps2_keyboard_exe = build_support.userBinary(b, .{
|
||||
.name = "ps2-keyboard",
|
||||
.root_source_file = b.path("keyboard.zig"),
|
||||
.imports = &.{
|
||||
"acpi-ids", "channel", "driver", "input-client", "input-protocol", "ipc",
|
||||
"logging", "memory", "process", "time", "xkeyboard-config",
|
||||
},
|
||||
});
|
||||
b.installArtifact(ps2_keyboard_exe);
|
||||
|
||||
const ps2_mouse_exe = build_support.userBinary(b, .{
|
||||
.name = "ps2-mouse",
|
||||
.root_source_file = b.path("mouse.zig"),
|
||||
.imports = &.{
|
||||
"acpi-ids", "channel", "driver", "input-client", "input-protocol", "ipc", "logging",
|
||||
"memory", "process", "time",
|
||||
},
|
||||
});
|
||||
b.installArtifact(ps2_mouse_exe);
|
||||
|
||||
// Standalone `zig build test`; the root aggregate depends on this step.
|
||||
const test_step = b.step("test", "Run the ps2-bus unit tests");
|
||||
for ([_][]const u8{
|
||||
"scancode.zig", // set-2 decode + keyboard state machine
|
||||
"mouse-packet.zig", // 3-byte mouse packet assembly
|
||||
}) |test_root| {
|
||||
const unit_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(test_root),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(unit_tests).step);
|
||||
}
|
||||
}
|
||||
@@ -1,18 +0,0 @@
|
||||
.{
|
||||
.name = .ps2_bus,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x642a365353bf7de9, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.client = .{ .path = "../../../library/client" },
|
||||
.device = .{ .path = "../../../library/device" },
|
||||
.protocol = .{ .path = "../../../library/protocol" },
|
||||
.@"xkeyboard-config" = .{ .path = "../../../library/xkeyboard-config" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -16,11 +16,10 @@
|
||||
|
||||
const std = @import("std");
|
||||
const device = @import("driver");
|
||||
const channel = @import("channel");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const time = @import("time");
|
||||
const input = @import("input-client");
|
||||
const input = @import("input");
|
||||
const memory = @import("memory");
|
||||
const logging = @import("logging");
|
||||
const xkb = @import("xkeyboard-config");
|
||||
@@ -28,12 +27,12 @@ const ps2 = @import("ps2-library.zig");
|
||||
const scancode = @import("scancode.zig");
|
||||
const input_protocol = @import("input-protocol");
|
||||
|
||||
/// Open `/protocol/ps2-bus`, retrying while the bus (which spawned us before
|
||||
/// binding) is still coming up.
|
||||
/// Look up the ps2-bus service, retrying while the bus (which spawned us before
|
||||
/// registering) is still coming up.
|
||||
fn lookupBus() ?ipc.Handle {
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 100) : (attempts += 1) {
|
||||
if (channel.openEndpoint("ps2-bus")) |handle| return handle;
|
||||
if (ipc.lookup(.ps2_bus)) |handle| return handle;
|
||||
time.sleepMillis(50);
|
||||
}
|
||||
return null;
|
||||
|
||||
@@ -16,23 +16,22 @@
|
||||
|
||||
const std = @import("std");
|
||||
const device = @import("driver");
|
||||
const channel = @import("channel");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const time = @import("time");
|
||||
const input = @import("input-client");
|
||||
const input = @import("input");
|
||||
const memory = @import("memory");
|
||||
const logging = @import("logging");
|
||||
const ps2 = @import("ps2-library.zig");
|
||||
const mouse_packet = @import("mouse-packet.zig");
|
||||
const input_protocol = @import("input-protocol");
|
||||
|
||||
/// Open `/protocol/ps2-bus`, retrying while the bus (which spawned us before
|
||||
/// binding) is still coming up.
|
||||
/// Look up the ps2-bus service, retrying while the bus (which spawned us before
|
||||
/// registering) is still coming up.
|
||||
fn lookupBus() ?ipc.Handle {
|
||||
var attempts: usize = 0;
|
||||
while (attempts < 100) : (attempts += 1) {
|
||||
if (channel.openEndpoint("ps2-bus")) |handle| return handle;
|
||||
if (ipc.lookup(.ps2_bus)) |handle| return handle;
|
||||
time.sleepMillis(50);
|
||||
}
|
||||
return null;
|
||||
|
||||
@@ -11,7 +11,6 @@
|
||||
//! - irq 0xc len 0x1
|
||||
const std = @import("std");
|
||||
const device = @import("driver");
|
||||
const channel = @import("channel");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
@@ -63,13 +62,7 @@ var port_device_types = [_]?ps2.DeviceType{ null, null };
|
||||
/// Handle a child driver's `AttachRequest`: record the endpoint capability it
|
||||
/// passed as the forwarding target for the port whose device matches its type.
|
||||
/// Writes an `AttachReply` into `out` and returns its length.
|
||||
/// A child driver's AttachRequest. The endpoint it hands over arrives under the
|
||||
/// same ownership rule the service harness states (`ipc.Arrival`): the turn owns
|
||||
/// it, and only the path that records it in `port_endpoints` says `take`. Every
|
||||
/// refusal here simply returns, and the loop closes what arrived — otherwise a
|
||||
/// stranger (this is a named contract, reachable by anyone) spends one of this
|
||||
/// driver's thirty-two handle slots per malformed attach.
|
||||
fn handleAttach(message: []const u8, out: []u8, arrived: *ipc.Arrival) usize {
|
||||
fn handleAttach(message: []const u8, got: ipc.Received, out: []u8) usize {
|
||||
const reply = struct {
|
||||
fn write(buffer: []u8, status: ps2.AttachStatus) usize {
|
||||
const header = ps2.AttachReply{ .status = @intFromEnum(status) };
|
||||
@@ -80,17 +73,12 @@ fn handleAttach(message: []const u8, out: []u8, arrived: *ipc.Arrival) usize {
|
||||
|
||||
if (message.len < @sizeOf(ps2.AttachRequest)) return reply.write(out, .invalid_request);
|
||||
const request = std.mem.bytesToValue(ps2.AttachRequest, message[0..@sizeOf(ps2.AttachRequest)]);
|
||||
const endpoint = arrived.peek() orelse return reply.write(out, .missing_endpoint);
|
||||
const endpoint = got.cap orelse return reply.write(out, .missing_endpoint);
|
||||
|
||||
for (&port_device_types, 0..) |maybe_type, port_index| {
|
||||
const device_type = maybe_type orelse continue;
|
||||
if (@intFromEnum(device_type) != request.device_type) continue;
|
||||
// Claimed. A re-attach supersedes the previous driver's endpoint, and the
|
||||
// one it displaces is closed: the slot holds exactly one reference.
|
||||
if (port_endpoints[port_index]) |previous| {
|
||||
if (previous != endpoint) _ = ipc.close(previous);
|
||||
}
|
||||
port_endpoints[port_index] = arrived.take();
|
||||
port_endpoints[port_index] = endpoint;
|
||||
std.log.info("{s} driver attached", .{@tagName(device_type)});
|
||||
return reply.write(out, .ok);
|
||||
}
|
||||
@@ -225,16 +213,15 @@ pub fn main() void {
|
||||
return;
|
||||
};
|
||||
|
||||
// The endpoint the child drivers attach to and IRQ1 wakes. Bound as the
|
||||
// `ps2-bus` contract so the children can find it by name, the way input
|
||||
// subscribers find the input service. This driver runs its own loop rather
|
||||
// than the service harness, so it binds by hand — same call the harness makes.
|
||||
// The endpoint the child drivers attach to and IRQ1 wakes. Registered under a
|
||||
// well-known id so the children can find it, the way input subscribers find
|
||||
// the input service.
|
||||
const endpoint = ipc.createIpcEndpoint() orelse {
|
||||
_ = logging.write("/system/drivers/ps2-bus: no endpoint\n");
|
||||
return;
|
||||
};
|
||||
if (!channel.bindPatiently("ps2-bus", endpoint)) {
|
||||
_ = logging.write("/system/drivers/ps2-bus: could not bind /protocol/ps2-bus\n");
|
||||
if (!ipc.register(.ps2_bus, endpoint)) {
|
||||
_ = logging.write("/system/drivers/ps2-bus: register failed\n");
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -286,13 +273,6 @@ pub fn main() void {
|
||||
var receive: [@sizeOf(ps2.AttachRequest)]u8 = undefined;
|
||||
while (true) {
|
||||
const got = ipc.replyWait(endpoint, reply_buffer[0..reply_len], &receive, null);
|
||||
// The turn owns whatever capability arrived and closes it unless
|
||||
// `handleAttach` claims it (`ipc.Arrival`) — the kernel installs one
|
||||
// whatever the message's length or kind, so this covers the notification
|
||||
// path and every refusal below it.
|
||||
var arrived: ipc.Arrival = .{ .handle = got.cap };
|
||||
defer arrived.release();
|
||||
|
||||
if (got.isNotification()) {
|
||||
reply_len = 0;
|
||||
if (got.isMessage() or got.isChildExit()) continue; // nothing sends us these
|
||||
@@ -321,6 +301,6 @@ pub fn main() void {
|
||||
}
|
||||
continue;
|
||||
}
|
||||
reply_len = handleAttach(receive[0..got.len], &reply_buffer, &arrived);
|
||||
reply_len = handleAttach(receive[0..got.len], got, &reply_buffer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,42 +0,0 @@
|
||||
//! The usb-hid driver as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! build-support resolves each name from the domains this zon declares.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const usb_hid_keyboard_exe = build_support.userBinary(b, .{
|
||||
.name = "usb-hid-keyboard",
|
||||
.root_source_file = b.path("keyboard.zig"),
|
||||
.imports = &.{
|
||||
"driver", "input-client", "input-protocol", "ipc", "logging", "process", "service",
|
||||
"usb", "usb-abi", "xkeyboard-config",
|
||||
},
|
||||
});
|
||||
b.installArtifact(usb_hid_keyboard_exe);
|
||||
|
||||
const usb_hid_mouse_exe = build_support.userBinary(b, .{
|
||||
.name = "usb-hid-mouse",
|
||||
.root_source_file = b.path("mouse.zig"),
|
||||
.imports = &.{
|
||||
"driver", "input-client", "input-protocol", "ipc", "logging", "process", "service",
|
||||
"usb", "usb-abi",
|
||||
},
|
||||
});
|
||||
b.installArtifact(usb_hid_mouse_exe);
|
||||
|
||||
// Standalone `zig build test`; the root aggregate depends on this step.
|
||||
const test_step = b.step("test", "Run the usb-hid unit tests");
|
||||
for ([_][]const u8{
|
||||
"hid-report.zig", // HID boot-report keyboard/mouse decode
|
||||
}) |test_root| {
|
||||
const unit_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(test_root),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(unit_tests).step);
|
||||
}
|
||||
}
|
||||
@@ -1,18 +0,0 @@
|
||||
.{
|
||||
.name = .usb_hid,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x66328b738fffff01, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.client = .{ .path = "../../../library/client" },
|
||||
.device = .{ .path = "../../../library/device" },
|
||||
.protocol = .{ .path = "../../../library/protocol" },
|
||||
.@"xkeyboard-config" = .{ .path = "../../../library/xkeyboard-config" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -18,7 +18,7 @@ const std = @import("std");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
const input = @import("input-client");
|
||||
const input = @import("input");
|
||||
const device_manager = @import("driver");
|
||||
const logging = @import("logging");
|
||||
const usb = @import("usb");
|
||||
@@ -112,10 +112,9 @@ pub fn main(init: process.Init) void {
|
||||
if (signals.has(.terminate)) return;
|
||||
continue;
|
||||
}
|
||||
if (!got.isMessage()) continue;
|
||||
// An `interrupt_report` event packet: the verb in its folded header, the
|
||||
// report after it. Anything else on this endpoint is not ours.
|
||||
const message = usb.reportOf(receive[0..got.len]) orelse continue;
|
||||
if (!got.isMessage() or got.len < @sizeOf(usb.InterruptReport)) continue;
|
||||
|
||||
const message = std.mem.bytesToValue(usb.InterruptReport, receive[0..@sizeOf(usb.InterruptReport)]);
|
||||
if (message.length < @sizeOf(hid.KeyboardReport)) continue;
|
||||
const report = std.mem.bytesToValue(hid.KeyboardReport, message.data[0..@sizeOf(hid.KeyboardReport)]);
|
||||
const transitions = decoder.feed(report);
|
||||
|
||||
@@ -14,7 +14,7 @@ const std = @import("std");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
const input = @import("input-client");
|
||||
const input = @import("input");
|
||||
const device_manager = @import("driver");
|
||||
const logging = @import("logging");
|
||||
const usb = @import("usb");
|
||||
@@ -74,10 +74,9 @@ pub fn main(init: process.Init) void {
|
||||
if (signals.has(.terminate)) return;
|
||||
continue;
|
||||
}
|
||||
if (!got.isMessage()) continue;
|
||||
// An `interrupt_report` event packet: the verb in its folded header, the
|
||||
// report after it. Anything else on this endpoint is not ours.
|
||||
const message = usb.reportOf(receive[0..got.len]) orelse continue;
|
||||
if (!got.isMessage() or got.len < @sizeOf(usb.InterruptReport)) continue;
|
||||
|
||||
const message = std.mem.bytesToValue(usb.InterruptReport, receive[0..@sizeOf(usb.InterruptReport)]);
|
||||
const length = @min(message.length, message.data.len);
|
||||
const report = hid.parseMouse(message.data[0..length]) orelse continue;
|
||||
const mask = buttonMask(report.buttons);
|
||||
|
||||
@@ -1,33 +0,0 @@
|
||||
//! The usb-storage driver as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! build-support resolves each name from the domains this zon declares.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "usb-storage",
|
||||
.root_source_file = b.path("usb-storage.zig"),
|
||||
.imports = &.{
|
||||
"block-protocol", "driver", "envelope", "ipc", "logging", "memory", "process",
|
||||
"service", "time", "usb",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
|
||||
// Standalone `zig build test`; the root aggregate depends on this step.
|
||||
const test_step = b.step("test", "Run the usb-storage unit tests");
|
||||
for ([_][]const u8{
|
||||
"bulk-only-transport.zig", // CBW/CSW wrapper sizes
|
||||
"scsi.zig", // SCSI CDB encodings (big-endian)
|
||||
}) |test_root| {
|
||||
const unit_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(test_root),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(unit_tests).step);
|
||||
}
|
||||
}
|
||||
@@ -1,16 +0,0 @@
|
||||
.{
|
||||
.name = .usb_storage,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0xce09fdc4c50bb4fe, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.device = .{ .path = "../../../library/device" },
|
||||
.protocol = .{ .path = "../../../library/protocol" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -22,16 +22,8 @@ const logging = @import("logging");
|
||||
const usb = @import("usb");
|
||||
const scsi = @import("scsi.zig");
|
||||
const bot = @import("bulk-only-transport.zig");
|
||||
const envelope = @import("envelope");
|
||||
const block_protocol = @import("block-protocol");
|
||||
|
||||
/// The generated block dispatch. One device per process, so the handler context
|
||||
/// is empty and the geometry stays in this file's globals.
|
||||
const Serve = block_protocol.Protocol.Provider(void);
|
||||
|
||||
const Invocation = envelope.Invocation;
|
||||
const Answer = envelope.Answer;
|
||||
|
||||
var device_id: u64 = 0;
|
||||
var device: usb.Device = undefined;
|
||||
var bulk_in: usb.Endpoint = undefined;
|
||||
@@ -152,65 +144,52 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
return true;
|
||||
}
|
||||
|
||||
// --- serving the block protocol ---------------------------------------------
|
||||
//
|
||||
// Geometry, and whole-block read/write to and from the caller's DMA buffer
|
||||
// (named by physical address). One device per process, so `Header.target` is
|
||||
// always 0 and no handler reads it.
|
||||
|
||||
/// A transfer the device refused. Every failure here is the same one — the SCSI
|
||||
/// command did not complete — so there is one errno for all of them.
|
||||
const refused: isize = -envelope.ENOENT;
|
||||
|
||||
fn onGeometry(_: void, _: Invocation(void), answer: Answer(block_protocol.Geometry)) isize {
|
||||
answer.set(.{ .block_size = block_size, .block_count = block_count });
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onRead(_: void, invocation: Invocation(block_protocol.Transfer), answer: Answer(block_protocol.Transferred)) isize {
|
||||
const request = invocation.request;
|
||||
const cdb = scsi.read10(@intCast(request.lba), @intCast(request.count));
|
||||
if (!transact(&cdb, true, request.physical, request.count * block_size)) return refused;
|
||||
answer.set(.{ .count = request.count });
|
||||
return 0;
|
||||
}
|
||||
|
||||
fn onWrite(_: void, invocation: Invocation(block_protocol.Transfer), answer: Answer(block_protocol.Transferred)) isize {
|
||||
const request = invocation.request;
|
||||
const cdb = scsi.write10(@intCast(request.lba), @intCast(request.count));
|
||||
if (!transact(&cdb, false, request.physical, request.count * block_size)) return refused;
|
||||
answer.set(.{ .count = request.count });
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// SYNCHRONIZE CACHE: commit the device's write cache to flash. No data stage.
|
||||
/// Makes prior writes durable before a caller (init at shutdown) cuts power. A
|
||||
/// device without a volatile cache reports success anyway.
|
||||
fn onFlush(_: void, _: Invocation(void), _: Answer(void)) isize {
|
||||
/// Serve the block protocol: geometry, and whole-block read/write to/from the
|
||||
/// caller's DMA buffer (named by physical address).
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||
_ = sender;
|
||||
if (message.len < block_protocol.request_size) return 0;
|
||||
const request = std.mem.bytesToValue(block_protocol.Request, message[0..block_protocol.request_size]);
|
||||
switch (request.operation) {
|
||||
@intFromEnum(block_protocol.Operation.attach) => {
|
||||
// The filesystem's DMA buffer: forward its capability to the controller so
|
||||
// the device can reach it, then release our copy (the binding holds a ref).
|
||||
const handle = capability orelse return writeReply(reply, .{ .status = -1, .block_size = 0, .block_count = 0 });
|
||||
const ok = device.attachDma(handle);
|
||||
_ = ipc.close(handle);
|
||||
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = 0, .block_count = 0 });
|
||||
},
|
||||
@intFromEnum(block_protocol.Operation.geometry) => {
|
||||
return writeReply(reply, .{ .status = 0, .block_size = block_size, .block_count = block_count });
|
||||
},
|
||||
@intFromEnum(block_protocol.Operation.read) => {
|
||||
const count: u16 = @intCast(request.count);
|
||||
const cdb = scsi.read10(@intCast(request.lba), count);
|
||||
const ok = transact(&cdb, true, request.physical, request.count * block_size);
|
||||
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = block_size, .block_count = if (ok) request.count else 0 });
|
||||
},
|
||||
@intFromEnum(block_protocol.Operation.write) => {
|
||||
const count: u16 = @intCast(request.count);
|
||||
const cdb = scsi.write10(@intCast(request.lba), count);
|
||||
const ok = transact(&cdb, false, request.physical, request.count * block_size);
|
||||
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = block_size, .block_count = if (ok) request.count else 0 });
|
||||
},
|
||||
@intFromEnum(block_protocol.Operation.flush) => {
|
||||
// SYNCHRONIZE CACHE: commit the device's write cache to flash. No data
|
||||
// stage. Makes prior writes durable before a caller (init at shutdown)
|
||||
// cuts power. A device without a volatile cache reports success anyway.
|
||||
const cdb = scsi.synchronizeCache10();
|
||||
return if (transact(&cdb, false, 0, 0)) 0 else refused;
|
||||
const ok = transact(&cdb, false, 0, 0);
|
||||
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = block_size, .block_count = 0 });
|
||||
},
|
||||
else => return 0,
|
||||
}
|
||||
}
|
||||
|
||||
/// The filesystem's DMA buffer: forward its capability to the controller so the
|
||||
/// device can reach it. Never claimed — the binding holds its own reference, so
|
||||
/// our copy is the turn's to close, on this path and on the refusal alike.
|
||||
fn onAttach(_: void, invocation: Invocation(void), _: Answer(void)) isize {
|
||||
const handle = invocation.capability orelse return -envelope.EPROTO;
|
||||
return if (device.attachDma(handle)) 0 else refused;
|
||||
}
|
||||
|
||||
const handlers = Serve.Handlers{
|
||||
.geometry = onGeometry,
|
||||
.read = onRead,
|
||||
.write = onWrite,
|
||||
.flush = onFlush,
|
||||
.attach = onAttach,
|
||||
};
|
||||
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, arrived: *ipc.Arrival) usize {
|
||||
// Peeked, never taken: `attach` forwards the capability and the controller's
|
||||
// binding takes its own reference, so this copy stays the turn's to close.
|
||||
return Serve.dispatch({}, handlers, message, sender, arrived.peek(), reply);
|
||||
fn writeReply(reply: []u8, value: block_protocol.Reply) usize {
|
||||
const bytes = std.mem.asBytes(&value);
|
||||
@memcpy(reply[0..bytes.len], bytes);
|
||||
return bytes.len;
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
@@ -223,7 +202,7 @@ pub fn main(init: process.Init) void {
|
||||
return;
|
||||
};
|
||||
service.run(block_protocol.message_maximum, .{
|
||||
.service = "block",
|
||||
.service = .block,
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
});
|
||||
|
||||
@@ -1,20 +0,0 @@
|
||||
//! The usb-xhci-bus driver as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! build-support resolves each name from the domains this zon declares.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "usb-xhci-bus",
|
||||
.root_source_file = b.path("usb-xhci-bus.zig"),
|
||||
.imports = &.{
|
||||
"channel", "device-manager-protocol", "driver", "envelope",
|
||||
"input-client", "ipc", "logging", "memory",
|
||||
"mmio", "pci", "process", "service",
|
||||
"time", "usb-abi", "usb-ids", "usb-transfer-protocol",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user