Compare commits
23
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
fa8203cdba | ||
|
|
e53d6ebafb | ||
|
|
4476208361 | ||
|
|
c621b649f6 | ||
|
|
d27670ec39 | ||
|
|
ab7594df6e | ||
|
|
d03942b543 | ||
|
|
3f9b6813f7 | ||
|
|
bc2eb67581 | ||
|
|
cb98a9844e | ||
|
|
4701fbd123 | ||
|
|
3b23b11b0e | ||
|
|
902e4a0a9e | ||
|
|
15575960bd | ||
|
|
6f4fdc2789 | ||
|
|
721288c516 | ||
|
|
4194bb6e32 | ||
|
|
fc0b934b7f | ||
|
|
6a687fbc2b | ||
|
|
4e7cbc9792 | ||
|
|
e94adcfc02 | ||
|
|
f477ef7d9f | ||
|
|
9e649178bf |
@@ -0,0 +1,173 @@
|
||||
//! The danos build API (docs/build-packages-plan.md): the one shared recipe
|
||||
//! for building a user-space binary. A binary package's build.zig names its
|
||||
//! binary and EXACTLY the modules its source imports — the moral equivalent
|
||||
//! of a C file's include list — and `userBinary` resolves each name from the
|
||||
//! library domain package that exports it. Nothing is pre-wired: an @import
|
||||
//! the package did not declare is a compile error, and a domain none of the
|
||||
//! imports come from never appears in the package's manifest. The only
|
||||
//! implicit dependency is the kernel package, because the shared root shim
|
||||
//! (root.zig, user.ld) lives there and itself reaches start + logging.
|
||||
//!
|
||||
//! Consumers declare this package in their build.zig.zon (as "build-support")
|
||||
//! and @import its build.zig from their own build.zig; nothing is compiled
|
||||
//! from this package itself — it exports build-time functions only.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
_ = b; // nothing to build: this package exports build-time functions only
|
||||
}
|
||||
|
||||
/// The freestanding x86-64 target every danos binary (kernel and user) is
|
||||
/// built for. SSE2 is part of the x86_64 baseline and UEFI leaves it enabled
|
||||
/// at handoff, so we keep it: disabling it forces soft-float and makes the
|
||||
/// compiler unable to encode the vector ops that std's formatting/runtime
|
||||
/// still emit.
|
||||
pub fn freestandingTarget(b: *std.Build) std.Build.ResolvedTarget {
|
||||
return b.resolveTargetQuery(.{
|
||||
.cpu_arch = .x86_64,
|
||||
.os_tag = .freestanding,
|
||||
.abi = .none,
|
||||
});
|
||||
}
|
||||
|
||||
/// Which library domain package exports each importable module — the one
|
||||
/// name -> home table. When a domain grows a module, it gets a row here; a
|
||||
/// binary naming a module whose home is missing from its own build.zig.zon
|
||||
/// fails loudly at dependency resolution.
|
||||
const ModuleHome = struct { name: []const u8, home: []const u8 };
|
||||
const module_homes = [_]ModuleHome{
|
||||
// library/kernel — the userspace private-ABI library, split by concern.
|
||||
.{ .name = "abi", .home = "kernel" },
|
||||
.{ .name = "system-call", .home = "kernel" },
|
||||
.{ .name = "ipc", .home = "kernel" },
|
||||
.{ .name = "time", .home = "kernel" },
|
||||
.{ .name = "thread", .home = "kernel" },
|
||||
.{ .name = "logging", .home = "kernel" },
|
||||
.{ .name = "process", .home = "kernel" },
|
||||
.{ .name = "file-system", .home = "kernel" },
|
||||
.{ .name = "memory", .home = "kernel" },
|
||||
.{ .name = "service", .home = "kernel" },
|
||||
.{ .name = "start", .home = "kernel" },
|
||||
// library/device — driver-side libraries + the flat reference data.
|
||||
.{ .name = "mmio", .home = "device" },
|
||||
.{ .name = "acpi-ids", .home = "device" },
|
||||
.{ .name = "device-abi", .home = "device" },
|
||||
.{ .name = "aml", .home = "device" },
|
||||
.{ .name = "usb-abi", .home = "device" },
|
||||
.{ .name = "usb-ids", .home = "device" },
|
||||
.{ .name = "usb", .home = "device" },
|
||||
.{ .name = "driver", .home = "device" },
|
||||
.{ .name = "block", .home = "device" },
|
||||
.{ .name = "pci", .home = "device" },
|
||||
.{ .name = "pci-class", .home = "device" },
|
||||
.{ .name = "device-registry", .home = "device" },
|
||||
// library/client — userspace service clients.
|
||||
.{ .name = "display-client", .home = "client" },
|
||||
.{ .name = "input-client", .home = "client" },
|
||||
// library/protocol — the wire protocols.
|
||||
.{ .name = "vfs-protocol", .home = "protocol" },
|
||||
.{ .name = "input-protocol", .home = "protocol" },
|
||||
.{ .name = "block-protocol", .home = "protocol" },
|
||||
.{ .name = "usb-transfer-protocol", .home = "protocol" },
|
||||
.{ .name = "device-manager-protocol", .home = "protocol" },
|
||||
.{ .name = "display-protocol", .home = "protocol" },
|
||||
.{ .name = "scanout-protocol", .home = "protocol" },
|
||||
.{ .name = "power-protocol", .home = "protocol" },
|
||||
// library/csv — the /etc/*.csv helpers.
|
||||
.{ .name = "csv", .home = "csv" },
|
||||
// library/xkeyboard-config — keycode -> keysym/character tables.
|
||||
.{ .name = "xkeyboard-config", .home = "xkeyboard-config" },
|
||||
};
|
||||
|
||||
fn moduleHome(name: []const u8) ?[]const u8 {
|
||||
for (module_homes) |entry| {
|
||||
if (std.mem.eql(u8, entry.name, name)) return entry.home;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// What `userBinary` needs to know about one user binary.
|
||||
pub const UserBinaryOptions = struct {
|
||||
name: []const u8,
|
||||
/// The program's own source file — it becomes the `program` module the
|
||||
/// root shim imports; a program only defines `pub fn main`.
|
||||
root_source_file: std.Build.LazyPath,
|
||||
/// Exactly the modules the program's source @imports (directly or through
|
||||
/// its same-directory files) — no more, no less. Order is free; sorted
|
||||
/// reads best. An undeclared @import fails the compile; a declared name no
|
||||
/// domain exports fails the build graph with a pointer to module_homes.
|
||||
imports: []const []const u8,
|
||||
/// Built multi-threaded (`single_threaded = false`) so real atomics/TLS
|
||||
/// work — required before a binary may call `Thread.spawn`
|
||||
/// (docs/threading.md). Threads are a deliberate per-binary opt-in.
|
||||
threaded: bool = false,
|
||||
};
|
||||
|
||||
/// Build one user-space binary the same way for every program (init, the
|
||||
/// services, the drivers): freestanding, ReleaseSmall, `.large` code model
|
||||
/// (the image base is above 4 GiB — smaller models emit 32-bit relocations
|
||||
/// that can't reach), linked with the shared user link script. Pinned to
|
||||
/// LLVM + LLD so the script's PHDRS (segment permissions) are authoritative —
|
||||
/// the kernel's W^X user-ELF loader requires exact perms.
|
||||
///
|
||||
/// The compilation root is not the program's own file but the shared shim
|
||||
/// (the kernel package's root.zig), which supplies the root declarations
|
||||
/// (`main` re-export, panic handler, `_start` pull) so a program only defines
|
||||
/// `pub fn main`. The program's file becomes the `program` module the shim
|
||||
/// imports; reach it through `programModule` to add per-binary non-library
|
||||
/// modules (compile-time options).
|
||||
pub fn userBinary(b: *std.Build, options: UserBinaryOptions) *std.Build.Step.Compile {
|
||||
const kernel = b.dependency("kernel", .{});
|
||||
var imports: std.ArrayListUnmanaged(std.Build.Module.Import) = .empty;
|
||||
for (options.imports) |name| {
|
||||
const home = moduleHome(name) orelse @panic(b.fmt(
|
||||
"no library domain exports a module named '{s}' — if a domain grew it, add its row to module_homes in build-support/build.zig",
|
||||
.{name},
|
||||
));
|
||||
const dependency = if (std.mem.eql(u8, home, "kernel")) kernel else b.dependency(home, .{});
|
||||
imports.append(b.allocator, .{ .name = name, .module = dependency.module(name) }) catch @panic("OOM");
|
||||
}
|
||||
// Settings (target, optimize, code model, ...) live on the root module
|
||||
// only; the program module inherits them.
|
||||
const program_module = b.createModule(.{
|
||||
.root_source_file = options.root_source_file,
|
||||
.imports = imports.items,
|
||||
});
|
||||
const exe = b.addExecutable(.{
|
||||
.name = options.name,
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = kernel.path("root.zig"),
|
||||
.target = freestandingTarget(b),
|
||||
.optimize = .ReleaseSmall,
|
||||
.code_model = .large,
|
||||
.single_threaded = !options.threaded, // a threaded binary needs real atomics/TLS
|
||||
.sanitize_c = .off,
|
||||
.stack_check = false,
|
||||
.stack_protector = false,
|
||||
// The root shim itself imports only start (_start + panic) and
|
||||
// logging (std_options) — straight from the kernel package, so a
|
||||
// program's own import list stays exactly its own.
|
||||
.imports = &.{
|
||||
.{ .name = "start", .module = kernel.module("start") },
|
||||
.{ .name = "logging", .module = kernel.module("logging") },
|
||||
.{ .name = "program", .module = program_module },
|
||||
},
|
||||
}),
|
||||
});
|
||||
exe.setLinkerScript(kernel.path("user.ld"));
|
||||
exe.entry = .{ .symbol_name = "_start" };
|
||||
exe.image_base = 0x7000_0000_0000;
|
||||
exe.use_llvm = true;
|
||||
exe.use_lld = true;
|
||||
return exe;
|
||||
}
|
||||
|
||||
/// The `program` module of a binary built by `userBinary` — the module rooted
|
||||
/// at the program's own source file. Per-binary non-library modules (an
|
||||
/// addOptions build_options) go here, not on the root shim: module imports
|
||||
/// are not transitive, so an import added to the root would be invisible to
|
||||
/// the program's code.
|
||||
pub fn programModule(exe: *std.Build.Step.Compile) *std.Build.Module {
|
||||
return exe.root_module.import_table.get("program").?;
|
||||
}
|
||||
@@ -0,0 +1,8 @@
|
||||
.{
|
||||
.name = .build_support,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0xad91962994f4be41, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{},
|
||||
.paths = .{""},
|
||||
}
|
||||
+43
-8
@@ -32,6 +32,48 @@
|
||||
// Once all dependencies are fetched, `zig build` no longer requires
|
||||
// internet connectivity.
|
||||
.dependencies = .{
|
||||
// The danos build API — the shared user-binary recipe every build file
|
||||
// (root and per-binary packages) consumes (docs/build-packages-plan.md).
|
||||
.@"build-support" = .{ .path = "build-support" },
|
||||
// The library domains, each a package exporting its modules.
|
||||
.kernel = .{ .path = "library/kernel" },
|
||||
.device = .{ .path = "library/device" },
|
||||
.client = .{ .path = "library/client" },
|
||||
.protocol = .{ .path = "library/protocol" },
|
||||
.csv = .{ .path = "library/csv" },
|
||||
.@"xkeyboard-config" = .{ .path = "library/xkeyboard-config" },
|
||||
// Binary packages (phase 2), consumed as artifacts for the boot image.
|
||||
.@"pci-bus" = .{ .path = "system/drivers/pci-bus" },
|
||||
.init = .{ .path = "system/services/init" },
|
||||
.fat = .{ .path = "system/services/fat" },
|
||||
.display = .{ .path = "system/services/display" },
|
||||
.@"display-demo" = .{ .path = "system/services/display-demo" },
|
||||
.@"device-manager" = .{ .path = "system/services/device-manager" },
|
||||
.input = .{ .path = "system/services/input" },
|
||||
.logger = .{ .path = "system/services/logger" },
|
||||
// The discovery pair and the /test fixtures are lazy: only what a
|
||||
// given build actually ships gets its build file loaded and compiled
|
||||
// (-Ddiscovery picks one of the pair; -Dtest-case pulls the fixtures).
|
||||
.acpi = .{ .path = "system/services/acpi", .lazy = true },
|
||||
.fdt = .{ .path = "system/services/fdt", .lazy = true },
|
||||
.@"ps2-bus" = .{ .path = "system/drivers/ps2-bus" },
|
||||
.@"usb-xhci-bus" = .{ .path = "system/drivers/usb-xhci-bus" },
|
||||
.@"usb-hid" = .{ .path = "system/drivers/usb-hid" },
|
||||
.@"usb-storage" = .{ .path = "system/drivers/usb-storage" },
|
||||
.@"virtio-gpu" = .{ .path = "system/drivers/virtio-gpu" },
|
||||
.@"vfs-test" = .{ .path = "test/system/services/vfs-test", .lazy = true },
|
||||
.@"fat-test" = .{ .path = "test/system/services/fat-test", .lazy = true },
|
||||
.@"shared-memory-server" = .{ .path = "test/system/services/shared-memory-server", .lazy = true },
|
||||
.@"shared-memory-client" = .{ .path = "test/system/services/shared-memory-client", .lazy = true },
|
||||
.@"crash-test" = .{ .path = "test/system/services/crash-test", .lazy = true },
|
||||
.@"device-list" = .{ .path = "test/system/services/device-list", .lazy = true },
|
||||
.@"pci-cap-test" = .{ .path = "test/system/services/pci-cap-test", .lazy = true },
|
||||
.@"iommu-fault-test" = .{ .path = "test/system/services/iommu-fault-test", .lazy = true },
|
||||
.@"input-source" = .{ .path = "test/system/services/input-source", .lazy = true },
|
||||
.@"input-test" = .{ .path = "test/system/services/input-test", .lazy = true },
|
||||
.@"args-echo" = .{ .path = "test/system/services/args-echo", .lazy = true },
|
||||
.@"process-test" = .{ .path = "test/system/services/process-test", .lazy = true },
|
||||
.@"thread-test" = .{ .path = "test/system/services/thread-test", .lazy = true },
|
||||
// See `zig fetch --save <url>` for a command-line interface for adding dependencies.
|
||||
//.example = .{
|
||||
// // When updating this field to a new URL, be sure to delete the corresponding
|
||||
@@ -70,12 +112,5 @@
|
||||
// Paths are relative to the build root. Use the empty string (`""`) to refer to
|
||||
// the build root itself.
|
||||
// A directory listed here means that all files within, recursively, are included.
|
||||
.paths = .{
|
||||
"build.zig",
|
||||
"build.zig.zon",
|
||||
"src",
|
||||
// For example...
|
||||
//"LICENSE",
|
||||
//"README.md",
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
|
||||
@@ -0,0 +1,166 @@
|
||||
//! Boot-image assembly (docs/build-packages-plan.md, phase 3): everything
|
||||
//! between "here are the built binaries" and "here is a bootable volume".
|
||||
//! The FHS-shaped zig-out install tree, the boot manifest, the boot capsule,
|
||||
//! the FAT32 USB image (+ its serial-enabled twin for the QEMU run steps),
|
||||
//! and the release ISO — with their check steps. The root build.zig decides
|
||||
//! WHAT ships (the bundled list); this file owns HOW it becomes an image.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// One user binary and its FHS home on the boot volume (and in zig-out).
|
||||
pub const BundledBinary = struct { path: []const u8, binary: std.Build.LazyPath };
|
||||
|
||||
pub const Options = struct {
|
||||
/// The installed/flashable kernel (serial follows the root -Dserial).
|
||||
kernel: *std.Build.Step.Compile,
|
||||
/// The serial-enabled kernel variant the `run-x86-64` image boots.
|
||||
kernel_serial: *std.Build.Step.Compile,
|
||||
/// The UEFI loader (BOOTX64).
|
||||
efi: *std.Build.Step.Compile,
|
||||
/// Every user binary and data file at its FHS path.
|
||||
bundled: []const BundledBinary,
|
||||
};
|
||||
|
||||
/// Wire up the install tree, both FAT32 boot images, the release ISO, and the
|
||||
/// check steps. Returns the serial-enabled FAT image for the QEMU run steps.
|
||||
pub fn addImageSteps(b: *std.Build, options: Options) std.Build.LazyPath {
|
||||
// Everything installs into a FHS-shaped zig-out: it IS the danos filesystem *and*
|
||||
// the boot volume. Each binary lands at its addressed, leaf-collapsed path — the
|
||||
// kernel at zig-out/system/kernel (from system/kernel/kernel.zig), init at
|
||||
// zig-out/system/services/init, and so on (see docs/README.md). The bootloader
|
||||
// then loads these FHS paths off the volume.
|
||||
const kernel_install = b.addInstallArtifact(options.kernel, .{ .dest_dir = .{ .override = .{ .custom = "system" } } });
|
||||
b.getInstallStep().dependOn(&kernel_install.step);
|
||||
|
||||
// UEFI firmware requires the removable-media loader at exactly \EFI\BOOT\BOOTX64.efi,
|
||||
// so that path is fixed by the firmware (it is /boot's EFI stub, conceptually).
|
||||
const efi_install = b.addInstallArtifact(options.efi, .{ .dest_dir = .{ .override = .{ .custom = "EFI/BOOT" } } });
|
||||
b.getInstallStep().dependOn(&efi_install.step);
|
||||
|
||||
// The boot manifest: the FHS path of every bundled binary, one per line. The
|
||||
// EFI loader reads THIS by name and opens each listed path by name — FAT
|
||||
// name lookup is case-insensitive and firmware-portable, unlike directory
|
||||
// ENUMERATION, whose returned names vary by firmware (bare 8.3 entries come
|
||||
// back uppercase on some FAT drivers). The tree walk remains only as the
|
||||
// loader's fallback for hand-assembled sticks without a manifest.
|
||||
var manifest_text: std.ArrayListUnmanaged(u8) = .empty;
|
||||
for (options.bundled) |item| {
|
||||
manifest_text.append(b.allocator, '/') catch @panic("OOM");
|
||||
manifest_text.appendSlice(b.allocator, item.path) catch @panic("OOM");
|
||||
manifest_text.append(b.allocator, '\n') catch @panic("OOM");
|
||||
}
|
||||
const manifest_files = b.addWriteFiles();
|
||||
const manifest_file = manifest_files.add("manifest", manifest_text.items);
|
||||
const manifest_install = b.addInstallFileWithDir(manifest_file, .prefix, "system/manifest");
|
||||
b.getInstallStep().dependOn(&manifest_install.step);
|
||||
|
||||
// The boot capsule: the same bundled list packed into ONE file (v2
|
||||
// initial_ramdisk format), because a single open + sequential read is the
|
||||
// only firmware file I/O shape that is fast everywhere — a per-file tree
|
||||
// walk measured MINUTES on real firmware. The loader tries this first,
|
||||
// then the manifest, then the walk; the running system cannot tell the
|
||||
// difference (it always receives the same in-RAM table). Derived from the
|
||||
// tree in the same build graph, so the two cannot drift.
|
||||
const mk_capsule = b.addSystemCommand(&.{"python3"});
|
||||
mk_capsule.addFileArg(b.path("tools/pack-system-image.py"));
|
||||
const capsule_img = mk_capsule.addOutputFileArg("system.img");
|
||||
for (options.bundled) |item| {
|
||||
mk_capsule.addArg(item.path);
|
||||
mk_capsule.addFileArg(item.binary);
|
||||
}
|
||||
const capsule_install = b.addInstallFile(capsule_img, "boot/system.img");
|
||||
b.getInstallStep().dependOn(&capsule_install.step);
|
||||
|
||||
// Install every bundled binary to its FHS home, so zig-out is a true image of
|
||||
// the filesystem — the same tree make-fat-image.py lays out on the boot volume.
|
||||
for (options.bundled) |item| {
|
||||
const install = b.addInstallFileWithDir(item.binary, .prefix, item.path);
|
||||
b.getInstallStep().dependOn(&install.step);
|
||||
}
|
||||
|
||||
// --- danos-usb.img: the bootable FAT32 USB image ---
|
||||
// Format a real FAT32 image (the in-repo Python builder, no external tools)
|
||||
// holding the EFI stub, the kernel, and the whole /system tree of user
|
||||
// binaries at their FHS paths. QEMU presents this image as a USB mass-storage
|
||||
// device the guest boots from (see run-x86-64 and the test harness), and the
|
||||
// danos fat driver mounts the same image at /mnt/usb.
|
||||
const fat_image = addBootImage(b, options.kernel.getEmittedBin(), options.efi.getEmittedBin(), manifest_file, capsule_img, options.bundled);
|
||||
const fat_image_install = b.addInstallFile(fat_image, "danos-usb.img");
|
||||
b.getInstallStep().dependOn(&fat_image_install.step);
|
||||
|
||||
// The image `run-x86-64` boots: identical to the flashable one but with the
|
||||
// serial log sink compiled in, so a developer always gets the machine-readable
|
||||
// log captured to serial0 — without baking serial into the image users flash.
|
||||
// Built lazily (only when `run-x86-64` is requested), and never installed.
|
||||
const fat_image_serial = addBootImage(b, options.kernel_serial.getEmittedBin(), options.efi.getEmittedBin(), manifest_file, capsule_img, options.bundled);
|
||||
|
||||
// `zig build check-fat-image` — validate the produced image is a real FAT32
|
||||
// with the EFI stub present (the builder's own --verify, no external tools).
|
||||
const check_fat = b.addSystemCommand(&.{"python3"});
|
||||
check_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||
check_fat.addArg("--verify");
|
||||
check_fat.addFileArg(fat_image);
|
||||
const check_fat_step = b.step("check-fat-image", "Verify the FAT32 USB image is valid and bootable");
|
||||
check_fat_step.dependOn(&check_fat.step);
|
||||
|
||||
// --- release-x86-64: danos-x86-64.iso, the flashable release image ---
|
||||
// Wrap the FAT32 boot volume in a hybrid ISO (the in-repo Python builder
|
||||
// again, no xorriso/isohybrid): an ISO9660 whose El Torito EFI boot entry
|
||||
// and MBR ESP partition entry both point at the embedded FAT image. One
|
||||
// file then boots every way release media is consumed — flashed raw to a
|
||||
// USB stick with Etcher or dd, or burned to optical media — while
|
||||
// danos-usb.img stays the raw superfloppy QEMU and the test harness boot.
|
||||
const mk_iso = b.addSystemCommand(&.{"python3"});
|
||||
mk_iso.addFileArg(b.path("tools/make-iso-image.py"));
|
||||
const iso_image = mk_iso.addOutputFileArg("danos-x86-64.iso");
|
||||
mk_iso.addFileArg(fat_image);
|
||||
const iso_install = b.addInstallFile(iso_image, "danos-x86-64.iso");
|
||||
const release_step = b.step("release-x86-64", "Build the flashable x86-64 release ISO (zig-out/danos-x86-64.iso; flash with Etcher or dd)");
|
||||
release_step.dependOn(&iso_install.step);
|
||||
|
||||
// `zig build check-iso-image` — the ISO builder's own --verify (mirroring
|
||||
// check-fat-image): the MBR partition, the El Torito catalog, and the
|
||||
// embedded FAT32 image must all agree.
|
||||
const check_iso = b.addSystemCommand(&.{"python3"});
|
||||
check_iso.addFileArg(b.path("tools/make-iso-image.py"));
|
||||
check_iso.addArg("--verify");
|
||||
check_iso.addFileArg(iso_image);
|
||||
const check_iso_step = b.step("check-iso-image", "Verify the release ISO is a valid hybrid (MBR ESP partition + El Torito EFI entry)");
|
||||
check_iso_step.dependOn(&check_iso.step);
|
||||
|
||||
return fat_image_serial;
|
||||
}
|
||||
|
||||
/// Assemble the bootable FAT32 image (the in-repo Python builder) holding the
|
||||
/// EFI stub, the kernel, and every user binary at its FHS path — the volume's
|
||||
/// /system tree IS the system image; the EFI loader walks it at boot and builds
|
||||
/// the in-RAM initial_ramdisk from it. Factored so the serial-enabled
|
||||
/// `run-x86-64` variant can bundle its own serial kernel while sharing the
|
||||
/// loader and user tree (the loader's boot breadcrumbs and init's heartbeat both
|
||||
/// follow the top-level -Dserial). Returns the image's LazyPath.
|
||||
fn addBootImage(
|
||||
b: *std.Build,
|
||||
kernel_bin: std.Build.LazyPath,
|
||||
efi_bin: std.Build.LazyPath,
|
||||
manifest: std.Build.LazyPath,
|
||||
capsule: std.Build.LazyPath,
|
||||
bundled: []const BundledBinary,
|
||||
) std.Build.LazyPath {
|
||||
const mk_fat = b.addSystemCommand(&.{"python3"});
|
||||
mk_fat.addFileArg(b.path("tools/make-fat-image.py"));
|
||||
const fat_image = mk_fat.addOutputFileArg("danos-usb.img");
|
||||
mk_fat.addArg("64"); // MiB
|
||||
mk_fat.addArg("EFI/BOOT/BOOTX64.efi");
|
||||
mk_fat.addFileArg(efi_bin);
|
||||
mk_fat.addArg("system/kernel");
|
||||
mk_fat.addFileArg(kernel_bin);
|
||||
mk_fat.addArg("system/manifest");
|
||||
mk_fat.addFileArg(manifest);
|
||||
mk_fat.addArg("boot/system.img");
|
||||
mk_fat.addFileArg(capsule);
|
||||
for (bundled) |item| {
|
||||
mk_fat.addArg(item.path);
|
||||
mk_fat.addFileArg(item.binary);
|
||||
}
|
||||
return fat_image;
|
||||
}
|
||||
+176
@@ -0,0 +1,176 @@
|
||||
//! The QEMU run steps (docs/build-packages-plan.md, phase 3): `run-x86-64`
|
||||
//! boots the serial-enabled FAT image via UEFI/OVMF; `run-x86-64-gpu` adds a
|
||||
//! virtio-gpu adapter for the native-present display path. OVMF firmware is
|
||||
//! probed across distro/OS layouts (-Dovmf-code / -Dovmf-vars override).
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// Wire up the `run-x86-64` and `run-x86-64-gpu` steps around the given
|
||||
/// serial-enabled boot image (the guest boots that self-contained image
|
||||
/// attached as USB storage, not the installed FHS zig-out).
|
||||
pub fn addRunSteps(b: *std.Build, fat_image_serial: std.Build.LazyPath) void {
|
||||
// Firmware lives in different places per OS/distro, so probe the known
|
||||
// layouts (Architecture, Debian/Ubuntu, Fedora, macOS Homebrew) and use the first
|
||||
// that exists. Override with -Dovmf-code / -Dovmf-vars if yours is elsewhere.
|
||||
const ovmf_code = b.option(
|
||||
[]const u8,
|
||||
"ovmf-code",
|
||||
"Path to the OVMF_CODE firmware image",
|
||||
) orelse firstExisting(b.graph.io, &.{
|
||||
"/usr/share/edk2/x64/OVMF_CODE.4m.fd", // Architecture
|
||||
"/usr/share/OVMF/OVMF_CODE_4M.fd", // Debian/Ubuntu
|
||||
"/usr/share/OVMF/OVMF_CODE.fd", // older Debian/Ubuntu
|
||||
"/usr/share/edk2-ovmf/x64/OVMF_CODE.fd", // Fedora
|
||||
"/opt/homebrew/share/qemu/edk2-x86_64-code.fd", // macOS Homebrew (Apple Silicon)
|
||||
"/usr/local/share/qemu/edk2-x86_64-code.fd", // macOS Homebrew (Intel)
|
||||
});
|
||||
const ovmf_vars = b.option(
|
||||
[]const u8,
|
||||
"ovmf-vars",
|
||||
"Path to the OVMF_VARS firmware image (a writable copy is made)",
|
||||
) orelse firstExisting(b.graph.io, &.{
|
||||
"/usr/share/edk2/x64/OVMF_VARS.4m.fd", // Architecture
|
||||
"/usr/share/OVMF/OVMF_VARS_4M.fd", // Debian/Ubuntu
|
||||
"/usr/share/OVMF/OVMF_VARS.fd", // older Debian/Ubuntu
|
||||
"/usr/share/edk2-ovmf/x64/OVMF_VARS.fd", // Fedora
|
||||
"/opt/homebrew/share/qemu/edk2-i386-vars.fd", // macOS Homebrew (Apple Silicon)
|
||||
"/usr/local/share/qemu/edk2-i386-vars.fd", // macOS Homebrew (Intel)
|
||||
});
|
||||
|
||||
// The firmware needs to write NVRAM, so give it a writable copy of the vars.
|
||||
const vars_copy = b.addSystemCommand(&.{ "cp", "-f", ovmf_vars });
|
||||
const vars_out = vars_copy.addOutputFileArg("OVMF_VARS.4m.fd");
|
||||
|
||||
// Capture the guest's serial0 (danos's machine-readable log) to the qemu-test
|
||||
// scratch area — a dev/host artifact, kept out of the FHS boot volume we mount.
|
||||
// (/var/log/system is reserved for the kernel's own logging system later.) One
|
||||
// timestamped file per run.
|
||||
const log_dir = b.fmt("{s}/qemu-test", .{b.install_path});
|
||||
const make_log_dir = b.addSystemCommand(&.{ "mkdir", "-p", log_dir });
|
||||
|
||||
// --- run-x86-64: boot the x86-64 kernel in QEMU via UEFI/OVMF ---
|
||||
const run_efi = b.addSystemCommand(&.{
|
||||
"qemu-system-x86_64",
|
||||
"-device",
|
||||
"qemu-xhci,id=xhci",
|
||||
"-device",
|
||||
"usb-mouse,bus=xhci.0",
|
||||
"-device",
|
||||
"usb-kbd,bus=xhci.0",
|
||||
"-machine",
|
||||
"q35",
|
||||
"-m",
|
||||
"128M",
|
||||
"-drive",
|
||||
b.fmt("if=pflash,format=raw,readonly=on,file={s}", .{ovmf_code}),
|
||||
});
|
||||
run_efi.addArg("-drive");
|
||||
run_efi.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
||||
// Boot off the FAT32 USB image: a mass-storage device on the same xHCI bus as
|
||||
// the keyboard and mouse. OVMF finds \EFI\BOOT\BOOTX64.efi on it and boots.
|
||||
// The serial-enabled variant, so serial0 carries the log for this dev boot.
|
||||
run_efi.addArg("-drive");
|
||||
run_efi.addPrefixedFileArg("if=none,id=bootusb,format=raw,file=", fat_image_serial);
|
||||
run_efi.addArgs(&.{
|
||||
"-device",
|
||||
"usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
||||
"-net",
|
||||
"none",
|
||||
// Emulated display advertising 1280x720 as its native (EDID preferred)
|
||||
// resolution, so the kernel's native-resolution switch has something to
|
||||
// find. `-vga none` avoids a second, default adapter.
|
||||
"-vga",
|
||||
"none",
|
||||
"-device",
|
||||
"VGA,edid=on,xres=1280,yres=720",
|
||||
});
|
||||
const serial_log = b.fmt("{s}/run-x86-64-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
||||
run_efi.addArgs(&.{ "-serial", b.fmt("file:{s}", .{serial_log}) });
|
||||
// We boot the self-contained `fat_image_serial` (added as a file arg above, so
|
||||
// it's already a dependency) — not the installed FHS zig-out — so `run-x86-64`
|
||||
// builds only the serial kernel, never the flashable one. Just make the serial
|
||||
// scratch dir first.
|
||||
run_efi.step.dependOn(&make_log_dir.step);
|
||||
|
||||
const run_efi_step = b.step("run-x86-64", "Boot the x86-64 kernel in QEMU (UEFI/OVMF); serial0 is logged to zig-out/qemu-test/run-x86-64-serial0-<timestamp>.log");
|
||||
run_efi_step.dependOn(&run_efi.step);
|
||||
|
||||
// --- run-x86-64-gpu: the same boot plus a virtio-gpu adapter ---
|
||||
// The VGA device still supplies the boot (GOP) framebuffer the compositor starts
|
||||
// on; the virtio-gpu function is discovered by the device-manager stack, its
|
||||
// driver announces a shared scanout, and the compositor upgrades off the GOP
|
||||
// floor to fenced, tear-free native presents (docs/display-v2.md).
|
||||
// This is the interactive twin of the `display-native` test case, and 512M
|
||||
// matches it (the whole driver stack + the compositor's surfaces at once).
|
||||
// QEMU shows one head per adapter: pick the virtio-gpu head in the View menu
|
||||
// to watch the native output.
|
||||
const run_gpu = b.addSystemCommand(&.{
|
||||
"qemu-system-x86_64",
|
||||
"-device",
|
||||
"qemu-xhci,id=xhci",
|
||||
"-device",
|
||||
"usb-mouse,bus=xhci.0",
|
||||
"-device",
|
||||
"usb-kbd,bus=xhci.0",
|
||||
"-machine",
|
||||
"q35",
|
||||
"-m",
|
||||
"512M",
|
||||
"-drive",
|
||||
b.fmt("if=pflash,format=raw,readonly=on,file={s}", .{ovmf_code}),
|
||||
});
|
||||
run_gpu.addArg("-drive");
|
||||
run_gpu.addPrefixedFileArg("if=pflash,format=raw,file=", vars_out);
|
||||
run_gpu.addArg("-drive");
|
||||
run_gpu.addPrefixedFileArg("if=none,id=bootusb,format=raw,file=", fat_image_serial);
|
||||
run_gpu.addArgs(&.{
|
||||
"-device",
|
||||
"usb-storage,bus=xhci.0,drive=bootusb,removable=on,bootindex=0",
|
||||
"-net",
|
||||
"none",
|
||||
"-vga",
|
||||
"none",
|
||||
"-device",
|
||||
"VGA,edid=on,xres=1280,yres=720",
|
||||
"-device",
|
||||
"virtio-gpu-pci",
|
||||
});
|
||||
const gpu_serial_log = b.fmt("{s}/run-x86-64-gpu-serial0-{s}.log", .{ log_dir, timestamp(b) });
|
||||
run_gpu.addArgs(&.{ "-serial", b.fmt("file:{s}", .{gpu_serial_log}) });
|
||||
run_gpu.step.dependOn(&make_log_dir.step);
|
||||
|
||||
const run_gpu_step = b.step("run-x86-64-gpu", "Boot in QEMU with a virtio-gpu adapter: the compositor upgrades to fenced (tear-free) native presents; watch the virtio-gpu head in QEMU's View menu");
|
||||
run_gpu_step.dependOn(&run_gpu.step);
|
||||
}
|
||||
|
||||
/// Return the first path in `candidates` that exists on the build host, else the
|
||||
/// first candidate as a fallback so a missing-firmware error still names a
|
||||
/// concrete (and, by convention, the primary) path. Used to locate OVMF firmware
|
||||
/// across distro/OS layouts without configuration.
|
||||
fn firstExisting(io: std.Io, candidates: []const []const u8) []const u8 {
|
||||
for (candidates) |path| {
|
||||
std.Io.Dir.accessAbsolute(io, path, .{}) catch continue;
|
||||
return path;
|
||||
}
|
||||
return candidates[0];
|
||||
}
|
||||
|
||||
/// A UTC timestamp like "20260708-153045", for naming a per-run artifact so
|
||||
/// repeated runs don't clobber each other's logs. Resolved when `zig build`
|
||||
/// runs, which is moments before QEMU launches.
|
||||
fn timestamp(b: *std.Build) []const u8 {
|
||||
const ns = std.Io.Clock.now(.real, b.graph.io).nanoseconds;
|
||||
const secs: u64 = @intCast(@divFloor(ns, std.time.ns_per_s));
|
||||
const es = std.time.epoch.EpochSeconds{ .secs = secs };
|
||||
const yd = es.getEpochDay().calculateYearDay();
|
||||
const md = yd.calculateMonthDay();
|
||||
const ds = es.getDaySeconds();
|
||||
return b.fmt("{d:0>4}{d:0>2}{d:0>2}-{d:0>2}{d:0>2}{d:0>2}", .{
|
||||
yd.year,
|
||||
md.month.numeric(),
|
||||
@as(u32, md.day_index) + 1,
|
||||
ds.getHoursIntoDay(),
|
||||
ds.getMinutesIntoHour(),
|
||||
ds.getSecondsIntoMinute(),
|
||||
});
|
||||
}
|
||||
+14
-1
@@ -263,9 +263,20 @@ test/ → /test the test tree: the QEMU harness (qemu_test.py, h
|
||||
system/services/ beside the on-image test fixtures — vfs-test/ thread-test/
|
||||
crash-test/ … — whose repo path IS their boot-volume path
|
||||
(/test/system/services/<name>)
|
||||
build-support/ the danos build API (build-time only, nothing on the image):
|
||||
the shared user-binary recipe + default-import wiring every
|
||||
build file consumes (docs/build-packages-plan.md)
|
||||
build/ root-build helpers: image assembly (images.zig) + the QEMU
|
||||
run steps (qemu.zig)
|
||||
tools/ host-side build scripts
|
||||
```
|
||||
|
||||
**Builds are packages** (docs/build-packages-plan.md): each `library/` domain owns a
|
||||
`build.zig`/`build.zig.zon` exporting its modules (with a standalone `zig build test`),
|
||||
every binary directory is a ~15-line package build, and the root `build.zig`
|
||||
orchestrates — the kernel + loader, what ships, and the aggregate test step — with
|
||||
image assembly in `build/images.zig` and the QEMU run steps in `build/qemu.zig`.
|
||||
|
||||
**Wire protocols live in `library/protocol/`**, one module per directory
|
||||
(`library/protocol/vfs/vfs-protocol.zig` is the `vfs-protocol` module), imported by module
|
||||
name. A protocol is the seam between a low-level driver and the higher-level service it
|
||||
@@ -320,5 +331,7 @@ exception in [coding-standards.md](coding-standards.md) applies to that seam.
|
||||
| System services (init, the `fat` filesystem, the device-manager) | `system/services/` |
|
||||
| Device drivers, one sub-project each (`pci-bus`, `ps2-bus`, `usb-xhci-bus` bus drivers) | `system/drivers/` |
|
||||
| On-image test fixtures for the QEMU cases (`vfs-test`, `crash-test`, `thread-test`, …) → `/test/system/services` | `test/system/services/` |
|
||||
| Build + `run-x86-64` (QEMU/OVMF) + `release-x86-64` (the flashable ISO) | `build.zig` |
|
||||
| Build orchestration (kernel + loader, what ships, the aggregate test step) | `build.zig` (root; the shared user-binary recipe is `build-support/`, and each `library/` domain + binary package carries its own `build.zig`) |
|
||||
| Image assembly + `release-x86-64` (the flashable ISO) | `build/images.zig` |
|
||||
| `run-x86-64` / `run-x86-64-gpu` (QEMU/OVMF) | `build/qemu.zig` |
|
||||
| QEMU integration test harness | `test/qemu_test.py` |
|
||||
|
||||
@@ -0,0 +1,177 @@
|
||||
# Plan: packages — hierarchical builds for libraries and binaries
|
||||
|
||||
**Status: complete** (branch `claude/build-packages-plan-174144`). Phase 0
|
||||
(`build-support`), phase 1 (all six library domains), phase 2 (every binary —
|
||||
the pci-bus pilot first, then services, drivers, and test fixtures in waves;
|
||||
multi-binary directories like ps2-bus and usb-hid are one package exporting
|
||||
several artifacts, and the acpi/fdt discovery pair each export an artifact
|
||||
named "discovery" that the root's -Ddiscovery picks between), and phase 3 (the
|
||||
root split into `build/images.zig` + `build/qemu.zig`; the root `build.zig` is
|
||||
~460 lines of orchestration, down from ~1,250). Every phase landed green: unit
|
||||
tests, the QEMU suite at parity with main, boot-image file list unchanged.
|
||||
The `lazyDependency` payoff (What-this-buys #4) is in too: the /test fixtures
|
||||
and the unselected discovery package are lazy — a build loads and compiles
|
||||
only what it ships. And imports are exact: the pre-wired default set is gone;
|
||||
every binary names precisely the modules its source imports and carries only
|
||||
those domains in its manifest (rule 1 below).
|
||||
|
||||
## Why
|
||||
|
||||
`build.zig` was ~1,250 lines, growing by three hand-written stanzas per binary;
|
||||
at a driver per device family that does not scale. More fundamentally: in one
|
||||
monolithic build every binary compiles against library *source*, so a library
|
||||
interface break is silently absorbed by whoever edits everything in one commit —
|
||||
the interface never has to be honest. danos is about isolation; the build should
|
||||
mirror it.
|
||||
|
||||
A **package** here is a build-time unit only — a directory owning a `build.zig`
|
||||
(recipe: what it exports, how to test it) and a `build.zig.zon` (manifest: name
|
||||
+ dependencies). Binaries remain fully static freestanding ELFs; packages change
|
||||
who declares what, not what links to what. Source code is untouched: `@import`
|
||||
uses module names (`"pci"`, `"service"`) exactly as today — only build files
|
||||
know where anything lives.
|
||||
|
||||
## Target shape
|
||||
|
||||
```
|
||||
build-support/ package: the danos build API (userBinary(), defaultImports(), targets)
|
||||
library/kernel/ package "kernel": modules abi, ipc, service, memory, process, logging, time, ... (depends on protocol)
|
||||
library/device/ package "device": modules driver, pci, usb-abi, model, ... (depends on kernel, protocol, csv)
|
||||
library/protocol/ package "protocol": the wire protocols
|
||||
library/client/ package "client" (depends on kernel, protocol)
|
||||
library/csv/ package "csv"
|
||||
library/xkeyboard-config/ package "xkeyboard-config"
|
||||
system/services/<name>/ one package per binary: ~15-line build.zig + zon
|
||||
system/drivers/<name>/ one package per binary
|
||||
build.zig (root) orchestrator: dependency() per binary, image assembly, QEMU, test steps
|
||||
```
|
||||
|
||||
The three shared contracts: `boot-handoff` stays a root module (only the
|
||||
loader↔kernel pair speaks it); `abi` is exported by the kernel package from
|
||||
`../../system/abi.zig` (the source stays with the kernel; userspace's one view
|
||||
of it lives in the package, so every consumer names the same module instance);
|
||||
`device-abi` is exported by device. Reaching outside the package root means the
|
||||
kernel package is valid only as an in-repo path dependency — it could never be
|
||||
fetched by hash — which is fine: path dependencies are the only way any of
|
||||
these packages is consumed.
|
||||
|
||||
Rules:
|
||||
|
||||
- **Imports are exact and per binary.** A binary's build.zig names precisely
|
||||
the modules its source `@import`s — the moral equivalent of a C file's
|
||||
include list — and its zon names only the domains those modules come from
|
||||
(plus `build-support` and `kernel`, which is implicit in every binary: the
|
||||
root shim and user link script live there). Nothing is pre-wired: an
|
||||
undeclared `@import` is a compile error, and build-support's one
|
||||
module-to-domain table (`module_homes`) resolves each name. Availability
|
||||
never meant bloat — Zig only compiles what a program actually imports — but
|
||||
exactness makes the declared interface honest and machine-checked.
|
||||
- **Modules export source, not artifacts** — each consumer compiles libraries
|
||||
with its own flags, so per-binary optimization choices keep working; Zig's
|
||||
cache deduplicates.
|
||||
- **Zon paths are relative and that is accepted.** Binaries sit exactly three
|
||||
levels deep, so the `../../../` prefix is a constant idiom; a library-domain
|
||||
move is a rare, already-breaking event fixed by one sed across manifests, and
|
||||
a stale path fails loudly before anything compiles.
|
||||
- **Cross-cutting build changes live in `build-support` only** — that is the
|
||||
contract that keeps per-binary build files declarative.
|
||||
|
||||
## What this buys
|
||||
|
||||
1. Library interfaces become machine-checked: a consumer can only import what
|
||||
it declared — per binary, down to the single module — and each domain's zon
|
||||
declares what it needs (claim-before-touch, applied to source). A keyboard
|
||||
driver carries `xkeyboard-config` in its manifest; nothing else does.
|
||||
2. Each library domain gets a standalone `zig build test` — runtime-library
|
||||
stability testing in isolation.
|
||||
3. Adding a binary = adding a directory (source + two small files), not editing
|
||||
three places in a 1,250-line file.
|
||||
4. `lazyDependency` lets an image target build only what it ships: the /test
|
||||
fixtures resolve only under -Dtest-case, and only the -Ddiscovery-selected
|
||||
discovery package ever loads.
|
||||
|
||||
## Phases
|
||||
|
||||
Each phase ends green: `zig build test` passes (88/88 QEMU) and the boot
|
||||
image's file list is unchanged. Byte-identical binaries are expected but not
|
||||
required (module reorganization can perturb symbol order); file list is the
|
||||
hard gate.
|
||||
|
||||
**Phase 0 — `build-support`.** Extract `addUserBinary`/`addThreadedUserBinary`,
|
||||
the freestanding target setup, and the default-import wiring into the
|
||||
`build-support` package. Root build consumes it; nothing else moves. This is
|
||||
the cross-cutting-change home, so it lands first.
|
||||
|
||||
**Phase 1 — library domains become packages.** In dependency order: `protocol`
|
||||
and `csv` (the roots) → `kernel` (depends on protocol: file-system speaks
|
||||
vfs-protocol) → `device`, `client`; `xkeyboard-config` stands alone. Each gets
|
||||
build.zig + zon + a standalone test step (client's is empty until its modules
|
||||
grow host tests — kept for uniformity, since the root aggregate depends on
|
||||
every domain's test step). The root build swaps its `createModule` calls for
|
||||
`b.dependency("<domain>").module("<name>")`. **No binary moves in this phase**
|
||||
— the root build is the pilot consumer, which proves the packages without
|
||||
touching 30 binaries.
|
||||
|
||||
**Phase 2 — binaries become packages, in waves.** The template was shaken out
|
||||
by the pci-bus pilot (see Status). Wave A: services (done). Wave B: the
|
||||
remaining drivers (done). Wave C: test fixtures (done). Root build shrank to
|
||||
orchestration per wave. init's `-Dserial` heartbeat flag rides a dependency
|
||||
option; a directory with several binaries (ps2-bus, usb-hid) is one package
|
||||
exporting several artifacts.
|
||||
|
||||
**Phase 3 — root cleanup (done).** What remained of the root build split into
|
||||
`build/images.zig` (the FHS install tree, boot manifest + capsule, FAT32
|
||||
images, release ISO, check steps) and `build/qemu.zig` (the run steps + OVMF
|
||||
probing), imported by a short root `build.zig`.
|
||||
|
||||
**Afterwards** (outside this plan): the intel-uhd-graphics-750 driver is
|
||||
(re)created as a greenfield package. The new-driver checklist's build step
|
||||
(docs/device-driver-development/new-driver-checklist.md, step 2) is already
|
||||
rewritten against the package template.
|
||||
|
||||
## Execution notes (the finished shape)
|
||||
|
||||
- The shared recipe lives in `build-support/build.zig`: `userBinary` (what
|
||||
every binary package calls, resolving each named import through the
|
||||
`module_homes` table) and `programModule` (for per-binary addOptions
|
||||
modules). The `start` root shim and `user.ld` are named through the kernel
|
||||
package (Dependency.path).
|
||||
- Adding a binary = adding a directory with source + a ~15-line build.zig +
|
||||
zon (copy any existing binary package, e.g.
|
||||
`system/drivers/pci-bus/build.zig`) listing exactly the modules the source
|
||||
imports and the domains they come from, then one dependency + one bundled
|
||||
entry in the root build.zig and one zon line.
|
||||
- The boot-tree array in the root (search `"etc/init.csv"` or
|
||||
`.getEmittedBin()`) is the image file list — the authoritative comparison
|
||||
target for any future build change.
|
||||
- Package unit tests live in each package's own `test` step; the root
|
||||
aggregate depends on every test-bearing package's step, so `zig build test`
|
||||
at the root still runs everything.
|
||||
|
||||
Verification per phase:
|
||||
|
||||
- Unit tests: `zig build test`.
|
||||
- QEMU integration suite: `python3 test/qemu_test.py` (docs/testing.md; the
|
||||
full suite, all cases must pass).
|
||||
- Image file list: the boot-tree array is the source of truth — snapshot it
|
||||
(paths only) before phase 0 and diff after each phase; `zig build
|
||||
check-fat-image` must also stay green.
|
||||
|
||||
Context a fresh session should read first: this doc, docs/testing.md,
|
||||
docs/coding-standards.md (kebab-case names, no abbreviations), and the
|
||||
`userBinary`/`userBinaryFromImports` bodies in build-support/build.zig. Commit
|
||||
style: no Co-Authored-By trailers.
|
||||
|
||||
## Risks / notes
|
||||
|
||||
- Zig version churn: the package API (`b.dependency`, zon schema) has moved
|
||||
between releases; the work pins against the repo's current Zig and any
|
||||
upgrade lands separately, never mid-phase.
|
||||
- The QEMU size-check tests hardcode source paths (e.g. virtio-gpu protocol
|
||||
struct sizes) — they moved into their binaries' packages with their waves,
|
||||
discharging the carry-along obligation.
|
||||
- Doc updates ride each phase: docs/README.md (repo layout + source map),
|
||||
docs/device-driver-development/new-driver-checklist.md (step 2) and
|
||||
devices-csv.md ("Adding a driver"), and the docs that cite the build recipe
|
||||
(driver-model.md, threading.md, system-requirements.md) reference build
|
||||
shapes that keep changing.
|
||||
@@ -96,7 +96,15 @@ in different namespaces — against the right `bus` column.
|
||||
|
||||
## Adding a driver
|
||||
|
||||
1. Build the driver binary and bundle it at `/system/drivers/<name>` (build.zig).
|
||||
(The step-by-step walkthrough with a worked example is
|
||||
[new-driver-checklist.md](new-driver-checklist.md).)
|
||||
|
||||
1. Create `system/drivers/<name>/` with the driver source plus a ~15-line
|
||||
package `build.zig` + `build.zig.zon` (copy an existing driver package,
|
||||
e.g. `system/drivers/pci-bus/`; per-driver extras go through
|
||||
`build_support.programModule`). Then bundle it at `/system/drivers/<name>`:
|
||||
one dependency + one bundled entry in the root `build.zig`, one line in the
|
||||
root `build.zig.zon`.
|
||||
2. Add a row to `etc/devices.csv` naming the identity it binds and its full path.
|
||||
|
||||
No device-manager change is required — the registry is the seam.
|
||||
|
||||
@@ -20,9 +20,10 @@ Read [display.md](display.md) first for the *why*; this is the *what* and the *o
|
||||
|
||||
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym abbreviations in
|
||||
full, kebab-case file names, no `Co-Authored-By` trailers on commits. New user binaries
|
||||
go through `addUserBinary` in [build.zig](../../build.zig) and get packed into the
|
||||
initial-ramdisk; protocols are `b.addModule("…-protocol", …)` and imported into the
|
||||
`runtime` module.
|
||||
go through build-support's shared user-binary recipe and get packed into the
|
||||
initial-ramdisk; protocols are modules exported by the `library/protocol` package.
|
||||
(This section predates the build-packages split; see
|
||||
[build-packages-plan.md](../build-packages-plan.md) for the current build shape.)
|
||||
|
||||
## How to verify along the way
|
||||
|
||||
|
||||
@@ -18,9 +18,11 @@ lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run,
|
||||
|
||||
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym abbreviations,
|
||||
kebab-case file names, no `Co-Authored-By` trailers. New user binaries go through
|
||||
`addUserBinary` and get packed into the initial-ramdisk; protocols are
|
||||
`b.addModule("…-protocol", …)` imported into `runtime`; new syscalls extend
|
||||
[abi.zig](../../system/abi.zig) `SystemCall` + a `library/runtime` wrapper.
|
||||
build-support's shared user-binary recipe and get packed into the initial-ramdisk;
|
||||
protocols are modules exported by the `library/protocol` package; new syscalls extend
|
||||
[abi.zig](../../system/abi.zig) `SystemCall` + a `library/kernel` wrapper.
|
||||
(This section predates the build-packages split; see
|
||||
[build-packages-plan.md](../build-packages-plan.md) for the current build shape.)
|
||||
|
||||
## How to verify along the way
|
||||
|
||||
|
||||
@@ -29,7 +29,7 @@ which one you're holding decides what you can do.
|
||||
|
||||
- **The PCI class-0x03 device is the raw controller** — BARs, config space, registers,
|
||||
IO ports. It is what you actually *own* after boot. On QEMU's emulated adapter
|
||||
([`-device VGA,edid=on`](../../build.zig), the Bochs VBE/DISPI model) the `base` GOP handed
|
||||
([`-device VGA,edid=on`](../../build/qemu.zig), the Bochs VBE/DISPI model) the `base` GOP handed
|
||||
you *is* that device's linear-framebuffer BAR — the same physical memory, seen through
|
||||
a different door. On a real discrete GPU, GOP's `base` is an aperture inside the GPU's
|
||||
VRAM BAR. danos already decodes this device
|
||||
|
||||
@@ -138,13 +138,17 @@ a higher-level service (block ↔ filesystem, a scanout driver ↔ the composito
|
||||
private wire to its *hardware* — virtio-gpu's command set — is not that; it stays a
|
||||
driver-private file, like the virtio-pci transport beside it.
|
||||
|
||||
The build side of this has since landed: [`addUserBinary`](build.zig) injects the
|
||||
default modules — the library/kernel concern modules (`ipc`, `memory`, `process`, `time`,
|
||||
`logging`, `file-system`, `thread`, `service`), the device/service clients (`driver`,
|
||||
`block`, `display`, `input`), plus `mmio`, `xkeyboard-config`, `acpi-ids` — into every user
|
||||
binary, and per-binary extras — protocol modules, bus logic — are added with
|
||||
`programModule(exe).addImport(...)`. That's the *entire* mechanism — Zig modules
|
||||
already give you everything else.
|
||||
The build side of this has since landed: every binary owns a package whose
|
||||
~15-line `build.zig` names EXACTLY the modules its source imports — the moral
|
||||
equivalent of a C file's include list — and the shared recipe in
|
||||
[`build-support/build.zig`](../../build-support/build.zig) (`userBinary`)
|
||||
resolves each name from the library domain that exports it (kernel's concern
|
||||
modules, the device driver libraries, the service clients, the protocols). An
|
||||
undeclared `@import` is a compile error, and a domain none of the imports come
|
||||
from never appears in the binary's manifest — a keyboard driver declares
|
||||
`xkeyboard-config`; nothing else does (see
|
||||
[build-packages-plan.md](../build-packages-plan.md)). That's the *entire*
|
||||
mechanism — Zig modules already give you everything else.
|
||||
|
||||
The discipline that makes this work: **a class driver must not import a bus's *hardware*
|
||||
logic module.** `usb-hid` imports `usb` (the transfer client) and `input-protocol`, never
|
||||
|
||||
@@ -0,0 +1,218 @@
|
||||
# New driver: the minimum steps
|
||||
|
||||
The shortest path from "a device shows up in the boot log" to "my process is
|
||||
running with its registers mapped". This is the checklist; the reasoning behind
|
||||
every step lives in [Writing a driver](drivers.md), the matching rules in
|
||||
[devices.csv](devices-csv.md), and interrupts in
|
||||
[device interrupts](device-interrupts.md).
|
||||
|
||||
Worked example throughout: the Intel UHD 750 iGPU, which the boot log reports as
|
||||
|
||||
```
|
||||
pci-bus: 0:2.0 bus=pci base=03 class=00 prog_if=00 vendor=8086 device=4C8A ...
|
||||
```
|
||||
|
||||
## 1. Create the source file
|
||||
|
||||
`system/drivers/<name>/<name>.zig` — kebab-case, abbreviations spelled out
|
||||
([coding standards](../coding-standards.md)). The directory name, the binary
|
||||
name, and the `devices.csv` driver path must all agree; a mismatch fails
|
||||
silently (the device-manager logs the spawn failure, nothing else happens).
|
||||
|
||||
The complete minimal driver — claims its device, logs every resource, maps the
|
||||
register window, then sleeps in the harness loop:
|
||||
|
||||
```zig
|
||||
//! /system/drivers/intel-uhd-graphics-750 — spawned by the device manager with
|
||||
//! the device-tree id as argv[1]; claims that device and no other.
|
||||
|
||||
const std = @import("std");
|
||||
const device = @import("driver");
|
||||
const ipc = @import("ipc");
|
||||
const memory = @import("memory");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
|
||||
/// No protocol yet: the kernel's IPC ceiling (MESSAGE_MAXIMUM) sizes the buffers.
|
||||
const message_maximum = 256;
|
||||
|
||||
var controller_id: u64 = 0;
|
||||
var register_base: usize = 0;
|
||||
|
||||
fn initialise(endpoint: ipc.Handle) bool {
|
||||
_ = endpoint; // needed later, for irq binding and timers
|
||||
|
||||
if (!device.claim(controller_id)) {
|
||||
std.log.err("unable to claim device {d}", .{controller_id});
|
||||
return false;
|
||||
}
|
||||
|
||||
// Fetch our own descriptor back for the device's resources.
|
||||
const buffer = memory.allocator().alloc(device.DeviceDescriptor, 64) catch return false;
|
||||
defer memory.allocator().free(buffer);
|
||||
const total = device.enumerate(buffer);
|
||||
const descriptor = for (buffer[0..@min(total, buffer.len)]) |d| {
|
||||
if (d.id == controller_id) break d;
|
||||
} else {
|
||||
std.log.err("device {d} not in the device tree", .{controller_id});
|
||||
return false;
|
||||
};
|
||||
|
||||
// Log every resource BEFORE choosing one (see step 5).
|
||||
var register_index: u64 = 0;
|
||||
for (descriptor.resources[0..@intCast(descriptor.resource_count)], 0..) |resource, index| {
|
||||
std.log.info("resource {d}: kind={d} start=0x{x} len=0x{x}", .{
|
||||
index, resource.kind, resource.start, resource.len,
|
||||
});
|
||||
// The 16 MiB window is GTTMMADR, the register BAR (this device also has
|
||||
// a 256 MiB memory BAR, GMADR — "first memory resource" would be wrong).
|
||||
if (resource.kind == @intFromEnum(device.ResourceKind.memory) and
|
||||
resource.len == 16 * 1024 * 1024) register_index = index;
|
||||
}
|
||||
if (register_index == 0) {
|
||||
std.log.err("register BAR not found", .{});
|
||||
return false;
|
||||
}
|
||||
|
||||
register_base = device.mmioMap(controller_id, register_index) orelse {
|
||||
std.log.err("mmio_map failed", .{});
|
||||
return false;
|
||||
};
|
||||
std.log.info("registers mapped at 0x{x}", .{register_base});
|
||||
return true;
|
||||
}
|
||||
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||
_ = message;
|
||||
_ = reply;
|
||||
_ = sender;
|
||||
_ = capability;
|
||||
return 0; // no protocol yet; the zero-length ping is answered by the harness
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
const argument = init.arguments.get(1) orelse {
|
||||
std.log.err("missing device id (argv[1])", .{});
|
||||
return;
|
||||
};
|
||||
controller_id = std.fmt.parseInt(u64, argument, 10) catch {
|
||||
std.log.err("malformed device id '{s}'", .{argument});
|
||||
return;
|
||||
};
|
||||
service.run(message_maximum, .{
|
||||
.init = initialise,
|
||||
.on_message = onMessage,
|
||||
// .on_notification only once an IRQ or timer is bound
|
||||
});
|
||||
}
|
||||
```
|
||||
|
||||
`claim` is the capability gate: MMIO mapping, DMA grants, and IRQ binding all
|
||||
require it, and it pins the IOMMU domain to this process
|
||||
([drivers.md — claim before touch](drivers.md#the-capability-claim-before-touch)).
|
||||
|
||||
## 2. Create the build package and register it in the root build
|
||||
|
||||
The driver directory is its own build package
|
||||
([build-packages-plan.md](../build-packages-plan.md)): a ~15-line `build.zig`
|
||||
plus a `build.zig.zon` beside the source. Copy both from an existing driver —
|
||||
`system/drivers/pci-bus/` is the template — and adjust the name, root source
|
||||
file, and the import list. The list names EXACTLY the modules the driver's
|
||||
source `@import`s (the moral equivalent of its include list; an undeclared
|
||||
import is a compile error):
|
||||
|
||||
```zig
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "intel-uhd-graphics-750",
|
||||
.root_source_file = b.path("intel-uhd-graphics-750.zig"),
|
||||
.imports = &.{ "driver", "ipc", "memory", "process", "service" },
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
}
|
||||
```
|
||||
|
||||
The zon declares `build-support`, `kernel` (implicit in every binary: the root
|
||||
shim lives there), and the homes of the listed imports — for the minimal
|
||||
driver above that is kernel alone plus `device` (for `driver`); add
|
||||
`protocol`, `client`, ... only when an import comes from them (again, copy
|
||||
pci-bus's zon and adjust). For the `.fingerprint` field, leave the copied
|
||||
value in place and `zig build` will reject it and suggest the fresh one to
|
||||
paste.
|
||||
|
||||
Then three one-liners in the root build register the package: the dependency
|
||||
and a row in the boot-tree array in `build.zig` (search for
|
||||
`virtio_gpu_package` to land in the right places),
|
||||
|
||||
```zig
|
||||
const intel_uhd_graphics_750_exe = b.dependency("intel-uhd-graphics-750", .{}).artifact("intel-uhd-graphics-750");
|
||||
```
|
||||
|
||||
```zig
|
||||
.{ .path = "system/drivers/intel-uhd-graphics-750", .binary = intel_uhd_graphics_750_exe.getEmittedBin() },
|
||||
```
|
||||
|
||||
and the path entry in the root `build.zig.zon`:
|
||||
|
||||
```zig
|
||||
.@"intel-uhd-graphics-750" = .{ .path = "system/drivers/intel-uhd-graphics-750" },
|
||||
```
|
||||
|
||||
Without the boot-tree row the binary never reaches the image and the
|
||||
device-manager has nothing to spawn. (The package also builds standalone:
|
||||
`cd system/drivers/intel-uhd-graphics-750 && zig build`.)
|
||||
|
||||
## 3. Add the match rule to `etc/devices.csv`
|
||||
|
||||
One row: bus, class triplet, vendor/device, driver path. **Copy the class
|
||||
triplet from the pci-bus boot log line, not from another row** — for the iGPU
|
||||
above the correct rule is
|
||||
|
||||
```
|
||||
pci, 03, 00, 00, 8086, 4C8A, *, *, /system/drivers/intel-uhd-graphics-750
|
||||
```
|
||||
|
||||
Field-by-field rules and the most-specific-wins policy: [devices.csv](devices-csv.md).
|
||||
The registry is authoritative: an unmatched device is logged unbound, never
|
||||
guessed — so a wrong nibble here means the driver simply never starts.
|
||||
|
||||
## 4. First contact: read, predict, verify
|
||||
|
||||
Before writing any register, read one whose value you can predict from state
|
||||
the firmware already programmed (for a display controller: the pipe source
|
||||
size of the live mode). Registers are volatile loads at `register_base +
|
||||
offset`, where `offset` is what the device's manual lists:
|
||||
|
||||
```zig
|
||||
fn read32(offset: usize) u32 {
|
||||
return @as(*volatile u32, @ptrFromInt(register_base + offset)).*;
|
||||
}
|
||||
```
|
||||
|
||||
A matching read proves the whole chain — CSV match, spawn, claim, BAR choice,
|
||||
mapping — with zero risk to the hardware.
|
||||
|
||||
## 5. Verify the plumbing
|
||||
|
||||
- `zig build test` still passes.
|
||||
- On the image: `/var/log/<boot-stamp>/system/services/device-manager.log`
|
||||
shows `spawned <name> for device <N>`, and
|
||||
`/var/log/<boot-stamp>/system/drivers/<name>.log` holds the resource list and
|
||||
your first read.
|
||||
- If the driver did not spawn, diagnose in this order: binary on the image
|
||||
(step 2) → CSV row matches the log line exactly (step 3) → path identical in
|
||||
both (step 1).
|
||||
|
||||
## Later, when the device needs them
|
||||
|
||||
- **Interrupts**: MSI/MSI-X via the `pci` module, delivered as notifications to
|
||||
`on_notification` — see [device interrupts](device-interrupts.md) and the
|
||||
xHCI driver's `setupMsi` (QEMU trap documented there: enable MSI-X before
|
||||
unmasking the device's own interrupt-enable bit).
|
||||
- **DMA**: grant-backed buffers, bounded by the IOMMU domain established at
|
||||
claim time ([driver model](driver-model.md)).
|
||||
- **Children**: a bus driver publishes what it finds via `device_register`
|
||||
([drivers.md — publishing children](drivers.md#publishing-children-device_register)).
|
||||
- **A protocol**: replace `message_maximum` with the protocol's own maximum and
|
||||
dispatch on the operation word in `onMessage` — every service under
|
||||
`system/services/` is an example.
|
||||
@@ -24,8 +24,9 @@ EFI/BOOT/BOOTX64.efi <- the "removable media" default for x86-64
|
||||
```
|
||||
|
||||
The boot volume is **FHS-shaped** (see the repository-layout note in
|
||||
[README.md](../README.md)): `build.zig` installs `boot/efi.zig` (built for the `uefi`
|
||||
target) at `EFI/BOOT/BOOTX64.efi` — the one path UEFI firmware fixes — and lays
|
||||
[README.md](../README.md)): the root `build.zig` compiles `boot/efi.zig` (built
|
||||
for the `uefi` target) and `build/images.zig` places it at
|
||||
`EFI/BOOT/BOOTX64.efi` — the one path UEFI firmware fixes — and lays
|
||||
the rest out by FHS path: the kernel at `system/kernel`, init at
|
||||
`system/services/init`, the pre-packed boot capsule at `boot/system.img`
|
||||
([system-image.md](system-image.md)).
|
||||
|
||||
@@ -22,11 +22,12 @@ lands on its own and ends in a **verifiable gate** — shaped for a `/loop` run,
|
||||
## Conventions
|
||||
|
||||
Follow [coding-standards.md](../coding-standards.md): spell out non-acronym abbreviations,
|
||||
kebab-case file names, no `Co-Authored-By` trailers. New user binaries go through
|
||||
`addUserBinary` (with the new `threaded` flag where a binary spawns threads) and get
|
||||
packed into the initial-ramdisk; new syscalls extend [abi.zig](../../system/abi.zig)
|
||||
`SystemCall` + a `library/runtime` wrapper; test services live beside the code they
|
||||
exercise and register a `ServiceId` if they must be looked up.
|
||||
kebab-case file names, no `Co-Authored-By` trailers. New user binaries are
|
||||
packages whose build.zig calls `build_support.userBinary` (with `.threaded =
|
||||
true` where a binary spawns threads) and get packed into the initial-ramdisk;
|
||||
new syscalls extend [abi.zig](../../system/abi.zig) `SystemCall` + a
|
||||
`library/kernel` wrapper; test services live beside the code they exercise and
|
||||
register a `ServiceId` if they must be looked up.
|
||||
|
||||
## How to verify along the way
|
||||
|
||||
|
||||
@@ -61,9 +61,10 @@ runtime — rebuilt in lockstep — knows the mapping.
|
||||
backend would either bake danos syscall numbers into std (breaking ABI privacy and
|
||||
renumbering) or fork std to route back through the runtime — a permanent rebase
|
||||
cost that buys nothing the native type doesn't.
|
||||
2. **Our user binaries are built `single_threaded = true`** ([build.zig](../../build.zig)
|
||||
`addUserBinary`), which compiles threading out entirely and makes atomics and TLS
|
||||
single-threaded. Threads need this flipped per binary regardless.
|
||||
2. **Our user binaries are built `single_threaded = true`** (the shared recipe in
|
||||
[build-support/build.zig](../../build-support/build.zig)), which compiles threading
|
||||
out entirely and makes atomics and TLS single-threaded. Threads need this flipped
|
||||
per binary regardless.
|
||||
|
||||
So we take the *shape* of `std.Thread`, not the *type*. The cost of replicating the
|
||||
surface (spawn/join/Mutex/Condition) is small; the cost of the std type is the ABI
|
||||
@@ -236,9 +237,10 @@ see the intro). Two scoped pieces, as built:
|
||||
|
||||
### Build: multi-threaded codegen, opt-in
|
||||
|
||||
A binary opts in by being added with `addThreadedUserBinary` — as `addUserBinary`,
|
||||
but the shared implementation builds it `single_threaded = false` — so atomics and
|
||||
(later) TLS are real. Threads and atomics are unsound in a `single_threaded` image,
|
||||
A binary opts in with `.threaded = true` in its package's
|
||||
`build_support.userBinary` call — the shared recipe in build-support then builds it
|
||||
`single_threaded = false` — so atomics
|
||||
and (later) TLS are real. Threads and atomics are unsound in a `single_threaded` image,
|
||||
so a binary must opt in **before** it may call `Thread.spawn`. Everyone else
|
||||
stays single-threaded and lean.
|
||||
|
||||
|
||||
@@ -95,9 +95,9 @@ hypervisor configured for UEFI firmware and an xHCI USB controller.
|
||||
|
||||
| Requirement | Detail | Source |
|
||||
|---|---|---|
|
||||
| **x86-64, 64-bit only** | Kernel and loader are built exclusively for `x86_64`; the loader rejects any non-x86-64 kernel ELF (`error.WrongArchitecture`). | `build.zig:481`, `boot/efi.zig:622` |
|
||||
| **x86-64, 64-bit only** | Kernel and loader are built exclusively for `x86_64`; the loader rejects any non-x86-64 kernel ELF (`error.WrongArchitecture`). | `build-support/build.zig` (`freestandingTarget`), `boot/efi.zig:622` |
|
||||
| **Long mode + PAE + NX** | AP trampoline sets `CR4.PAE`, `EFER.LME`, `EFER.NXE`; NX is used in kernel page-table entries. | `system/kernel/architecture/x86_64/trampoline.s:62` |
|
||||
| **SSE / SSE2** | Baseline: the compiler emits SSE for ordinary struct copies. Trampoline enables `CR4.OSFXSR` + `OSXMMEXCPT` and clears `CR0.EM`. | `build.zig:477`, `trampoline.s:62` |
|
||||
| **SSE / SSE2** | Baseline: the compiler emits SSE for ordinary struct copies. Trampoline enables `CR4.OSFXSR` + `OSXMMEXCPT` and clears `CR0.EM`. | `build-support/build.zig` (`freestandingTarget`), `trampoline.s:62` |
|
||||
| **`syscall` / `sysret`** | Primary user↔kernel entry path. `EFER.SCE` enabled; `STAR`/`LSTAR`/`SFMASK` programmed per core. (`int 0x80` exists as a parallel gate.) | `architecture/x86_64/per-cpu.zig:71`, `isr.s:196` |
|
||||
| **Local APIC (xAPIC)** | LAPIC accessed via MMIO at `0xFEE00000`. LAPIC ID read as a `u8` — classic xAPIC. **x2APIC is not supported** (no MSR path). | `apic.zig:67`, `apic.zig:646` |
|
||||
| **CPUID + RDTSC** | CPUID leaf `0x15` for TSC frequency; RDTSC is the monotonic clock. | `apic.zig:333`, `apic.zig:113` |
|
||||
@@ -108,7 +108,7 @@ hypervisor configured for UEFI firmware and an xHCI USB controller.
|
||||
- **UEFI only.** A custom UEFI application loader is installed to
|
||||
`\EFI\BOOT\BOOTX64.efi`. There is **no BIOS, multiboot, or limine** path. The
|
||||
loader tolerates UEFI Class-3 machines with no legacy PIC/PIT.
|
||||
(`build.zig:246`, `boot/efi.zig`)
|
||||
(`build/images.zig` — the EFI/BOOT install — and `boot/efi.zig`)
|
||||
- **ACPI is the hardware-discovery mechanism.** The RSDP is taken from the UEFI
|
||||
configuration table (ACPI 2.0 GUID preferred, 1.0 fallback). Without a valid
|
||||
RSDP there is **no device discovery** — no SMP, no IOAPIC routing, no PCI/USB.
|
||||
|
||||
+3
-1
@@ -12,7 +12,9 @@ There are two layers:
|
||||
`system/abi.zig`, `library/device/model/device-abi.zig`) now spans ~26 modules:
|
||||
protocol and on-wire definitions (VFS, USB, virtio-gpu), the FAT engine, the
|
||||
display compositor, PS/2 and HID decoding, the kernel log ring, and the
|
||||
runtime's `time`/`thread` — the full list is the test step in `build.zig`.
|
||||
runtime's `time`/`thread` — the list is distributed across the library-domain
|
||||
and binary packages' own `test` steps, which the root `zig build test`
|
||||
aggregates (docs/build-packages-plan.md).
|
||||
These compile for the host and run natively.
|
||||
- **QEMU integration tests** (`python3 test/qemu_test.py`) — boot the real kernel
|
||||
and check its behaviour. This is the interesting part.
|
||||
|
||||
@@ -0,0 +1,38 @@
|
||||
//! The "client" library domain (library/client): userspace-service clients —
|
||||
//! they talk to services over IPC, not to the kernel. Client modules end in
|
||||
//! `-client` the way wire protocols end in `-protocol`, so a service, its
|
||||
//! protocol, and its client never share a name (`display` the service,
|
||||
//! `display-protocol` the wire contract, `display-client` a program's view).
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const kernel = b.dependency("kernel", .{});
|
||||
const protocol = b.dependency("protocol", .{});
|
||||
|
||||
const ipc = kernel.module("ipc");
|
||||
const time = kernel.module("time");
|
||||
|
||||
_ = b.addModule("display-client", .{
|
||||
.root_source_file = b.path("display/display-client.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "time", .module = time },
|
||||
.{ .name = "display-protocol", .module = protocol.module("display-protocol") },
|
||||
},
|
||||
});
|
||||
_ = b.addModule("input-client", .{
|
||||
.root_source_file = b.path("input/input-client.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "time", .module = time },
|
||||
.{ .name = "input-protocol", .module = protocol.module("input-protocol") },
|
||||
},
|
||||
});
|
||||
|
||||
// Standalone `zig build test`, kept for uniformity across the domains (the
|
||||
// root aggregate depends on every domain's test step). The clients have no
|
||||
// host-runnable unit tests yet — they are thin IPC conversation wrappers —
|
||||
// so the step is empty until one grows some.
|
||||
_ = b.step("test", "Run the client unit tests (none yet)");
|
||||
}
|
||||
@@ -0,0 +1,13 @@
|
||||
.{
|
||||
.name = .client,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0xc74404553e73d4ff, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// The clients converse over ipc with time-bounded waits.
|
||||
.kernel = .{ .path = "../kernel" },
|
||||
// Each client speaks its service's wire protocol.
|
||||
.protocol = .{ .path = "../protocol" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -0,0 +1,20 @@
|
||||
//! The "csv" library domain: shared CSV helpers (comment stripping, field
|
||||
//! iteration) for the /etc/*.csv config files — the device registry and the
|
||||
//! init service list both parse them.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
_ = b.addModule("csv", .{ .root_source_file = b.path("csv.zig") });
|
||||
|
||||
// Standalone `zig build test` for this domain alone; the root build keeps
|
||||
// its aggregate test step.
|
||||
const test_step = b.step("test", "Run the csv unit tests");
|
||||
const csv_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("csv.zig"),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(csv_tests).step);
|
||||
}
|
||||
@@ -0,0 +1,8 @@
|
||||
.{
|
||||
.name = .csv,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x8a4525791f4e5b6, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -28,6 +28,18 @@ pub const Device = struct {
|
||||
return .{ .block_size = result.block_size, .block_count = result.block_count };
|
||||
}
|
||||
|
||||
/// Hand the block server a DMA-region capability (`handle` — from a `shareable`
|
||||
/// dma_alloc) so it forwards it to the controller and the buffer's physical
|
||||
/// addresses become reachable by the device. Call once per buffer before naming it
|
||||
/// in `read`/`write`. Harmless success when no IOMMU is enforcing.
|
||||
pub fn attach(self: Device, handle: ipc.Handle) bool {
|
||||
var request = block_protocol.Request{ .operation = @intFromEnum(block_protocol.Operation.attach), .lba = 0, .count = 0, .physical = 0 };
|
||||
var reply: [block_protocol.reply_size]u8 = undefined;
|
||||
const result = ipc.callCap(self.endpoint, std.mem.asBytes(&request), &reply, handle) catch return false;
|
||||
if (result.len < block_protocol.reply_size) return false;
|
||||
return std.mem.bytesToValue(block_protocol.Reply, reply[0..block_protocol.reply_size]).status == 0;
|
||||
}
|
||||
|
||||
/// Read `count` blocks starting at `lba` into the DMA buffer at `physical`.
|
||||
pub fn read(self: Device, lba: u64, count: u32, physical: u64) bool {
|
||||
return self.transfer(.read, lba, count, physical);
|
||||
|
||||
@@ -0,0 +1,137 @@
|
||||
//! The "device" library domain (library/device): what a driver author imports.
|
||||
//! The flat reference data (device-abi, pci-class, acpi-ids, usb-abi, usb-ids),
|
||||
//! typed MMIO access, the driver-side client libraries (driver, pci, usb,
|
||||
//! block), the AML interpreter, and the data-driven device registry.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const kernel = b.dependency("kernel", .{});
|
||||
const protocol = b.dependency("protocol", .{});
|
||||
const csv = b.dependency("csv", .{});
|
||||
|
||||
const abi = kernel.module("abi");
|
||||
const system_call = kernel.module("system-call");
|
||||
const ipc = kernel.module("ipc");
|
||||
const time = kernel.module("time");
|
||||
|
||||
// The devices sub-project's public interface (the flat wire types),
|
||||
// importable by user space, unlike the kernel-internal device model it
|
||||
// also feeds (system/kernel/device-model.zig).
|
||||
const device_abi = b.addModule("device-abi", .{
|
||||
.root_source_file = b.path("model/device-abi.zig"),
|
||||
});
|
||||
// PCI class-code decoding (class/subclass/prog-IF -> names). Pure reference
|
||||
// data, shared by kernel discovery and any user-space PCI tool.
|
||||
const pci_class = b.addModule("pci-class", .{
|
||||
.root_source_file = b.path("pci/pci-class.zig"),
|
||||
});
|
||||
// ACPI/PnP hardware-ID (_HID) names — the flat analog of pci-class.
|
||||
_ = b.addModule("acpi-ids", .{
|
||||
.root_source_file = b.path("acpi/acpi-ids.zig"),
|
||||
});
|
||||
// The AML interpreter, a build module so the ring-3 acpi service can run
|
||||
// the same parser the kernel does (docs/discovery.md). Pure Zig, no kernel
|
||||
// imports — one source, two builds.
|
||||
_ = b.addModule("aml", .{
|
||||
.root_source_file = b.path("acpi/aml/aml.zig"),
|
||||
});
|
||||
// The USB device-framework wire ABI (chapter-9 set-up packets, standard +
|
||||
// class requests, descriptors) and the USB class-code taxonomy.
|
||||
const usb_abi = b.addModule("usb-abi", .{
|
||||
.root_source_file = b.path("usb/usb-abi.zig"),
|
||||
});
|
||||
const usb_ids = b.addModule("usb-ids", .{
|
||||
.root_source_file = b.path("usb/usb-ids.zig"),
|
||||
});
|
||||
// Typed volatile MMIO register access + memory-ordering barriers, for
|
||||
// drivers on top of an mmio_map grant. Depends only on `builtin`.
|
||||
const mmio = b.addModule("mmio", .{
|
||||
.root_source_file = b.path("mmio/mmio.zig"),
|
||||
});
|
||||
// The driver author's interface: device access + the device-manager hello
|
||||
// handshake, folded together.
|
||||
const driver = b.addModule("driver", .{
|
||||
.root_source_file = b.path("driver/driver.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi },
|
||||
.{ .name = "device-abi", .module = device_abi },
|
||||
.{ .name = "system-call", .module = system_call },
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "time", .module = time },
|
||||
.{ .name = "device-manager-protocol", .module = protocol.module("device-manager-protocol") },
|
||||
},
|
||||
});
|
||||
// A device driver's view of its claimed PCI function: config-space header
|
||||
// fields, BAR decode + map, capability walks (legacy + extended), MSI/MSI-X
|
||||
// programming, power state, and function-level reset — the generic PCI
|
||||
// mechanics every leaf PCI driver used to re-derive inline.
|
||||
_ = b.addModule("pci", .{
|
||||
.root_source_file = b.path("pci/pci.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "driver", .module = driver },
|
||||
.{ .name = "mmio", .module = mmio },
|
||||
.{ .name = "pci-class", .module = pci_class },
|
||||
.{ .name = "time", .module = time },
|
||||
},
|
||||
});
|
||||
// The USB class-driver transfer client: open a device on the xHCI bus and
|
||||
// drive it (control / interrupt / bulk). Re-exports usb-abi / usb-ids as
|
||||
// usb.abi / usb.ids for a single USB import.
|
||||
_ = b.addModule("usb", .{
|
||||
.root_source_file = b.path("usb/usb.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "time", .module = time },
|
||||
.{ .name = "usb-transfer-protocol", .module = protocol.module("usb-transfer-protocol") },
|
||||
.{ .name = "usb-abi", .module = usb_abi },
|
||||
.{ .name = "usb-ids", .module = usb_ids },
|
||||
},
|
||||
});
|
||||
// The block-device client — a device type, so it lives here.
|
||||
_ = b.addModule("block", .{
|
||||
.root_source_file = b.path("block/block.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "time", .module = time },
|
||||
.{ .name = "block-protocol", .module = protocol.module("block-protocol") },
|
||||
},
|
||||
});
|
||||
// The device registry: parse /etc/devices.csv into match rules and bind a
|
||||
// reported device to a driver. Pure logic (no hardware, no syscalls), so it
|
||||
// unit-tests on the host; the device manager imports it.
|
||||
_ = b.addModule("device-registry", .{
|
||||
.root_source_file = b.path("registry/device-registry.zig"),
|
||||
.imports = &.{.{ .name = "csv", .module = csv.module("csv") }},
|
||||
});
|
||||
|
||||
// Standalone `zig build test` for this domain alone; the root build keeps
|
||||
// its aggregate test step.
|
||||
const test_step = b.step("test", "Run the device library unit tests");
|
||||
for ([_][]const u8{
|
||||
"model/device-abi.zig", // wire-type sizes
|
||||
"pci/pci-class.zig", // class/subclass/prog-IF name decoding
|
||||
"acpi/acpi-ids.zig", // _HID name decoding
|
||||
"acpi/aml/aml.zig", // AML parse + interpret, incl. Notify dispatch
|
||||
"usb/usb-abi.zig", // wire sizes + bit packings + set-up packet encodings
|
||||
"usb/usb-ids.zig", // class/subclass/protocol code assignments
|
||||
"mmio/mmio.zig", // barriers assemble + registers round-trip
|
||||
}) |root| {
|
||||
const device_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(root),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(device_tests).step);
|
||||
}
|
||||
// The registry needs its csv import wired, so it doesn't fit the loop.
|
||||
const registry_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("registry/device-registry.zig"),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
.imports = &.{.{ .name = "csv", .module = csv.module("csv") }},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(registry_tests).step);
|
||||
}
|
||||
@@ -0,0 +1,15 @@
|
||||
.{
|
||||
.name = .device,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x92fb68eace23a4f, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// driver/block/usb/pci build on the kernel library's concern modules.
|
||||
.kernel = .{ .path = "../kernel" },
|
||||
// driver speaks device-manager-protocol; block/usb their transfer protocols.
|
||||
.protocol = .{ .path = "../protocol" },
|
||||
// device-registry parses /etc/devices.csv with the shared csv helpers.
|
||||
.csv = .{ .path = "../csv" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -99,6 +99,27 @@ pub fn msiBind(device_id: u64, endpoint: usize) ?Msi {
|
||||
return .{ .address = rax, .data = @intCast(rdx) };
|
||||
}
|
||||
|
||||
/// Map a delegated DMA-region (or shared-memory) capability into a claimed device's
|
||||
/// IOMMU domain, so the device may DMA to that buffer. The caller must own `device_id`
|
||||
/// and hold `handle` (received over IPC or from its own `dma.alloc(.. | shareable)`).
|
||||
/// Idempotent. Returns true on success (and trivially when no IOMMU is present).
|
||||
pub fn dmaBind(device_id: u64, handle: usize) bool {
|
||||
return !failed(sc.systemCall2(.dma_bind, device_id, handle));
|
||||
}
|
||||
|
||||
/// Unmap a previously `dmaBind`'d buffer from the device's domain.
|
||||
pub fn dmaUnbind(device_id: u64, handle: usize) bool {
|
||||
return !failed(sc.systemCall2(.dma_unbind, device_id, handle));
|
||||
}
|
||||
|
||||
/// Drain and log any pending IOMMU translation faults, returning the count seen. A
|
||||
/// diagnostic: a driver that suspects its device attempted an out-of-domain DMA (or a
|
||||
/// test proving enforcement) forces the hardware's fault records to the log now. Returns
|
||||
/// 0 when no IOMMU is present.
|
||||
pub fn iommuFaultDrain() usize {
|
||||
return sc.systemCall0(.iommu_fault_drain);
|
||||
}
|
||||
|
||||
/// Read `width` bytes (1, 2, or 4) from a port in a claimed device's `io_port`
|
||||
/// resource, at byte `offset` within it. Ring 3 has no direct `in`/`out`, so a legacy
|
||||
/// driver (PS/2, 16550 UART) reaches its ports through this claim-gated call — each
|
||||
|
||||
@@ -52,16 +52,134 @@ pub const config_vendor_id: usize = 0x00;
|
||||
pub const config_device_id: usize = 0x02;
|
||||
pub const config_command: usize = 0x04;
|
||||
pub const config_status: usize = 0x06;
|
||||
pub const config_capabilities_pointer: usize = 0x34;
|
||||
pub const config_revision_id: usize = 0x08;
|
||||
pub const config_class_code: usize = 0x09; // 3 bytes: prog-IF 0x09, subclass 0x0A, base class 0x0B
|
||||
pub const config_bar0: usize = 0x10; // BAR0; BAR n is at config_bar0 + n*4
|
||||
pub const config_subsystem_vendor_id: usize = 0x2C;
|
||||
pub const config_subsystem_id: usize = 0x2E;
|
||||
pub const config_expansion_rom: usize = 0x30;
|
||||
pub const config_capabilities_pointer: usize = 0x34;
|
||||
pub const config_interrupt_line: usize = 0x3C;
|
||||
pub const config_interrupt_pin: usize = 0x3D; // 0 = none, 1..4 = INTA..INTD
|
||||
|
||||
/// Command register: Memory-Space enable (bit 1) | Bus-Master enable (bit 2).
|
||||
pub const command_memory_and_bus_master: u16 = 0x06;
|
||||
/// Command register bits.
|
||||
pub const command_io_space: u16 = 0x0001; // bit 0: I/O-space decode enable
|
||||
pub const command_memory_space: u16 = 0x0002; // bit 1: memory-space decode enable
|
||||
pub const command_bus_master: u16 = 0x0004; // bit 2: bus-master (DMA) enable
|
||||
pub const command_interrupt_disable: u16 = 0x0400; // bit 10: suppress legacy INTx (MSI/MSI-X unaffected)
|
||||
/// The pair a bus-mastering driver enables together: decode my BARs, let me DMA.
|
||||
pub const command_memory_and_bus_master: u16 = command_memory_space | command_bus_master;
|
||||
|
||||
/// Status register bit 3: legacy INTx is asserted (upstream of the command bit-10 gate).
|
||||
pub const status_interrupt: u16 = 0x0008;
|
||||
/// Status register bit 4: a capability list is present at config_capabilities_pointer.
|
||||
pub const status_capabilities_list: u16 = 0x10;
|
||||
/// Capability pointers are dword-aligned; the low two bits are reserved.
|
||||
pub const capability_pointer_mask: u8 = 0xFC;
|
||||
|
||||
/// Capability IDs — the first byte of each entry in the legacy capability list.
|
||||
/// Non-exhaustive: hardware may report IDs not named here.
|
||||
pub const CapabilityId = enum(u8) {
|
||||
power_management = 0x01,
|
||||
msi = 0x05,
|
||||
vendor_specific = 0x09,
|
||||
pci_express = 0x10,
|
||||
msix = 0x11,
|
||||
_,
|
||||
};
|
||||
|
||||
/// MSI capability (id 0x05) register layout. Offsets are relative to the capability
|
||||
/// header; whether the address is one or two dwords (and therefore where the data word
|
||||
/// sits) depends on `control_64bit_capable`.
|
||||
pub const msi = struct {
|
||||
pub const control: usize = 0x02; // u16 Message Control
|
||||
pub const control_enable: u16 = 0x0001;
|
||||
pub const control_multiple_message_capable_mask: u16 = 0x000E; // bits 3:1, log2(vectors requested)
|
||||
pub const control_multiple_message_enable_mask: u16 = 0x0070; // bits 6:4, log2(vectors granted)
|
||||
pub const control_64bit_capable: u16 = 0x0080; // bit 7: address is 64-bit (layout shifts)
|
||||
pub const control_per_vector_masking: u16 = 0x0100; // bit 8
|
||||
pub const address: usize = 0x04; // u32 low address dword (both layouts)
|
||||
pub const address_high: usize = 0x08; // u32, present only when 64-bit capable
|
||||
pub const data_32: usize = 0x08; // u16 message data, 32-bit layout
|
||||
pub const data_64: usize = 0x0C; // u16 message data, 64-bit layout
|
||||
pub const mask_bits_32: usize = 0x0C; // u32, only with per-vector masking
|
||||
pub const mask_bits_64: usize = 0x10;
|
||||
};
|
||||
|
||||
/// MSI-X capability (id 0x11) register layout, plus the 16-byte vector table entry that
|
||||
/// lives in BAR space (not configuration space) at the decoded (BIR, offset).
|
||||
pub const msix = struct {
|
||||
pub const control: usize = 0x02; // u16 Message Control
|
||||
pub const control_table_size_mask: u16 = 0x07FF; // bits 10:0, encoded as N-1
|
||||
pub const control_function_mask: u16 = 0x4000; // bit 14: mask every vector
|
||||
pub const control_enable: u16 = 0x8000; // bit 15
|
||||
pub const table_offset_word: usize = 0x04; // u32: BIR in bits 2:0, table offset in bits 31:3
|
||||
pub const pba_offset_word: usize = 0x08; // u32: same encoding, pending-bit array
|
||||
pub const bir_mask: u32 = 0x0000_0007;
|
||||
pub const offset_mask: u32 = 0xFFFF_FFF8;
|
||||
pub const entry_size: usize = 16; // table entry stride; offsets within an entry:
|
||||
pub const entry_address: usize = 0x0; // u32 low
|
||||
pub const entry_address_high: usize = 0x4; // u32 high
|
||||
pub const entry_data: usize = 0x8; // u32
|
||||
pub const entry_vector_control: usize = 0xC; // u32
|
||||
pub const entry_vector_control_masked: u32 = 0x1; // bit 0; entries reset to masked
|
||||
|
||||
/// Where the table (or pending-bit array) lives, decoded from its offset/BIR dword.
|
||||
pub const TableLocation = struct { bar: u8, offset: u32 };
|
||||
pub fn tableLocation(word: u32) TableLocation {
|
||||
return .{ .bar = @intCast(word & bir_mask), .offset = word & offset_mask };
|
||||
}
|
||||
/// Number of table entries (the control field encodes N-1).
|
||||
pub fn tableSize(control_value: u16) u16 {
|
||||
return (control_value & control_table_size_mask) + 1;
|
||||
}
|
||||
};
|
||||
|
||||
/// Power-management capability (id 0x01) register layout.
|
||||
pub const power_management = struct {
|
||||
pub const capabilities: usize = 0x02; // u16 PMC (read-only: version, D-state support)
|
||||
pub const control_status: usize = 0x04; // u16 PMCSR
|
||||
pub const control_status_power_state_mask: u16 = 0x0003; // bits 1:0
|
||||
pub const power_state_d0: u16 = 0x0;
|
||||
pub const power_state_d3_hot: u16 = 0x3;
|
||||
pub const control_status_pme_enable: u16 = 0x0100; // bit 8: plain RW — preserve on writes
|
||||
pub const control_status_pme_status: u16 = 0x8000; // bit 15: RW1C — write 0 or you clear it
|
||||
};
|
||||
|
||||
/// PCI Express capability (id 0x10) register layout — the slice function-level reset
|
||||
/// needs; the full capability is much larger.
|
||||
pub const pci_express = struct {
|
||||
pub const capabilities: usize = 0x02; // u16 PCIe Capabilities register
|
||||
pub const device_capabilities: usize = 0x04; // u32
|
||||
pub const device_capabilities_flr: u32 = 1 << 28; // Function Level Reset supported
|
||||
pub const device_control: usize = 0x08; // u16
|
||||
pub const device_control_initiate_flr: u16 = 1 << 15;
|
||||
pub const device_status: usize = 0x0A; // u16
|
||||
pub const device_status_transactions_pending: u16 = 1 << 5;
|
||||
};
|
||||
|
||||
/// Extended (PCI Express) capabilities start here in the 4 KiB configuration space; a
|
||||
/// conventional-PCI function has nothing there (the space reads as all-ones).
|
||||
pub const extended_capability_start: usize = 0x100;
|
||||
/// Extended-capability next pointers are dword-aligned within the 4 KiB space.
|
||||
pub const extended_capability_pointer_mask: u16 = 0xFFC;
|
||||
|
||||
/// The 32-bit header at the start of each extended capability: ID in bits 15:0,
|
||||
/// version in 19:16, next offset in 31:20 (0 = end of list).
|
||||
pub const ExtendedCapabilityHeader = struct {
|
||||
id: u16,
|
||||
version: u4,
|
||||
next: u16,
|
||||
|
||||
pub fn decode(word: u32) ExtendedCapabilityHeader {
|
||||
return .{
|
||||
.id = @truncate(word),
|
||||
.version = @truncate(word >> 16),
|
||||
.next = @intCast((word >> 20) & extended_capability_pointer_mask),
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
/// BAR bit layout: bit 0 selects I/O (1) vs memory (0) space; for a memory BAR, bits 2:1
|
||||
/// give the type (00 = 32-bit, 10 = 64-bit spanning the next BAR), and the base address is
|
||||
/// the dword with the low 4 flag bits masked off.
|
||||
@@ -595,3 +713,37 @@ test "named parts pack to the raw triple" {
|
||||
};
|
||||
try std.testing.expectEqual(@as(u24, 0x0C_03_30), xhci.pack());
|
||||
}
|
||||
|
||||
test "MSI-X table word decodes to BIR and offset" {
|
||||
const eq = std.testing.expectEqual;
|
||||
// BIR 3, table at 0x2000 within that BAR.
|
||||
try eq(msix.TableLocation{ .bar = 3, .offset = 0x2000 }, msix.tableLocation(0x0000_2003));
|
||||
// BIR 0, offset 0 — the degenerate-but-common "table at BAR start" case.
|
||||
try eq(msix.TableLocation{ .bar = 0, .offset = 0 }, msix.tableLocation(0));
|
||||
// Table size encodes N-1 in bits 10:0; enable/function-mask bits must not leak in.
|
||||
try eq(@as(u16, 11), msix.tableSize(msix.control_enable | 0x000A));
|
||||
try eq(@as(u16, 1), msix.tableSize(0));
|
||||
try eq(@as(u16, 2048), msix.tableSize(msix.control_table_size_mask));
|
||||
}
|
||||
|
||||
test "extended capability header unpacks id, version, next" {
|
||||
const eq = std.testing.expectEqual;
|
||||
// AER (id 0x0001), version 1, next capability at 0x140.
|
||||
const aer = ExtendedCapabilityHeader.decode(0x1401_0001);
|
||||
try eq(@as(u16, 0x0001), aer.id);
|
||||
try eq(@as(u4, 1), aer.version);
|
||||
try eq(@as(u16, 0x140), aer.next);
|
||||
// A zero header is the "nothing here" terminator.
|
||||
const none = ExtendedCapabilityHeader.decode(0);
|
||||
try eq(@as(u16, 0), none.id);
|
||||
try eq(@as(u16, 0), none.next);
|
||||
}
|
||||
|
||||
test "command bits and capability ids compose" {
|
||||
const eq = std.testing.expectEqual;
|
||||
try eq(command_memory_space | command_bus_master, command_memory_and_bus_master);
|
||||
try eq(@as(u8, 0x05), @intFromEnum(CapabilityId.msi));
|
||||
try eq(@as(u8, 0x11), @intFromEnum(CapabilityId.msix));
|
||||
try eq(@as(u8, 0x01), @intFromEnum(CapabilityId.power_management));
|
||||
try eq(@as(u8, 0x10), @intFromEnum(CapabilityId.pci_express));
|
||||
}
|
||||
|
||||
+283
-7
@@ -1,7 +1,8 @@
|
||||
//! library/device/pci/pci.zig — a device driver's view of the ONE PCI function it has
|
||||
//! claimed. Config space is mapped as resource 0; this gives header-field accessors, BAR
|
||||
//! decode + map, and a capability-list iterator, so a driver never re-derives the
|
||||
//! config-space layout by hand.
|
||||
//! claimed. Config space is mapped as resource 0 (a full 4 KiB ECAM page); this gives
|
||||
//! header-field accessors, BAR decode + map, capability walks (legacy and extended),
|
||||
//! MSI/MSI-X programming, power-state handling, and function-level reset, so a driver
|
||||
//! never re-derives the config-space layout by hand.
|
||||
//!
|
||||
//! This is the *device-owned* view: read my own function's live config, map my own BARs.
|
||||
//! The bus enumerator's view — probing arbitrary, not-yet-claimed functions and sizing
|
||||
@@ -13,6 +14,18 @@ const std = @import("std");
|
||||
const mmio = @import("mmio");
|
||||
const pci_class = @import("pci-class");
|
||||
const device = @import("driver");
|
||||
const time = @import("time");
|
||||
|
||||
/// Spec recovery time after a D3hot -> D0 transition.
|
||||
const d0_recovery_millis: u64 = 10;
|
||||
/// How long to wait for in-flight transactions to drain before a function-level reset
|
||||
/// (then reset anyway — resetting a stuck function is the point of FLR).
|
||||
const flr_pending_timeout_millis: u64 = 100;
|
||||
/// The spec's maximum FLR completion time.
|
||||
const flr_settle_millis: u64 = 100;
|
||||
/// How long to wait for the function to become readable again after an FLR.
|
||||
const flr_ready_timeout_millis: u64 = 1000;
|
||||
const flr_poll_interval_millis: u64 = 10;
|
||||
|
||||
/// A claimed PCI function whose configuration space is mapped (resource 0). `descriptor`
|
||||
/// must outlive the Function — the driver's `device.enumerate` buffer does, for the whole
|
||||
@@ -42,12 +55,61 @@ pub const Function = struct {
|
||||
pub fn status(self: *const Function) u16 {
|
||||
return mmio.readRegister(u16, self.config + pci_class.config_status);
|
||||
}
|
||||
pub fn revisionId(self: *const Function) u8 {
|
||||
return mmio.readRegister(u8, self.config + pci_class.config_revision_id);
|
||||
}
|
||||
/// Subsystem vendor ID (config 0x2C) — with `subsystemId`, the standard key for
|
||||
/// board-level quirk matching.
|
||||
pub fn subsystemVendorId(self: *const Function) u16 {
|
||||
return mmio.readRegister(u16, self.config + pci_class.config_subsystem_vendor_id);
|
||||
}
|
||||
pub fn subsystemId(self: *const Function) u16 {
|
||||
return mmio.readRegister(u16, self.config + pci_class.config_subsystem_id);
|
||||
}
|
||||
/// Interrupt pin (config 0x3D): 0 = none, 1..4 = INTA..INTD.
|
||||
pub fn interruptPin(self: *const Function) u8 {
|
||||
return mmio.readRegister(u8, self.config + pci_class.config_interrupt_pin);
|
||||
}
|
||||
/// The live class-code triple (config 0x09..0x0B), same shape discovery records.
|
||||
pub fn classCode(self: *const Function) pci_class.ClassCode {
|
||||
return .{
|
||||
.prog_if = mmio.readRegister(u8, self.config + pci_class.config_class_code),
|
||||
.subclass = mmio.readRegister(u8, self.config + pci_class.config_class_code + 1),
|
||||
.base = mmio.readRegister(u8, self.config + pci_class.config_class_code + 2),
|
||||
};
|
||||
}
|
||||
|
||||
/// Set Memory-Space + Bus-Master enable in the command register. Firmware often leaves
|
||||
/// a secondary display's decode off; a bus-mastering device must enable both.
|
||||
pub fn enableMemoryAndBusMaster(self: *const Function) void {
|
||||
fn commandSetBits(self: *const Function, bits: u16) void {
|
||||
const at = self.config + pci_class.config_command;
|
||||
mmio.writeRegister(u16, at, mmio.readRegister(u16, at) | pci_class.command_memory_and_bus_master);
|
||||
mmio.writeRegister(u16, at, mmio.readRegister(u16, at) | bits);
|
||||
}
|
||||
fn commandClearBits(self: *const Function, bits: u16) void {
|
||||
const at = self.config + pci_class.config_command;
|
||||
mmio.writeRegister(u16, at, mmio.readRegister(u16, at) & ~bits);
|
||||
}
|
||||
|
||||
/// Set Memory-Space + Bus-Master Enable in the command register. Firmware only enables
|
||||
/// memory decode on devices it used at boot; any other device has dead BARs until its
|
||||
/// driver sets it. Bus mastering is separately required for the device to do DMA.
|
||||
pub fn enableMemoryAndBusMaster(self: *const Function) void {
|
||||
self.commandSetBits(pci_class.command_memory_and_bus_master);
|
||||
}
|
||||
|
||||
/// Clear Bus-Master Enable — stop the device initiating DMA. The quiesce half of a
|
||||
/// driver's shutdown (or a supervisor restart): after this the device can no longer
|
||||
/// write memory the process is about to stop owning.
|
||||
pub fn disableBusMaster(self: *const Function) void {
|
||||
self.commandClearBits(pci_class.command_bus_master);
|
||||
}
|
||||
|
||||
/// Set command bit 10: suppress legacy INTx assertion. MSI/MSI-X are unaffected —
|
||||
/// set this when enabling either, so the device cannot also raise the shared pin.
|
||||
pub fn setInterruptDisable(self: *const Function) void {
|
||||
self.commandSetBits(pci_class.command_interrupt_disable);
|
||||
}
|
||||
/// Clear command bit 10, re-allowing legacy INTx assertion.
|
||||
pub fn clearInterruptDisable(self: *const Function) void {
|
||||
self.commandClearBits(pci_class.command_interrupt_disable);
|
||||
}
|
||||
|
||||
/// Decode BAR `bar` (0..5) and map it: read the BAR register, reject I/O-space BARs,
|
||||
@@ -86,6 +148,129 @@ pub const Function = struct {
|
||||
0;
|
||||
return .{ .config = self.config, .cursor = first };
|
||||
}
|
||||
|
||||
/// First capability with `id`, or null.
|
||||
pub fn findCapability(self: *const Function, id: pci_class.CapabilityId) ?Capability {
|
||||
var walk = self.capabilities();
|
||||
while (walk.next()) |capability| {
|
||||
if (capability.id == @intFromEnum(id)) return capability;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Program the MSI capability with the kernel's `msi_bind` result and enable it —
|
||||
/// one vector (multiple-message-enable 0, matching the kernel's single-vector
|
||||
/// grant), INTx suppressed. false if the function has no MSI capability.
|
||||
pub fn programMsi(self: *const Function, message: device.Msi) bool {
|
||||
const cap = self.findCapability(.msi) orelse return false;
|
||||
const control_at = cap.offset + pci_class.msi.control;
|
||||
const control = mmio.readRegister(u16, control_at);
|
||||
// Program the registers while the capability is disabled.
|
||||
mmio.writeRegister(u16, control_at, control & ~pci_class.msi.control_enable);
|
||||
mmio.writeRegister(u32, cap.offset + pci_class.msi.address, @truncate(message.address));
|
||||
const data_offset = if (control & pci_class.msi.control_64bit_capable != 0) offset: {
|
||||
mmio.writeRegister(u32, cap.offset + pci_class.msi.address_high, @intCast(message.address >> 32));
|
||||
break :offset pci_class.msi.data_64;
|
||||
} else pci_class.msi.data_32;
|
||||
// Message data is a 16-bit register in both layouts.
|
||||
mmio.writeRegister(u16, cap.offset + data_offset, @truncate(message.data));
|
||||
mmio.writeRegister(u16, control_at, (control & ~pci_class.msi.control_multiple_message_enable_mask) | pci_class.msi.control_enable);
|
||||
self.setInterruptDisable();
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Clear the MSI enable bit. No-op if the function has no MSI capability.
|
||||
pub fn disableMsi(self: *const Function) void {
|
||||
const cap = self.findCapability(.msi) orelse return;
|
||||
const control_at = cap.offset + pci_class.msi.control;
|
||||
mmio.writeRegister(u16, control_at, mmio.readRegister(u16, control_at) & ~pci_class.msi.control_enable);
|
||||
}
|
||||
|
||||
/// The function's MSI-X capability with its vector table mapped: the table's BIR is
|
||||
/// resolved through `mapBar` (a free cache hit when it is a BAR the driver already
|
||||
/// mapped). null if the capability is absent or the table's BAR cannot be mapped.
|
||||
pub fn msix(self: *Function) ?MsiX {
|
||||
const cap = self.findCapability(.msix) orelse return null;
|
||||
const control = mmio.readRegister(u16, cap.offset + pci_class.msix.control);
|
||||
const word = mmio.readRegister(u32, cap.offset + pci_class.msix.table_offset_word);
|
||||
const location = pci_class.msix.tableLocation(word);
|
||||
const bar_base = self.mapBar(location.bar) orelse return null;
|
||||
return .{
|
||||
.capability = cap.offset,
|
||||
.table = bar_base + location.offset,
|
||||
.entry_count = pci_class.msix.tableSize(control),
|
||||
};
|
||||
}
|
||||
|
||||
/// Bring the function to D0. Firmware can leave a non-boot device in D3hot, where
|
||||
/// its BARs and MSI registers do not decode; call this before touching either. No
|
||||
/// power-management capability means the function is always at D0: nothing to do.
|
||||
/// Preserves PME-Enable and never clears the write-1-to-clear PME-Status bit.
|
||||
pub fn ensurePowerStateD0(self: *const Function) void {
|
||||
const cap = self.findCapability(.power_management) orelse return;
|
||||
const at = cap.offset + pci_class.power_management.control_status;
|
||||
const pmcsr = mmio.readRegister(u16, at);
|
||||
if (pmcsr & pci_class.power_management.control_status_power_state_mask == pci_class.power_management.power_state_d0) return;
|
||||
// PME-Status is RW1C: echoing a read 1 back would clear it, so write it as 0.
|
||||
mmio.writeRegister(u16, at, (pmcsr & ~pci_class.power_management.control_status_power_state_mask & ~pci_class.power_management.control_status_pme_status) | pci_class.power_management.power_state_d0);
|
||||
time.sleepMillis(d0_recovery_millis);
|
||||
}
|
||||
|
||||
/// Function Level Reset via the PCI Express capability: return the hardware to a
|
||||
/// known state (a supervisor re-claiming a device after its driver died, or a driver
|
||||
/// recovering a wedged function). The six BAR dwords are saved and restored — FLR
|
||||
/// clears them, and the bus enumerator's assignment must survive for the descriptor
|
||||
/// correlation and `mapBar` cache to stay valid. Everything else is reset: command
|
||||
/// enables and MSI/MSI-X programming are gone, so the caller re-runs its whole
|
||||
/// bring-up afterwards. false if the function has no PCI Express capability, does
|
||||
/// not advertise FLR (conventional-PCI Advanced Features FLR is a possible
|
||||
/// follow-up), or never became readable again. Blocks for at least 100 ms.
|
||||
pub fn functionLevelReset(self: *const Function) bool {
|
||||
const cap = self.findCapability(.pci_express) orelse return false;
|
||||
const device_capabilities = mmio.readRegister(u32, cap.offset + pci_class.pci_express.device_capabilities);
|
||||
if (device_capabilities & pci_class.pci_express.device_capabilities_flr == 0) return false;
|
||||
|
||||
// Stop new DMA, then give in-flight transactions a bounded chance to drain —
|
||||
// and reset anyway on timeout, since resetting a stuck function is the point.
|
||||
self.disableBusMaster();
|
||||
var waited: u64 = 0;
|
||||
while (mmio.readRegister(u16, cap.offset + pci_class.pci_express.device_status) & pci_class.pci_express.device_status_transactions_pending != 0) {
|
||||
if (waited >= flr_pending_timeout_millis) break;
|
||||
time.sleepMillis(flr_poll_interval_millis);
|
||||
waited += flr_poll_interval_millis;
|
||||
}
|
||||
|
||||
var bars: [6]u32 = undefined;
|
||||
for (&bars, 0..) |*bar, index| bar.* = mmio.readRegister(u32, self.config + pci_class.config_bar0 + index * 4);
|
||||
|
||||
const control_at = cap.offset + pci_class.pci_express.device_control;
|
||||
mmio.writeRegister(u16, control_at, mmio.readRegister(u16, control_at) | pci_class.pci_express.device_control_initiate_flr);
|
||||
time.sleepMillis(flr_settle_millis);
|
||||
|
||||
waited = 0;
|
||||
while (self.vendorId() == 0xFFFF) {
|
||||
if (waited >= flr_ready_timeout_millis) return false;
|
||||
time.sleepMillis(flr_poll_interval_millis);
|
||||
waited += flr_poll_interval_millis;
|
||||
}
|
||||
for (bars, 0..) |bar, index| mmio.writeRegister(u32, self.config + pci_class.config_bar0 + index * 4, bar);
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Iterate the extended (PCI Express) capability list at 0x100.. in the 4 KiB ECAM
|
||||
/// page. Empty on a conventional-PCI function (the space reads as all-ones).
|
||||
pub fn extendedCapabilities(self: *const Function) ExtendedCapabilityIterator {
|
||||
return .{ .config = self.config };
|
||||
}
|
||||
|
||||
/// First extended capability with `id`, or null.
|
||||
pub fn findExtendedCapability(self: *const Function, id: u16) ?ExtendedCapability {
|
||||
var walk = self.extendedCapabilities();
|
||||
while (walk.next()) |capability| {
|
||||
if (capability.id == id) return capability;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
};
|
||||
|
||||
/// One capability header. `offset` is the ABSOLUTE virtual address of the header, so the
|
||||
@@ -106,3 +291,94 @@ pub const CapabilityIterator = struct {
|
||||
return .{ .id = id, .offset = at };
|
||||
}
|
||||
};
|
||||
|
||||
/// A resolved MSI-X capability from `Function.msix`: `capability` is the absolute
|
||||
/// virtual address of the config-space header, `table` of vector-table entry 0 (in BAR
|
||||
/// space — table writes are MMIO, not config space). Entries reset masked; bring-up
|
||||
/// order is programEntry per vector, unmaskEntry per used vector, `enable`, then
|
||||
/// `Function.setInterruptDisable`.
|
||||
pub const MsiX = struct {
|
||||
capability: usize,
|
||||
table: usize,
|
||||
entry_count: u16,
|
||||
|
||||
/// Write `message` into table entry `entry`, leaving the entry masked (its reset
|
||||
/// state) — the spec requires masking while address/data change. false if `entry`
|
||||
/// is out of range.
|
||||
pub fn programEntry(self: *const MsiX, entry: u16, message: device.Msi) bool {
|
||||
if (entry >= self.entry_count) return false;
|
||||
const at = self.table + @as(usize, entry) * pci_class.msix.entry_size;
|
||||
mmio.writeRegister(u32, at + pci_class.msix.entry_vector_control, pci_class.msix.entry_vector_control_masked);
|
||||
mmio.writeRegister(u32, at + pci_class.msix.entry_address, @truncate(message.address));
|
||||
mmio.writeRegister(u32, at + pci_class.msix.entry_address_high, @intCast(message.address >> 32));
|
||||
mmio.writeRegister(u32, at + pci_class.msix.entry_data, message.data);
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Set the entry's vector-control mask bit — its interrupt is held off (pended in
|
||||
/// the PBA, not lost). false if `entry` is out of range.
|
||||
pub fn maskEntry(self: *const MsiX, entry: u16) bool {
|
||||
return self.writeEntryMask(entry, true);
|
||||
}
|
||||
/// Clear the entry's vector-control mask bit. false if `entry` is out of range.
|
||||
pub fn unmaskEntry(self: *const MsiX, entry: u16) bool {
|
||||
return self.writeEntryMask(entry, false);
|
||||
}
|
||||
fn writeEntryMask(self: *const MsiX, entry: u16, masked: bool) bool {
|
||||
if (entry >= self.entry_count) return false;
|
||||
const at = self.table + @as(usize, entry) * pci_class.msix.entry_size + pci_class.msix.entry_vector_control;
|
||||
const control = mmio.readRegister(u32, at);
|
||||
mmio.writeRegister(u32, at, if (masked)
|
||||
control | pci_class.msix.entry_vector_control_masked
|
||||
else
|
||||
control & ~pci_class.msix.entry_vector_control_masked);
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Set the function-mask control bit: every vector masked regardless of entry bits.
|
||||
pub fn setFunctionMask(self: *const MsiX) void {
|
||||
self.writeControl(pci_class.msix.control_function_mask, true);
|
||||
}
|
||||
/// Clear the function-mask control bit.
|
||||
pub fn clearFunctionMask(self: *const MsiX) void {
|
||||
self.writeControl(pci_class.msix.control_function_mask, false);
|
||||
}
|
||||
/// Set MSI-X Enable. The caller also calls `Function.setInterruptDisable` (INTx off).
|
||||
pub fn enable(self: *const MsiX) void {
|
||||
self.writeControl(pci_class.msix.control_enable, true);
|
||||
}
|
||||
/// Clear MSI-X Enable.
|
||||
pub fn disable(self: *const MsiX) void {
|
||||
self.writeControl(pci_class.msix.control_enable, false);
|
||||
}
|
||||
fn writeControl(self: *const MsiX, bit: u16, set: bool) void {
|
||||
const at = self.capability + pci_class.msix.control;
|
||||
const control = mmio.readRegister(u16, at);
|
||||
mmio.writeRegister(u16, at, if (set) control | bit else control & ~bit);
|
||||
}
|
||||
};
|
||||
|
||||
/// One extended capability. `offset` is the ABSOLUTE virtual address of its header,
|
||||
/// like `Capability.offset`.
|
||||
pub const ExtendedCapability = struct { id: u16, version: u4, offset: usize };
|
||||
|
||||
pub const ExtendedCapabilityIterator = struct {
|
||||
config: usize,
|
||||
cursor: u16 = @intCast(pci_class.extended_capability_start),
|
||||
guard: u32 = 0, // bounds a malformed chain (480 = the 0xF00-byte space / 8-byte minimum spacing)
|
||||
|
||||
pub fn next(self: *ExtendedCapabilityIterator) ?ExtendedCapability {
|
||||
if (self.cursor == 0 or self.guard >= 480) return null;
|
||||
self.guard += 1;
|
||||
const at = self.config + self.cursor;
|
||||
const header = pci_class.ExtendedCapabilityHeader.decode(mmio.readRegister(u32, at));
|
||||
// Id 0 marks an empty list; all-ones is a conventional-PCI function (no
|
||||
// extended space — reads come back as FFs).
|
||||
if (header.id == 0 or header.id == 0xFFFF) return null;
|
||||
// A next pointer below 0x100 would walk into the legacy header; treat it as the
|
||||
// terminator it must be (0 is the normal one). The 0xFFC decode mask already
|
||||
// keeps `config + cursor + 4` inside the 4 KiB page.
|
||||
self.cursor = if (header.next >= pci_class.extended_capability_start) header.next else 0;
|
||||
return .{ .id = header.id, .version = header.version, .offset = at };
|
||||
}
|
||||
};
|
||||
|
||||
@@ -99,6 +99,19 @@ pub const Device = struct {
|
||||
return std.mem.bytesToValue(usb_transfer_protocol.InterruptSubscribeReply, reply[0..@sizeOf(usb_transfer_protocol.InterruptSubscribeReply)]).status == 0;
|
||||
}
|
||||
|
||||
/// Hand the controller a DMA-region capability (`handle` — from a `shareable`
|
||||
/// dma_alloc, or forwarded from another process) so it binds that buffer into its
|
||||
/// IOMMU domain. Must be called for every buffer whose physical address this device
|
||||
/// will name in a `bulk` transfer, before the transfer. Harmless (and a no-op
|
||||
/// success) when no IOMMU is enforcing. Returns false on failure.
|
||||
pub fn attachDma(self: *Device, handle: ipc.Handle) bool {
|
||||
var request = usb_transfer_protocol.DmaAttachRequest{ .device_token = self.token };
|
||||
var reply: [@sizeOf(usb_transfer_protocol.DmaAttachReply)]u8 = undefined;
|
||||
const result = ipc.callCap(self.bus, std.mem.asBytes(&request), &reply, handle) catch return false;
|
||||
if (result.len < @sizeOf(usb_transfer_protocol.DmaAttachReply)) return false;
|
||||
return std.mem.bytesToValue(usb_transfer_protocol.DmaAttachReply, reply[0..@sizeOf(usb_transfer_protocol.DmaAttachReply)]).status == 0;
|
||||
}
|
||||
|
||||
/// One bulk transfer (IN or OUT per `endpoint_address`'s direction bit) to or
|
||||
/// from the caller's own DMA buffer at `physical`. Returns the bytes moved.
|
||||
pub fn bulk(self: *Device, endpoint_address: u8, physical: u64, length: u32) ?u32 {
|
||||
|
||||
@@ -0,0 +1,104 @@
|
||||
//! The "kernel" library domain (library/kernel): the userspace private-ABI
|
||||
//! library (kernel32-style), split by concern into directly-importable
|
||||
//! modules. The graph is a DAG: memory depends on thread (heap needs
|
||||
//! Thread.Mutex), and thread does its own raw mmap so there is no cycle.
|
||||
//!
|
||||
//! This package also exports `abi` — the kernel <-> user contract (SystemCall
|
||||
//! numbers, mmap prot flags, page_size). Its source lives with the kernel in
|
||||
//! system/abi.zig, outside this directory, but userspace's one view of it is
|
||||
//! exported here so every consumer names the same module instance. Reaching
|
||||
//! outside the package root means this package is valid only as an in-repo
|
||||
//! path dependency (never fetchable by hash) — fine, since path dependencies
|
||||
//! are the only way danos packages are consumed.
|
||||
//!
|
||||
//! The root shim (root.zig) and the user link script (user.ld) are plain
|
||||
//! files, not modules; build-support reaches them through this package's
|
||||
//! directory (Dependency.path).
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const protocol = b.dependency("protocol", .{});
|
||||
|
||||
const abi = b.addModule("abi", .{
|
||||
.root_source_file = b.path("../../system/abi.zig"),
|
||||
});
|
||||
const system_call = b.addModule("system-call", .{
|
||||
.root_source_file = b.path("system-call.zig"),
|
||||
.imports = &.{.{ .name = "abi", .module = abi }},
|
||||
});
|
||||
const ipc = b.addModule("ipc", .{
|
||||
.root_source_file = b.path("ipc.zig"),
|
||||
.imports = &.{ .{ .name = "abi", .module = abi }, .{ .name = "system-call", .module = system_call } },
|
||||
});
|
||||
const time = b.addModule("time", .{
|
||||
.root_source_file = b.path("time.zig"),
|
||||
.imports = &.{.{ .name = "system-call", .module = system_call }},
|
||||
});
|
||||
const thread = b.addModule("thread", .{
|
||||
.root_source_file = b.path("thread.zig"),
|
||||
.imports = &.{ .{ .name = "abi", .module = abi }, .{ .name = "system-call", .module = system_call } },
|
||||
});
|
||||
const logging = b.addModule("logging", .{
|
||||
.root_source_file = b.path("logging.zig"),
|
||||
.imports = &.{ .{ .name = "abi", .module = abi }, .{ .name = "system-call", .module = system_call } },
|
||||
});
|
||||
const process = b.addModule("process", .{
|
||||
.root_source_file = b.path("process.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi },
|
||||
.{ .name = "system-call", .module = system_call },
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "time", .module = time },
|
||||
},
|
||||
});
|
||||
_ = b.addModule("file-system", .{
|
||||
.root_source_file = b.path("file-system.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi },
|
||||
.{ .name = "system-call", .module = system_call },
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "vfs-protocol", .module = protocol.module("vfs-protocol") },
|
||||
},
|
||||
});
|
||||
_ = b.addModule("memory", .{
|
||||
.root_source_file = b.path("memory/memory.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi },
|
||||
.{ .name = "system-call", .module = system_call },
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "thread", .module = thread },
|
||||
},
|
||||
});
|
||||
_ = b.addModule("service", .{
|
||||
.root_source_file = b.path("service.zig"),
|
||||
.imports = &.{
|
||||
.{ .name = "abi", .module = abi },
|
||||
.{ .name = "ipc", .module = ipc },
|
||||
.{ .name = "process", .module = process },
|
||||
},
|
||||
});
|
||||
_ = b.addModule("start", .{
|
||||
.root_source_file = b.path("start.zig"),
|
||||
.imports = &.{ .{ .name = "process", .module = process }, .{ .name = "logging", .module = logging } },
|
||||
});
|
||||
|
||||
// Standalone `zig build test` for this domain alone; the root build keeps
|
||||
// its aggregate test step. time and thread pull in the syscall wrappers,
|
||||
// which need the `abi` module; their danos seams fall back to host
|
||||
// primitives off the danos target, so they run with real host threads.
|
||||
const test_step = b.step("test", "Run the kernel library unit tests");
|
||||
for ([_][]const u8{
|
||||
"time.zig", // Instant/Duration arithmetic
|
||||
"thread.zig", // Mutex/Condition/RwLock/WaitGroup state machines
|
||||
}) |root| {
|
||||
const kernel_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(root),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
.imports = &.{.{ .name = "abi", .module = abi }},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(kernel_tests).step);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
.{
|
||||
.name = .kernel,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x5dd29aab36503453, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// file-system speaks the VFS wire protocol.
|
||||
.protocol = .{ .path = "../protocol" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -34,6 +34,14 @@ pub fn register(id: abi.ServiceId, h: Handle) bool {
|
||||
return !failed(sc.systemCall2(.ipc_register, @intFromEnum(id), h));
|
||||
}
|
||||
|
||||
/// Drop a capability handle (endpoint, shared-memory, or DMA-region) and free its table
|
||||
/// slot. A forwarding hop closes a cap it passed on; a binder closes a DMA-region cap
|
||||
/// once the binding holds its own reference — the 32-slot table is otherwise consumed by
|
||||
/// repeated cap-passing.
|
||||
pub fn close(h: Handle) bool {
|
||||
return !failed(sc.systemCall1(.handle_close, h));
|
||||
}
|
||||
|
||||
/// Find the endpoint published under `id`, installing a handle to it in this
|
||||
/// process.
|
||||
pub fn lookup(id: abi.ServiceId) ?Handle {
|
||||
|
||||
@@ -12,12 +12,18 @@ const sc = @import("system-call");
|
||||
pub const coherent: usize = abi.dma_coherent;
|
||||
pub const write_combining: usize = abi.dma_write_combining;
|
||||
pub const below_4g: usize = abi.dma_below_4g;
|
||||
/// Ask for a capability handle (in `Region.handle`) so the buffer can be delegated to
|
||||
/// another driver and bound into a device's IOMMU domain (`driver.dmaBind`). A driver's
|
||||
/// private rings don't need it; a buffer whose physical address crosses IPC does.
|
||||
pub const shareable: usize = abi.dma_shareable;
|
||||
|
||||
/// A DMA allocation: the `virtual` address the CPU touches, and the `physical` address
|
||||
/// to program into the device's descriptor-ring / base registers.
|
||||
/// A DMA allocation: the `virtual` address the CPU touches, the `physical` address to
|
||||
/// program into the device's registers, and — when `shareable` was requested — a
|
||||
/// capability `handle` naming the region for delegation (null otherwise).
|
||||
pub const Region = struct {
|
||||
virtual: usize,
|
||||
physical: usize,
|
||||
handle: ?usize = null,
|
||||
};
|
||||
|
||||
inline fn failed(r: usize) bool {
|
||||
@@ -25,21 +31,23 @@ inline fn failed(r: usize) bool {
|
||||
}
|
||||
|
||||
/// Allocate `len` bytes of DMA-capable memory with `flags` (e.g. `coherent`, or
|
||||
/// `coherent | below_4g`). Returns the virtual/physical pair, or null on failure. Two
|
||||
/// return values — the virtual address in rax, the physical address in rdx — so it
|
||||
/// needs a hand-written stub.
|
||||
/// `coherent | shareable`). Returns virtual/physical (and a handle when `shareable`), or
|
||||
/// null on failure. Three return values — virtual in rax, physical in rdx, handle in r8
|
||||
/// — so it needs a hand-written stub.
|
||||
pub fn alloc(len: usize, flags: usize) ?Region {
|
||||
var rax: usize = undefined;
|
||||
var rdx: usize = undefined; // out: physical address
|
||||
var r8: usize = undefined; // out: capability handle (abi.no_cap unless shareable)
|
||||
asm volatile ("syscall"
|
||||
: [rax] "={rax}" (rax),
|
||||
[rdx] "={rdx}" (rdx),
|
||||
[r8] "={r8}" (r8),
|
||||
: [n] "{rax}" (@intFromEnum(abi.SystemCall.dma_alloc)),
|
||||
[a0] "{rdi}" (len),
|
||||
[a1] "{rsi}" (flags),
|
||||
: .{ .rcx = true, .r11 = true, .memory = true });
|
||||
if (failed(rax)) return null;
|
||||
return .{ .virtual = rax, .physical = rdx };
|
||||
return .{ .virtual = rax, .physical = rdx, .handle = if (r8 == abi.no_cap) null else r8 };
|
||||
}
|
||||
|
||||
/// Release a region from a prior `alloc` (`virtual` and the same `len`).
|
||||
|
||||
@@ -38,6 +38,7 @@ pub const DmaRegion = dma.Region;
|
||||
pub const dma_coherent = dma.coherent;
|
||||
pub const dma_write_combining = dma.write_combining;
|
||||
pub const dma_below_4g = dma.below_4g;
|
||||
pub const dma_shareable = dma.shareable;
|
||||
pub const dmaAlloc = dma.alloc;
|
||||
pub const dmaFree = dma.free;
|
||||
|
||||
|
||||
@@ -7,7 +7,9 @@
|
||||
//! buffer**, named by its physical address — the same physical-address handoff
|
||||
//! usb-storage already uses toward the controller, one layer up. So a 512-byte
|
||||
//! sector never has to cross the 256-byte IPC boundary; only the small request /
|
||||
//! reply headers do. (Safe while the IOMMU is unenforced; see docs/driver-model.md.)
|
||||
//! reply headers do. Under an enforcing IOMMU the buffer's physical addresses are
|
||||
//! only reachable by the device once the filesystem has `attach`ed the buffer's
|
||||
//! capability (the block server forwards it to the controller); see docs/driver-model.md.
|
||||
|
||||
pub const Operation = enum(u32) {
|
||||
/// geometry() -> { block_size, block_count }
|
||||
@@ -20,6 +22,11 @@ pub const Operation = enum(u32) {
|
||||
/// A filesystem calls this to make prior writes durable — e.g. before power-off,
|
||||
/// so a shutdown-time write isn't lost in the USB flash controller's cache.
|
||||
flush = 3,
|
||||
/// attach(): the caller's DMA-region capability rides the call's cap slot; the
|
||||
/// block server forwards it to the controller so the buffer's physical addresses
|
||||
/// (named in later read/write) are reachable by the device under an enforcing
|
||||
/// IOMMU. Call once per buffer before using it in a transfer.
|
||||
attach = 4,
|
||||
};
|
||||
|
||||
pub const Request = extern struct {
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
//! The "protocol" library domain: the wire protocols — each service's public
|
||||
//! interface, exposed as its own module (docs/driver-model.md). Both sides of
|
||||
//! every conversation depend on the contract by name; neither reaches into the
|
||||
//! other's files. Pure flat wire types: no protocol module imports anything.
|
||||
//!
|
||||
//! vfs-protocol : the VFS server <-> the file layer (unistd/stdio)
|
||||
//! input-protocol : the input fan-out service <-> sources + subscribers
|
||||
//! block-protocol : a filesystem <-> a block driver (usb-storage)
|
||||
//! usb-transfer-protocol : a USB class driver <-> the xHCI bus driver
|
||||
//! device-manager-protocol : the device manager <-> drivers + discovery
|
||||
//! display-protocol : the compositor's client-facing surface
|
||||
//! scanout-protocol : the compositor -> a native scanout driver (docs/display-v2.md)
|
||||
//! power-protocol : system power's domain-named surface (docs/power.md)
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
for ([_]struct { name: []const u8, root: []const u8 }{
|
||||
.{ .name = "vfs-protocol", .root = "vfs/vfs-protocol.zig" },
|
||||
.{ .name = "input-protocol", .root = "input/input-protocol.zig" },
|
||||
.{ .name = "block-protocol", .root = "block/block-protocol.zig" },
|
||||
.{ .name = "usb-transfer-protocol", .root = "usb-transfer/usb-transfer-protocol.zig" },
|
||||
.{ .name = "device-manager-protocol", .root = "device-manager/device-manager-protocol.zig" },
|
||||
.{ .name = "display-protocol", .root = "display/display-protocol.zig" },
|
||||
.{ .name = "scanout-protocol", .root = "scanout/scanout-protocol.zig" },
|
||||
.{ .name = "power-protocol", .root = "power/power-protocol.zig" },
|
||||
}) |protocol| {
|
||||
_ = b.addModule(protocol.name, .{ .root_source_file = b.path(protocol.root) });
|
||||
}
|
||||
|
||||
// Standalone `zig build test` for this domain alone; the root build keeps
|
||||
// its aggregate test step.
|
||||
const test_step = b.step("test", "Run the protocol unit tests");
|
||||
for ([_][]const u8{
|
||||
"vfs/vfs-protocol.zig", // NodeKind / DirectoryEntry sizes + op values
|
||||
"display/display-protocol.zig", // pack(): native pixel encoding per format
|
||||
}) |root| {
|
||||
const protocol_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(root),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(protocol_tests).step);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,8 @@
|
||||
.{
|
||||
.name = .protocol,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0xc8c0bc4c4d551283, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -45,6 +45,11 @@ pub const Operation = enum(u32) {
|
||||
control = 1,
|
||||
interrupt_subscribe = 2,
|
||||
bulk = 3,
|
||||
/// dma_attach: a class driver hands the controller a DMA-region capability (riding
|
||||
/// the call's cap slot) so the controller binds that buffer into its IOMMU domain
|
||||
/// and may then DMA to the physical addresses inside it. Needed once per buffer the
|
||||
/// class driver will name in a `bulk` transfer (its own, or one forwarded to it).
|
||||
dma_attach = 4,
|
||||
};
|
||||
|
||||
/// The endpoint facts a class driver needs, lifted from the endpoint descriptor
|
||||
@@ -138,6 +143,19 @@ pub const BulkReply = extern struct {
|
||||
actual_length: u32,
|
||||
};
|
||||
|
||||
/// dma_attach: the region capability rides the call's cap slot; the body only carries
|
||||
/// the device token (scoping) so the controller knows which caller is attaching.
|
||||
pub const DmaAttachRequest = extern struct {
|
||||
operation: u32 = @intFromEnum(Operation.dma_attach),
|
||||
reserved: u32 = 0,
|
||||
device_token: u64,
|
||||
};
|
||||
|
||||
pub const DmaAttachReply = extern struct {
|
||||
status: i32,
|
||||
reserved: u32 = 0,
|
||||
};
|
||||
|
||||
/// An asynchronous interrupt report, pushed with `ipc.send` to a subscriber's
|
||||
/// endpoint. `Received.isMessage()` is set; there is no reply owed.
|
||||
pub const InterruptReport = extern struct {
|
||||
|
||||
@@ -0,0 +1,31 @@
|
||||
//! The "xkeyboard-config" library domain: keyboard layouts compiled from the
|
||||
//! X11 xkeyboard-config database into native Zig (keycode + modifiers ->
|
||||
//! keysym/character). The `layouts` tables are generated by
|
||||
//! tools/make-xkeyboard-config.py; `xkeyboard-config` is the hand-written API
|
||||
//! over them.
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const layouts = b.addModule("layouts", .{
|
||||
.root_source_file = b.path("generated/layouts.zig"),
|
||||
});
|
||||
_ = b.addModule("xkeyboard-config", .{
|
||||
.root_source_file = b.path("xkeyboard-config.zig"),
|
||||
.imports = &.{.{ .name = "layouts", .module = layouts }},
|
||||
});
|
||||
|
||||
// Standalone `zig build test` for this domain alone; the root build keeps
|
||||
// its aggregate test step. The keycode->character assertions are the
|
||||
// end-to-end proof that the xkb-data -> generator -> Zig-lookup pipeline
|
||||
// is correct.
|
||||
const test_step = b.step("test", "Run the xkeyboard-config unit tests");
|
||||
const xkb_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("xkeyboard-config.zig"),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
.imports = &.{.{ .name = "layouts", .module = layouts }},
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(xkb_tests).step);
|
||||
}
|
||||
@@ -0,0 +1,8 @@
|
||||
.{
|
||||
.name = .xkeyboard_config,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0xea5abe82f08b6eae, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -76,6 +76,10 @@ pub const SystemCall = enum(u64) {
|
||||
fs_node = 47, // fs_node(op, node_token, offset, buf_ptr, buf_len) -> bytes/0/-errno: read/status/readdir on a kernel-served node (op values mirror the vfs-protocol Operation numbers)
|
||||
fs_mount = 48, // fs_mount(prefix_ptr, prefix_len, backend_handle, rewrite_ptr, rewrite_len) -> 0/-errno: mount a userspace filesystem's endpoint at an absolute prefix (possession of the handle is the capability)
|
||||
fs_unmount = 49, // fs_unmount(prefix_ptr, prefix_len) -> 0/-errno: remove a backend mount
|
||||
iommu_fault_drain = 50, // iommu_fault_drain() -> count: drain + log pending IOMMU translation faults (a diagnostic; the count of faults seen this call)
|
||||
dma_bind = 51, // dma_bind(device_id, region_handle) -> 0/-errno: map a DMA-region capability into the claimed device's IOMMU domain (idempotent). The caller must own the device and hold the handle
|
||||
dma_unbind = 52, // dma_unbind(device_id, region_handle) -> 0/-errno: unmap a previously bound region from the device's domain and invalidate
|
||||
handle_close = 53, // handle_close(handle) -> 0/-errno: drop one capability handle and free its table slot (endpoints, shared-memory, DMA regions)
|
||||
_,
|
||||
};
|
||||
|
||||
@@ -113,6 +117,7 @@ pub const msi_address_base: u64 = 0xFEE0_0000;
|
||||
pub const dma_coherent: u64 = 1; // strong-uncacheable — the default, the only portable one
|
||||
pub const dma_write_combining: u64 = 2; // write-combining (framebuffers); needs PAT
|
||||
pub const dma_below_4g: u64 = 4; // physical address must fit 32 bits (legacy DMA engines)
|
||||
pub const dma_shareable: u64 = 8; // return a capability handle (r8) so the region can be delegated + dma_bound
|
||||
|
||||
/// Set in the badge returned by `ipc_reply_wait` when what arrived is an
|
||||
/// **asynchronous notification** (a device interrupt bound with `irq_bind`, or a
|
||||
|
||||
@@ -0,0 +1,18 @@
|
||||
//! The pci-bus driver as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! the shared recipe and the module-to-domain map live in build-support.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "pci-bus",
|
||||
.root_source_file = b.path("pci-bus.zig"),
|
||||
.imports = &.{
|
||||
"device-manager-protocol", "driver", "ipc", "logging", "memory", "pci-class",
|
||||
"process", "service",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
.{
|
||||
.name = .pci_bus,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x283fca121f0bb145, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.device = .{ .path = "../../../library/device" },
|
||||
.protocol = .{ .path = "../../../library/protocol" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -0,0 +1,50 @@
|
||||
//! The ps2-bus driver as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! the shared recipe and the module-to-domain map live in build-support.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const ps2_bus_exe = build_support.userBinary(b, .{
|
||||
.name = "ps2-bus",
|
||||
.root_source_file = b.path("ps2-bus.zig"),
|
||||
.imports = &.{ "acpi-ids", "driver", "ipc", "logging", "memory", "process", "service", "time" },
|
||||
});
|
||||
b.installArtifact(ps2_bus_exe);
|
||||
|
||||
const ps2_keyboard_exe = build_support.userBinary(b, .{
|
||||
.name = "ps2-keyboard",
|
||||
.root_source_file = b.path("keyboard.zig"),
|
||||
.imports = &.{
|
||||
"acpi-ids", "driver", "input-client", "input-protocol", "ipc", "logging", "memory",
|
||||
"process", "time", "xkeyboard-config",
|
||||
},
|
||||
});
|
||||
b.installArtifact(ps2_keyboard_exe);
|
||||
|
||||
const ps2_mouse_exe = build_support.userBinary(b, .{
|
||||
.name = "ps2-mouse",
|
||||
.root_source_file = b.path("mouse.zig"),
|
||||
.imports = &.{
|
||||
"acpi-ids", "driver", "input-client", "input-protocol", "ipc", "logging", "memory",
|
||||
"process", "time",
|
||||
},
|
||||
});
|
||||
b.installArtifact(ps2_mouse_exe);
|
||||
|
||||
// Standalone `zig build test`; the root aggregate depends on this step.
|
||||
const test_step = b.step("test", "Run the ps2-bus unit tests");
|
||||
for ([_][]const u8{
|
||||
"scancode.zig", // set-2 decode + keyboard state machine
|
||||
"mouse-packet.zig", // 3-byte mouse packet assembly
|
||||
}) |test_root| {
|
||||
const unit_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(test_root),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(unit_tests).step);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,18 @@
|
||||
.{
|
||||
.name = .ps2_bus,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x642a365353bf7de9, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.client = .{ .path = "../../../library/client" },
|
||||
.device = .{ .path = "../../../library/device" },
|
||||
.protocol = .{ .path = "../../../library/protocol" },
|
||||
.@"xkeyboard-config" = .{ .path = "../../../library/xkeyboard-config" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -19,7 +19,7 @@ const device = @import("driver");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const time = @import("time");
|
||||
const input = @import("input");
|
||||
const input = @import("input-client");
|
||||
const memory = @import("memory");
|
||||
const logging = @import("logging");
|
||||
const xkb = @import("xkeyboard-config");
|
||||
|
||||
@@ -19,7 +19,7 @@ const device = @import("driver");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const time = @import("time");
|
||||
const input = @import("input");
|
||||
const input = @import("input-client");
|
||||
const memory = @import("memory");
|
||||
const logging = @import("logging");
|
||||
const ps2 = @import("ps2-library.zig");
|
||||
|
||||
@@ -0,0 +1,42 @@
|
||||
//! The usb-hid driver as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! the shared recipe and the module-to-domain map live in build-support.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const usb_hid_keyboard_exe = build_support.userBinary(b, .{
|
||||
.name = "usb-hid-keyboard",
|
||||
.root_source_file = b.path("keyboard.zig"),
|
||||
.imports = &.{
|
||||
"driver", "input-client", "input-protocol", "ipc", "logging", "process", "service",
|
||||
"usb", "usb-abi", "xkeyboard-config",
|
||||
},
|
||||
});
|
||||
b.installArtifact(usb_hid_keyboard_exe);
|
||||
|
||||
const usb_hid_mouse_exe = build_support.userBinary(b, .{
|
||||
.name = "usb-hid-mouse",
|
||||
.root_source_file = b.path("mouse.zig"),
|
||||
.imports = &.{
|
||||
"driver", "input-client", "input-protocol", "ipc", "logging", "process", "service",
|
||||
"usb", "usb-abi",
|
||||
},
|
||||
});
|
||||
b.installArtifact(usb_hid_mouse_exe);
|
||||
|
||||
// Standalone `zig build test`; the root aggregate depends on this step.
|
||||
const test_step = b.step("test", "Run the usb-hid unit tests");
|
||||
for ([_][]const u8{
|
||||
"hid-report.zig", // HID boot-report keyboard/mouse decode
|
||||
}) |test_root| {
|
||||
const unit_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(test_root),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(unit_tests).step);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,18 @@
|
||||
.{
|
||||
.name = .usb_hid,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x66328b738fffff01, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.client = .{ .path = "../../../library/client" },
|
||||
.device = .{ .path = "../../../library/device" },
|
||||
.protocol = .{ .path = "../../../library/protocol" },
|
||||
.@"xkeyboard-config" = .{ .path = "../../../library/xkeyboard-config" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -18,7 +18,7 @@ const std = @import("std");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
const input = @import("input");
|
||||
const input = @import("input-client");
|
||||
const device_manager = @import("driver");
|
||||
const logging = @import("logging");
|
||||
const usb = @import("usb");
|
||||
|
||||
@@ -14,7 +14,7 @@ const std = @import("std");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
const input = @import("input");
|
||||
const input = @import("input-client");
|
||||
const device_manager = @import("driver");
|
||||
const logging = @import("logging");
|
||||
const usb = @import("usb");
|
||||
|
||||
@@ -0,0 +1,33 @@
|
||||
//! The usb-storage driver as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! the shared recipe and the module-to-domain map live in build-support.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "usb-storage",
|
||||
.root_source_file = b.path("usb-storage.zig"),
|
||||
.imports = &.{
|
||||
"block-protocol", "driver", "ipc", "logging", "memory", "process", "service",
|
||||
"time", "usb",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
|
||||
// Standalone `zig build test`; the root aggregate depends on this step.
|
||||
const test_step = b.step("test", "Run the usb-storage unit tests");
|
||||
for ([_][]const u8{
|
||||
"bulk-only-transport.zig", // CBW/CSW wrapper sizes
|
||||
"scsi.zig", // SCSI CDB encodings (big-endian)
|
||||
}) |test_root| {
|
||||
const unit_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(test_root),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(unit_tests).step);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
.{
|
||||
.name = .usb_storage,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0xce09fdc4c50bb4fe, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.device = .{ .path = "../../../library/device" },
|
||||
.protocol = .{ .path = "../../../library/protocol" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -90,9 +90,22 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
bring_up_failed = true;
|
||||
return false;
|
||||
};
|
||||
command_wrapper = memory.dmaAlloc(4096, memory.dma_coherent) orelse return false;
|
||||
status_wrapper = memory.dmaAlloc(4096, memory.dma_coherent) orelse return false;
|
||||
command_data = memory.dmaAlloc(4096, memory.dma_coherent) orelse return false;
|
||||
// Shareable, so each buffer's capability can be handed to the controller: usb-storage
|
||||
// owns no device, so its buffers are not auto-bound anywhere — the controller reaches
|
||||
// them only once attached. (No-op binding when no IOMMU is enforcing.)
|
||||
command_wrapper = memory.dmaAlloc(4096, memory.dma_coherent | memory.dma_shareable) orelse return false;
|
||||
status_wrapper = memory.dmaAlloc(4096, memory.dma_coherent | memory.dma_shareable) orelse return false;
|
||||
command_data = memory.dmaAlloc(4096, memory.dma_coherent | memory.dma_shareable) orelse return false;
|
||||
for ([_]memory.DmaRegion{ command_wrapper, status_wrapper, command_data }) |region| {
|
||||
if (region.handle) |handle| {
|
||||
if (!device.attachDma(handle)) {
|
||||
_ = logging.write("/system/drivers/usb-storage: could not attach a DMA buffer to the controller\n");
|
||||
bring_up_failed = true;
|
||||
return false;
|
||||
}
|
||||
_ = ipc.close(handle); // the binding holds its own reference now
|
||||
}
|
||||
}
|
||||
|
||||
// Bring the LUN up: wait for it to be ready (clearing the initial unit-attention
|
||||
// with REQUEST SENSE), identify it, and read its capacity.
|
||||
@@ -135,10 +148,17 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
/// caller's DMA buffer (named by physical address).
|
||||
fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Handle) usize {
|
||||
_ = sender;
|
||||
_ = capability;
|
||||
if (message.len < block_protocol.request_size) return 0;
|
||||
const request = std.mem.bytesToValue(block_protocol.Request, message[0..block_protocol.request_size]);
|
||||
switch (request.operation) {
|
||||
@intFromEnum(block_protocol.Operation.attach) => {
|
||||
// The filesystem's DMA buffer: forward its capability to the controller so
|
||||
// the device can reach it, then release our copy (the binding holds a ref).
|
||||
const handle = capability orelse return writeReply(reply, .{ .status = -1, .block_size = 0, .block_count = 0 });
|
||||
const ok = device.attachDma(handle);
|
||||
_ = ipc.close(handle);
|
||||
return writeReply(reply, .{ .status = if (ok) 0 else -1, .block_size = 0, .block_count = 0 });
|
||||
},
|
||||
@intFromEnum(block_protocol.Operation.geometry) => {
|
||||
return writeReply(reply, .{ .status = 0, .block_size = block_size, .block_count = block_count });
|
||||
},
|
||||
|
||||
@@ -0,0 +1,19 @@
|
||||
//! The usb-xhci-bus driver as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! the shared recipe and the module-to-domain map live in build-support.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "usb-xhci-bus",
|
||||
.root_source_file = b.path("usb-xhci-bus.zig"),
|
||||
.imports = &.{
|
||||
"device-manager-protocol", "driver", "input-client", "ipc", "logging", "memory",
|
||||
"mmio", "pci", "process", "service", "time", "usb-abi", "usb-ids",
|
||||
"usb-transfer-protocol",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
}
|
||||
@@ -0,0 +1,17 @@
|
||||
.{
|
||||
.name = .usb_xhci_bus,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x46d21373f05f6b20, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.client = .{ .path = "../../../library/client" },
|
||||
.device = .{ .path = "../../../library/device" },
|
||||
.protocol = .{ .path = "../../../library/protocol" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -19,7 +19,7 @@ const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
const time = @import("time");
|
||||
const input = @import("input");
|
||||
const input = @import("input-client");
|
||||
const device_manager = @import("driver");
|
||||
const memory = @import("memory");
|
||||
const logging = @import("logging");
|
||||
@@ -28,18 +28,37 @@ const usb_ids = @import("usb-ids");
|
||||
const usb_abi = @import("usb-abi");
|
||||
const usb_transfer_protocol = @import("usb-transfer-protocol");
|
||||
const library = @import("usb-xhci-library.zig");
|
||||
const pci = @import("pci");
|
||||
|
||||
/// The controller engine (reset, rings, transfers), stood up in `initialise`.
|
||||
var controller: ?library.Controller = null;
|
||||
|
||||
/// This driver's service endpoint (registered as `.usb_bus`), where class-driver
|
||||
/// requests, signals, and the interrupt-poll timer all arrive.
|
||||
/// requests, signals, MSI notifications, and the poll/reconcile timer all arrive.
|
||||
var service_endpoint: ipc.Handle = 0;
|
||||
|
||||
/// How often the driver drains the event ring for interrupt reports (~125 Hz),
|
||||
/// re-armed each tick. Frequent enough for responsive input.
|
||||
/// How often the driver drains the event ring for interrupt reports (~125 Hz) when
|
||||
/// polling, re-armed each tick. Frequent enough for responsive input.
|
||||
const poll_interval_ms: u64 = 8;
|
||||
|
||||
/// The timer interval in MSI mode: the ring is drained at interrupt time, and the tick
|
||||
/// only reconciles root ports (real hardware delivers late USB2 companion-hub debounce
|
||||
/// with no reliable port-change event — see onNotification) and un-wedges a lost MSI
|
||||
/// edge (edge-triggered, no kernel mask/ack: a missed IP clear stalls, never storms).
|
||||
const reconcile_interval_ms: u64 = 250;
|
||||
|
||||
/// The controller's own descriptor, kept at file scope because `pci.Function` holds a
|
||||
/// pointer to it for the whole bring-up.
|
||||
var controller_descriptor: device.DeviceDescriptor = undefined;
|
||||
|
||||
/// Non-null iff MSI mode is active: the vector whose notification badge means "the
|
||||
/// controller interrupted". Null means the 8 ms polling fallback is running.
|
||||
var msi_vector: ?u32 = null;
|
||||
|
||||
fn timerInterval() u64 {
|
||||
return if (msi_vector != null) reconcile_interval_ms else poll_interval_ms;
|
||||
}
|
||||
|
||||
/// The class driver endpoints that opened each device, so interrupt reports can
|
||||
/// be pushed back to them. Keyed by the device token (the interface's device id).
|
||||
const Open = struct {
|
||||
@@ -95,6 +114,7 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
std.log.info("device {d} not in the device tree", .{controller_id});
|
||||
return false;
|
||||
};
|
||||
controller_descriptor = descriptor;
|
||||
|
||||
// The xHC's registers live behind the first memory BAR. Resource 0 is the
|
||||
// function's ECAM configuration space (M15), so the walk starts at 1.
|
||||
@@ -118,6 +138,14 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
return false;
|
||||
};
|
||||
|
||||
// Message-signalled interrupt setup comes BEFORE the controller bring-up, not
|
||||
// after: Controller.init writes IMAN.IE, and QEMU's xhci only registers the MSI-X
|
||||
// vector as in-use when that write happens with MSI-X already enabled (its
|
||||
// intr_update callback early-outs on !msix_enabled, and msix_notify silently
|
||||
// drops interrupts for an unused vector). Real hardware does not care about the
|
||||
// order; QEMU requires it.
|
||||
setupMsi();
|
||||
|
||||
// Bring the controller up: reset it, stand up the command and event rings,
|
||||
// and start it running (the hardware half lives in usb-xhci-library.zig).
|
||||
controller = library.Controller.init(register_base) orelse {
|
||||
@@ -146,12 +174,52 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
|
||||
scanPorts(handle);
|
||||
|
||||
// Arm the poll timer that drains interrupt reports from the event ring. It is
|
||||
// re-armed on each tick in onNotification; class drivers subscribe later.
|
||||
_ = time.timerOnce(service_endpoint, poll_interval_ms);
|
||||
// Arm the timer: in polling mode it drains the event ring; in MSI mode it is the
|
||||
// slower port-reconcile/safety-net tick. Re-armed on each tick in onNotification.
|
||||
_ = time.timerOnce(service_endpoint, timerInterval());
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Switch the event ring from timer polling to message-signalled interrupts, if the
|
||||
/// whole path is available: map the function's config space, bind a vector, program
|
||||
/// the MSI capability — or, on an MSI-X-only function (QEMU's qemu-xhci is one: it
|
||||
/// advertises MSI-X and PCIe but no plain MSI), entry 0 of the MSI-X table, which
|
||||
/// takes the same kernel (address, data) pair (xHCI interrupter 0 raises vector 0).
|
||||
/// Any step failing leaves `msi_vector` null and the 8 ms polling path exactly as it
|
||||
/// was. The controller side needs nothing extra — IMAN.IE and USBCMD.INTE are already
|
||||
/// set (see Controller.init: QEMU only writes runtime events with the interrupter
|
||||
/// enabled).
|
||||
///
|
||||
/// After a supervised kill, the kernel drops the vector binding but the device still
|
||||
/// has the interrupt enabled and fires the stale vector; the kernel EOIs it
|
||||
/// harmlessly, and the respawned driver re-runs this with its fresh vector.
|
||||
fn setupMsi() void {
|
||||
var function = pci.Function.map(controller_id, &controller_descriptor) orelse {
|
||||
std.log.info("config-space map failed; polling at {d} ms", .{poll_interval_ms});
|
||||
return;
|
||||
};
|
||||
function.enableMemoryAndBusMaster();
|
||||
const message = device.msiBind(controller_id, service_endpoint) orelse {
|
||||
std.log.info("msi_bind unavailable; polling at {d} ms", .{poll_interval_ms});
|
||||
return;
|
||||
};
|
||||
if (function.programMsi(message)) {
|
||||
msi_vector = message.data;
|
||||
std.log.info("msi active (vector {d}); reconcile tick at {d} ms", .{ message.data, reconcile_interval_ms });
|
||||
return;
|
||||
}
|
||||
if (function.msix()) |table| {
|
||||
if (table.programEntry(0, message) and table.unmaskEntry(0)) {
|
||||
table.enable();
|
||||
function.setInterruptDisable();
|
||||
msi_vector = message.data;
|
||||
std.log.info("msix active (vector {d}); reconcile tick at {d} ms", .{ message.data, reconcile_interval_ms });
|
||||
return;
|
||||
}
|
||||
}
|
||||
std.log.info("no msi/msi-x capability; polling at {d} ms", .{poll_interval_ms});
|
||||
}
|
||||
|
||||
var register_base: usize = 0;
|
||||
|
||||
/// The xHCI default Protocol Speed IDs (the PORTSC port-speed field, bits 13:10)
|
||||
@@ -416,10 +484,22 @@ fn onMessage(message: []const u8, reply: []u8, sender: u32, capability: ?ipc.Han
|
||||
@intFromEnum(usb_transfer_protocol.Operation.control) => handleControl(message, reply),
|
||||
@intFromEnum(usb_transfer_protocol.Operation.interrupt_subscribe) => handleSubscribe(message, reply),
|
||||
@intFromEnum(usb_transfer_protocol.Operation.bulk) => handleBulk(message, reply),
|
||||
@intFromEnum(usb_transfer_protocol.Operation.dma_attach) => handleDmaAttach(message, reply, capability),
|
||||
else => 0,
|
||||
};
|
||||
}
|
||||
|
||||
/// dma_attach: bind the class driver's DMA-region capability into the controller's IOMMU
|
||||
/// domain, so the controller may DMA to the physical addresses inside that buffer. The
|
||||
/// binding holds its own kernel reference, so the forwarded capability is closed here.
|
||||
fn handleDmaAttach(message: []const u8, reply: []u8, capability: ?ipc.Handle) usize {
|
||||
if (message.len < @sizeOf(usb_transfer_protocol.DmaAttachRequest)) return writeReply(reply, usb_transfer_protocol.DmaAttachReply{ .status = -1 });
|
||||
const handle = capability orelse return writeReply(reply, usb_transfer_protocol.DmaAttachReply{ .status = -1 });
|
||||
const ok = device.dmaBind(controller_id, handle);
|
||||
_ = ipc.close(handle);
|
||||
return writeReply(reply, usb_transfer_protocol.DmaAttachReply{ .status = if (ok) 0 else -1 });
|
||||
}
|
||||
|
||||
fn writeReply(reply: []u8, value: anytype) usize {
|
||||
const bytes = std.mem.asBytes(&value);
|
||||
@memcpy(reply[0..bytes.len], bytes);
|
||||
@@ -501,10 +581,27 @@ fn handleBulk(message: []const u8, reply: []u8) usize {
|
||||
return writeReply(reply, usb_transfer_protocol.BulkReply{ .status = if (transferred != null) 0 else -1, .actual_length = transferred orelse 0 });
|
||||
}
|
||||
|
||||
/// The poll timer landed: drain any interrupt reports off the event ring and push
|
||||
/// each to the class driver that subscribed, then re-arm the timer.
|
||||
/// A timer tick or an MSI landed: drain the event ring, reconcile ports, and fan out.
|
||||
/// The timer arm re-arms itself (8 ms drain when polling, 250 ms reconcile under MSI);
|
||||
/// the MSI arm clears the interrupter's pending bit FIRST, then drains — so an event
|
||||
/// arriving after the drain takes IP 0→1 and fires a fresh edge instead of being
|
||||
/// swallowed until the reconcile tick.
|
||||
fn onNotification(badge: u64) void {
|
||||
if (badge & ipc.notify_timer_bit == 0) return;
|
||||
if (badge & ipc.notify_timer_bit != 0) {
|
||||
serviceController();
|
||||
_ = time.timerOnce(service_endpoint, timerInterval());
|
||||
return;
|
||||
}
|
||||
const vector = msi_vector orelse return;
|
||||
if (badge & ~ipc.notify_badge_bit != vector) return;
|
||||
if (controller) |*engine| engine.acknowledgeInterrupt();
|
||||
serviceController();
|
||||
}
|
||||
|
||||
/// Everything one servicing pass does, shared verbatim by the poll/reconcile tick and
|
||||
/// the MSI notification: drain the event ring, reconcile root ports, service hub
|
||||
/// changes, and push interrupt reports to their class drivers.
|
||||
fn serviceController() void {
|
||||
if (controller) |*engine| {
|
||||
engine.pump();
|
||||
// Poll every root port and reconcile — a device present but not yet
|
||||
@@ -564,7 +661,6 @@ fn onNotification(badge: u64) void {
|
||||
_ = ipc.send(report.report_endpoint, std.mem.asBytes(&message));
|
||||
}
|
||||
}
|
||||
_ = time.timerOnce(service_endpoint, poll_interval_ms);
|
||||
}
|
||||
|
||||
pub fn main(init: process.Init) void {
|
||||
|
||||
@@ -1563,6 +1563,15 @@ pub const Controller = struct {
|
||||
return report;
|
||||
}
|
||||
|
||||
/// Clear interrupter 0's pending bit (IMAN.IP). IP is write-1-to-clear, and the
|
||||
/// read-back carries IE (plain read-write) through unchanged. In MSI mode the
|
||||
/// driver clears IP **before** draining the ring: an event that lands after the
|
||||
/// drain then takes IP 0→1 and fires a fresh edge, where clearing afterwards would
|
||||
/// leave a race in which a new event finds IP already set and raises nothing.
|
||||
pub fn acknowledgeInterrupt(self: *const Controller) void {
|
||||
write32(self.interrupter(interrupter_management), read32(self.interrupter(interrupter_management)) | 1);
|
||||
}
|
||||
|
||||
/// Drain any events currently on the event ring: interrupt reports into the
|
||||
/// report queue, PORT STATUS CHANGES into the port-change queue (hot-plug —
|
||||
/// these were silently dropped before M20). Non-blocking — called on the
|
||||
|
||||
@@ -0,0 +1,33 @@
|
||||
//! The virtio-gpu driver as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! the shared recipe and the module-to-domain map live in build-support.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "virtio-gpu",
|
||||
.root_source_file = b.path("virtio-gpu.zig"),
|
||||
.imports = &.{
|
||||
"display-protocol", "driver", "ipc", "logging", "memory", "mmio", "pci", "process",
|
||||
"scanout-protocol", "service", "time",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
|
||||
// Standalone `zig build test`; the root aggregate depends on this step.
|
||||
const test_step = b.step("test", "Run the virtio-gpu unit tests");
|
||||
for ([_][]const u8{
|
||||
"virtio-gpu-protocol.zig", // virtio-gpu command struct sizes
|
||||
"virtio-pci.zig", // virtio 1.0 PCI transport struct sizes
|
||||
}) |test_root| {
|
||||
const unit_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(test_root),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(unit_tests).step);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
.{
|
||||
.name = .virtio_gpu,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0xfb704899c18b9a23, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.device = .{ .path = "../../../library/device" },
|
||||
.protocol = .{ .path = "../../../library/protocol" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -323,6 +323,15 @@ fn initialise(endpoint: ipc.Handle) bool {
|
||||
std.log.info("could not resolve the scanout surface physical address", .{});
|
||||
return false;
|
||||
};
|
||||
// Bind the shared surface into this device's IOMMU domain so the GPU may DMA the
|
||||
// framebuffer (attach_backing points it here). The ring/command buffers are
|
||||
// dma_alloc'd and auto-bound; a shared-memory surface needs an explicit bind. We keep
|
||||
// the handle (it is also passed to the display service), so do not close it. No-op
|
||||
// without an IOMMU.
|
||||
if (!device.dmaBind(device_id, surface.handle)) {
|
||||
std.log.info("could not bind the scanout surface for DMA", .{});
|
||||
return false;
|
||||
}
|
||||
{
|
||||
const request = requestAt(vg.ResourceAttachBacking);
|
||||
request.* = .{
|
||||
|
||||
+153
-25
@@ -95,18 +95,46 @@ pub const PlatformInformation = struct {
|
||||
override_count: usize = 0,
|
||||
/// Whether an IOMMU (VT-d DMA-remapping unit) was found in the ACPI DMAR table.
|
||||
/// When false, `device_claim` on a DMA-capable device is equivalent to granting
|
||||
/// ring 0 — a device can DMA to any physical address (docs/driver-model.md M16).
|
||||
/// Detection is the first step; per-device domain enforcement lands with the first
|
||||
/// DMA driver.
|
||||
/// ring 0 — a device can DMA to any physical address (docs/driver-model.md M16),
|
||||
/// and the kernel says so at every boot (the fail-open platform log line). When
|
||||
/// true, the IOMMU core builds per-device translation domains from this record.
|
||||
iommu_present: bool = false,
|
||||
/// MMIO base of the first DMA-remapping hardware unit (DMAR DRHD), when present.
|
||||
/// true when the present unit is AMD-Vi (from IVRS) rather than Intel VT-d (DMAR).
|
||||
/// The two are mutually exclusive on real hardware; the IOMMU core picks the backend.
|
||||
iommu_is_amd: bool = false,
|
||||
/// MMIO base of the selected DMA-remapping hardware unit — the VT-d DRHD with
|
||||
/// INCLUDE_PCI_ALL (the catch-all unit; falls back to the first), or the AMD-Vi
|
||||
/// IOMMU's control-register base from the first IVHD.
|
||||
iommu_base: u64 = 0,
|
||||
/// The unit's Version register (offset 0x00) — its low byte is major.minor;
|
||||
/// reading it back nonzero confirms a real, mappable VT-d unit.
|
||||
iommu_version: u32 = 0,
|
||||
/// The unit's Capability register (offset 0x08): supported address widths, number
|
||||
/// of domains, etc. Recorded now; consumed when enforcement is built.
|
||||
iommu_capabilities: u64 = 0,
|
||||
/// Whether the selected unit carries INCLUDE_PCI_ALL. False means every unit is
|
||||
/// device-scoped (unusual) — the core still enables on the selected unit but
|
||||
/// devices outside its scope remain untranslated.
|
||||
iommu_include_all: bool = false,
|
||||
/// DRHD units in the DMAR beyond the selected one. Devices scoped to those units
|
||||
/// (typically the integrated GPU) are NOT translated by v1 — the boot log warns.
|
||||
iommu_extra_units: u8 = 0,
|
||||
/// Reserved-memory regions (DMAR RMRRs): firmware-owned buffers a named device
|
||||
/// keeps DMAing into across the OS handoff (classically the xHC keyboard-emulation
|
||||
/// buffer). These must be identity-mapped in the device's domain BEFORE translation
|
||||
/// enables, or platform firmware breaks. Only single-path endpoint scopes are
|
||||
/// recorded; anything fancier is skipped with a loud log at parse time.
|
||||
rmrr: [maximum_rmrr]RmrrRegion = undefined,
|
||||
rmrr_count: usize = 0,
|
||||
/// RMRR device scopes the parser could not record (multi-hop paths, sub-hierarchy
|
||||
/// types, or table overflow). Non-zero means a device keeps an unmapped firmware
|
||||
/// buffer — the kernel boot log warns loudly (the platform module itself is
|
||||
/// log-free by design; it records, the kernel reports).
|
||||
rmrr_skipped: u8 = 0,
|
||||
};
|
||||
|
||||
pub const maximum_rmrr = 8;
|
||||
|
||||
/// One recorded RMRR: the device (requester id) and the inclusive physical range it
|
||||
/// must always be allowed to reach.
|
||||
pub const RmrrRegion = struct {
|
||||
bdf: u16,
|
||||
base: u64,
|
||||
limit: u64,
|
||||
};
|
||||
|
||||
/// Filled in by `discover`; the architecture layer reads it during bring-up.
|
||||
@@ -248,6 +276,8 @@ const SLIT: [4]u8 = "SLIT".*;
|
||||
const SRAT: [4]u8 = "SRAT".*;
|
||||
/// Secondary System Description Table (SSDT)
|
||||
const DMAR: [4]u8 = "DMAR".*;
|
||||
/// I/O Virtualization Reporting Structure (IVRS) — the AMD-Vi analogue of DMAR.
|
||||
const IVRS: [4]u8 = "IVRS".*;
|
||||
const SSDT: [4]u8 = "SSDT".*;
|
||||
/// Serial Port Console Redirection table (SPCR) — the firmware's console UART.
|
||||
const SPCR: [4]u8 = "SPCR".*;
|
||||
@@ -466,7 +496,9 @@ fn handleTable(device_tree: *DeviceTree, hal: Hal, sdt_physical: u64) !void {
|
||||
} else if (std.mem.eql(u8, &sig, &SPCR)) {
|
||||
parseSpcr(header);
|
||||
} else if (std.mem.eql(u8, &sig, &DMAR)) {
|
||||
parseDmar(hal, header);
|
||||
parseDmar(header);
|
||||
} else if (std.mem.eql(u8, &sig, &IVRS)) {
|
||||
parseIvrs(header);
|
||||
} else if (std.mem.eql(u8, &sig, &SSDT)) {
|
||||
// Secondary namespace bytecode — collect it to publish for the ring-3 parse.
|
||||
addAmlBlock(sdt_physical);
|
||||
@@ -757,19 +789,34 @@ fn parseSpcr(header: *const SystemDescriptorTableHeader) void {
|
||||
|
||||
// DMAR remapping-structure layout (Intel VT-d spec §8): the DMAR-specific header is 12
|
||||
// bytes (host-address-width, flags, 10 reserved), then a list of {type u16, length u16}
|
||||
// structures. Type 0 is a DRHD (DMA Remapping Hardware Unit Definition), whose 64-bit
|
||||
// register base sits at offset 8 within it.
|
||||
// structures. Type 0 is a DRHD (DMA Remapping Hardware Unit Definition): flags byte at
|
||||
// offset 4 (bit 0 = INCLUDE_PCI_ALL, the catch-all unit), 64-bit register base at
|
||||
// offset 8. Type 1 is an RMRR (Reserved Memory Region Reporting): a physical range at
|
||||
// offsets 8/16 (base / inclusive limit) that the device(s) named by the trailing
|
||||
// device-scope entries keep DMAing into across the firmware→OS handoff.
|
||||
const dmar_structures_offset = 48; // 36-byte ACPI header + 12-byte DMAR header
|
||||
const dmar_type_drhd: u16 = 0;
|
||||
const dmar_type_rmrr: u16 = 1;
|
||||
const drhd_flags_offset = 4;
|
||||
const drhd_include_pci_all: u8 = 1;
|
||||
const drhd_register_base_offset = 8;
|
||||
const rmrr_base_offset = 8;
|
||||
const rmrr_limit_offset = 16;
|
||||
const rmrr_scopes_offset = 24;
|
||||
// Device-scope entry (within DRHD/RMRR structures): type 1 = PCI endpoint; the path is
|
||||
// (device, function) byte pairs from offset 6, one pair per bridge hop plus the leaf.
|
||||
const scope_type_pci_endpoint: u8 = 1;
|
||||
const scope_start_bus_offset = 5;
|
||||
const scope_path_offset = 6;
|
||||
|
||||
/// DMAR -> detect the IOMMU. Find the first DMA-remapping hardware unit, map its
|
||||
/// register block, and record its version and capabilities. This is *detection only*:
|
||||
/// it tells the system an IOMMU exists (so `device_claim` on a DMA device could one day
|
||||
/// be gated by a per-device translation domain), but no domains are programmed yet —
|
||||
/// enforcement is built with the first DMA driver, which is what there is to protect and
|
||||
/// test against. See docs/driver-model.md (M16), the honest caveat.
|
||||
fn parseDmar(hal: Hal, header: *const SystemDescriptorTableHeader) void {
|
||||
/// DMAR -> the VT-d unit(s) and reserved memory regions. Walks every remapping
|
||||
/// structure: selects the INCLUDE_PCI_ALL DRHD (the catch-all covering all devices not
|
||||
/// scoped elsewhere — commonly the SECOND unit on real machines, after an iGPU-scoped
|
||||
/// one), counts the rest so the boot log can warn that their devices stay untranslated,
|
||||
/// and records single-path endpoint RMRRs for the IOMMU core to pre-map before it
|
||||
/// enables translation. Multi-hop RMRR scopes are skipped loudly: better a named gap
|
||||
/// than a silent one.
|
||||
fn parseDmar(header: *const SystemDescriptorTableHeader) void {
|
||||
const base: [*]align(1) const u8 = @ptrCast(header);
|
||||
const total: usize = header.length;
|
||||
|
||||
@@ -780,13 +827,94 @@ fn parseDmar(hal: Hal, header: *const SystemDescriptorTableHeader) void {
|
||||
if (length < 4 or off + length > total) break; // malformed; stop rather than loop
|
||||
if (kind == dmar_type_drhd) {
|
||||
const register_base = fadt(u64, base, total, off + drhd_register_base_offset) orelse 0;
|
||||
const include_all = ((fadt(u8, base, total, off + drhd_flags_offset) orelse 0) & drhd_include_pci_all) != 0;
|
||||
if (register_base != 0) {
|
||||
const regs = hal.mapMmio(register_base, abi.page_size, true);
|
||||
// Selection: the INCLUDE_PCI_ALL unit wins; otherwise keep the first
|
||||
// seen. A later catch-all replaces an earlier scoped unit.
|
||||
const replace = !platform_information.iommu_present or
|
||||
(include_all and !platform_information.iommu_include_all);
|
||||
if (replace) {
|
||||
// Table facts only: the unit's registers are the IOMMU
|
||||
// backend's business (it maps and validates them at
|
||||
// detect) — discovery records where they live, never
|
||||
// reads them.
|
||||
if (platform_information.iommu_present) platform_information.iommu_extra_units += 1;
|
||||
platform_information.iommu_present = true;
|
||||
platform_information.iommu_base = register_base;
|
||||
platform_information.iommu_include_all = include_all;
|
||||
} else {
|
||||
platform_information.iommu_extra_units += 1;
|
||||
}
|
||||
}
|
||||
} else if (kind == dmar_type_rmrr) {
|
||||
parseRmrr(base, total, off, length);
|
||||
}
|
||||
off += length;
|
||||
}
|
||||
}
|
||||
|
||||
/// One RMRR structure: record a {bdf, base, limit} per single-path endpoint scope.
|
||||
fn parseRmrr(base: [*]align(1) const u8, total: usize, off: usize, length: usize) void {
|
||||
const range_base = fadt(u64, base, total, off + rmrr_base_offset) orelse return;
|
||||
const range_limit = fadt(u64, base, total, off + rmrr_limit_offset) orelse return;
|
||||
if (range_limit < range_base) return;
|
||||
|
||||
var scope = off + rmrr_scopes_offset;
|
||||
const end = off + length;
|
||||
while (scope + 6 <= end) {
|
||||
const scope_type = fadt(u8, base, total, scope) orelse break;
|
||||
const scope_length = fadt(u8, base, total, scope + 1) orelse break;
|
||||
if (scope_length < 6 or scope + scope_length > end) break;
|
||||
if (scope_type == scope_type_pci_endpoint and scope_length == scope_path_offset + 2) {
|
||||
// Single (device, function) pair: a directly-reachable endpoint.
|
||||
const bus = fadt(u8, base, total, scope + scope_start_bus_offset) orelse 0;
|
||||
const device = fadt(u8, base, total, scope + scope_path_offset) orelse 0;
|
||||
const function = fadt(u8, base, total, scope + scope_path_offset + 1) orelse 0;
|
||||
if (platform_information.rmrr_count < maximum_rmrr) {
|
||||
platform_information.rmrr[platform_information.rmrr_count] = .{
|
||||
.bdf = (@as(u16, bus) << 8) | (@as(u16, device) << 3) | function,
|
||||
.base = range_base,
|
||||
.limit = range_limit,
|
||||
};
|
||||
platform_information.rmrr_count += 1;
|
||||
} else {
|
||||
platform_information.rmrr_skipped +|= 1; // table full
|
||||
}
|
||||
} else {
|
||||
platform_information.rmrr_skipped +|= 1; // multi-hop path or non-endpoint scope
|
||||
}
|
||||
scope += scope_length;
|
||||
}
|
||||
}
|
||||
|
||||
// IVRS layout (AMD I/O Virtualization spec): 36-byte ACPI header, IVinfo u32 @36,
|
||||
// 8 reserved @40, then IVHD/IVMD blocks from @48. An IVHD common header is type u8 @0,
|
||||
// flags u8 @1, length u16 @2, device id u16 @4, capability offset u16 @6, IOMMU base
|
||||
// address u64 @8, PCI segment u16 @16, IOMMU info u16 @18.
|
||||
const ivrs_blocks_offset = 48;
|
||||
const ivhd_type_10: u8 = 0x10;
|
||||
const ivhd_type_11: u8 = 0x11;
|
||||
const ivhd_base_offset = 8;
|
||||
|
||||
/// IVRS -> detect an AMD-Vi IOMMU. Record the control-register base from the first IVHD
|
||||
/// of type 0x10/0x11. Per-device entries and IVMD (the AMD analogue of RMRR) are ignored
|
||||
/// in v1 — the default-deny device table is what we build anyway, and QEMU emits no IVMD;
|
||||
/// a real machine that needs them is flagged untested on AMD regardless.
|
||||
fn parseIvrs(header: *const SystemDescriptorTableHeader) void {
|
||||
const base: [*]align(1) const u8 = @ptrCast(header);
|
||||
const total: usize = header.length;
|
||||
var off: usize = ivrs_blocks_offset;
|
||||
while (off + 4 <= total) {
|
||||
const kind = fadt(u8, base, total, off) orelse break;
|
||||
const length = fadt(u16, base, total, off + 2) orelse break;
|
||||
if (length < 4 or off + length > total) break;
|
||||
if (kind == ivhd_type_10 or kind == ivhd_type_11) {
|
||||
const iommu_base = fadt(u64, base, total, off + ivhd_base_offset) orelse 0;
|
||||
if (iommu_base != 0) {
|
||||
platform_information.iommu_present = true;
|
||||
platform_information.iommu_base = register_base;
|
||||
platform_information.iommu_version = @as(*const volatile u32, @ptrFromInt(regs + 0x00)).*;
|
||||
platform_information.iommu_capabilities = @as(*const volatile u64, @ptrFromInt(regs + 0x08)).*;
|
||||
return; // first unit is enough for detection; multi-unit is future
|
||||
platform_information.iommu_is_amd = true;
|
||||
platform_information.iommu_base = iommu_base;
|
||||
return; // first IVHD is enough; multi-unit is future work
|
||||
}
|
||||
}
|
||||
off += length;
|
||||
|
||||
@@ -17,6 +17,11 @@ const io = @import("io.zig");
|
||||
const smp = @import("smp.zig");
|
||||
const pcpu = @import("per-cpu.zig");
|
||||
|
||||
/// The x86-64 IOMMU backends (Intel VT-d, AMD-Vi) behind their dispatch
|
||||
/// surface — the architecture-neutral IOMMU core (system/kernel/iommu.zig)
|
||||
/// reaches the hardware only through this.
|
||||
pub const iommu = @import("iommu.zig");
|
||||
|
||||
/// The saved register/trap frame passed to a fault handler.
|
||||
pub const CpuState = idt.CpuState;
|
||||
|
||||
|
||||
@@ -0,0 +1,265 @@
|
||||
//! AMD-Vi (AMD I/O Virtualization) backend for the IOMMU core, behind the
|
||||
//! architecture boundary. The AMD analogue of iommu-intel.zig: it supplies
|
||||
//! the architecture-neutral core's `Backend` vtable (iommu.zig beside this file)
|
||||
//! with AMD-Vi's page-table entry bits and drives the device table, command buffer, and
|
||||
//! event log.
|
||||
//!
|
||||
//! **UNTESTED on real AMD hardware.** danos is developed on Intel; this backend is
|
||||
//! validated only against QEMU's `-device amd-iommu,dma-remap=on`, whose AMD-Vi
|
||||
//! emulation is far less exercised than its Intel one. Every code path here should be
|
||||
//! read as "QEMU-verified, real-AMD-unverified" until an AMD machine confirms it.
|
||||
//!
|
||||
//! Interrupt remapping is left off (the DTE forwards interrupts unmapped), so MSI writes
|
||||
//! to the 0xFEE00000 range reach the APIC untranslated, exactly as on the Intel path.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const paging = @import("paging.zig");
|
||||
const iommu = @import("iommu.zig");
|
||||
|
||||
const page_size = abi.page_size;
|
||||
|
||||
// MMIO register offsets from the IOMMU control-register base.
|
||||
const reg_device_table_base = 0x00; // base | (pages-1) in bits 8:0
|
||||
const reg_command_buffer_base = 0x08; // base | (ComLen << 56)
|
||||
const reg_event_log_base = 0x10; // base | (EventLen << 56)
|
||||
const reg_control = 0x18;
|
||||
const reg_command_head = 0x2000;
|
||||
const reg_command_tail = 0x2008;
|
||||
const reg_event_head = 0x2010;
|
||||
const reg_event_tail = 0x2018;
|
||||
const reg_status = 0x2020;
|
||||
|
||||
const control_iommu_enable: u64 = 1 << 0;
|
||||
const control_event_log_enable: u64 = 1 << 2;
|
||||
const control_command_buffer_enable: u64 = 1 << 12;
|
||||
|
||||
// Device table: one 32-byte DTE (4 qwords) per requester id, indexed by bdf.
|
||||
const device_table_pages = 512; // 2 MiB = 65536 entries (a full 256-bus segment)
|
||||
const dte_qwords = 4;
|
||||
const dte_valid: u64 = 1 << 0; // V
|
||||
const dte_translation_valid: u64 = 1 << 1; // TV
|
||||
const dte_mode_shift = 9; // bits 11:9 — page-table levels
|
||||
const dte_read: u64 = 1 << 61; // IR
|
||||
const dte_write: u64 = 1 << 62; // IW
|
||||
const dte_intctl_forward: u64 = @as(u64, 1) << 60; // qword2 bits 61:60 = 01b: forward interrupts unmapped
|
||||
|
||||
// Page-table entry bits (AMD native format).
|
||||
const pte_present: u64 = 1 << 0; // PR
|
||||
const pte_next_level_shift = 9; // bits 11:9: 0 = leaf, N = pointer to a level-N table
|
||||
const pte_read: u64 = 1 << 61; // IR
|
||||
const pte_write: u64 = 1 << 62; // IW
|
||||
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||
|
||||
// Command buffer / event log: one 4 KiB frame each = 256 entries.
|
||||
const ring_entries = 256;
|
||||
const ring_length_code: u64 = 8; // log2(256), the ComLen/EventLen field value
|
||||
const command_opcode_shift = 60; // opcode in bits 63:60 of qword 0
|
||||
const command_completion_wait: u64 = 0x01;
|
||||
const command_invalidate_devtab: u64 = 0x02;
|
||||
const command_invalidate_pages: u64 = 0x03;
|
||||
const completion_wait_store: u64 = 1 << 1; // S: store `data` to the supplied address
|
||||
const invalidate_pages_all: u64 = 0x000F_FFFF_FFFF_F000 | 1; // address bits 51:12 all-ones + S
|
||||
|
||||
const levels: u8 = 4; // 48-bit IOVA, matching the Intel 4-level path
|
||||
|
||||
var register_base: usize = 0;
|
||||
var device_table: u64 = 0; // physical base of the device table
|
||||
var command_buffer: u64 = 0;
|
||||
var event_log: u64 = 0;
|
||||
var completion_frame: u64 = 0; // COMPLETION_WAIT store target
|
||||
var command_tail: u32 = 0; // our software copy of the command tail (bytes)
|
||||
|
||||
var fault_log_budget: u32 = 32;
|
||||
var completion_warned = false;
|
||||
|
||||
fn read64(offset: usize) u64 {
|
||||
return @as(*const volatile u64, @ptrFromInt(register_base + offset)).*;
|
||||
}
|
||||
fn write64(offset: usize, value: u64) void {
|
||||
@as(*volatile u64, @ptrFromInt(register_base + offset)).* = value;
|
||||
}
|
||||
fn ram(physical: u64) [*]volatile u64 {
|
||||
return @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
||||
}
|
||||
|
||||
/// Map the register window, allocate the device table / command buffer / event log.
|
||||
/// Returns the vtable, or null if the boot-time allocations fail.
|
||||
pub fn detect(discovery: iommu.Discovery) ?iommu.Backend {
|
||||
register_base = paging.mapMmio(discovery.register_base, 16 * 1024, true);
|
||||
|
||||
device_table = iommu.environment.allocateContiguous(device_table_pages, ~@as(u64, 0)) orelse return null;
|
||||
zero(device_table, device_table_pages); // all-zero DTE = V=0 = deny every device
|
||||
command_buffer = allocZeroedFrame() orelse return null;
|
||||
event_log = allocZeroedFrame() orelse return null;
|
||||
completion_frame = allocZeroedFrame() orelse return null;
|
||||
|
||||
return iommu.Backend{
|
||||
.levels = levels,
|
||||
.supports_huge_pages = false, // 4 KiB leaves only (AMD superpage encoding deferred)
|
||||
.enable = enable,
|
||||
.makeLeaf = makeLeaf,
|
||||
.makeTable = makeTable,
|
||||
.isPresent = isPresent,
|
||||
.flushStructure = flushStructure,
|
||||
.attach = attach,
|
||||
.detach = detach,
|
||||
.invalidateDomain = invalidateDomain,
|
||||
.faultDrain = faultDrain,
|
||||
};
|
||||
}
|
||||
|
||||
/// Program the base registers and enable translation. The device table is already
|
||||
/// zeroed (every device denied) except any entries `attach` wrote for RMRR/claimed
|
||||
/// devices, so turning translation on blocks all other DMA and logs it.
|
||||
fn enable() void {
|
||||
write64(reg_device_table_base, (device_table & address_mask) | (device_table_pages - 1));
|
||||
write64(reg_command_buffer_base, (command_buffer & address_mask) | (ring_length_code << 56));
|
||||
write64(reg_event_log_base, (event_log & address_mask) | (ring_length_code << 56));
|
||||
write64(reg_command_head, 0);
|
||||
write64(reg_command_tail, 0);
|
||||
write64(reg_event_head, 0);
|
||||
write64(reg_event_tail, 0);
|
||||
command_tail = 0;
|
||||
// Buffers first, then the master enable.
|
||||
write64(reg_control, control_command_buffer_enable | control_event_log_enable);
|
||||
write64(reg_control, control_command_buffer_enable | control_event_log_enable | control_iommu_enable);
|
||||
|
||||
iommu.environment.write("/system/kernel: iommu online (AMD-Vi) — UNTESTED on real AMD hardware (QEMU-verified only)\n");
|
||||
var buffer: [48]u8 = undefined;
|
||||
if (std.fmt.bufPrint(&buffer, " levels : {d} (48-bit)\n", .{levels})) |line|
|
||||
iommu.environment.write(line)
|
||||
else |_| {}
|
||||
}
|
||||
|
||||
// --- Backend vtable ------------------------------------------------------------------
|
||||
|
||||
fn makeLeaf(physical: u64, huge: bool) u64 {
|
||||
_ = huge; // 4 KiB only
|
||||
return (physical & address_mask) | pte_present | pte_read | pte_write; // Next Level 0 = leaf
|
||||
}
|
||||
fn makeTable(table_physical: u64, level: u8) u64 {
|
||||
// This entry (at `level`) points to a table one level down; AMD's Next Level names
|
||||
// the pointed-to table's level.
|
||||
const next_level: u64 = @as(u64, level) - 1;
|
||||
return (table_physical & address_mask) | pte_present | pte_read | pte_write | (next_level << pte_next_level_shift);
|
||||
}
|
||||
fn isPresent(entry: u64) bool {
|
||||
return (entry & pte_present) != 0;
|
||||
}
|
||||
fn flushStructure(address: usize) void {
|
||||
_ = address; // AMD-Vi reads its structures coherently
|
||||
}
|
||||
|
||||
fn attach(bdf: u16, domain: u16, page_table_root: u64) void {
|
||||
const dte = ram(device_table) + @as(usize, bdf) * dte_qwords;
|
||||
dte[0] = (page_table_root & address_mask) | dte_valid | dte_translation_valid |
|
||||
(@as(u64, levels) << dte_mode_shift) | dte_read | dte_write;
|
||||
dte[1] = @as(u64, domain); // DomainID in bits 15:0
|
||||
dte[2] = dte_intctl_forward; // forward interrupts unmapped (no remapping)
|
||||
dte[3] = 0;
|
||||
invalidateDevice(bdf);
|
||||
invalidateDomain(domain);
|
||||
}
|
||||
|
||||
fn detach(bdf: u16) void {
|
||||
const dte = ram(device_table) + @as(usize, bdf) * dte_qwords;
|
||||
dte[0] = 0; // V=0: deny
|
||||
dte[1] = 0;
|
||||
dte[2] = 0;
|
||||
dte[3] = 0;
|
||||
invalidateDevice(bdf);
|
||||
}
|
||||
|
||||
fn invalidateDomain(domain: u16) void {
|
||||
submitCommand((command_invalidate_pages << command_opcode_shift) | (@as(u64, domain) << 32), invalidate_pages_all);
|
||||
completeAndWait();
|
||||
}
|
||||
|
||||
fn faultDrain() usize {
|
||||
const tail = read64(reg_event_tail) & 0xFFFF_FFF0;
|
||||
var head = read64(reg_event_head) & 0xFFFF_FFF0;
|
||||
if (head == tail) return 0;
|
||||
var seen: usize = 0;
|
||||
while (head != tail) {
|
||||
const entry = ram(event_log) + (head / 8);
|
||||
const code: u4 = @truncate(entry[0] >> command_opcode_shift);
|
||||
if (code == 0x2) { // IO_PAGE_FAULT
|
||||
const source: u16 = @truncate(entry[0]);
|
||||
logFault(source, entry[1]);
|
||||
}
|
||||
seen += 1;
|
||||
head += 16;
|
||||
if (head >= ring_entries * 16) head = 0;
|
||||
}
|
||||
write64(reg_event_head, head);
|
||||
return seen;
|
||||
}
|
||||
|
||||
fn logFault(source: u16, address: u64) void {
|
||||
if (fault_log_budget == 0) return;
|
||||
fault_log_budget -= 1;
|
||||
var buffer: [128]u8 = undefined;
|
||||
if (std.fmt.bufPrint(&buffer, "DANOS-IOMMU-FAULT: bdf={x:0>2}:{x:0>2}.{d} addr=0x{x} reason=amd-io-page-fault\n", .{
|
||||
source >> 8,
|
||||
(source >> 3) & 0x1F,
|
||||
source & 0x7,
|
||||
address,
|
||||
})) |line| iommu.environment.write(line) else |_| {}
|
||||
if (fault_log_budget == 0) iommu.environment.write("DANOS-IOMMU-FAULT: (further faults suppressed)\n");
|
||||
}
|
||||
|
||||
// --- command ring --------------------------------------------------------------------
|
||||
|
||||
fn invalidateDevice(bdf: u16) void {
|
||||
submitCommand((command_invalidate_devtab << command_opcode_shift) | @as(u64, bdf), 0);
|
||||
completeAndWait();
|
||||
}
|
||||
|
||||
/// Append a 128-bit command (two qwords) to the ring and advance the tail.
|
||||
fn submitCommand(qword0: u64, qword1: u64) void {
|
||||
const slot = ram(command_buffer) + (command_tail / 8);
|
||||
slot[0] = qword0;
|
||||
slot[1] = qword1;
|
||||
command_tail += 16;
|
||||
if (command_tail >= ring_entries * 16) command_tail = 0;
|
||||
write64(reg_command_tail, command_tail);
|
||||
}
|
||||
|
||||
/// Append a COMPLETION_WAIT (store form) and spin until the IOMMU writes our sentinel to
|
||||
/// the completion frame. QEMU consumes the command buffer synchronously on the tail-
|
||||
/// register write, so by the time we poll the prior invalidation is already applied; the
|
||||
/// store confirmation is belt-and-suspenders for real hardware. If it never lands
|
||||
/// (QEMU's amd-iommu does not implement the store form), warn ONCE and proceed — the
|
||||
/// invalidation itself has happened.
|
||||
fn completeAndWait() void {
|
||||
const sentinel: u64 = 0xC0FFEE;
|
||||
ram(completion_frame)[0] = 0;
|
||||
submitCommand(
|
||||
(command_completion_wait << command_opcode_shift) | (completion_frame & 0x000F_FFFF_FFFF_FFF8) | completion_wait_store,
|
||||
sentinel,
|
||||
);
|
||||
var spins: u64 = 0;
|
||||
while (@as(*const volatile u64, @ptrFromInt(boot_handoff.physicalToVirtual(completion_frame))).* != sentinel) {
|
||||
spins += 1;
|
||||
if (spins > 100_000) {
|
||||
if (!completion_warned) {
|
||||
completion_warned = true;
|
||||
iommu.environment.write("/system/kernel: AMD-Vi COMPLETION_WAIT store not observed — proceeding (QEMU processes commands synchronously)\n");
|
||||
}
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn allocZeroedFrame() ?u64 {
|
||||
const frame = iommu.environment.allocateFrame() orelse return null;
|
||||
zero(frame, 1);
|
||||
return frame;
|
||||
}
|
||||
fn zero(physical: u64, pages: usize) void {
|
||||
const words = ram(physical);
|
||||
var i: usize = 0;
|
||||
while (i < pages * page_size / 8) : (i += 1) words[i] = 0;
|
||||
}
|
||||
@@ -0,0 +1,327 @@
|
||||
//! Intel VT-d backend for the IOMMU core, behind the architecture boundary.
|
||||
//!
|
||||
//! Provides the architecture-neutral core (system/kernel/iommu.zig) with the VT-d
|
||||
//! hardware specifics behind the `Backend` vtable (iommu.zig beside this file): second-level page-table entry bits, the root/context table structure, the
|
||||
//! translation-enable and invalidation register sequences, and the fault drain. The
|
||||
//! core owns the domain table and the page-table walk; this file owns the registers.
|
||||
//!
|
||||
//! Register model (VT-d spec §10-11): offsets from the DRHD register base. The unit is
|
||||
//! programmed once at enable (root table + Translation Enable), then touched only for
|
||||
//! per-device context changes, per-domain invalidations, and fault draining. Interrupt
|
||||
//! remapping is deliberately left OFF (GCMD.IRE stays 0): with it off, upstream writes
|
||||
//! to 0xFEE0_0000-0xFEEF_FFFF are treated as interrupt requests and bypass second-level
|
||||
//! translation, so the existing MSI contract survives unchanged.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const paging = @import("paging.zig");
|
||||
const iommu = @import("iommu.zig");
|
||||
|
||||
const page_size = abi.page_size;
|
||||
|
||||
// Register offsets from the unit's base.
|
||||
const reg_cap = 0x08; // Capability (64)
|
||||
const reg_ecap = 0x10; // Extended Capability (64)
|
||||
const reg_gcmd = 0x18; // Global Command (32, write-only)
|
||||
const reg_gsts = 0x1C; // Global Status (32, read-only)
|
||||
const reg_rtaddr = 0x20; // Root Table Address (64)
|
||||
const reg_ccmd = 0x28; // Context Command (64)
|
||||
const reg_fsts = 0x34; // Fault Status (32)
|
||||
|
||||
const gcmd_te: u32 = 1 << 31; // Translation Enable
|
||||
const gcmd_srtp: u32 = 1 << 30; // Set Root Table Pointer
|
||||
const gsts_tes: u32 = 1 << 31; // Translation Enable Status
|
||||
const gsts_rtps: u32 = 1 << 30; // Root Table Pointer Status
|
||||
|
||||
const cap_cm: u64 = 1 << 7; // Caching Mode
|
||||
const cap_sagaw_shift = 8; // Supported Adjusted Guest Address Widths, bits 12:8
|
||||
const cap_sagaw_39bit: u64 = 1 << 9; // 3-level
|
||||
const cap_sagaw_48bit: u64 = 1 << 10; // 4-level
|
||||
const cap_fro_shift = 24; // Fault-Recording Register Offset, bits 33:24 (×16)
|
||||
const cap_nfr_shift = 40; // Number of Fault Recording regs, bits 47:40 (+1)
|
||||
const ecap_coherent: u64 = 1 << 0; // hardware snoops CPU caches reading its structures
|
||||
const ecap_iro_shift = 8; // IOTLB Register Offset, bits 17:8 (×16)
|
||||
|
||||
const ccmd_icc: u64 = 1 << 63; // Invalidate Context-Cache
|
||||
const ccmd_cirg_global: u64 = @as(u64, 1) << 61; // global granularity
|
||||
const ccmd_cirg_device: u64 = @as(u64, 3) << 61; // device-selective
|
||||
|
||||
const iotlb_ivt: u64 = 1 << 63; // Invalidate IOTLB
|
||||
const iotlb_iirg_global: u64 = @as(u64, 1) << 60;
|
||||
const iotlb_iirg_domain: u64 = @as(u64, 2) << 60;
|
||||
const iotlb_dr: u64 = 1 << 49; // drain reads
|
||||
const iotlb_dw: u64 = 1 << 48; // drain writes
|
||||
|
||||
const fsts_ppf: u32 = 1 << 1; // Primary Pending Fault
|
||||
|
||||
// Second-level PTE bits.
|
||||
const slpte_read: u64 = 1 << 0;
|
||||
const slpte_write: u64 = 1 << 1;
|
||||
const slpte_page_size: u64 = 1 << 7; // a 2 MiB leaf (== iommu.huge_leaf_bit)
|
||||
|
||||
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||
|
||||
var register_base: usize = 0;
|
||||
var version: u32 = 0;
|
||||
var capabilities: u64 = 0;
|
||||
var extended_capabilities: u64 = 0;
|
||||
var coherent: bool = true; // ECAP.C — whether clflush is unnecessary
|
||||
var levels: u8 = 4;
|
||||
var context_aw: u64 = 2; // context-entry AW field (001=3-level, 010=4-level)
|
||||
var gcmd_shadow: u32 = 0; // sticky GCMD bits (TE etc.), for the write-only register
|
||||
|
||||
var root_table: u64 = 0; // physical base of the 256-entry root table
|
||||
var context_table: [256]u64 = .{0} ** 256; // per-bus context table physical, 0 = none
|
||||
|
||||
var fault_log_budget: u32 = 32; // rate-limit: log this many faults, then just count
|
||||
var faults_suppressed: u64 = 0;
|
||||
|
||||
fn read32(offset: usize) u32 {
|
||||
return @as(*const volatile u32, @ptrFromInt(register_base + offset)).*;
|
||||
}
|
||||
fn write32(offset: usize, value: u32) void {
|
||||
@as(*volatile u32, @ptrFromInt(register_base + offset)).* = value;
|
||||
}
|
||||
fn read64(offset: usize) u64 {
|
||||
return @as(*const volatile u64, @ptrFromInt(register_base + offset)).*;
|
||||
}
|
||||
fn write64(offset: usize, value: u64) void {
|
||||
@as(*volatile u64, @ptrFromInt(register_base + offset)).* = value;
|
||||
}
|
||||
|
||||
fn tableAt(physical: u64) [*]volatile u64 {
|
||||
return @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
||||
}
|
||||
|
||||
/// Map the register window, read caps, pick the address width. Returns the vtable, or
|
||||
/// null when the unit is not live or advertises no address width danos can drive.
|
||||
pub fn detect(discovery: iommu.Discovery) ?iommu.Backend {
|
||||
// Map 16 KiB: FRCD and IOTLB registers can sit past the first page (CAP.FRO /
|
||||
// ECAP.IRO are 16-byte-unit offsets).
|
||||
register_base = paging.mapMmio(discovery.register_base, 16 * 1024, true);
|
||||
// The Version register's low byte is major.minor; reading it back nonzero is
|
||||
// the live-mappable-unit sanity check (previously a kernel-test assertion).
|
||||
version = read32(0x00);
|
||||
if (version == 0) return null;
|
||||
capabilities = read64(reg_cap);
|
||||
extended_capabilities = read64(reg_ecap);
|
||||
coherent = (extended_capabilities & ecap_coherent) != 0;
|
||||
|
||||
const sagaw = capabilities >> cap_sagaw_shift;
|
||||
if (sagaw & cap_sagaw_48bit != 0) {
|
||||
levels = 4;
|
||||
context_aw = 2; // 010b
|
||||
} else if (sagaw & cap_sagaw_39bit != 0) {
|
||||
levels = 3;
|
||||
context_aw = 1; // 001b
|
||||
} else {
|
||||
return null; // no width we build tables for
|
||||
}
|
||||
|
||||
root_table = allocZeroed() orelse return null;
|
||||
|
||||
return iommu.Backend{
|
||||
.levels = levels,
|
||||
.supports_huge_pages = true,
|
||||
.enable = enable,
|
||||
.makeLeaf = makeLeaf,
|
||||
.makeTable = makeTable,
|
||||
.isPresent = isPresent,
|
||||
.flushStructure = flushStructure,
|
||||
.attach = attach,
|
||||
.detach = detach,
|
||||
.invalidateDomain = invalidateDomain,
|
||||
.faultDrain = faultDrain,
|
||||
};
|
||||
}
|
||||
|
||||
/// Program the root table and turn Translation Enable on. The core has already created
|
||||
/// and populated the RMRR domains (their context entries are live via `attach`), so at
|
||||
/// this instant every OTHER device's context entry is not-present and will fault — which
|
||||
/// for stale firmware bus-mastering is the desired evidence, not a bug.
|
||||
fn enable() void {
|
||||
write64(reg_rtaddr, root_table); // legacy mode (bits 11:10 = 00)
|
||||
setGlobalCommand(gcmd_srtp);
|
||||
spinStatus(gsts_rtps);
|
||||
globalInvalidate();
|
||||
setGlobalCommand(gcmd_te);
|
||||
spinStatus(gsts_tes);
|
||||
gcmd_shadow |= gcmd_te;
|
||||
|
||||
iommu.environment.write("/system/kernel: iommu online (Intel VT-d)\n");
|
||||
var buffer: [64]u8 = undefined;
|
||||
if (std.fmt.bufPrint(&buffer, " version : 0x{x}\n", .{version})) |line|
|
||||
iommu.environment.write(line)
|
||||
else |_| {}
|
||||
if (std.fmt.bufPrint(&buffer, " agaw : {d} levels\n", .{levels})) |line|
|
||||
iommu.environment.write(line)
|
||||
else |_| {}
|
||||
}
|
||||
|
||||
// --- Backend vtable ------------------------------------------------------------------
|
||||
|
||||
fn makeLeaf(physical: u64, huge: bool) u64 {
|
||||
return (physical & address_mask) | slpte_read | slpte_write | (if (huge) slpte_page_size else 0);
|
||||
}
|
||||
fn makeTable(table_physical: u64, level: u8) u64 {
|
||||
_ = level;
|
||||
return (table_physical & address_mask) | slpte_read | slpte_write;
|
||||
}
|
||||
fn isPresent(entry: u64) bool {
|
||||
return (entry & (slpte_read | slpte_write)) != 0;
|
||||
}
|
||||
|
||||
fn flushStructure(address: usize) void {
|
||||
if (coherent) return; // the unit snoops CPU caches; no flush needed (QEMU)
|
||||
asm volatile ("clflush (%[p])"
|
||||
:
|
||||
: [p] "r" (address),
|
||||
: .{ .memory = true });
|
||||
}
|
||||
|
||||
fn attach(bdf: u16, domain: u16, page_table_root: u64) void {
|
||||
const bus: u8 = @intCast(bdf >> 8);
|
||||
const devfn: u8 = @intCast(bdf & 0xFF);
|
||||
|
||||
// Lazily allocate this bus's context table and link it into the root table.
|
||||
if (context_table[bus] == 0) {
|
||||
const table = allocZeroed() orelse return;
|
||||
context_table[bus] = table;
|
||||
const root_entry = &tableAt(root_table)[@as(usize, bus) * 2]; // 16-byte entries
|
||||
root_entry.* = (table & address_mask) | 1; // present
|
||||
flushStructure(@intFromPtr(root_entry));
|
||||
}
|
||||
|
||||
const context = tableAt(context_table[bus]);
|
||||
const low = &context[@as(usize, devfn) * 2];
|
||||
const high = &context[@as(usize, devfn) * 2 + 1];
|
||||
high.* = (context_aw & 0x7) | (@as(u64, domain) << 8); // AW + DID
|
||||
low.* = (page_table_root & address_mask) | 1; // present, TT=00 (use second-level)
|
||||
flushStructure(@intFromPtr(high));
|
||||
flushStructure(@intFromPtr(low));
|
||||
|
||||
invalidateContextDevice(bdf, domain);
|
||||
invalidateDomain(domain);
|
||||
}
|
||||
|
||||
fn detach(bdf: u16) void {
|
||||
const bus: u8 = @intCast(bdf >> 8);
|
||||
const devfn: u8 = @intCast(bdf & 0xFF);
|
||||
if (context_table[bus] == 0) return;
|
||||
const context = tableAt(context_table[bus]);
|
||||
context[@as(usize, devfn) * 2] = 0; // not present
|
||||
context[@as(usize, devfn) * 2 + 1] = 0;
|
||||
flushStructure(@intFromPtr(&context[@as(usize, devfn) * 2]));
|
||||
invalidateContextDevice(bdf, 0);
|
||||
globalIotlb();
|
||||
}
|
||||
|
||||
fn invalidateDomain(domain: u16) void {
|
||||
const iotlb_offset = iotlbOffset();
|
||||
write64(iotlb_offset, iotlb_ivt | iotlb_iirg_domain | iotlb_dr | iotlb_dw | (@as(u64, domain) << 32));
|
||||
spin64(iotlb_offset, iotlb_ivt);
|
||||
}
|
||||
|
||||
fn faultDrain() usize {
|
||||
const fsts = read32(reg_fsts);
|
||||
if (fsts & fsts_ppf == 0) return 0;
|
||||
|
||||
const fro = (capabilities >> cap_fro_shift) & 0x3FF;
|
||||
const nfr = ((capabilities >> cap_nfr_shift) & 0xFF) + 1;
|
||||
const frcd_base = @as(usize, @intCast(fro)) * 16;
|
||||
|
||||
var seen: usize = 0;
|
||||
var i: usize = 0;
|
||||
while (i < nfr) : (i += 1) {
|
||||
const off = frcd_base + i * 16;
|
||||
const high = read64(off + 8);
|
||||
if (high & (@as(u64, 1) << 63) == 0) continue; // F: no fault recorded here
|
||||
const low = read64(off);
|
||||
const address = low & ~@as(u64, 0xFFF);
|
||||
const source: u16 = @intCast(high & 0xFFFF);
|
||||
const reason: u8 = @intCast((high >> 32) & 0xFF);
|
||||
const is_read = (high >> 62) & 1; // T: 1 = read request
|
||||
logFault(source, address, reason, is_read == 1);
|
||||
write64(off + 8, @as(u64, 1) << 63); // RW1C: clear F
|
||||
seen += 1;
|
||||
}
|
||||
write32(reg_fsts, fsts); // clear PPF/PFO (write-1-to-clear)
|
||||
return seen;
|
||||
}
|
||||
|
||||
fn logFault(source: u16, address: u64, reason: u8, is_read: bool) void {
|
||||
if (fault_log_budget > 0) {
|
||||
fault_log_budget -= 1;
|
||||
var buffer: [128]u8 = undefined;
|
||||
if (std.fmt.bufPrint(&buffer, "DANOS-IOMMU-FAULT: bdf={x:0>2}:{x:0>2}.{d} addr=0x{x} reason=0x{x} write={d}\n", .{
|
||||
source >> 8,
|
||||
(source >> 3) & 0x1F,
|
||||
source & 0x7,
|
||||
address,
|
||||
reason,
|
||||
@intFromBool(!is_read),
|
||||
})) |line| iommu.environment.write(line) else |_| {}
|
||||
if (fault_log_budget == 0)
|
||||
iommu.environment.write("DANOS-IOMMU-FAULT: (further faults suppressed)\n");
|
||||
} else {
|
||||
faults_suppressed += 1;
|
||||
}
|
||||
}
|
||||
|
||||
// --- register helpers ----------------------------------------------------------------
|
||||
|
||||
fn setGlobalCommand(one_shot: u32) void {
|
||||
// GCMD is write-only: every write must carry the full sticky state plus the one-shot
|
||||
// bit being requested, or a set sticky bit (TE) would be cleared as a side effect.
|
||||
write32(reg_gcmd, gcmd_shadow | one_shot);
|
||||
}
|
||||
|
||||
fn spinStatus(bit: u32) void {
|
||||
var spins: u64 = 0;
|
||||
while (read32(reg_gsts) & bit == 0) {
|
||||
spins += 1;
|
||||
if (spins > 10_000_000) {
|
||||
iommu.environment.write("/system/kernel: WARNING VT-d status bit never set — translation may be incomplete\n");
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn spin64(offset: usize, bit: u64) void {
|
||||
var spins: u64 = 0;
|
||||
while (read64(offset) & bit != 0) {
|
||||
spins += 1;
|
||||
if (spins > 10_000_000) return;
|
||||
}
|
||||
}
|
||||
|
||||
fn globalInvalidate() void {
|
||||
write64(reg_ccmd, ccmd_icc | ccmd_cirg_global);
|
||||
spin64(reg_ccmd, ccmd_icc);
|
||||
globalIotlb();
|
||||
}
|
||||
|
||||
fn globalIotlb() void {
|
||||
const iotlb_offset = iotlbOffset();
|
||||
write64(iotlb_offset, iotlb_ivt | iotlb_iirg_global | iotlb_dr | iotlb_dw);
|
||||
spin64(iotlb_offset, iotlb_ivt);
|
||||
}
|
||||
|
||||
fn invalidateContextDevice(bdf: u16, domain: u16) void {
|
||||
write64(reg_ccmd, ccmd_icc | ccmd_cirg_device | (@as(u64, bdf) << 16) | domain);
|
||||
spin64(reg_ccmd, ccmd_icc);
|
||||
}
|
||||
|
||||
fn iotlbOffset() usize {
|
||||
const iro = (extended_capabilities >> ecap_iro_shift) & 0x3FF;
|
||||
return @as(usize, @intCast(iro)) * 16 + 8; // IOTLB register sits at IRO*16 + 8
|
||||
}
|
||||
|
||||
fn allocZeroed() ?u64 {
|
||||
const frame = iommu.environment.allocateFrame() orelse return null;
|
||||
const table = tableAt(frame);
|
||||
var i: usize = 0;
|
||||
while (i < 512) : (i += 1) table[i] = 0;
|
||||
return frame;
|
||||
}
|
||||
@@ -0,0 +1,77 @@
|
||||
//! x86-64 IOMMU backends, behind the architecture boundary: Intel VT-d
|
||||
//! (iommu-intel.zig) and AMD-Vi (iommu-amd.zig). The architecture-neutral
|
||||
//! core (system/kernel/iommu.zig) owns the domain table and the shared
|
||||
//! page-table walker; it hands this file the firmware discovery facts and an
|
||||
//! environment (frame allocation + the log sink, injected the same way
|
||||
//! enablePaging receives its frame hooks), and gets back a hardware vtable.
|
||||
//! A new architecture supplies its own unit (ARM: the SMMU) from its own
|
||||
//! directory with no core change.
|
||||
|
||||
const intel = @import("iommu-intel.zig");
|
||||
const amd = @import("iommu-amd.zig");
|
||||
|
||||
/// What the platform's firmware tables reported: where the unit's registers
|
||||
/// live, and which programming model its table implies (an IVRS table
|
||||
/// describes AMD-Vi; a DMAR table describes Intel VT-d).
|
||||
pub const Discovery = struct {
|
||||
register_base: u64,
|
||||
amd: bool,
|
||||
};
|
||||
|
||||
/// What the backends need from the generic kernel, injected at detect so this
|
||||
/// module never imports kernel internals: physical-frame allocation for the
|
||||
/// hardware structures, and the kernel log sink (fault reports, warnings, the
|
||||
/// enable banner).
|
||||
pub const Environment = struct {
|
||||
allocateFrame: *const fn () ?u64,
|
||||
allocateContiguous: *const fn (count: usize, max_physical: u64) ?u64,
|
||||
write: *const fn (bytes: []const u8) void,
|
||||
};
|
||||
|
||||
/// The bit encodings and hardware operations a backend supplies to the shared
|
||||
/// core. Entry helpers build the raw page-table entries for the backend's
|
||||
/// format; the core walks the tree with them. The hardware ops act on a whole
|
||||
/// domain (identified by its hardware domain id = core index + 1) or device
|
||||
/// (by requester id / bdf).
|
||||
pub const Backend = struct {
|
||||
/// Number of page-table levels (3 or 4) the backend selected from hardware caps.
|
||||
levels: u8,
|
||||
/// Largest leaf the walker may emit: 4 KiB always, 2 MiB when the backend allows.
|
||||
supports_huge_pages: bool,
|
||||
|
||||
/// Raw entry bits for a leaf mapping `physical` (with the given size), and for a
|
||||
/// non-leaf entry pointing at `table_physical` at `level` (level counts down to 1
|
||||
/// at the leaf's parent). `isPresent` tests a read-back entry.
|
||||
makeLeaf: *const fn (physical: u64, huge: bool) u64,
|
||||
makeTable: *const fn (table_physical: u64, level: u8) u64,
|
||||
isPresent: *const fn (entry: u64) bool,
|
||||
/// Flush a cache line holding IOMMU structures the hardware reads non-coherently
|
||||
/// (VT-d with ECAP.C==0). A no-op where the unit snoops caches.
|
||||
flushStructure: *const fn (address: usize) void,
|
||||
|
||||
/// Turn translation on (the core has already seeded any pre-claim domains)
|
||||
/// and write the unit's identity lines to the log — the core follows with
|
||||
/// the neutral posture lines.
|
||||
enable: *const fn () void,
|
||||
/// Point `bdf`'s translation structure at `domain` (hardware id) and invalidate the
|
||||
/// context/device caches so the change takes effect.
|
||||
attach: *const fn (bdf: u16, domain: u16, page_table_root: u64) void,
|
||||
/// Return `bdf`'s translation structure to not-present + invalidate — all its DMA
|
||||
/// faults afterward.
|
||||
detach: *const fn (bdf: u16) void,
|
||||
/// Invalidate cached translations for `domain` (after a map or unmap).
|
||||
invalidateDomain: *const fn (domain: u16) void,
|
||||
/// Pull pending faults out of the hardware, log them (rate-limited), return the
|
||||
/// count seen this call.
|
||||
faultDrain: *const fn () usize,
|
||||
};
|
||||
|
||||
/// The injected kernel services, stored for the backends at detect time.
|
||||
pub var environment: Environment = undefined;
|
||||
|
||||
/// Probe the discovered unit and return its vtable, or null when it is
|
||||
/// unusable (the core stays fail-open and says so).
|
||||
pub fn detect(discovery: Discovery, injected: Environment) ?Backend {
|
||||
environment = injected;
|
||||
return if (discovery.amd) amd.detect(discovery) else intel.detect(discovery);
|
||||
}
|
||||
@@ -167,6 +167,20 @@ pub fn releaseAllOwnedBy(owner: u32) void {
|
||||
}
|
||||
}
|
||||
|
||||
/// Release the claim on `id` iff `owner` holds it — the rollback for a claim that
|
||||
/// cannot be confined (the IOMMU domain could not be created/attached). Returns true
|
||||
/// when a claim was actually cleared.
|
||||
pub fn unclaim(id: u64, owner: u32) bool {
|
||||
if (id >= count) return false;
|
||||
if (claimed[@intCast(id)]) |o| {
|
||||
if (o == owner) {
|
||||
claimed[@intCast(id)] = null;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Resource `index` of device `id`, or null if out of range.
|
||||
pub fn resourceOf(id: u64, index: u64) ?device_abi.ResourceDescriptor {
|
||||
if (id >= count) return null;
|
||||
@@ -175,6 +189,50 @@ pub fn resourceOf(id: u64, index: u64) ?device_abi.ResourceDescriptor {
|
||||
return d.resources[@intCast(index)];
|
||||
}
|
||||
|
||||
/// The PCI requester id (bus<<8 | device<<3 | function) of device `id`, derived from
|
||||
/// its config-space slice against its host bridge's ECAM window — the identity a VT-d
|
||||
/// context entry / AMD-Vi DTE is keyed by. null when `id` is not a PCI function or the
|
||||
/// geometry doesn't decode. The kernel never stored the BDF (the descriptor has no such
|
||||
/// field); pci-bus encodes it into resource 0's physical base as
|
||||
/// `ecam_base + ((bus - start_bus) << 20 | device << 15 | function << 12)`, and the
|
||||
/// requester id the device emits uses the absolute bus, so we add `start_bus << 8` back.
|
||||
pub fn pciAddressOf(id: u64) ?u16 {
|
||||
if (id >= count) return null;
|
||||
const d = &devices[@intCast(id)];
|
||||
if (d.class != @intFromEnum(device_abi.DeviceClass.pci_device)) return null;
|
||||
if (d.resource_count == 0) return null;
|
||||
const config = d.resources[0];
|
||||
if (config.kind != @intFromEnum(device_abi.ResourceKind.memory) or config.len != 4096) return null;
|
||||
|
||||
// Walk up to the host bridge, whose resource 0 is the segment's ECAM window and
|
||||
// resource 1 the bus_range (start_bus, bus_count).
|
||||
var parent = d.parent;
|
||||
while (parent != device_abi.no_parent and parent < count) {
|
||||
const p = &devices[@intCast(parent)];
|
||||
if (p.class == @intFromEnum(device_abi.DeviceClass.pci_host_bridge)) {
|
||||
if (p.resource_count < 2) return null;
|
||||
const ecam = p.resources[0];
|
||||
const bus_range = p.resources[1];
|
||||
if (config.start < ecam.start or config.start >= ecam.start + ecam.len) return null;
|
||||
const offset = config.start - ecam.start;
|
||||
const start_bus: u16 = @intCast(bus_range.start & 0xFF);
|
||||
return @intCast((offset >> 12) + (@as(u64, start_bus) << 8));
|
||||
}
|
||||
parent = p.parent;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Call `visit(id, bdf)` for every PCI function in the table — the IOMMU core's boot
|
||||
/// sweep to place every device under a domain. Only functions whose BDF decodes are
|
||||
/// visited.
|
||||
pub fn forEachPciFunction(visit: *const fn (id: u64, bdf: u16) void) void {
|
||||
var id: u64 = 0;
|
||||
while (id < count) : (id += 1) {
|
||||
if (pciAddressOf(id)) |bdf| visit(id, bdf);
|
||||
}
|
||||
}
|
||||
|
||||
/// Is `child` wholly inside `parent`? For a range (memory, io_port, bus_range) that's
|
||||
/// interval containment; for an irq it's equality, since an interrupt line is not
|
||||
/// divisible. Zero-length child ranges are refused — an empty window is meaningless
|
||||
|
||||
@@ -0,0 +1,397 @@
|
||||
//! system/kernel/iommu.zig — architecture-neutral IOMMU core: per-device DMA
|
||||
//! translation domains over a backend the architecture module supplies
|
||||
//! (x86-64: Intel VT-d or AMD-Vi, behind architecture/x86_64/iommu.zig).
|
||||
//!
|
||||
//! The problem this closes: without an IOMMU, a claimed bus-mastering device can DMA to
|
||||
//! ANY physical address, so a compromised or buggy driver reaches all of memory through
|
||||
//! its device — driver isolation stops at the CPU's MMU. This core gives each claimed
|
||||
//! PCI function its own translation domain; a device reaches only the physical ranges
|
||||
//! mapped into its domain, and nothing else (kernel, page tables, other processes) is
|
||||
//! visible to it.
|
||||
//!
|
||||
//! Design:
|
||||
//! - **Identity mappings** (IOVA == physical). `dma_alloc` already hands drivers the
|
||||
//! physical address they program into hardware; a domain simply makes that same
|
||||
//! address the ONLY thing the device can reach. No IOVA allocator, and every
|
||||
//! driver's register-programming code is untouched.
|
||||
//! - **Architecture-neutral**: this file owns the domain table and a shared
|
||||
//! 512-entry page-table walker; the architecture module's `Backend` vtable
|
||||
//! supplies the hardware specifics — the entry-bit encodings, the
|
||||
//! enable/invalidate register dances, and the fault drain — with the frame
|
||||
//! allocator and log sink injected the other way.
|
||||
//! - **Fail-open**: when no IOMMU is found, nothing activates and every entry point is a
|
||||
//! success no-op, so callers in process.zig stay unconditional and behavior is
|
||||
//! byte-for-byte the pre-IOMMU kernel. The boot log states the posture.
|
||||
//!
|
||||
//! All entry points run under the big kernel lock (the caller holds it); no internal
|
||||
//! locking. All memory comes from `pmm` reached through the physmap, like paging.zig.
|
||||
|
||||
const std = @import("std");
|
||||
const abi = @import("abi");
|
||||
const boot_handoff = @import("boot-handoff");
|
||||
const pmm = @import("pmm.zig");
|
||||
const platform = @import("platform");
|
||||
const architecture = @import("architecture");
|
||||
const devices_broker = @import("devices-broker.zig");
|
||||
const log = @import("log.zig");
|
||||
|
||||
const page_size: u64 = abi.page_size;
|
||||
const page_mask: u64 = page_size - 1;
|
||||
const huge_page_size: u64 = 2 * 1024 * 1024;
|
||||
|
||||
/// One domain per claimed PCI function. 64 mirrors devices-broker's device cap.
|
||||
pub const maximum_domains = 64;
|
||||
pub const invalid_domain: u16 = 0xFFFF;
|
||||
|
||||
const Domain = struct {
|
||||
in_use: bool = false,
|
||||
owner: u32 = 0, // task that owns the attached device
|
||||
bdf: u16 = 0, // requester id of the attached device
|
||||
page_table_root: u64 = 0, // physical address of the top-level table
|
||||
rmrr: bool = false, // a firmware reserved-region domain (persists across claims)
|
||||
};
|
||||
|
||||
var active: bool = false;
|
||||
var backend: architecture.iommu.Backend = undefined;
|
||||
var domains: [maximum_domains]Domain = .{Domain{}} ** maximum_domains;
|
||||
|
||||
pub fn enabled() bool {
|
||||
return active;
|
||||
}
|
||||
|
||||
/// Detect the IOMMU (the architecture module probes the discovered unit and
|
||||
/// returns its backend), pre-map firmware reserved regions, and enable
|
||||
/// translation. Fail-open (nothing activates) when no usable unit exists — the
|
||||
/// caller logs the posture. Must run after platform discovery and before any
|
||||
/// user process starts.
|
||||
pub fn init() void {
|
||||
const info = platform.platformInformation();
|
||||
if (!info.iommu_present) return;
|
||||
// A present-but-unusable unit stays fail-open with a logged reason rather
|
||||
// than half-enabling. The backend receives the kernel services it needs
|
||||
// (frames, the log sink) here — it never imports kernel internals.
|
||||
backend = architecture.iommu.detect(.{
|
||||
.register_base = info.iommu_base,
|
||||
.amd = info.iommu_is_amd,
|
||||
}, .{
|
||||
.allocateFrame = pmm.alloc,
|
||||
.allocateContiguous = pmm.allocContiguous,
|
||||
.write = log.write,
|
||||
}) orelse {
|
||||
log.write("/system/kernel: WARNING IOMMU present but unusable — staying fail-open\n");
|
||||
return;
|
||||
};
|
||||
active = true;
|
||||
|
||||
// The translation structures start empty: every device is denied until its driver
|
||||
// claims it (confineDevice gives it a private domain). PCI functions are enumerated
|
||||
// post-boot by the ring-3 pci-bus driver, so there is nothing to attach at init.
|
||||
// The backend writes its identity lines; the neutral posture lines follow.
|
||||
backend.enable();
|
||||
logPosture(info);
|
||||
}
|
||||
|
||||
/// Per-claimed-device record: its private domain, so a driver's death tears down
|
||||
/// exactly the domains it held.
|
||||
const Confined = struct { active: bool = false, owner: u32 = 0, bdf: u16 = 0, domain: u16 = invalid_domain };
|
||||
var confined: [maximum_domains]Confined = .{Confined{}} ** maximum_domains;
|
||||
|
||||
/// Place a just-claimed PCI function under IOMMU translation on behalf of `owner`: give
|
||||
/// it a private empty domain, seed it with the device's own firmware reserved region,
|
||||
/// and attach. Its DMA buffers arrive afterward as explicit grants — the owner's own
|
||||
/// `dma_alloc`'d regions are bound by the claim path (`mapForDevice`), and cross-process
|
||||
/// buffers by `dma_bind`. false only if a domain can't be allocated — the caller rolls
|
||||
/// the claim back (a claim that can't be confined must not stand). No-op success when no
|
||||
/// IOMMU exists (fail-open).
|
||||
pub fn confineDevice(device_id: u64, bdf: u16, owner: u32) bool {
|
||||
if (!active) return true;
|
||||
if (device_id >= confined.len) return true; // unusual id; leave it to fail-open
|
||||
const domain = domainCreate(owner, bdf) orelse return false;
|
||||
|
||||
// Firmware reserved region for this device, if any (real hardware; QEMU has none).
|
||||
const info = platform.platformInformation();
|
||||
var i: usize = 0;
|
||||
while (i < info.rmrr_count) : (i += 1) {
|
||||
if (info.rmrr[i].bdf == bdf)
|
||||
_ = map(domain, info.rmrr[i].base, info.rmrr[i].limit - info.rmrr[i].base + 1);
|
||||
}
|
||||
|
||||
attachDevice(domain, bdf);
|
||||
confined[@intCast(device_id)] = .{ .active = true, .owner = owner, .bdf = bdf, .domain = domain };
|
||||
return true;
|
||||
}
|
||||
|
||||
/// The confined record for `device_id`, or null if the device is not confined.
|
||||
fn confinedOf(device_id: u64) ?*Confined {
|
||||
if (device_id >= confined.len) return null;
|
||||
const c = &confined[@intCast(device_id)];
|
||||
return if (c.active) c else null;
|
||||
}
|
||||
|
||||
/// Map a DMA region into a specific claimed device's domain (the device owner binding a
|
||||
/// granted buffer). false if the device is not confined. No-op success without an IOMMU.
|
||||
pub fn mapForDevice(device_id: u64, physical: u64, len: u64) bool {
|
||||
if (!active) return true;
|
||||
const c = confinedOf(device_id) orelse return false;
|
||||
return map(c.domain, physical, len);
|
||||
}
|
||||
|
||||
/// Unmap a DMA region from a specific claimed device's domain. No-op if not confined.
|
||||
pub fn unmapForDevice(device_id: u64, physical: u64, len: u64) void {
|
||||
if (!active) return;
|
||||
const c = confinedOf(device_id) orelse return;
|
||||
unmap(c.domain, physical, len);
|
||||
}
|
||||
|
||||
/// Map a region into every claimed device owned by `owner` — the auto-bind of a task's
|
||||
/// own freshly-`dma_alloc`'d buffer into the devices it drives.
|
||||
pub fn mapRegionForOwner(owner: u32, physical: u64, len: u64) void {
|
||||
if (!active) return;
|
||||
for (&confined) |*c| {
|
||||
if (c.active and c.owner == owner) _ = map(c.domain, physical, len);
|
||||
}
|
||||
}
|
||||
|
||||
/// Unmap a region from EVERY claimed device's domain — the freed-region sweep. MUST run
|
||||
/// before the frames return to pmm: a device translating to a reallocated frame is the
|
||||
/// use-after-free this prevents. Cross-device because a granted buffer may be bound in a
|
||||
/// domain other than its owner's.
|
||||
pub fn unmapRegionEverywhere(physical: u64, len: u64) void {
|
||||
if (!active) return;
|
||||
for (&confined) |*c| {
|
||||
if (c.active) unmap(c.domain, physical, len);
|
||||
}
|
||||
}
|
||||
|
||||
/// A driver died or released its devices: tear down every domain it held (detach the
|
||||
/// device, free the tables) so their DMA is blocked again and a restarted driver
|
||||
/// re-claims cleanly. Runs BEFORE the broker claims and the DMA frames are released.
|
||||
pub fn releaseAllOwnedBy(owner: u32) void {
|
||||
if (!active) return;
|
||||
for (&confined) |*c| {
|
||||
if (c.active and c.owner == owner) {
|
||||
detachDevice(c.bdf);
|
||||
domainDestroy(c.domain);
|
||||
c.* = .{};
|
||||
}
|
||||
}
|
||||
_ = faultDrain(); // log any faults a mid-DMA device raised as it was cut off
|
||||
}
|
||||
|
||||
/// Allocate an empty domain (an empty top-level table). null when the table is full.
|
||||
pub fn domainCreate(owner: u32, bdf: u16) ?u16 {
|
||||
if (!active) return 0; // fail-open: a dummy id the no-op ops ignore
|
||||
for (&domains, 0..) |*d, index| {
|
||||
if (d.in_use) continue;
|
||||
const root = allocTable() orelse return null;
|
||||
d.* = .{ .in_use = true, .owner = owner, .bdf = bdf, .page_table_root = root };
|
||||
return @intCast(index);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Free a domain's page-table frames and its slot. Precondition: no device attached
|
||||
/// (detach first).
|
||||
pub fn domainDestroy(domain: u16) void {
|
||||
if (!active) return;
|
||||
const d = &domains[domain];
|
||||
if (!d.in_use) return;
|
||||
freeTables(d.page_table_root, backend.levels);
|
||||
d.* = .{};
|
||||
}
|
||||
|
||||
/// Attach `bdf`'s device to `domain` and pre-load any RMRR range recorded for it.
|
||||
pub fn attachDevice(domain: u16, bdf: u16) void {
|
||||
if (!active) return;
|
||||
const d = &domains[domain];
|
||||
d.bdf = bdf;
|
||||
backend.attach(bdf, hardwareId(domain), d.page_table_root);
|
||||
}
|
||||
|
||||
/// Return `bdf`'s device to not-present + invalidate.
|
||||
pub fn detachDevice(bdf: u16) void {
|
||||
if (!active) return;
|
||||
backend.detach(bdf);
|
||||
}
|
||||
|
||||
/// Identity-map [physical, physical+len) into `domain` (read+write) and invalidate.
|
||||
/// Unconditional domain-selective invalidation after every map — correct under VT-d
|
||||
/// caching-mode and free otherwise.
|
||||
pub fn map(domain: u16, physical: u64, len: u64) bool {
|
||||
if (!active) return true;
|
||||
const d = &domains[domain];
|
||||
if (!d.in_use) return false;
|
||||
if (!mapRange(d.page_table_root, physical, len)) return false;
|
||||
backend.invalidateDomain(hardwareId(domain));
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Unmap [physical, physical+len) from `domain` and invalidate. MUST finish its
|
||||
/// invalidation before the caller returns the frames to pmm — a stale IOTLB entry
|
||||
/// pointing at a reallocated frame is the use-after-free this ordering prevents.
|
||||
pub fn unmap(domain: u16, physical: u64, len: u64) void {
|
||||
if (!active) return;
|
||||
const d = &domains[domain];
|
||||
if (!d.in_use) return;
|
||||
unmapRange(d.page_table_root, physical, len);
|
||||
backend.invalidateDomain(hardwareId(domain));
|
||||
}
|
||||
|
||||
/// Poll the hardware for translation faults, log them, return the count. Called by the
|
||||
/// IOMMU test case and opportunistically after a device detaches.
|
||||
pub fn faultDrain() usize {
|
||||
if (!active) return 0;
|
||||
return backend.faultDrain();
|
||||
}
|
||||
|
||||
/// The physical address `virtual` maps to in `domain`, or null if unmapped — a test
|
||||
/// helper that walks the domain's page tables (identity mappings return `virtual`).
|
||||
pub fn translationOf(domain: u16, virtual: u64) ?u64 {
|
||||
if (!active) return virtual;
|
||||
const d = &domains[domain];
|
||||
if (!d.in_use) return null;
|
||||
var table = d.page_table_root;
|
||||
var level = backend.levels;
|
||||
while (level > 1) : (level -= 1) {
|
||||
const entry = tableAt(table)[indexAt(virtual, level)];
|
||||
if (!backend.isPresent(entry)) return null;
|
||||
if (level == 2 and isHugeLeaf(entry))
|
||||
return (entry & address_mask) | (virtual & (huge_page_size - 1));
|
||||
table = entry & address_mask;
|
||||
}
|
||||
const leaf = tableAt(table)[indexAt(virtual, 1)];
|
||||
if (!backend.isPresent(leaf)) return null;
|
||||
return (leaf & address_mask) | (virtual & page_mask);
|
||||
}
|
||||
|
||||
// --- the shared page-table walker -----------------------------------------------------
|
||||
// 512-entry, 9-bits-per-level, 4 KiB tables reached through the physmap — the shape both
|
||||
// VT-d second-level and AMD-Vi native tables share. The backend supplies the entry bits.
|
||||
|
||||
fn tableAt(physical: u64) [*]volatile u64 {
|
||||
return @ptrFromInt(boot_handoff.physicalToVirtual(physical));
|
||||
}
|
||||
|
||||
fn allocTable() ?u64 {
|
||||
const frame = pmm.alloc() orelse return null;
|
||||
const table = tableAt(frame);
|
||||
var i: usize = 0;
|
||||
while (i < 512) : (i += 1) table[i] = 0;
|
||||
return frame;
|
||||
}
|
||||
|
||||
const address_mask: u64 = 0x000F_FFFF_FFFF_F000;
|
||||
|
||||
fn indexAt(virtual: u64, level: u8) usize {
|
||||
// level 1 is the leaf table; shift = 12 + 9*(level-1).
|
||||
const shift: u6 = @intCast(12 + 9 * (@as(u32, level) - 1));
|
||||
return @intCast((virtual >> shift) & 0x1FF);
|
||||
}
|
||||
|
||||
/// Descend to (allocating) the next-level table below `entry_ptr`, returning its
|
||||
/// physical base. null on out-of-memory.
|
||||
fn descend(entry_ptr: *volatile u64, level: u8) ?u64 {
|
||||
const entry = entry_ptr.*;
|
||||
if (backend.isPresent(entry)) return entry & address_mask;
|
||||
const table = allocTable() orelse return null;
|
||||
backend.flushStructure(@intFromPtr(tableAt(table)));
|
||||
entry_ptr.* = backend.makeTable(table, level);
|
||||
backend.flushStructure(@intFromPtr(entry_ptr));
|
||||
return table;
|
||||
}
|
||||
|
||||
fn mapRange(root: u64, physical: u64, len: u64) bool {
|
||||
const start = physical & ~page_mask;
|
||||
const end = (physical + len + page_mask) & ~page_mask;
|
||||
var addr = start;
|
||||
while (addr < end) {
|
||||
// 2 MiB leaf when the backend allows it and both address and remaining span are
|
||||
// huge-aligned — keeps table memory sane for the blanket-identity and real-PC
|
||||
// cases without a separate superpage path per backend.
|
||||
const huge = backend.supports_huge_pages and
|
||||
addr % huge_page_size == 0 and (end - addr) >= huge_page_size;
|
||||
if (!mapOne(root, addr, huge)) return false;
|
||||
addr += if (huge) huge_page_size else page_size;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
fn mapOne(root: u64, addr: u64, huge: bool) bool {
|
||||
const leaf_level: u8 = if (huge) 2 else 1;
|
||||
var table = root;
|
||||
var level = backend.levels;
|
||||
while (level > leaf_level) : (level -= 1) {
|
||||
const entry_ptr = &tableAt(table)[indexAt(addr, level)];
|
||||
table = descend(entry_ptr, level) orelse return false;
|
||||
}
|
||||
const leaf_ptr = &tableAt(table)[indexAt(addr, leaf_level)];
|
||||
leaf_ptr.* = backend.makeLeaf(addr, huge);
|
||||
backend.flushStructure(@intFromPtr(leaf_ptr));
|
||||
return true;
|
||||
}
|
||||
|
||||
fn unmapRange(root: u64, physical: u64, len: u64) void {
|
||||
const start = physical & ~page_mask;
|
||||
const end = (physical + len + page_mask) & ~page_mask;
|
||||
var addr = start;
|
||||
while (addr < end) {
|
||||
const huge = backend.supports_huge_pages and
|
||||
addr % huge_page_size == 0 and (end - addr) >= huge_page_size;
|
||||
unmapOne(root, addr, huge);
|
||||
addr += if (huge) huge_page_size else page_size;
|
||||
}
|
||||
}
|
||||
|
||||
fn unmapOne(root: u64, addr: u64, huge: bool) void {
|
||||
const leaf_level: u8 = if (huge) 2 else 1;
|
||||
var table = root;
|
||||
var level = backend.levels;
|
||||
while (level > leaf_level) : (level -= 1) {
|
||||
const entry = tableAt(table)[indexAt(addr, level)];
|
||||
if (!backend.isPresent(entry)) return; // nothing mapped here
|
||||
table = entry & address_mask;
|
||||
}
|
||||
const leaf_ptr = &tableAt(table)[indexAt(addr, leaf_level)];
|
||||
leaf_ptr.* = 0;
|
||||
backend.flushStructure(@intFromPtr(leaf_ptr));
|
||||
}
|
||||
|
||||
/// Post-order free of a domain's whole table tree.
|
||||
fn freeTables(root: u64, level: u8) void {
|
||||
if (level > 1) {
|
||||
const table = tableAt(root);
|
||||
var i: usize = 0;
|
||||
while (i < 512) : (i += 1) {
|
||||
const entry = table[i];
|
||||
if (!backend.isPresent(entry)) continue;
|
||||
// A 2 MiB leaf sits at level 2 and points at RAM, not a sub-table.
|
||||
if (level == 2 and isHugeLeaf(entry)) continue;
|
||||
freeTables(entry & address_mask, level - 1);
|
||||
}
|
||||
}
|
||||
pmm.free(root);
|
||||
}
|
||||
|
||||
fn isHugeLeaf(entry: u64) bool {
|
||||
// Both backends set a page-size bit (VT-d bit 7, AMD leaf next-level=0 at level 2).
|
||||
// The backend's makeLeaf encodes it; the walker only needs "is this a leaf, not a
|
||||
// pointer" at level 2, which huge leaves are by construction.
|
||||
return entry & huge_leaf_bit != 0;
|
||||
}
|
||||
|
||||
/// The size-bit the backends set on a 2 MiB leaf (VT-d SL-PTE PS bit 7; AMD encodes a
|
||||
/// leaf as next-level 0, so the core marks huge leaves with this software bit — an
|
||||
/// ignored bit in both formats — to tell them apart from table pointers when freeing).
|
||||
const huge_leaf_bit: u64 = 1 << 7;
|
||||
|
||||
fn hardwareId(domain: u16) u16 {
|
||||
return domain + 1; // id 0 is reserved by both architectures
|
||||
}
|
||||
|
||||
fn logPosture(info: platform.PlatformInformation) void {
|
||||
log.print(" rmrr : {d} region(s) premapped\n", .{info.rmrr_count});
|
||||
if (info.rmrr_skipped > 0)
|
||||
log.print(" rmrr : WARNING {d} scope(s) skipped — a device keeps an unmapped firmware buffer\n", .{info.rmrr_skipped});
|
||||
if (info.iommu_extra_units > 0)
|
||||
log.print(" units : WARNING {d} other unit(s) — their scoped devices are NOT translated\n", .{info.iommu_extra_units});
|
||||
}
|
||||
@@ -154,6 +154,66 @@ pub fn dropRef(endpoint: *Endpoint) void {
|
||||
/// Defined here (not in scheduler) because the meaning is the IPC/capability layer's.
|
||||
pub const handle_kind_endpoint: u8 = 0;
|
||||
pub const handle_kind_shared_memory: u8 = 1;
|
||||
pub const handle_kind_dma_region: u8 = 2;
|
||||
|
||||
/// A DMA buffer's delegation token: names the `pages` contiguous frames at `phys` that
|
||||
/// `dma_alloc` handed its `owner`, and can be passed across processes as a capability so
|
||||
/// the driver that owns a device can bind it into that device's IOMMU domain
|
||||
/// (`dma_bind`). Unlike `SharedMemoryObject` this does NOT own the frames — the
|
||||
/// allocating address space still does, and frees them on `dma_free` or teardown — so
|
||||
/// this is a pure token: `dead` is set when the allocator frees the region, after which
|
||||
/// a stale handle can no longer bind it. `refcount` counts the allocator's registry
|
||||
/// entry plus every outstanding handle; the object is freed when the last drops.
|
||||
pub const DmaRegionObject = struct {
|
||||
refcount: u32 = 1,
|
||||
phys: u64,
|
||||
pages: usize,
|
||||
owner: u32,
|
||||
dead: bool = false,
|
||||
};
|
||||
|
||||
/// Create a DMA-region token for `pages` frames at `phys` owned by task `owner`. The
|
||||
/// frames are already allocated and mapped by the caller; this only wraps them for
|
||||
/// delegation. null if the heap is out of room.
|
||||
pub fn createDmaRegion(phys: u64, pages: usize, owner: u32) ?*DmaRegionObject {
|
||||
const region = heap.allocator().create(DmaRegionObject) catch return null;
|
||||
region.* = .{ .phys = phys, .pages = pages, .owner = owner };
|
||||
return region;
|
||||
}
|
||||
|
||||
/// Drop a DMA-region reference; free the token when the last (registry + handles) goes.
|
||||
/// Never frees frames — the allocator owns those.
|
||||
pub fn dropDmaRegionReference(region: *DmaRegionObject) void {
|
||||
if (region.refcount > 1) {
|
||||
region.refcount -= 1;
|
||||
} else {
|
||||
heap.allocator().destroy(region);
|
||||
}
|
||||
}
|
||||
|
||||
/// Resolve a handle to its DMA-region token, or null if out of range, unused, or a
|
||||
/// different kind.
|
||||
pub fn resolveDmaRegion(t: *Task, h: u64) ?*DmaRegionObject {
|
||||
if (h >= t.handles.len) return null;
|
||||
const entry = t.handles[@intCast(h)] orelse return null;
|
||||
if (entry.kind != handle_kind_dma_region) return null;
|
||||
return @ptrCast(@alignCast(entry.ptr));
|
||||
}
|
||||
|
||||
/// Install a DMA-region handle in task `t`'s table (the slot owns a reference).
|
||||
pub fn installDmaRegionHandle(t: *Task, region: *DmaRegionObject) i64 {
|
||||
return installEntry(t, .{ .kind = handle_kind_dma_region, .ptr = @ptrCast(region) });
|
||||
}
|
||||
|
||||
/// Drop the handle at slot `h` of task `t` (handle_close): release its reference and
|
||||
/// free the slot. Returns 0 or -EBADF.
|
||||
pub fn closeHandle(t: *Task, h: u64) i64 {
|
||||
if (h >= t.handles.len) return -EBADF;
|
||||
const entry = t.handles[@intCast(h)] orelse return -EBADF;
|
||||
dropEntry(entry);
|
||||
t.handles[@intCast(h)] = null;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/// A page-aligned block of **shared cacheable RAM** (docs/display-v2.md), referenced by
|
||||
/// capability handles across processes and freed when the last one drops. `phys` is its
|
||||
@@ -307,6 +367,10 @@ fn shareCapability(from: *Task, to: *Task, cap: u64) i64 {
|
||||
const s: *SharedMemoryObject = @ptrCast(@alignCast(entry.ptr));
|
||||
s.refcount += 1;
|
||||
},
|
||||
handle_kind_dma_region => {
|
||||
const r: *DmaRegionObject = @ptrCast(@alignCast(entry.ptr));
|
||||
r.refcount += 1;
|
||||
},
|
||||
else => return -EBADF,
|
||||
}
|
||||
const handle = installEntry(to, entry);
|
||||
@@ -574,6 +638,7 @@ fn dropEntry(entry: scheduler.HandleObject) void {
|
||||
switch (entry.kind) {
|
||||
handle_kind_endpoint => dropRef(@ptrCast(@alignCast(entry.ptr))),
|
||||
handle_kind_shared_memory => dropSharedMemoryReference(@ptrCast(@alignCast(entry.ptr))),
|
||||
handle_kind_dma_region => dropDmaRegionReference(@ptrCast(@alignCast(entry.ptr))),
|
||||
else => {},
|
||||
}
|
||||
}
|
||||
|
||||
@@ -13,6 +13,7 @@ const sync = @import("sync.zig");
|
||||
const process = @import("process.zig");
|
||||
const devices_broker = @import("devices-broker.zig");
|
||||
const irq = @import("irq.zig");
|
||||
const iommu = @import("iommu.zig");
|
||||
const platform = @import("platform");
|
||||
const tests = @import("tests.zig");
|
||||
const build_options = @import("build_options");
|
||||
@@ -215,6 +216,12 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
// land on. Every line stays masked until something binds it (ioapic.init).
|
||||
irq.init();
|
||||
|
||||
// Bring up DMA translation: build the IOMMU domains and enable it (or record
|
||||
// fail-open when no unit exists). Must run before any driver claims a device —
|
||||
// an unclaimed device's DMA is blocked once translation is on. Logs its own
|
||||
// enable block; the fail-open posture is stated in the platform block below.
|
||||
iommu.init();
|
||||
|
||||
// Power register map, from the FADT (the SLP_TYP sleep values live in AML,
|
||||
// which the kernel doesn't parse — the ring-3 acpi service owns soft-off).
|
||||
const pw = platform.powerInformation();
|
||||
@@ -272,6 +279,11 @@ fn kmain(boot_information: *const BootInformation) noreturn {
|
||||
log.print(" cpus : {d} usable core(s); 1 running (BSP), {d} AP(s) parked (SMP bring-up pending)\n", .{ cores.len, if (cores.len > 0) cores.len - 1 else 0 });
|
||||
if (platform.cpusDropped() > 0)
|
||||
log.print(" cpus : WARNING {d} core(s) beyond pool cap dropped\n", .{platform.cpusDropped()});
|
||||
// The DMA-isolation posture, stated plainly at every boot. When a unit exists,
|
||||
// iommu.init already logged its enable block above; here we only state the
|
||||
// fail-open case, so a boot without the line is a boot with translation on.
|
||||
if (!iommu.enabled())
|
||||
log.write(" iommu : none present - DMA fail-open (unisolated)\n");
|
||||
} else |err| {
|
||||
log.print("\n/system/kernel: device discovery failed: {s}\n", .{@errorName(err)});
|
||||
}
|
||||
|
||||
@@ -33,6 +33,7 @@ const sync = @import("sync.zig");
|
||||
const ipc = @import("ipc-synchronous.zig");
|
||||
const devices_broker = @import("devices-broker.zig");
|
||||
const irq = @import("irq.zig");
|
||||
const iommu = @import("iommu.zig");
|
||||
const initial_ramdisk = @import("initial-ramdisk");
|
||||
const vfs = @import("vfs.zig");
|
||||
const log = @import("log.zig");
|
||||
@@ -250,6 +251,10 @@ fn system_call(state: *architecture.CpuState) void {
|
||||
.fs_node => systemFsNode(state),
|
||||
.fs_mount => systemFsMount(state),
|
||||
.fs_unmount => systemFsUnmount(state),
|
||||
.iommu_fault_drain => systemIommuFaultDrain(state),
|
||||
.dma_bind => systemDmaBind(state),
|
||||
.dma_unbind => systemDmaUnbind(state),
|
||||
.handle_close => systemHandleClose(state),
|
||||
.wall_clock => systemWallClock(state),
|
||||
.shared_memory_create => systemSharedMemoryCreate(state),
|
||||
.shared_memory_map => systemSharedMemoryMap(state),
|
||||
@@ -390,6 +395,21 @@ fn systemDeviceClaim(state: *architecture.CpuState) void {
|
||||
const claim_flags = sync.enter();
|
||||
defer sync.leave(claim_flags);
|
||||
if (devices_broker.claim(device_id, scheduler.current().id)) {
|
||||
// Confine the device's DMA before the driver can program it: a PCI function
|
||||
// becomes reachable to the IOMMU only once claimed (until now its DMA is
|
||||
// blocked). A claim that cannot be confined must not stand — roll it back —
|
||||
// since the whole point is that claiming a DMA device is no longer equivalent
|
||||
// to ring 0. No-op when no IOMMU exists (fail-open).
|
||||
if (devices_broker.pciAddressOf(device_id)) |bdf| {
|
||||
const owner = scheduler.current().id;
|
||||
if (!iommu.confineDevice(device_id, bdf, owner)) {
|
||||
_ = devices_broker.unclaim(device_id, owner);
|
||||
return fail(state);
|
||||
}
|
||||
// Bind the buffers this task allocated before claiming the device (a driver
|
||||
// that dma_alloc'd its rings, then claimed the controller).
|
||||
dmaBindOwnerRegionsInto(owner, device_id);
|
||||
}
|
||||
// A display service just took the framebuffer — quiesce the bootstrap console
|
||||
// so the kernel and the service don't scribble over each other's pixels. The
|
||||
// claim releases (and the console resumes) automatically if the service dies;
|
||||
@@ -497,6 +517,137 @@ fn systemIoWrite(state: *architecture.CpuState) void {
|
||||
/// their physical address is never disclosed. `dma_below_4g` caps the physical address
|
||||
/// for legacy engines; `dma_write_combining` is accepted but falls back to coherent
|
||||
/// until PAT is programmed. See docs/driver-model.md (M14).
|
||||
// --- DMA-region registry ---------------------------------------------------------------
|
||||
// Every dma_alloc'd region is tracked here so it can be (a) auto-bound into the devices
|
||||
// its owner claims, (b) delegated across processes as a capability and bound into a
|
||||
// device's IOMMU domain by dma_bind, and (c) unmapped from every domain before its frames
|
||||
// return to the allocator. Only *shareable* regions carry a heap `object` (the delegation
|
||||
// token); a driver's private rings are tracked without one. All access under the big lock.
|
||||
const DmaRegistryEntry = struct {
|
||||
active: bool = false,
|
||||
object: ?*ipc.DmaRegionObject = null,
|
||||
physical: u64 = 0,
|
||||
len: u64 = 0,
|
||||
owner: u32 = 0,
|
||||
};
|
||||
const maximum_dma_regions = 256;
|
||||
var dma_registry: [maximum_dma_regions]DmaRegistryEntry = .{DmaRegistryEntry{}} ** maximum_dma_regions;
|
||||
|
||||
fn dmaRegistryAdd(object: ?*ipc.DmaRegionObject, physical: u64, len: u64, owner: u32) void {
|
||||
for (&dma_registry) |*e| {
|
||||
if (!e.active) {
|
||||
e.* = .{ .active = true, .object = object, .physical = physical, .len = len, .owner = owner };
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Retire the region based at `physical` owned by `owner`: unmap it from every device
|
||||
/// domain (before the frames are freed), mark its token dead so a stale downstream handle
|
||||
/// can no longer bind it, drop the registry's reference, and clear the slot.
|
||||
fn dmaRegistryRemove(owner: u32, physical: u64) void {
|
||||
for (&dma_registry) |*e| {
|
||||
if (e.active and e.owner == owner and e.physical == physical) {
|
||||
iommu.unmapRegionEverywhere(e.physical, e.len);
|
||||
if (e.object) |object| {
|
||||
object.dead = true;
|
||||
ipc.dropDmaRegionReference(object);
|
||||
}
|
||||
e.* = .{};
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Retire every region owned by `owner` (task death) — same discipline as a per-region
|
||||
/// free, run before the address space is torn down and its DMA frames reclaimed.
|
||||
fn dmaRegistryReleaseOwner(owner: u32) void {
|
||||
for (&dma_registry) |*e| {
|
||||
if (e.active and e.owner == owner) {
|
||||
iommu.unmapRegionEverywhere(e.physical, e.len);
|
||||
if (e.object) |object| {
|
||||
object.dead = true;
|
||||
ipc.dropDmaRegionReference(object);
|
||||
}
|
||||
e.* = .{};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Bind every region `owner` allocated into the domain of the device it just claimed
|
||||
/// (its rings allocated before the claim). Regions allocated *after* the claim are bound
|
||||
/// by dma_alloc's own auto-bind.
|
||||
fn dmaBindOwnerRegionsInto(owner: u32, device_id: u64) void {
|
||||
for (&dma_registry) |*e| {
|
||||
if (e.active and e.owner == owner) _ = iommu.mapForDevice(device_id, e.physical, e.len);
|
||||
}
|
||||
}
|
||||
|
||||
/// iommu_fault_drain() -> count: drain and log any pending IOMMU translation faults,
|
||||
/// returning how many were seen. A diagnostic hook — a driver (or a test) that suspects
|
||||
/// its device faulted can force the fault records to be logged now rather than waiting
|
||||
/// for the next device-release drain. Harmless without an IOMMU (returns 0).
|
||||
fn systemIommuFaultDrain(state: *architecture.CpuState) void {
|
||||
const flags = sync.enter();
|
||||
const count = iommu.faultDrain();
|
||||
sync.leave(flags);
|
||||
architecture.setSystemCallResult(state, count);
|
||||
}
|
||||
|
||||
/// Resolve a capability handle the caller holds to the physical range it names — either
|
||||
/// a DMA-region token or a shared-memory object (both are bindable buffers). null if the
|
||||
/// handle is neither, or names a region already freed by its allocator.
|
||||
fn bindableRange(t: *scheduler.Task, handle: u64) ?struct { physical: u64, len: u64 } {
|
||||
if (ipc.resolveDmaRegion(t, handle)) |region| {
|
||||
if (region.dead) return null;
|
||||
return .{ .physical = region.phys, .len = region.pages * page_size };
|
||||
}
|
||||
if (ipc.resolveSharedMemory(t, handle)) |shared| {
|
||||
return .{ .physical = shared.phys, .len = shared.pages * page_size };
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/// dma_bind(device_id, handle) -> 0/-errno: map the buffer named by `handle` into the
|
||||
/// claimed device's IOMMU domain. The caller must own the device (the mmio_map gate) and
|
||||
/// hold the handle. Idempotent: re-binding is a harmless success, so a driver may re-bind
|
||||
/// after a restart without tracking what it already bound. Success (no-op) without an IOMMU.
|
||||
fn systemDmaBind(state: *architecture.CpuState) void {
|
||||
const device_id = architecture.systemCallArg(state, 0);
|
||||
const handle = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
if (devices_broker.ownerOf(device_id) != t.id) return fail(state);
|
||||
const range = bindableRange(t, handle) orelse return fail(state);
|
||||
if (!iommu.mapForDevice(device_id, range.physical, range.len)) return fail(state);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// dma_unbind(device_id, handle) -> 0/-errno: unmap a previously bound buffer from the
|
||||
/// device's domain and invalidate.
|
||||
fn systemDmaUnbind(state: *architecture.CpuState) void {
|
||||
const device_id = architecture.systemCallArg(state, 0);
|
||||
const handle = architecture.systemCallArg(state, 1);
|
||||
const t = scheduler.current();
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
if (devices_broker.ownerOf(device_id) != t.id) return fail(state);
|
||||
const range = bindableRange(t, handle) orelse return fail(state);
|
||||
iommu.unmapForDevice(device_id, range.physical, range.len);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
/// handle_close(handle) -> 0/-errno: drop one capability handle and free its slot.
|
||||
fn systemHandleClose(state: *architecture.CpuState) void {
|
||||
const handle = architecture.systemCallArg(state, 0);
|
||||
const t = scheduler.current();
|
||||
const flags = sync.enter();
|
||||
defer sync.leave(flags);
|
||||
if (ipc.closeHandle(t, handle) < 0) return fail(state);
|
||||
architecture.setSystemCallResult(state, 0);
|
||||
}
|
||||
|
||||
fn systemDmaAlloc(state: *architecture.CpuState) void {
|
||||
const len = architecture.systemCallArg(state, 0);
|
||||
const flags = architecture.systemCallArg(state, 1);
|
||||
@@ -543,8 +694,33 @@ fn systemDmaAlloc(state: *architecture.CpuState) void {
|
||||
architecture.mapUserDmaInto(t.address_space, base_v + i * page_size, phys + i * page_size, page_size);
|
||||
sync.leave(lock_flags);
|
||||
}
|
||||
// Register the region and bind it into every device this task already drives (its
|
||||
// own buffers reach its own devices). When `dma_shareable` is set, also wrap it in a
|
||||
// capability token and return a handle so it can be delegated to another driver and
|
||||
// dma_bound there. All under the lock (the registry + IOMMU tables are shared).
|
||||
var handle: u64 = abi.no_cap;
|
||||
{
|
||||
const lock_flags = sync.enter();
|
||||
defer sync.leave(lock_flags);
|
||||
var object: ?*ipc.DmaRegionObject = null;
|
||||
if (flags & abi.dma_shareable != 0) {
|
||||
if (ipc.createDmaRegion(phys, pages, t.id)) |region| {
|
||||
const h = ipc.installDmaRegionHandle(t, region);
|
||||
if (h >= 0) {
|
||||
region.refcount += 1; // the handle's reference (registry holds the first)
|
||||
handle = @intCast(h);
|
||||
object = region;
|
||||
} else {
|
||||
ipc.dropDmaRegionReference(region); // no table slot; drop it
|
||||
}
|
||||
}
|
||||
}
|
||||
dmaRegistryAdd(object, phys, pages * page_size, t.id);
|
||||
iommu.mapRegionForOwner(t.id, phys, pages * page_size);
|
||||
}
|
||||
architecture.setSystemCallResult(state, base_v); // virtual address for the CPU
|
||||
architecture.setSystemCallResult2(state, phys); // physical address for the device
|
||||
architecture.setSystemCallResult3(state, handle); // capability handle (no_cap unless shareable)
|
||||
}
|
||||
|
||||
/// dma_free(virtual_address, len) -> 0: release a prior `dma_alloc`. Bounded to the DMA arena so
|
||||
@@ -560,6 +736,16 @@ fn systemDmaFree(state: *architecture.CpuState) void {
|
||||
const pages: usize = @intCast((len + page_size - 1) / page_size);
|
||||
if (base_v < dma_arena_base or base_v + pages * page_size > dma_arena_end) return fail(state);
|
||||
|
||||
// Retire the region — unmap it from every device domain and mark its token dead —
|
||||
// BEFORE any frame returns to the allocator, so no device can still translate to a
|
||||
// reallocated frame. dma_alloc's frames are contiguous, so the base names the region.
|
||||
{
|
||||
const lock_flags = sync.enter();
|
||||
if (architecture.translate(t.address_space, base_v)) |base_phys|
|
||||
dmaRegistryRemove(t.id, base_phys);
|
||||
sync.leave(lock_flags);
|
||||
}
|
||||
|
||||
for (0..pages) |i| {
|
||||
const va = base_v + i * page_size;
|
||||
// Per-page lock hold: the translate/unmap walks the shared page tables
|
||||
@@ -991,6 +1177,14 @@ pub var fault_kill_count: u64 = 0;
|
||||
fn releaseTaskResourcesLocked(t: *scheduler.Task) void {
|
||||
recordExitLocked(t);
|
||||
irq.releaseOwner(t.id);
|
||||
// Detach the task's devices from their IOMMU domains BEFORE the broker clears the
|
||||
// claims (the detach reads ownership) and before the address space is torn down and
|
||||
// its DMA frames return to the allocator — a device must stop translating to a frame
|
||||
// before that frame can be handed to someone else. Then retire the buffers this task
|
||||
// allocated, unmapping them from any *other* driver's domain they were granted into,
|
||||
// before those frames are freed too.
|
||||
iommu.releaseAllOwnedBy(t.id);
|
||||
dmaRegistryReleaseOwner(t.id);
|
||||
devices_broker.releaseAllOwnedBy(t.id);
|
||||
// If that dropped the framebuffer claim (this task was the display service), let the
|
||||
// bootstrap console draw again — the screen is nobody's now, so panics/status land.
|
||||
|
||||
@@ -149,7 +149,10 @@ pub const maximum_task_name = abi.maximum_process_name;
|
||||
|
||||
/// Size of each task's IPC handle table. Kept here (not in ipc_sync.zig) because
|
||||
/// it dimensions a field of `Task`; ipc_sync.zig re-exports it.
|
||||
pub const ipc_maximum_handles = 16;
|
||||
// Raised from 16 with DMA-region capabilities: a driver now holds its per-device
|
||||
// channel endpoints plus received DMA-region handles (a storage driver forwards several
|
||||
// buffer caps), and repeated cap-passing consumes slots until handle_close.
|
||||
pub const ipc_maximum_handles = 32;
|
||||
|
||||
/// One handle-table entry: a capability object plus a `kind` tag saying what `ptr` points
|
||||
/// at (an ipc endpoint or a shared-memory object), so a task's exit path and the
|
||||
|
||||
+131
-15
@@ -18,6 +18,7 @@ const wall_clock = @import("wall-clock.zig");
|
||||
const devices_broker = @import("devices-broker.zig");
|
||||
const platform = @import("platform");
|
||||
const pmm = @import("pmm.zig");
|
||||
const iommu = @import("iommu.zig");
|
||||
const heap = @import("heap.zig");
|
||||
const scheduler = @import("scheduler.zig");
|
||||
const ipc = @import("ipc.zig");
|
||||
@@ -212,6 +213,10 @@ pub fn run(case: []const u8, boot_information: *const BootInformation) void {
|
||||
deviceListTest(boot_information);
|
||||
} else if (eql(case, "pci-scan")) {
|
||||
pciScanTest(boot_information);
|
||||
} else if (eql(case, "pci-caps")) {
|
||||
pciCapsTest(boot_information);
|
||||
} else if (eql(case, "iommu-fault")) {
|
||||
iommuFaultTest(boot_information);
|
||||
} else if (eql(case, "acpi-parse")) {
|
||||
acpiParseTest(boot_information);
|
||||
} else if (eql(case, "acpi-report")) {
|
||||
@@ -1164,6 +1169,11 @@ fn dmaTest() void {
|
||||
log("DANOS-TEST-BEGIN: dma\n", .{});
|
||||
const base_free = pmm.stats().free_frames;
|
||||
|
||||
// This case boots WITHOUT an IOMMU device, so it is the explicit witness of the
|
||||
// fail-open posture: no unit found, and the kernel said so at boot (the harness
|
||||
// asserts the boot line; this check pins the recorded state to it).
|
||||
check("no IOMMU present: DMA runs fail-open", !platform.platformInformation().iommu_present);
|
||||
|
||||
// A contiguous run: aligned, and it consumed exactly that many frames.
|
||||
const frames = 4;
|
||||
const phys = pmm.allocContiguous(frames, ~@as(u64, 0)) orelse {
|
||||
@@ -1234,16 +1244,40 @@ fn msiTest() void {
|
||||
|
||||
/// IOMMU (M16): with an emulated VT-d unit present (the harness boots this case with
|
||||
/// `-device intel-iommu`), danos must find it in the ACPI DMAR table, map its register
|
||||
/// block, and read back a real version. This is *detection*, the honest first step —
|
||||
/// no translation domains are programmed yet, so DMA is still unprotected; enforcement
|
||||
/// lands with the first DMA driver (docs/driver-model.md M16).
|
||||
/// block, and read back a real version. Detection was M16's honest first step; the
|
||||
/// IOVA-enforcement track extends this case milestone by milestone (translation
|
||||
/// enabled, then per-device domains) — see the plan in docs and the fail-open witness
|
||||
/// in dmaTest.
|
||||
fn iommuTest() void {
|
||||
log("DANOS-TEST-BEGIN: iommu\n", .{});
|
||||
const pinfo = platform.platformInformation();
|
||||
check("IOMMU found in the DMAR table", pinfo.iommu_present);
|
||||
check("VT-d unit has a register base", pinfo.iommu_base != 0);
|
||||
check("VT-d version register reads back nonzero (real, mappable unit)", pinfo.iommu_version != 0);
|
||||
log("DANOS-IOMMU: base=0x{x} version=0x{x} capabilities=0x{x}\n", .{ pinfo.iommu_base, pinfo.iommu_version, pinfo.iommu_capabilities });
|
||||
check("IOMMU found in the firmware tables", pinfo.iommu_present);
|
||||
check("IOMMU unit has a register base", pinfo.iommu_base != 0);
|
||||
// The live-unit sanity (the version register reading back nonzero) now
|
||||
// gates detect itself: an unusable unit stays fail-open, so `enabled()`
|
||||
// below subsumes the old vendor-gated register check.
|
||||
log("DANOS-IOMMU: base=0x{x}\n", .{pinfo.iommu_base});
|
||||
|
||||
// Translation was enabled at boot (kernel.zig: iommu.init before any driver claims
|
||||
// a device). The blanket domain keeps every device identity-mapped, so DMA still
|
||||
// works, but the unit is live — and with only the boot-time mappings present, no
|
||||
// device should have faulted yet.
|
||||
check("IOMMU enabled (translation on)", iommu.enabled());
|
||||
check("no spurious translation faults at idle", iommu.faultDrain() == 0);
|
||||
|
||||
// A scratch domain proves the walker + invalidation path end to end: create it,
|
||||
// identity-map a page, and confirm the mapping resolves; then unmap and destroy.
|
||||
if (iommu.domainCreate(0, 0)) |scratch| {
|
||||
const scratch_phys: u64 = 0x0010_0000; // 1 MiB, page-aligned
|
||||
check("map into a scratch domain succeeds", iommu.map(scratch, scratch_phys, abi.page_size));
|
||||
check("scratch domain resolves the mapping", iommu.translationOf(scratch, scratch_phys) == scratch_phys);
|
||||
iommu.unmap(scratch, scratch_phys, abi.page_size);
|
||||
check("scratch domain drops the mapping", iommu.translationOf(scratch, scratch_phys) == null);
|
||||
iommu.domainDestroy(scratch);
|
||||
} else {
|
||||
check("scratch domain allocated", false);
|
||||
}
|
||||
log("DANOS-IOMMU: enabled base=0x{x} domains active\n", .{pinfo.iommu_base});
|
||||
result();
|
||||
}
|
||||
|
||||
@@ -1935,15 +1969,20 @@ fn initTest(boot_information: *const BootInformation) void {
|
||||
};
|
||||
check("init loaded and spawned as a process", spawned);
|
||||
|
||||
// Wait (real time) for at least two heartbeats — proving it runs, writes,
|
||||
// and sleeps repeatedly (init sleeps ~1 s between beats).
|
||||
// Wait (real time) until the LAST write is a heartbeat — proving init got
|
||||
// through its boot chatter (heap ok, the /etc/init.csv lookup) and settled
|
||||
// into its beat-and-sleep loop (~1 s between beats). Waiting on the text
|
||||
// rather than a raw write count: the boot chatter alone satisfies a count,
|
||||
// which is exactly the too-early check that used to fail here.
|
||||
scheduler.setPriority(1);
|
||||
const deadline = architecture.millis() + 8000;
|
||||
while (process.write_count < 2 and architecture.millis() < deadline) scheduler.yield();
|
||||
scheduler.setPriority(4);
|
||||
|
||||
const prefix = "init: heartbeat";
|
||||
const beat_ok = bufferHas(prefix);
|
||||
const deadline = architecture.millis() + 8000;
|
||||
var beat_ok = false;
|
||||
while (!beat_ok and architecture.millis() < deadline) {
|
||||
beat_ok = bufferHas(prefix);
|
||||
scheduler.yield();
|
||||
}
|
||||
scheduler.setPriority(4);
|
||||
check("init produced repeated heartbeats (>=2)", process.write_count >= 2);
|
||||
check("heartbeat text arrived intact", beat_ok);
|
||||
check("heartbeats came from user mode (CPL 3)", process.write_from_user);
|
||||
@@ -2388,6 +2427,74 @@ fn deviceListTest(boot_information: *const BootInformation) void {
|
||||
result();
|
||||
}
|
||||
|
||||
/// The driver-side PCI library against a real function: the pci-caps QEMU case adds an
|
||||
/// e1000e NIC no danos driver claims; the pci-cap-test fixture claims it and exercises
|
||||
/// header accessors, command bits, the capability walks, MSI programming (the first
|
||||
/// driver-side `msi_bind` use), the MSI-X table, power state, and FLR. The kernel side
|
||||
/// only spawns the manager (which spawns pci-bus itself) and the fixture; the substance
|
||||
/// is asserted by the harness on the fixture's own serial lines.
|
||||
fn pciCapsTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: pci-caps\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
process.setInitialRamdisk(image);
|
||||
// Plain mode — no restart drill, whose kill would race the fixture's claim.
|
||||
var manager: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
|
||||
break;
|
||||
}
|
||||
check("device-manager spawned", manager != 0);
|
||||
check("pci-cap-test spawned", spawnNamed(rd, "pci-cap-test"));
|
||||
result();
|
||||
}
|
||||
|
||||
/// IOMMU enforcement, the negative proof: boot with VT-d on and an unclaimed e1000e.
|
||||
/// The manager spawns pci-bus, the fixture claims the NIC and fires a DMA at an
|
||||
/// unmapped page; the unit must fault it and the system survive. Substance is asserted
|
||||
/// by the harness on the kernel's DANOS-IOMMU-FAULT line and the fixture's markers.
|
||||
fn iommuFaultTest(boot_information: *const BootInformation) void {
|
||||
log("DANOS-TEST-BEGIN: iommu-fault\n", .{});
|
||||
if (boot_information.initial_ramdisk_len == 0) {
|
||||
check("bootloader handed over an initial_ramdisk", false);
|
||||
result();
|
||||
return;
|
||||
}
|
||||
const image = @as([*]const u8, @ptrFromInt(boot_handoff.physicalToVirtual(boot_information.initial_ramdisk_base)))[0..boot_information.initial_ramdisk_len];
|
||||
const rd = initial_ramdisk.Reader.init(image) orelse {
|
||||
check("initial_ramdisk image is valid", false);
|
||||
result();
|
||||
return;
|
||||
};
|
||||
|
||||
check("IOMMU enabled for the enforcement test", iommu.enabled());
|
||||
process.setInitialRamdisk(image);
|
||||
var manager: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
if (!eql(initial_ramdisk.basename(item.name), "device-manager")) continue;
|
||||
manager = process.spawnProcessSupervised(item.blob, 4, &.{"device-manager"}, scheduler.currentId(), null) catch 0;
|
||||
break;
|
||||
}
|
||||
check("device-manager spawned", manager != 0);
|
||||
check("iommu-fault-test spawned", spawnNamed(rd, "iommu-fault-test"));
|
||||
result();
|
||||
}
|
||||
|
||||
/// M19.1: the ring-3 PCI scan agrees with the kernel's. The manager spawns
|
||||
/// pci-bus for the host bridge; the driver walks the same ECAM window through
|
||||
/// its mmio_map grant and must find exactly the functions the kernel's own
|
||||
@@ -2689,15 +2796,24 @@ fn initialRamdiskTest(boot_information: *const BootInformation) void {
|
||||
|
||||
process.write_count = 0;
|
||||
process.write_from_user = false;
|
||||
var programs: u32 = 0;
|
||||
var spawned: u32 = 0;
|
||||
var i: u32 = 0;
|
||||
while (i < rd.count) : (i += 1) {
|
||||
const item = rd.entry(i) orelse continue;
|
||||
// The FHS boot tree ferries data files too (/etc/devices.csv,
|
||||
// /etc/init.csv — served read-only by the kernel VFS, never spawned);
|
||||
// only the /system and /test trees hold programs, so only those count
|
||||
// toward the spawn-everything sweep.
|
||||
const is_program = std.mem.startsWith(u8, item.name, "/system/") or
|
||||
std.mem.startsWith(u8, item.name, "/test/");
|
||||
if (!is_program) continue;
|
||||
programs += 1;
|
||||
if (process.spawnProcess(item.blob, 4, &.{item.name})) spawned += 1 else |err| {
|
||||
log("DANOS-INITRD-ERR: {s}: {s}\n", .{ item.name, @errorName(err) });
|
||||
}
|
||||
}
|
||||
check("every initial_ramdisk binary spawned", spawned == rd.count);
|
||||
check("every initial_ramdisk program spawned", programs >= 1 and spawned == programs);
|
||||
|
||||
// Wait for the spawned programs to run and make syscalls (they write + sleep).
|
||||
scheduler.setPriority(1);
|
||||
|
||||
@@ -0,0 +1,22 @@
|
||||
//! The acpi service as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! the shared recipe and the module-to-domain map live in build-support.
|
||||
//!
|
||||
//! The artifact is named "discovery": one swappable process per firmware
|
||||
//! fills the ramdisk's neutral `discovery` slot (docs/discovery.md); the
|
||||
//! root's -Ddiscovery picks this package or `fdt`.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "discovery",
|
||||
.root_source_file = b.path("acpi.zig"),
|
||||
.imports = &.{
|
||||
"acpi-ids", "aml", "device-manager-protocol", "driver", "ipc", "logging", "memory",
|
||||
"power-protocol", "process", "service", "time",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
.{
|
||||
.name = .acpi,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0xf31e9a0903256b64, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.device = .{ .path = "../../../library/device" },
|
||||
.protocol = .{ .path = "../../../library/protocol" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -0,0 +1,18 @@
|
||||
//! The device-manager service as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! the shared recipe and the module-to-domain map live in build-support.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "device-manager",
|
||||
.root_source_file = b.path("device-manager.zig"),
|
||||
.imports = &.{
|
||||
"device-manager-protocol", "device-registry", "driver", "file-system", "ipc",
|
||||
"logging", "memory", "process", "service", "time",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
.{
|
||||
.name = .device_manager,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x7092fc24905cd147, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.device = .{ .path = "../../../library/device" },
|
||||
.protocol = .{ .path = "../../../library/protocol" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -0,0 +1,15 @@
|
||||
//! The display-demo service as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! the shared recipe and the module-to-domain map live in build-support.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "display-demo",
|
||||
.root_source_file = b.path("display-demo.zig"),
|
||||
.imports = &.{ "display-client", "logging", "time" },
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
}
|
||||
@@ -0,0 +1,15 @@
|
||||
.{
|
||||
.name = .display_demo,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x5d2832e1e2880143, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.client = .{ .path = "../../../library/client" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -11,7 +11,7 @@
|
||||
//! is deliberately independent of the mouse.
|
||||
|
||||
|
||||
const display = @import("display");
|
||||
const display = @import("display-client");
|
||||
const time = @import("time");
|
||||
const logging = @import("logging");
|
||||
pub fn main() void {
|
||||
|
||||
@@ -0,0 +1,33 @@
|
||||
//! The display service as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! the shared recipe and the module-to-domain map live in build-support.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "display",
|
||||
.root_source_file = b.path("display.zig"),
|
||||
.imports = &.{
|
||||
"display-client", "display-protocol", "driver", "input-client", "ipc", "logging",
|
||||
"memory", "scanout-protocol", "service", "thread", "time",
|
||||
},
|
||||
.threaded = true, // real atomics/TLS (docs/threading.md)
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
|
||||
// Standalone `zig build test`; the root aggregate depends on this step.
|
||||
const test_step = b.step("test", "Run the display unit tests");
|
||||
for ([_][]const u8{
|
||||
"compositor.zig", // Rect math + fill/composite/blit-tile
|
||||
}) |test_root| {
|
||||
const unit_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(test_root),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(unit_tests).step);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,17 @@
|
||||
.{
|
||||
.name = .display,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0xcd172a34b127a19, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.client = .{ .path = "../../../library/client" },
|
||||
.device = .{ .path = "../../../library/device" },
|
||||
.protocol = .{ .path = "../../../library/protocol" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -17,11 +17,11 @@
|
||||
|
||||
const std = @import("std");
|
||||
const ipc = @import("ipc");
|
||||
const input = @import("input");
|
||||
const input = @import("input-client");
|
||||
const Thread = @import("thread").Thread;
|
||||
const service = @import("service");
|
||||
const time = @import("time");
|
||||
const display = @import("display");
|
||||
const display = @import("display-client");
|
||||
const memory = @import("memory");
|
||||
const logging = @import("logging");
|
||||
const compositor = @import("compositor.zig");
|
||||
|
||||
@@ -0,0 +1,33 @@
|
||||
//! The fat service as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! the shared recipe and the module-to-domain map live in build-support.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "fat",
|
||||
.root_source_file = b.path("fat.zig"),
|
||||
.imports = &.{
|
||||
"block", "file-system", "ipc", "logging", "memory", "process", "service", "time",
|
||||
"vfs-protocol",
|
||||
},
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
|
||||
// Standalone `zig build test`; the root aggregate depends on this step.
|
||||
const test_step = b.step("test", "Run the fat unit tests");
|
||||
for ([_][]const u8{
|
||||
"on-disk.zig", // FAT on-disk struct sizes + type detection
|
||||
"engine.zig", // FAT read/write over a RAM-backed image
|
||||
}) |test_root| {
|
||||
const unit_tests = b.addTest(.{
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path(test_root),
|
||||
.target = b.resolveTargetQuery(.{}),
|
||||
}),
|
||||
});
|
||||
test_step.dependOn(&b.addRunArtifact(unit_tests).step);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
.{
|
||||
.name = .fat,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x98958143ecfe1636, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.device = .{ .path = "../../../library/device" },
|
||||
.protocol = .{ .path = "../../../library/protocol" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -118,7 +118,17 @@ fn tryBringUp() void {
|
||||
_ = logging.write("/system/services/fat: block geometry unavailable\n");
|
||||
return;
|
||||
};
|
||||
const bounce = memory.dmaAlloc(engine.max_transfer_sectors * 512, memory.dma_coherent) orelse return;
|
||||
// Shareable so the buffer's capability can be attached down the chain (block server
|
||||
// -> controller), making its physical addresses reachable by the device under an
|
||||
// enforcing IOMMU. No-op binding otherwise.
|
||||
const bounce = memory.dmaAlloc(engine.max_transfer_sectors * 512, memory.dma_coherent | memory.dma_shareable) orelse return;
|
||||
if (bounce.handle) |handle| {
|
||||
if (!device.attach(handle)) {
|
||||
_ = logging.write("/system/services/fat: could not attach the DMA bounce buffer\n");
|
||||
return;
|
||||
}
|
||||
_ = ipc.close(handle); // the binding holds its own reference now
|
||||
}
|
||||
ipc_block = .{ .device = device, .bounce = bounce };
|
||||
|
||||
const block_device = engine.BlockDevice{
|
||||
|
||||
@@ -0,0 +1,20 @@
|
||||
//! The fdt service as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! the shared recipe and the module-to-domain map live in build-support.
|
||||
//!
|
||||
//! The artifact is named "discovery" like acpi's: the Raspberry Pis hand over
|
||||
//! a flattened device tree, and the aarch64 target flips the root's
|
||||
//! -Ddiscovery default when it lands (docs/arm.md). A placeholder until the
|
||||
//! ARM bring-up.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "discovery",
|
||||
.root_source_file = b.path("fdt.zig"),
|
||||
.imports = &.{ "process" },
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
.{
|
||||
.name = .fdt,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0xe5e27506c7fc2eea, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -0,0 +1,28 @@
|
||||
//! init (PID 1) as a binary package (docs/build-packages-plan.md): this file
|
||||
//! names the binary and EXACTLY the modules its source imports — the shared
|
||||
//! recipe and the module-to-domain map live in build-support. The root build
|
||||
//! consumes the artifact for the boot image and forwards its -Dserial here.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "init",
|
||||
.root_source_file = b.path("init.zig"),
|
||||
.imports = &.{
|
||||
"csv", "file-system", "ipc", "logging", "memory", "power-protocol",
|
||||
"process", "time",
|
||||
},
|
||||
});
|
||||
// init reads the same `serial` flag the kernel does: its liveness heartbeat
|
||||
// is a serial/test-build diagnostic (the QEMU harness's init tests assert
|
||||
// on it, and -Dserial images emit it), so a flashable image runs a purely
|
||||
// event-driven PID 1 that wakes only for real work. The root build forwards
|
||||
// its top-level -Dserial as this dependency option.
|
||||
const serial = b.option(bool, "serial", "Compile the serial liveness heartbeat in (forwarded from the root -Dserial)") orelse false;
|
||||
const init_options = b.addOptions();
|
||||
init_options.addOption(bool, "serial", serial);
|
||||
build_support.programModule(exe).addImport("build_options", init_options.createModule());
|
||||
b.installArtifact(exe);
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
.{
|
||||
.name = .init,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0xc674e474eeeced43, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.csv = .{ .path = "../../../library/csv" },
|
||||
.protocol = .{ .path = "../../../library/protocol" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -0,0 +1,15 @@
|
||||
//! The input service as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! the shared recipe and the module-to-domain map live in build-support.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "input",
|
||||
.root_source_file = b.path("input.zig"),
|
||||
.imports = &.{ "input-client", "input-protocol", "ipc", "logging", "process", "service" },
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
.{
|
||||
.name = .input,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0xd82832d7113ed94e, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
.client = .{ .path = "../../../library/client" },
|
||||
.protocol = .{ .path = "../../../library/protocol" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
@@ -23,7 +23,7 @@ const std = @import("std");
|
||||
const ipc = @import("ipc");
|
||||
const process = @import("process");
|
||||
const service = @import("service");
|
||||
const input = @import("input");
|
||||
const input = @import("input-client");
|
||||
const logging = @import("logging");
|
||||
const input_protocol = @import("input-protocol");
|
||||
|
||||
|
||||
@@ -0,0 +1,15 @@
|
||||
//! The logger service as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! the shared recipe and the module-to-domain map live in build-support.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "logger",
|
||||
.root_source_file = b.path("logger.zig"),
|
||||
.imports = &.{ "file-system", "ipc", "logging", "service", "time" },
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
.{
|
||||
.name = .logger,
|
||||
.version = "0.0.0",
|
||||
.fingerprint = 0x987e13f37b0eaea2, // Changing this has security and trust implications.
|
||||
.minimum_zig_version = "0.16.0",
|
||||
.dependencies = .{
|
||||
// build-support supplies the shared recipe; kernel is implicit in
|
||||
// every binary (the root shim + link script live there). The rest
|
||||
// are exactly the homes of this binary's declared imports.
|
||||
.@"build-support" = .{ .path = "../../../build-support" },
|
||||
.kernel = .{ .path = "../../../library/kernel" },
|
||||
},
|
||||
.paths = .{""},
|
||||
}
|
||||
+67
-5
@@ -146,20 +146,70 @@ CASES = [
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# DMA memory (M14): contiguous frame allocation, below-4G cap, coherent mapping,
|
||||
# and reclaim on teardown.
|
||||
# Boots with no IOMMU device, so it also asserts the explicit fail-open boot line
|
||||
# (the DMA-isolation posture must be stated, never silent).
|
||||
{"name": "dma",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"expect": r"(?s)(?=.*iommu : none present - DMA fail-open \(unisolated\))(?=.*DANOS-TEST-RESULT: PASS)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# MSI (M15): allocate a per-device vector and deliver it as a notification (a
|
||||
# self-IPI stands in for the device's MSI write, since the HPET has no MSI).
|
||||
{"name": "msi",
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
# IOMMU (M16): boot with an emulated VT-d unit and confirm danos parses the DMAR
|
||||
# table and reads the unit's registers. Detection only — enforcement is future.
|
||||
# IOMMU: boot with an emulated VT-d unit; danos parses the DMAR table, enables
|
||||
# translation, and proves the domain walker (map/resolve/unmap on a scratch domain)
|
||||
# with no spurious faults. The `enabled` line is a lookahead so a silently-dead unit
|
||||
# cannot fake a pass.
|
||||
{"name": "iommu",
|
||||
"qemu_extra": ["-device", "intel-iommu,intremap=off"],
|
||||
"expect": r"DANOS-TEST-RESULT: PASS",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL"},
|
||||
"expect": r"(?s)(?=.*DANOS-IOMMU: enabled base=0x[0-9a-f]+ domains active)(?=.*DANOS-TEST-RESULT: PASS)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||
# DMA under translation: the full USB storage stack (xHC ring DMA + BOT/SCSI + fat's
|
||||
# cross-process bounce buffer) runs with VT-d enabled. Every device is identity-
|
||||
# mapped in the blanket domain, so DMA works, but through real second-level walks.
|
||||
{"name": "iommu-usb-storage",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "intel-iommu,intremap=off"],
|
||||
"expect": r"(?s)(?=.*/system/kernel: iommu online)(?=.*fat: mounted /mnt/usb)(?=.*fat-test: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||
# DMA + MSI under translation: interrupt-IN reports arrive through translated DMA and
|
||||
# the xHC's MSI/MSI-X still delivers (the 0xFEE00000 interrupt window bypasses second-
|
||||
# level translation with interrupt remapping off).
|
||||
{"name": "iommu-usb-hid",
|
||||
"build_case": "usb-hid",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "intel-iommu,intremap=off"],
|
||||
"expect": r"(?s)(?=.*/system/kernel: iommu online)(?=.*usb-hid-keyboard: ok)(?=.*usb-hid-mouse: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||
# Enforcement, the negative proof: a claimed e1000e fires a DMA at an unmapped page;
|
||||
# VT-d must fault it (logged) and the system must stay alive. The fault line is the
|
||||
# point here, so unlike the positive cases it appears in `expect`, not `fail`.
|
||||
{"name": "iommu-fault",
|
||||
"smp": 4,
|
||||
"timeout": 120,
|
||||
"qemu_extra": ["-device", "intel-iommu,intremap=off", "-device", "e1000e"],
|
||||
"expect": r"(?s)(?=.*DANOS-IOMMU-FAULT: bdf=)(?=.*iommu-fault-test: system alive)(?=.*DANOS-TEST-RESULT: PASS)",
|
||||
"fail": r"iommu-fault-test: FAIL|DANOS-TEST-RESULT: FAIL|CPU EXCEPTION|KERNEL PANIC"},
|
||||
# AMD-Vi: the same detection + scratch-domain walker proof as the `iommu` case, but on
|
||||
# the AMD backend (IVRS parse, device table, command buffer). QEMU's amd-iommu needs
|
||||
# dma-remap=on (default off = translation silently ignored). UNTESTED on real AMD.
|
||||
{"name": "amd-iommu",
|
||||
"build_case": "iommu",
|
||||
"qemu_extra": ["-device", "amd-iommu,dma-remap=on,intremap=off"],
|
||||
"expect": r"(?s)(?=.*iommu online \(AMD-Vi\))(?=.*DANOS-IOMMU: enabled base=0x[0-9a-f]+ domains active)(?=.*DANOS-TEST-RESULT: PASS)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||
# DMA under AMD-Vi translation: the full USB storage stack through AMD device-table
|
||||
# translation. Tentative — land depends on QEMU amd-iommu behaving. UNTESTED on real AMD.
|
||||
{"name": "amd-iommu-usb-storage",
|
||||
"build_case": "fat-mount",
|
||||
"smp": 4,
|
||||
"timeout": 150,
|
||||
"qemu_extra": ["-device", "amd-iommu,dma-remap=on,intremap=off"],
|
||||
"expect": r"(?s)(?=.*iommu online \(AMD-Vi\))(?=.*fat: mounted /mnt/usb)(?=.*fat-test: ok)",
|
||||
"fail": r"DANOS-TEST-RESULT: FAIL|DANOS-IOMMU-FAULT"},
|
||||
# Port I/O grants: a claimed device's io_port resource lets a driver read/write its
|
||||
# ports (PS/2 status 0x64), gated by the claim; out-of-range/unclaimed is refused.
|
||||
{"name": "ioport",
|
||||
@@ -687,6 +737,18 @@ CASES = [
|
||||
# duplicates); this ordered regex asserts the drill itself over the whole serial
|
||||
# log — the backreference requires the respawn to re-scan the same count, and the
|
||||
# full-capture match is immune to the transient-line races an in-kernel poll hits.
|
||||
# The driver-side PCI library (library/device/pci) against a real function: an extra
|
||||
# e1000e NIC — PM + MSI + PCIe + MSI-X capabilities, claimed by no danos driver — is
|
||||
# claimed by the pci-cap-test fixture, which exercises the capability walks, MSI
|
||||
# programming (the first driver-side msi_bind use), the MSI-X table, power state,
|
||||
# and FLR, printing a marker per check.
|
||||
{"name": "pci-caps",
|
||||
"smp": 4,
|
||||
"timeout": 120,
|
||||
"qemu_extra": ["-device", "e1000e"],
|
||||
"expect": r"(?s)(?=.*DANOS-TEST-RESULT: PASS)(?=.*pci-cap-test: all checks passed)",
|
||||
"fail": r"pci-cap-test: FAIL|DANOS-TEST-RESULT: FAIL"},
|
||||
|
||||
{"name": "pci-scan",
|
||||
"smp": 4,
|
||||
"timeout": 60,
|
||||
|
||||
@@ -0,0 +1,15 @@
|
||||
//! The args-echo test fixture as a binary package (docs/build-packages-plan.md):
|
||||
//! this file names the binary and EXACTLY the modules its source imports —
|
||||
//! the shared recipe and the module-to-domain map live in build-support.
|
||||
|
||||
const std = @import("std");
|
||||
const build_support = @import("build-support");
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
const exe = build_support.userBinary(b, .{
|
||||
.name = "args-echo",
|
||||
.root_source_file = b.path("args-echo.zig"),
|
||||
.imports = &.{ "logging", "process" },
|
||||
});
|
||||
b.installArtifact(exe);
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user